kordoc 4.0.8 → 4.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (88) hide show
  1. package/README.md +25 -11
  2. package/dist/{-LE4RLXVA.js → -AS4IABL2.js} +23 -10
  3. package/dist/{chunk-2DZQF6YJ.js → chunk-37HJHCPA.js} +24 -15
  4. package/dist/chunk-37HJHCPA.js.map +1 -0
  5. package/dist/{chunk-6XAAYAWL.cjs → chunk-37VKMUST.cjs} +47 -37
  6. package/dist/chunk-37VKMUST.cjs.map +1 -0
  7. package/dist/{chunk-OQPILS7B.js → chunk-4KI23VHA.js} +33 -23
  8. package/dist/chunk-4KI23VHA.js.map +1 -0
  9. package/dist/{chunk-D35VBACN.js → chunk-AX2R5Q2N.js} +7 -4
  10. package/dist/chunk-AX2R5Q2N.js.map +1 -0
  11. package/dist/chunk-DZIXKL7E.js +145 -0
  12. package/dist/chunk-DZIXKL7E.js.map +1 -0
  13. package/dist/chunk-F2ZU3IZG.js +82 -0
  14. package/dist/chunk-F2ZU3IZG.js.map +1 -0
  15. package/dist/{chunk-BNU5QGIZ.js → chunk-GQVSHGG4.js} +12 -4
  16. package/dist/chunk-GQVSHGG4.js.map +1 -0
  17. package/dist/{chunk-7S3M4N4E.js → chunk-L2XQQTOY.js} +566 -189
  18. package/dist/chunk-L2XQQTOY.js.map +1 -0
  19. package/dist/{chunk-HMUEU7D7.js → chunk-O5VS7RIR.js} +2 -2
  20. package/dist/{chunk-UEIYETCQ.js → chunk-SSTK6IKK.js} +18 -160
  21. package/dist/chunk-SSTK6IKK.js.map +1 -0
  22. package/dist/chunk-UOO7LSDG.js +152 -0
  23. package/dist/chunk-UOO7LSDG.js.map +1 -0
  24. package/dist/{chunk-RROH5WHO.js → chunk-WO52HVQJ.js} +2 -2
  25. package/dist/{chunk-MEPHGCPQ.js → chunk-YOO6ET6M.js} +7 -3
  26. package/dist/chunk-YOO6ET6M.js.map +1 -0
  27. package/dist/chunks-WVMGF7JL.js +10 -0
  28. package/dist/cli.js +170 -35
  29. package/dist/cli.js.map +1 -1
  30. package/dist/{detect-RI2MQ33K.js → detect-YEZX2XCF.js} +2 -2
  31. package/dist/index.cjs +1142 -531
  32. package/dist/index.cjs.map +1 -1
  33. package/dist/index.d.cts +105 -1
  34. package/dist/index.d.ts +105 -1
  35. package/dist/index.js +802 -191
  36. package/dist/index.js.map +1 -1
  37. package/dist/mcp.js +273 -65
  38. package/dist/mcp.js.map +1 -1
  39. package/dist/{parser-ITMU2WN5.cjs → parser-N4A7UGDK.cjs} +127 -45
  40. package/dist/parser-N4A7UGDK.cjs.map +1 -0
  41. package/dist/{parser-TCTCSBYZ.js → parser-PEOYBXBO.js} +118 -34
  42. package/dist/parser-PEOYBXBO.js.map +1 -0
  43. package/dist/{parser-KMMUS73Q.js → parser-XX4QTDFQ.js} +114 -32
  44. package/dist/parser-XX4QTDFQ.js.map +1 -0
  45. package/dist/{profile-io-SLN4P76T.js → profile-io-PLDQFPJA.js} +3 -3
  46. package/dist/{provider-G4C2V2PD.cjs → provider-4FAYHJ6N.cjs} +3 -3
  47. package/dist/provider-4FAYHJ6N.cjs.map +1 -0
  48. package/dist/{provider-4ZJKV3DC.js → provider-C6IOGIS5.js} +3 -3
  49. package/dist/provider-C6IOGIS5.js.map +1 -0
  50. package/dist/{provider-AKROB7WQ.js → provider-UIBW2MIH.js} +3 -3
  51. package/dist/provider-UIBW2MIH.js.map +1 -0
  52. package/dist/rasterize-WWNQLKYW.js +40 -0
  53. package/dist/rasterize-WWNQLKYW.js.map +1 -0
  54. package/dist/redact-7ILEAUTS.js +12 -0
  55. package/dist/render-T7S6H6AN.js +10 -0
  56. package/dist/render-T7S6H6AN.js.map +1 -0
  57. package/dist/seal-7PTL6MXH.js +10 -0
  58. package/dist/seal-7PTL6MXH.js.map +1 -0
  59. package/dist/{setup-57FB3LSP.js → setup-Q2PRE7UA.js} +5 -3
  60. package/dist/setup-Q2PRE7UA.js.map +1 -0
  61. package/dist/{watch-OHQWSVPE.js → watch-SFHXZALC.js} +77 -23
  62. package/dist/watch-SFHXZALC.js.map +1 -0
  63. package/package.json +1 -1
  64. package/dist/chunk-2DZQF6YJ.js.map +0 -1
  65. package/dist/chunk-6XAAYAWL.cjs.map +0 -1
  66. package/dist/chunk-7S3M4N4E.js.map +0 -1
  67. package/dist/chunk-BNU5QGIZ.js.map +0 -1
  68. package/dist/chunk-D35VBACN.js.map +0 -1
  69. package/dist/chunk-MEPHGCPQ.js.map +0 -1
  70. package/dist/chunk-OQPILS7B.js.map +0 -1
  71. package/dist/chunk-UEIYETCQ.js.map +0 -1
  72. package/dist/parser-ITMU2WN5.cjs.map +0 -1
  73. package/dist/parser-KMMUS73Q.js.map +0 -1
  74. package/dist/parser-TCTCSBYZ.js.map +0 -1
  75. package/dist/provider-4ZJKV3DC.js.map +0 -1
  76. package/dist/provider-AKROB7WQ.js.map +0 -1
  77. package/dist/provider-G4C2V2PD.cjs.map +0 -1
  78. package/dist/render-25BT623I.js +0 -10
  79. package/dist/seal-R4TETA4C.js +0 -10
  80. package/dist/setup-57FB3LSP.js.map +0 -1
  81. package/dist/watch-OHQWSVPE.js.map +0 -1
  82. /package/dist/{-LE4RLXVA.js.map → -AS4IABL2.js.map} +0 -0
  83. /package/dist/{chunk-HMUEU7D7.js.map → chunk-O5VS7RIR.js.map} +0 -0
  84. /package/dist/{chunk-RROH5WHO.js.map → chunk-WO52HVQJ.js.map} +0 -0
  85. /package/dist/{detect-RI2MQ33K.js.map → chunks-WVMGF7JL.js.map} +0 -0
  86. /package/dist/{profile-io-SLN4P76T.js.map → detect-YEZX2XCF.js.map} +0 -0
  87. /package/dist/{render-25BT623I.js.map → profile-io-PLDQFPJA.js.map} +0 -0
  88. /package/dist/{seal-R4TETA4C.js.map → redact-7ILEAUTS.js.map} +0 -0
@@ -3,22 +3,14 @@ import {
3
3
  HEADING_RATIO_H1,
4
4
  HEADING_RATIO_H2,
5
5
  HEADING_RATIO_H3,
6
- MAX_COLS,
7
- MAX_ROWS,
8
- blocksToMarkdown,
9
- buildTable,
10
- convertTableToText,
11
- dedupeRunningHeaders,
12
- flattenLayoutTables,
13
- inlineImagesIntoMarkdown,
14
- mapPuaText
15
- } from "./chunk-UEIYETCQ.js";
6
+ inlineImagesIntoMarkdown
7
+ } from "./chunk-UOO7LSDG.js";
16
8
  import {
17
9
  detectFormat,
18
10
  detectOle2Format,
19
11
  detectZipFormat,
20
12
  parseLenientCfb
21
- } from "./chunk-MEPHGCPQ.js";
13
+ } from "./chunk-YOO6ET6M.js";
22
14
  import {
23
15
  parsePageRange
24
16
  } from "./chunk-MOL7MDBG.js";
@@ -33,25 +25,36 @@ import {
33
25
  paraTextPureT,
34
26
  patchZipEntries,
35
27
  scanSectionXml
36
- } from "./chunk-D35VBACN.js";
28
+ } from "./chunk-AX2R5Q2N.js";
29
+ import {
30
+ MAX_COLS,
31
+ MAX_ROWS,
32
+ blocksToMarkdown,
33
+ buildTable,
34
+ convertTableToText,
35
+ dedupeRunningHeaders,
36
+ flattenLayoutTables,
37
+ mapPuaText
38
+ } from "./chunk-SSTK6IKK.js";
37
39
  import {
38
40
  SPACE_EM_FIXED,
39
41
  charWidthEm1000,
40
42
  fitRatioForFewerLines,
41
43
  measureTextWidth,
42
44
  simulateWrap
43
- } from "./chunk-2DZQF6YJ.js";
45
+ } from "./chunk-37HJHCPA.js";
44
46
  import {
45
47
  MAX_DECOMPRESS_SIZE,
46
48
  MAX_XML_DEPTH,
47
49
  MAX_ZIP_ENTRIES,
50
+ ZipBombError,
48
51
  applyPageText,
49
52
  clampSpan,
50
53
  createSectionShared,
51
54
  createXmlParser,
52
55
  extractTextFromNode,
53
56
  findChildByLocalName
54
- } from "./chunk-BNU5QGIZ.js";
57
+ } from "./chunk-GQVSHGG4.js";
55
58
  import {
56
59
  KordocError,
57
60
  classifyError,
@@ -59,10 +62,11 @@ import {
59
62
  isPathTraversal,
60
63
  normalizeSectionHref,
61
64
  precheckZipSize,
65
+ sanitizeError,
62
66
  sanitizeHref,
63
67
  stripDtd,
64
68
  toArrayBuffer
65
- } from "./chunk-HMUEU7D7.js";
69
+ } from "./chunk-O5VS7RIR.js";
66
70
 
67
71
  // src/index.ts
68
72
  import { readFile } from "fs/promises";
@@ -1277,6 +1281,7 @@ function toRoman(n) {
1277
1281
  return out;
1278
1282
  }
1279
1283
  function formatHeadNumber(n, numFormat) {
1284
+ if (n === 0 && numFormat === "DIGIT") return "0";
1280
1285
  if (n <= 0) n = 1;
1281
1286
  switch (numFormat) {
1282
1287
  case "DIGIT":
@@ -1334,25 +1339,25 @@ function resolveParaHeading(paraEl, ctx) {
1334
1339
  if (!numDef) return headingLevel ? { headingLevel } : null;
1335
1340
  let counters = ctx.shared.numState.get(numId);
1336
1341
  if (!counters) {
1337
- counters = new Array(11).fill(0);
1342
+ counters = new Array(11).fill(-1);
1338
1343
  ctx.shared.numState.set(numId, counters);
1339
1344
  }
1340
1345
  const head = numDef.heads.get(level);
1341
- counters[level] = counters[level] === 0 ? head?.start ?? 1 : counters[level] + 1;
1342
- for (let l = level + 1; l <= 10; l++) counters[l] = 0;
1346
+ counters[level] = counters[level] < 0 ? head?.start ?? 1 : counters[level] + 1;
1347
+ for (let l = level + 1; l <= 10; l++) counters[l] = -1;
1343
1348
  const fmtText = head ? head.text.trim() : `^${level}.`;
1344
1349
  const prefix = fmtText.replace(/\^(10|[1-9])/g, (_, d) => {
1345
1350
  const lv = parseInt(d, 10);
1346
1351
  const refHead = numDef.heads.get(lv);
1347
- const n = counters[lv] || refHead?.start || 1;
1352
+ const n = counters[lv] >= 0 ? counters[lv] : refHead?.start ?? 1;
1348
1353
  return formatHeadNumber(n, refHead?.numFormat || "DIGIT");
1349
1354
  });
1350
1355
  return { prefix: prefix || void 0, headingLevel };
1351
1356
  }
1352
1357
 
1353
1358
  // src/hwpx/table-build.ts
1354
- function buildTableWithCellMeta(state) {
1355
- const table2 = buildTable(state.rows);
1359
+ function buildTableWithCellMeta(state, keepAnchoredEmptyCols) {
1360
+ const table2 = buildTable(state.rows, { keepAnchoredEmptyCols });
1356
1361
  if (state.caption) table2.caption = state.caption;
1357
1362
  const anchors = [];
1358
1363
  {
@@ -1416,7 +1421,7 @@ function completeTable(newTable, tableStack, blocks, ctx) {
1416
1421
  if (newTable.caption) blocks.push({ type: "paragraph", text: newTable.caption, pageNumber: ctx.sectionNum });
1417
1422
  return parentTable;
1418
1423
  }
1419
- const ir = buildTableWithCellMeta(newTable);
1424
+ const ir = buildTableWithCellMeta(newTable, ctx.shared.keepTrailingEmptyCols);
1420
1425
  const block = { type: "table", table: ir, pageNumber: ctx.sectionNum };
1421
1426
  if (parentTable?.cell) {
1422
1427
  const cell2 = parentTable.cell;
@@ -1450,7 +1455,8 @@ function parseSectionXml(xml, styleMap, warnings, sectionNum, shared) {
1450
1455
  walkSection(doc.documentElement, blocks, null, [], ctx);
1451
1456
  return blocks;
1452
1457
  }
1453
- function extractImageRef(el) {
1458
+ function extractImageRef(el, depth = 0) {
1459
+ if (depth > MAX_XML_DEPTH) return null;
1454
1460
  const children = el.childNodes;
1455
1461
  if (!children) return null;
1456
1462
  for (let i = 0; i < children.length; i++) {
@@ -1461,7 +1467,7 @@ function extractImageRef(el) {
1461
1467
  const ref = child.getAttribute("binaryItemIDRef") || child.getAttribute("href") || "";
1462
1468
  if (ref) return ref;
1463
1469
  }
1464
- const nested = extractImageRef(child);
1470
+ const nested = extractImageRef(child, depth + 1);
1465
1471
  if (nested) return nested;
1466
1472
  }
1467
1473
  const directRef = el.getAttribute("binaryItemIDRef") || "";
@@ -1568,7 +1574,7 @@ function walkSection(node, blocks, tableCtx, tableStack, ctx, depth = 0) {
1568
1574
  ;
1569
1575
  (cell2.blocks ??= []).push(cellBlock);
1570
1576
  } else if (!tableCtx) {
1571
- if (/^─{10,}$/.test(text)) {
1577
+ if (ctx.shared.kordocLayout && /^─{10,}$/.test(text)) {
1572
1578
  blocks.push({ type: "separator", pageNumber: ctx.sectionNum });
1573
1579
  tableCtx = walkParagraphChildren(el, blocks, tableCtx, tableStack, ctx, depth + 1);
1574
1580
  break;
@@ -1844,8 +1850,8 @@ function extractDrawTextBlocks(drawTextNode, blocks, ctx) {
1844
1850
  } else {
1845
1851
  const info = extractParagraphInfo(child, ctx.styleMap, ctx);
1846
1852
  let text = info.text.trim();
1853
+ const ph = resolveParaHeading(child, ctx);
1847
1854
  if (text) {
1848
- const ph = resolveParaHeading(child, ctx);
1849
1855
  if (ph?.prefix) text = ph.prefix + " " + text;
1850
1856
  const block = { type: "paragraph", text, style: info.style ?? void 0, pageNumber: ctx.sectionNum };
1851
1857
  if (info.href) block.href = info.href;
@@ -1958,7 +1964,8 @@ function extractParagraphInfo(para2, styleMap, ctx) {
1958
1964
  }
1959
1965
  }
1960
1966
  };
1961
- const walk = (node) => {
1967
+ const walk = (node, depth = 0) => {
1968
+ if (depth > MAX_XML_DEPTH) return;
1962
1969
  const children = node.childNodes;
1963
1970
  if (!children) return;
1964
1971
  for (let i = 0; i < children.length; i++) {
@@ -1979,7 +1986,7 @@ function extractParagraphInfo(para2, styleMap, ctx) {
1979
1986
  const tag = (child.tagName || child.localName || "").replace(/^[^:]+:/, "");
1980
1987
  switch (tag) {
1981
1988
  case "t":
1982
- walk(child);
1989
+ walk(child, depth + 1);
1983
1990
  break;
1984
1991
  // 자식 순회 (tab 등 하위 요소 처리)
1985
1992
  case "tab": {
@@ -2012,7 +2019,7 @@ function extractParagraphInfo(para2, styleMap, ctx) {
2012
2019
  const safe = sanitizeHref(url);
2013
2020
  if (safe) href = safe;
2014
2021
  }
2015
- walk(child);
2022
+ walk(child, depth + 1);
2016
2023
  break;
2017
2024
  }
2018
2025
  // 각주/미주
@@ -2083,11 +2090,11 @@ function extractParagraphInfo(para2, styleMap, ctx) {
2083
2090
  case "r": {
2084
2091
  const runCharPr = child.getAttribute("charPrIDRef");
2085
2092
  if (runCharPr && !charPrId) charPrId = runCharPr;
2086
- walk(child);
2093
+ walk(child, depth + 1);
2087
2094
  break;
2088
2095
  }
2089
2096
  default:
2090
- walk(child);
2097
+ walk(child, depth + 1);
2091
2098
  break;
2092
2099
  }
2093
2100
  }
@@ -2098,7 +2105,7 @@ function extractParagraphInfo(para2, styleMap, ctx) {
2098
2105
  let cleanText = text.replace(/[ \t]+/g, " ").trim();
2099
2106
  if (/^그림입니다\.?\s*원본\s*그림의\s*(이름|크기)/.test(cleanText)) cleanText = "";
2100
2107
  cleanText = cleanText.replace(/그림입니다\.?\s*원본\s*그림의\s*(이름|크기)[^\n]*(\n[^\n]*원본\s*그림의\s*(이름|크기)[^\n]*)*/g, "").trim();
2101
- cleanText = cleanText.replace(/(?:모서리가 둥근 |둥근 )?(?:사각형|직사각형|정사각형|원|타원|삼각형|선|직선|곡선|화살표|오각형|육각형|팔각형|별|십자|구름|마름모|도넛|평행사변형|사다리꼴|개체|그리기\s?개체|묶음\s?개체|글상자|표|그림|OLE\s?개체)\s?입니다\.?/g, "").trim();
2108
+ cleanText = cleanText.replace(/^(?:모서리가 둥근 |둥근 )?(?:사각형|직사각형|정사각형|원|타원|삼각형|선|직선|곡선|화살표|오각형|육각형|팔각형|별|십자|구름|마름모|도넛|평행사변형|사다리꼴|개체|그리기\s?개체|묶음\s?개체|글상자|표|그림|OLE\s?개체)\s?입니다\.?$/gm, "").trim();
2102
2109
  let style;
2103
2110
  if (styleMap && charPrId) {
2104
2111
  const charProp = styleMap.charProperties.get(charPrId);
@@ -2222,10 +2229,9 @@ var CHAR_LINE = 0;
2222
2229
  var CHAR_SECTION_BREAK = 10;
2223
2230
  var CHAR_PARA = 13;
2224
2231
  var CHAR_TAB = 9;
2225
- var CHAR_HYPHEN = 30;
2226
- var CHAR_NBSP = 31;
2227
- var CHAR_FIXED_NBSP = 24;
2228
- var CHAR_FIXED_WIDTH = 25;
2232
+ var CHAR_HYPHEN = 24;
2233
+ var CHAR_NBSP = 30;
2234
+ var CHAR_FIXED_WIDTH = 31;
2229
2235
  var FLAG_COMPRESSED = 1 << 0;
2230
2236
  var FLAG_ENCRYPTED = 1 << 1;
2231
2237
  var FLAG_DISTRIBUTION = 1 << 2;
@@ -2424,9 +2430,6 @@ function appendParaText(state, data, resolveControl) {
2424
2430
  result += "-";
2425
2431
  break;
2426
2432
  case CHAR_NBSP:
2427
- result += " ";
2428
- break;
2429
- case CHAR_FIXED_NBSP:
2430
2433
  result += "\xA0";
2431
2434
  break;
2432
2435
  // 진짜 NBSP
@@ -2725,8 +2728,8 @@ async function extractImagesFromZip(zip, blocks, decompressed, warnings, sweepUn
2725
2728
  const ext = path.includes(".") ? path.split(".").pop() || "png" : "png";
2726
2729
  const mimeType = imageExtToMime(ext);
2727
2730
  imageIndex++;
2728
- const filename = `image_${String(imageIndex).padStart(3, "0")}.${mimeToExt(mimeType)}`;
2729
- img = { filename, data, mimeType };
2731
+ const filename2 = `image_${String(imageIndex).padStart(3, "0")}.${mimeToExt(mimeType)}`;
2732
+ img = { filename: filename2, data, mimeType };
2730
2733
  images.push(img);
2731
2734
  usedPaths.add(path);
2732
2735
  break;
@@ -2740,12 +2743,13 @@ async function extractImagesFromZip(zip, blocks, decompressed, warnings, sweepUn
2740
2743
  if (!img) {
2741
2744
  block.type = "paragraph";
2742
2745
  block.text = `[\uC774\uBBF8\uC9C0: ${ref}]`;
2743
- if (ownerCell) ownerCell.text = ownerCell.text.replace(`![image](${ref})`, `[\uC774\uBBF8\uC9C0: ${ref}]`);
2746
+ if (ownerCell) ownerCell.text = ownerCell.text.replace(`![image](${ref})`, () => `[\uC774\uBBF8\uC9C0: ${ref}]`);
2744
2747
  continue;
2745
2748
  }
2746
- block.text = img.filename;
2749
+ const filename = img.filename;
2750
+ block.text = filename;
2747
2751
  block.imageData = { data: img.data, mimeType: img.mimeType, filename: ref };
2748
- if (ownerCell) ownerCell.text = ownerCell.text.replace(`![image](${ref})`, `![image](${img.filename})`);
2752
+ if (ownerCell) ownerCell.text = ownerCell.text.replace(`![image](${ref})`, () => `![image](${filename})`);
2749
2753
  }
2750
2754
  if (sweepUnreferenced) {
2751
2755
  const binEntries = zip.file(/(?:^|\/)BinData\//i);
@@ -2991,6 +2995,7 @@ async function parseHwpxDocument(buffer, options) {
2991
2995
  const blocks = [];
2992
2996
  const shared = createSectionShared();
2993
2997
  shared.kordocLayout = await readKordocLayout(zip);
2998
+ shared.keepTrailingEmptyCols = options?.keepTrailingEmptyCols;
2994
2999
  let parsedSections = 0;
2995
3000
  for (let si = 0; si < sectionPaths.length; si++) {
2996
3001
  if (pageFilter && !pageFilter.has(si + 1)) continue;
@@ -2999,12 +3004,12 @@ async function parseHwpxDocument(buffer, options) {
2999
3004
  try {
3000
3005
  const xml = await file.async("text");
3001
3006
  decompressed.total += xml.length * 2;
3002
- if (decompressed.total > MAX_DECOMPRESS_SIZE) throw new KordocError("ZIP \uC555\uCD95 \uD574\uC81C \uD06C\uAE30 \uCD08\uACFC (ZIP bomb \uC758\uC2EC)");
3007
+ if (decompressed.total > MAX_DECOMPRESS_SIZE) throw new ZipBombError("ZIP \uC555\uCD95 \uD574\uC81C \uD06C\uAE30 \uCD08\uACFC (ZIP bomb \uC758\uC2EC)");
3003
3008
  blocks.push(...parseSectionXml(xml, styleMap, warnings, si + 1, shared));
3004
3009
  parsedSections++;
3005
3010
  options?.onProgress?.(parsedSections, totalTarget);
3006
3011
  } catch (secErr) {
3007
- if (secErr instanceof KordocError) throw secErr;
3012
+ if (secErr instanceof ZipBombError) throw secErr;
3008
3013
  warnings.push({ page: si + 1, message: `\uC139\uC158 ${si + 1} \uD30C\uC2F1 \uC2E4\uD328: ${secErr instanceof Error ? secErr.message : "\uC54C \uC218 \uC5C6\uB294 \uC624\uB958"}`, code: "PARTIAL_PARSE" });
3009
3014
  }
3010
3015
  }
@@ -4221,6 +4226,7 @@ function parseHwp5Document(buffer, options) {
4221
4226
  const totalTarget = pageFilter ? pageFilter.size : sections.length;
4222
4227
  const bodyBlocks = [];
4223
4228
  const doc = createHwp5DocState();
4229
+ doc.keepTrailingEmptyCols = options?.keepTrailingEmptyCols;
4224
4230
  let totalDecompressed = 0;
4225
4231
  let parsedSections = 0;
4226
4232
  for (let si = 0; si < sections.length; si++) {
@@ -4836,7 +4842,7 @@ function parseTableControl(ctrl, records, ctx) {
4836
4842
  return table3;
4837
4843
  }
4838
4844
  const cellRows = arrangeCells(rows, cols, cells);
4839
- const table2 = buildTable(cellRows);
4845
+ const table2 = buildTable(cellRows, { keepAnchoredEmptyCols: ctx.doc.keepTrailingEmptyCols });
4840
4846
  if (caption && table2.rows > 0) table2.caption = caption;
4841
4847
  return table2.rows > 0 ? table2 : null;
4842
4848
  }
@@ -17030,6 +17036,7 @@ function readHeader(reader) {
17030
17036
  }
17031
17037
 
17032
17038
  // src/hwp3/parser.ts
17039
+ var MAX_DECOMPRESS_SIZE3 = 100 * 1024 * 1024;
17033
17040
  var PARA_SHAPE_SIZE = 187;
17034
17041
  var LINE_INFO_SIZE = 14;
17035
17042
  var INLINE_CHAR_SHAPE_SIZE = 31;
@@ -17065,8 +17072,11 @@ function parseHwp3Document(buffer, _options) {
17065
17072
  const warnings = [];
17066
17073
  if (header.compressed !== 0) {
17067
17074
  try {
17068
- body = inflateRawSync3(tail);
17075
+ body = inflateRawSync3(tail, { maxOutputLength: MAX_DECOMPRESS_SIZE3 });
17069
17076
  } catch (err) {
17077
+ if (err?.code === "ERR_BUFFER_TOO_LARGE") {
17078
+ throw new Error(`HWP3 \uC555\uCD95 \uD574\uC81C \uACB0\uACFC\uAC00 \uCD5C\uB300 \uD5C8\uC6A9 \uD06C\uAE30(${MAX_DECOMPRESS_SIZE3 / 1024 / 1024}MB)\uB97C \uCD08\uACFC\uD588\uC2B5\uB2C8\uB2E4`);
17079
+ }
17070
17080
  const msg2 = err instanceof Error ? err.message : String(err);
17071
17081
  throw new Error(`HWP3 \uC555\uCD95 \uD574\uC81C \uC2E4\uD328: ${msg2}`);
17072
17082
  }
@@ -17263,7 +17273,7 @@ function isDistributionSentinel(markdown) {
17263
17273
  import JSZip3 from "jszip";
17264
17274
  import { DOMParser } from "@xmldom/xmldom";
17265
17275
  var MAX_SHEETS = 100;
17266
- var MAX_DECOMPRESS_SIZE3 = 100 * 1024 * 1024;
17276
+ var MAX_DECOMPRESS_SIZE4 = 100 * 1024 * 1024;
17267
17277
  var MAX_ROWS2 = 1e4;
17268
17278
  var MAX_COLS2 = 200;
17269
17279
  function cleanNumericValue(raw) {
@@ -17303,16 +17313,87 @@ function getTextContent(el) {
17303
17313
  function parseXml(text) {
17304
17314
  return new DOMParser().parseFromString(stripDtd(text), "text/xml");
17305
17315
  }
17316
+ function collectRichText(root) {
17317
+ let out = "";
17318
+ const walk = (node) => {
17319
+ const children = node.childNodes;
17320
+ for (let i = 0; i < children.length; i++) {
17321
+ if (children[i].nodeType !== 1) continue;
17322
+ const el = children[i];
17323
+ const local = el.localName || el.tagName?.replace(/^[^:]+:/, "") || "";
17324
+ if (local === "rPh") continue;
17325
+ if (local === "t") out += el.textContent ?? "";
17326
+ else walk(el);
17327
+ }
17328
+ };
17329
+ walk(root);
17330
+ return out;
17331
+ }
17306
17332
  function parseSharedStrings(xml) {
17307
17333
  const doc = parseXml(xml);
17308
17334
  const strings = [];
17309
17335
  const siList = getElements(doc.documentElement, "si");
17310
17336
  for (const si of siList) {
17311
- const tElements = getElements(si, "t");
17312
- strings.push(tElements.map((t) => t.textContent ?? "").join(""));
17337
+ strings.push(collectRichText(si));
17313
17338
  }
17314
17339
  return strings;
17315
17340
  }
17341
+ var BUILTIN_DATE_FMT = /* @__PURE__ */ new Map([
17342
+ [14, "date"],
17343
+ [15, "date"],
17344
+ [16, "date"],
17345
+ [17, "date"],
17346
+ [18, "datetime"],
17347
+ [19, "datetime"],
17348
+ [20, "datetime"],
17349
+ [21, "datetime"],
17350
+ [22, "datetime"],
17351
+ [45, "datetime"],
17352
+ [46, "datetime"],
17353
+ [47, "datetime"]
17354
+ ]);
17355
+ function classifyDateFormat(code) {
17356
+ const stripped = code.replace(/"[^"]*"/g, "").replace(/\[[^\]]*\]/g, "").replace(/\\./g, "");
17357
+ if (!/[ymdh]/i.test(stripped)) return null;
17358
+ return /[hs]/i.test(stripped) ? "datetime" : "date";
17359
+ }
17360
+ function dateKindOfFmt(fmtId, customFormats) {
17361
+ const builtin = BUILTIN_DATE_FMT.get(fmtId);
17362
+ if (builtin) return builtin;
17363
+ const code = customFormats.get(fmtId);
17364
+ return code !== void 0 ? classifyDateFormat(code) : null;
17365
+ }
17366
+ function dateSerialToIso(serial, date1904, kind) {
17367
+ if (!isFinite(serial) || serial < 0) return null;
17368
+ let days = serial;
17369
+ if (date1904) days += 1462;
17370
+ else if (days < 60) days += 1;
17371
+ const ms = Math.round((days - 25569) * 864e5);
17372
+ const d = new Date(ms);
17373
+ if (isNaN(d.getTime()) || d.getUTCFullYear() > 9999) return null;
17374
+ const iso = d.toISOString();
17375
+ return kind === "datetime" ? iso.slice(0, 19) : iso.slice(0, 10);
17376
+ }
17377
+ function parseStyleDateXfs(xml) {
17378
+ const doc = parseXml(xml);
17379
+ const customFormats = /* @__PURE__ */ new Map();
17380
+ for (const el of getElements(doc.documentElement, "numFmt")) {
17381
+ const id = parseInt(el.getAttribute("numFmtId") ?? "", 10);
17382
+ if (isNaN(id)) continue;
17383
+ customFormats.set(id, el.getAttribute("formatCode") ?? "");
17384
+ }
17385
+ const dateXfs = /* @__PURE__ */ new Map();
17386
+ const cellXfsEls = getElements(doc.documentElement, "cellXfs");
17387
+ if (cellXfsEls.length === 0) return dateXfs;
17388
+ const xfs = getElements(cellXfsEls[0], "xf");
17389
+ for (let i = 0; i < xfs.length; i++) {
17390
+ const fmtId = parseInt(xfs[i].getAttribute("numFmtId") ?? "", 10);
17391
+ if (isNaN(fmtId)) continue;
17392
+ const kind = dateKindOfFmt(fmtId, customFormats);
17393
+ if (kind) dateXfs.set(i, kind);
17394
+ }
17395
+ return dateXfs;
17396
+ }
17316
17397
  function parseWorkbook(xml) {
17317
17398
  const doc = parseXml(xml);
17318
17399
  const sheets = [];
@@ -17324,7 +17405,9 @@ function parseWorkbook(xml) {
17324
17405
  rId: el.getAttribute("r:id") ?? ""
17325
17406
  });
17326
17407
  }
17327
- return sheets;
17408
+ const prEls = getElements(doc.documentElement, "workbookPr");
17409
+ const d1904 = prEls.length > 0 ? prEls[0].getAttribute("date1904") : null;
17410
+ return { sheets, date1904: d1904 === "1" || d1904 === "true" };
17328
17411
  }
17329
17412
  function parseRels(xml) {
17330
17413
  const doc = parseXml(xml);
@@ -17337,21 +17420,25 @@ function parseRels(xml) {
17337
17420
  }
17338
17421
  return map;
17339
17422
  }
17340
- function parseWorksheet(xml, sharedStrings) {
17423
+ function parseWorksheet(xml, sharedStrings, dateXfs, date1904) {
17341
17424
  const doc = parseXml(xml);
17342
17425
  const grid = [];
17343
17426
  let maxRow = 0;
17344
17427
  let maxCol = 0;
17345
17428
  const rows = getElements(doc.documentElement, "row");
17429
+ let prevRow = -1;
17346
17430
  for (const rowEl of rows) {
17347
- const rowNum = parseInt(rowEl.getAttribute("r") ?? "0", 10) - 1;
17431
+ const rAttr = rowEl.getAttribute("r");
17432
+ const rowNum = rAttr !== null ? parseInt(rAttr, 10) - 1 : prevRow + 1;
17348
17433
  if (rowNum < 0 || rowNum >= MAX_ROWS2) continue;
17434
+ if (Number.isFinite(rowNum)) prevRow = rowNum;
17349
17435
  const cells = getElements(rowEl, "c");
17436
+ let prevCol = -1;
17350
17437
  for (const cellEl of cells) {
17351
17438
  const ref = cellEl.getAttribute("r");
17352
- if (!ref) continue;
17353
- const pos = parseCellRef(ref);
17354
- if (!pos || pos.col >= MAX_COLS2) continue;
17439
+ const pos = ref !== null ? parseCellRef(ref) : { col: prevCol + 1, row: rowNum };
17440
+ if (!pos || !Number.isFinite(pos.row) || pos.row < 0 || pos.row >= MAX_ROWS2 || pos.col >= MAX_COLS2) continue;
17441
+ prevCol = pos.col;
17355
17442
  const type = cellEl.getAttribute("t");
17356
17443
  const vElements = getElements(cellEl, "v");
17357
17444
  const fElements = getElements(cellEl, "f");
@@ -17365,12 +17452,19 @@ function parseWorksheet(xml, sharedStrings) {
17365
17452
  value = raw === "1" ? "TRUE" : "FALSE";
17366
17453
  } else {
17367
17454
  value = cleanNumericValue(raw);
17455
+ if (type === null || type === "n") {
17456
+ const sAttr = cellEl.getAttribute("s");
17457
+ const kind = sAttr !== null ? dateXfs.get(parseInt(sAttr, 10)) : void 0;
17458
+ if (kind) {
17459
+ const iso = dateSerialToIso(parseFloat(raw), date1904, kind);
17460
+ if (iso) value = iso;
17461
+ }
17462
+ }
17368
17463
  }
17369
17464
  } else if (type === "inlineStr") {
17370
17465
  const isEl = getElements(cellEl, "is");
17371
17466
  if (isEl.length > 0) {
17372
- const tElements = getElements(isEl[0], "t");
17373
- value = tElements.map((t) => t.textContent ?? "").join("");
17467
+ value = collectRichText(isEl[0]);
17374
17468
  }
17375
17469
  }
17376
17470
  if (!value && fElements.length > 0) {
@@ -17389,7 +17483,14 @@ function parseWorksheet(xml, sharedStrings) {
17389
17483
  const ref = el.getAttribute("ref");
17390
17484
  if (!ref) continue;
17391
17485
  const m = parseMergeRef(ref);
17392
- if (m) merges.push(m);
17486
+ if (m) {
17487
+ merges.push({
17488
+ startCol: Math.min(m.startCol, MAX_COLS2 - 1),
17489
+ startRow: Math.min(m.startRow, MAX_ROWS2 - 1),
17490
+ endCol: Math.min(m.endCol, MAX_COLS2 - 1),
17491
+ endRow: Math.min(m.endRow, MAX_ROWS2 - 1)
17492
+ });
17493
+ }
17393
17494
  }
17394
17495
  return { grid, merges, maxRow, maxCol };
17395
17496
  }
@@ -17453,7 +17554,7 @@ function sheetToBlocks(sheetName, grid, merges, maxRow, maxCol, sheetIndex) {
17453
17554
  return blocks;
17454
17555
  }
17455
17556
  async function parseXlsxDocument(buffer, options) {
17456
- precheckZipSize(buffer, MAX_DECOMPRESS_SIZE3);
17557
+ precheckZipSize(buffer, MAX_DECOMPRESS_SIZE4);
17457
17558
  const zip = await JSZip3.loadAsync(buffer);
17458
17559
  const warnings = [];
17459
17560
  const workbookFile = zip.file("xl/workbook.xml");
@@ -17465,10 +17566,18 @@ async function parseXlsxDocument(buffer, options) {
17465
17566
  if (ssFile) {
17466
17567
  sharedStrings = parseSharedStrings(await ssFile.async("text"));
17467
17568
  }
17468
- const sheets = parseWorkbook(await workbookFile.async("text"));
17569
+ const { sheets, date1904 } = parseWorkbook(await workbookFile.async("text"));
17469
17570
  if (sheets.length === 0) {
17470
17571
  throw new KordocError("XLSX \uD30C\uC77C\uC5D0 \uC2DC\uD2B8\uAC00 \uC5C6\uC2B5\uB2C8\uB2E4");
17471
17572
  }
17573
+ let dateXfs = /* @__PURE__ */ new Map();
17574
+ const stylesFile = zip.file("xl/styles.xml");
17575
+ if (stylesFile) {
17576
+ try {
17577
+ dateXfs = parseStyleDateXfs(await stylesFile.async("text"));
17578
+ } catch {
17579
+ }
17580
+ }
17472
17581
  let relsMap = /* @__PURE__ */ new Map();
17473
17582
  const relsFile = zip.file("xl/_rels/workbook.xml.rels");
17474
17583
  if (relsFile) {
@@ -17506,7 +17615,7 @@ async function parseXlsxDocument(buffer, options) {
17506
17615
  }
17507
17616
  try {
17508
17617
  const sheetXml = await sheetFile.async("text");
17509
- const { grid, merges, maxRow, maxCol } = parseWorksheet(sheetXml, sharedStrings);
17618
+ const { grid, merges, maxRow, maxCol } = parseWorksheet(sheetXml, sharedStrings, dateXfs, date1904);
17510
17619
  const sheetBlocks = sheetToBlocks(sheet.name, grid, merges, maxRow, maxCol, i);
17511
17620
  blocks.push(...sheetBlocks);
17512
17621
  } catch (err) {
@@ -17550,7 +17659,10 @@ var OP_CONTINUE = 60;
17550
17659
  var OP_BOUNDSHEET8 = 133;
17551
17660
  var OP_SST = 252;
17552
17661
  var OP_CODEPAGE = 66;
17662
+ var OP_DATE1904 = 34;
17553
17663
  var OP_FILEPASS = 47;
17664
+ var OP_FORMAT = 1054;
17665
+ var OP_XF = 224;
17554
17666
  var OP_NUMBER = 515;
17555
17667
  var OP_RK = 638;
17556
17668
  var OP_MULRK = 189;
@@ -17562,6 +17674,8 @@ var OP_BOOLERR = 517;
17562
17674
  var OP_BLANK = 513;
17563
17675
  var OP_MULBLANK = 190;
17564
17676
  var OP_MERGECELLS = 229;
17677
+ var OP_SHRFMLA = 1212;
17678
+ var OP_ARRAY = 545;
17565
17679
  var DT_GLOBALS = 5;
17566
17680
  var DT_WORKSHEET = 16;
17567
17681
  var MAX_RECORDS2 = 1e6;
@@ -17657,7 +17771,7 @@ function decodeUtf16Le(buf) {
17657
17771
  }
17658
17772
 
17659
17773
  // src/xls/sst.ts
17660
- function parseString(buf, offset, segments) {
17774
+ function parseString(buf, offset, segments, segCursor) {
17661
17775
  if (offset + 3 > buf.length) return null;
17662
17776
  const cch = buf.readUInt16LE(offset);
17663
17777
  let flags = buf.readUInt8(offset + 2);
@@ -17680,7 +17794,15 @@ function parseString(buf, offset, segments) {
17680
17794
  const charBytes = [];
17681
17795
  let charsRead = 0;
17682
17796
  while (charsRead < cch) {
17683
- const nextBoundary = segments.find((s) => s > off) ?? buf.length;
17797
+ while (segCursor.idx < segments.length && segments[segCursor.idx] < off) segCursor.idx++;
17798
+ if (segCursor.idx < segments.length && segments[segCursor.idx] === off) {
17799
+ if (off >= buf.length) return null;
17800
+ flags = buf.readUInt8(off);
17801
+ highByte = (flags & 1) !== 0;
17802
+ off += 1;
17803
+ segCursor.idx++;
17804
+ }
17805
+ const nextBoundary = segCursor.idx < segments.length ? segments[segCursor.idx] : buf.length;
17684
17806
  const remainChars = cch - charsRead;
17685
17807
  const bytesPerChar = highByte ? 2 : 1;
17686
17808
  const bytesAvail = nextBoundary - off;
@@ -17691,12 +17813,10 @@ function parseString(buf, offset, segments) {
17691
17813
  charBytes.push(highByte ? slice : padToUtf16(slice));
17692
17814
  off += bytesToRead;
17693
17815
  charsRead += charsInThisRun;
17694
- }
17695
- if (charsRead < cch) {
17696
- if (off >= buf.length) return null;
17697
- flags = buf.readUInt8(off);
17698
- highByte = (flags & 1) !== 0;
17699
- off += 1;
17816
+ } else if (nextBoundary < buf.length) {
17817
+ off = nextBoundary;
17818
+ } else {
17819
+ return null;
17700
17820
  }
17701
17821
  }
17702
17822
  const text = decodeUtf16Le(Buffer.concat(charBytes));
@@ -17721,8 +17841,9 @@ function decodeSST(records) {
17721
17841
  const cstUnique = combined.readUInt32LE(4);
17722
17842
  const strings = [];
17723
17843
  let off = 8;
17844
+ const segCursor = { idx: 0 };
17724
17845
  for (let i = 0; i < cstUnique && off < combined.length; i++) {
17725
- const r = parseString(combined, off, segments);
17846
+ const r = parseString(combined, off, segments, segCursor);
17726
17847
  if (!r) break;
17727
17848
  strings.push(r.text);
17728
17849
  off += r.consumed;
@@ -17797,7 +17918,7 @@ function decodeFormulaStringRecord(data) {
17797
17918
  return decodeUtf16Le(padded);
17798
17919
  }
17799
17920
  }
17800
- function extractSheetCells(records, bofIndex, sst) {
17921
+ function extractSheetCells(records, bofIndex, sst, convertNum) {
17801
17922
  const cells = [];
17802
17923
  const merges = [];
17803
17924
  const bofOffset = records[bofIndex].offset;
@@ -17815,14 +17936,16 @@ function extractSheetCells(records, bofIndex, sst) {
17815
17936
  case OP_NUMBER: {
17816
17937
  const h = readCellHeader(rec.data);
17817
17938
  if (h && rec.data.length >= 14) {
17818
- cells.push({ row: h.row, col: h.col, value: rec.data.readDoubleLE(6) });
17939
+ const n = rec.data.readDoubleLE(6);
17940
+ cells.push({ row: h.row, col: h.col, value: convertNum ? convertNum(n, h.ixfe) : n });
17819
17941
  }
17820
17942
  break;
17821
17943
  }
17822
17944
  case OP_RK: {
17823
17945
  const h = readCellHeader(rec.data);
17824
17946
  if (h && rec.data.length >= 10) {
17825
- cells.push({ row: h.row, col: h.col, value: decodeRk(rec.data.readInt32LE(6)) });
17947
+ const n = decodeRk(rec.data.readInt32LE(6));
17948
+ cells.push({ row: h.row, col: h.col, value: convertNum ? convertNum(n, h.ixfe) : n });
17826
17949
  }
17827
17950
  break;
17828
17951
  }
@@ -17830,7 +17953,7 @@ function extractSheetCells(records, bofIndex, sst) {
17830
17953
  const m = decodeMulRk(rec.data);
17831
17954
  if (m) {
17832
17955
  for (const c of m.cells) {
17833
- cells.push({ row: m.row, col: c.col, value: c.value });
17956
+ cells.push({ row: m.row, col: c.col, value: convertNum ? convertNum(c.value, c.ixfe) : c.value });
17834
17957
  }
17835
17958
  }
17836
17959
  break;
@@ -17855,19 +17978,26 @@ function extractSheetCells(records, bofIndex, sst) {
17855
17978
  if (h && rec.data.length >= 14) {
17856
17979
  const result = decodeFormulaResult(rec.data.subarray(6, 14));
17857
17980
  if (result.kind === "stringRef") {
17858
- const next = records[i + 1];
17981
+ let j = i + 1;
17982
+ while (j < records.length && (records[j].opcode === OP_SHRFMLA || records[j].opcode === OP_ARRAY)) j++;
17983
+ const next = records[j];
17859
17984
  if (next && next.opcode === OP_STRING) {
17860
17985
  cells.push({
17861
17986
  row: h.row,
17862
17987
  col: h.col,
17863
17988
  value: decodeFormulaStringRecord(next.data)
17864
17989
  });
17865
- i++;
17990
+ i = j;
17866
17991
  } else {
17867
17992
  cells.push({ row: h.row, col: h.col, value: "" });
17868
17993
  }
17869
17994
  } else {
17870
- cells.push({ row: h.row, col: h.col, value: result.value });
17995
+ const v = result.value;
17996
+ cells.push({
17997
+ row: h.row,
17998
+ col: h.col,
17999
+ value: convertNum && typeof v === "number" ? convertNum(v, h.ixfe) : v
18000
+ });
17871
18001
  }
17872
18002
  }
17873
18003
  break;
@@ -17917,7 +18047,7 @@ function extractSheetCells(records, bofIndex, sst) {
17917
18047
 
17918
18048
  // src/xls/parser.ts
17919
18049
  var MAX_SHEETS2 = 100;
17920
- var MAX_ROWS3 = 1e5;
18050
+ var MAX_ROWS3 = 65536;
17921
18051
  var MAX_COLS3 = 1e3;
17922
18052
  function decodeBoundSheet(data) {
17923
18053
  if (data.length < 8) return null;
@@ -17940,10 +18070,33 @@ function decodeBoundSheet(data) {
17940
18070
  }
17941
18071
  return { name, lbPlyPos, dt };
17942
18072
  }
18073
+ function decodeFormatRecord(data) {
18074
+ if (data.length < 5) return null;
18075
+ const ifmt = data.readUInt16LE(0);
18076
+ const cch = data.readUInt16LE(2);
18077
+ const flags = data.readUInt8(4);
18078
+ const highByte = (flags & 1) !== 0;
18079
+ const start = 5;
18080
+ let code;
18081
+ if (highByte) {
18082
+ const end = Math.min(start + cch * 2, data.length);
18083
+ code = decodeUtf16Le(data.subarray(start, end));
18084
+ } else {
18085
+ const end = Math.min(start + cch, data.length);
18086
+ const slice = data.subarray(start, end);
18087
+ const padded = Buffer.alloc(slice.length * 2);
18088
+ for (let i = 0; i < slice.length; i++) padded[i * 2] = slice[i];
18089
+ code = decodeUtf16Le(padded);
18090
+ }
18091
+ return { ifmt, code };
18092
+ }
17943
18093
  function processGlobals(records) {
17944
18094
  const sheets = [];
17945
18095
  let codePage = 1200;
17946
18096
  let encrypted = false;
18097
+ let date1904 = false;
18098
+ const customFormats = /* @__PURE__ */ new Map();
18099
+ const xfFmtIds = [];
17947
18100
  const firstBof = records[0];
17948
18101
  if (!firstBof || firstBof.opcode !== OP_BOF) {
17949
18102
  throw new KordocError("XLS: \uCCAB \uB808\uCF54\uB4DC\uAC00 BOF\uAC00 \uC544\uB2D8");
@@ -17966,21 +18119,41 @@ function processGlobals(records) {
17966
18119
  codePage = r.data.readUInt16LE(0);
17967
18120
  } else if (r.opcode === OP_FILEPASS) {
17968
18121
  encrypted = true;
18122
+ } else if (r.opcode === OP_DATE1904 && r.data.length >= 2) {
18123
+ date1904 = r.data.readUInt16LE(0) === 1;
18124
+ } else if (r.opcode === OP_FORMAT) {
18125
+ const f = decodeFormatRecord(r.data);
18126
+ if (f) customFormats.set(f.ifmt, f.code);
18127
+ } else if (r.opcode === OP_XF && r.data.length >= 4) {
18128
+ xfFmtIds.push(r.data.readUInt16LE(2));
17969
18129
  }
17970
18130
  i++;
17971
18131
  }
18132
+ const dateXfs = /* @__PURE__ */ new Map();
18133
+ for (let k = 0; k < xfFmtIds.length; k++) {
18134
+ const kind = dateKindOfFmt(xfFmtIds[k], customFormats);
18135
+ if (kind) dateXfs.set(k, kind);
18136
+ }
17972
18137
  const globalsRecords = records.slice(0, i);
17973
18138
  const sst = decodeSST(globalsRecords);
17974
- return { sheets, sst, codePage, encrypted, endIndex: i };
18139
+ return { sheets, sst, codePage, encrypted, dateXfs, date1904, endIndex: i };
17975
18140
  }
17976
18141
  function findSheetBofIndex(records, lbPlyPos) {
17977
18142
  const exact = records.findIndex(
17978
18143
  (r) => r.opcode === OP_BOF && r.offset === lbPlyPos
17979
18144
  );
17980
18145
  if (exact >= 0) return exact;
17981
- const bofIndices = records.map((r, idx) => r.opcode === OP_BOF ? idx : -1).filter((idx) => idx >= 0);
17982
- if (bofIndices.length === 0) return -1;
17983
- return bofIndices.length > 1 ? bofIndices[1] : -1;
18146
+ let best = -1;
18147
+ let bestOffset = Infinity;
18148
+ for (let idx = 1; idx < records.length; idx++) {
18149
+ const r = records[idx];
18150
+ if (r.opcode !== OP_BOF) continue;
18151
+ if (r.offset >= lbPlyPos && r.offset < bestOffset) {
18152
+ best = idx;
18153
+ bestOffset = r.offset;
18154
+ }
18155
+ }
18156
+ return best;
17984
18157
  }
17985
18158
  function cellValueToText(v) {
17986
18159
  if (v === null || v === void 0) return "";
@@ -18111,6 +18284,14 @@ async function parseXlsDocument(buffer, options) {
18111
18284
  ]
18112
18285
  };
18113
18286
  }
18287
+ const convertNum = globals.dateXfs.size > 0 ? (n, ixfe) => {
18288
+ const kind = globals.dateXfs.get(ixfe);
18289
+ if (kind) {
18290
+ const iso = dateSerialToIso(n, globals.date1904, kind);
18291
+ if (iso) return iso;
18292
+ }
18293
+ return n;
18294
+ } : void 0;
18114
18295
  const totalSheets = Math.min(globals.sheets.length, MAX_SHEETS2);
18115
18296
  let pageFilter = null;
18116
18297
  if (options?.pages) {
@@ -18137,7 +18318,7 @@ async function parseXlsDocument(buffer, options) {
18137
18318
  continue;
18138
18319
  }
18139
18320
  try {
18140
- const { sheet } = extractSheetCells(records, bofIdx, globals.sst);
18321
+ const { sheet } = extractSheetCells(records, bofIdx, globals.sst, convertNum);
18141
18322
  const blocks = sheetToBlocks2(meta.name, sheet, i);
18142
18323
  allBlocks.push(...blocks);
18143
18324
  } catch (e) {
@@ -18265,6 +18446,9 @@ var NARY_MAP = {
18265
18446
  "\u2A02": "\\bigotimes",
18266
18447
  "\u2A00": "\\bigodot"
18267
18448
  };
18449
+ function onOffVal(v) {
18450
+ return v !== "0" && v !== "false" && v !== "off";
18451
+ }
18268
18452
  function mapDelim(ch, isLeft) {
18269
18453
  const l = {
18270
18454
  "(": "(",
@@ -18291,10 +18475,23 @@ function mapDelim(ch, isLeft) {
18291
18475
  const map = isLeft ? l : r;
18292
18476
  return map[ch] ?? ch;
18293
18477
  }
18478
+ function isBalancedWrap(s) {
18479
+ let depth = 0;
18480
+ for (let i = 0; i < s.length; i++) {
18481
+ const ch = s[i];
18482
+ if (ch === "{" && s[i - 1] !== "\\") depth++;
18483
+ else if (ch === "}" && s[i - 1] !== "\\") {
18484
+ depth--;
18485
+ if (depth === 0) return i === s.length - 1;
18486
+ if (depth < 0) return false;
18487
+ }
18488
+ }
18489
+ return false;
18490
+ }
18294
18491
  function grp(body) {
18295
18492
  const s = body.trim();
18296
18493
  if (s.length === 0) return "{}";
18297
- if (s.startsWith("{") && s.endsWith("}")) return s;
18494
+ if (s.startsWith("{") && s.endsWith("}") && isBalancedWrap(s)) return s;
18298
18495
  return "{" + s + "}";
18299
18496
  }
18300
18497
  function childrenToLatex(parent) {
@@ -18388,8 +18585,8 @@ function nodeToLatex(el) {
18388
18585
  }
18389
18586
  const sh = firstKid(naryPr, "subHide");
18390
18587
  const ph = firstKid(naryPr, "supHide");
18391
- if (sh) subHide = (sh.getAttribute("m:val") ?? sh.getAttribute("val")) !== "0";
18392
- if (ph) supHide = (ph.getAttribute("m:val") ?? ph.getAttribute("val")) !== "0";
18588
+ if (sh) subHide = onOffVal(sh.getAttribute("m:val") ?? sh.getAttribute("val"));
18589
+ if (ph) supHide = onOffVal(ph.getAttribute("m:val") ?? ph.getAttribute("val"));
18393
18590
  const ll = firstKid(naryPr, "limLoc");
18394
18591
  if (ll) limLoc = ll.getAttribute("m:val") ?? ll.getAttribute("val") ?? "";
18395
18592
  }
@@ -18536,7 +18733,7 @@ function isDisplayMath(el) {
18536
18733
  }
18537
18734
 
18538
18735
  // src/docx/parser.ts
18539
- var MAX_DECOMPRESS_SIZE4 = 100 * 1024 * 1024;
18736
+ var MAX_DECOMPRESS_SIZE5 = 100 * 1024 * 1024;
18540
18737
  function matchesLocal(el, localName2) {
18541
18738
  return el.localName === localName2 || (el.tagName?.endsWith(`:${localName2}`) ?? false);
18542
18739
  }
@@ -18554,6 +18751,8 @@ function effectiveChildElements(parent) {
18554
18751
  result.push(...effectiveChildElements(c));
18555
18752
  }
18556
18753
  }
18754
+ } else if (matchesLocal(el, "ins") || matchesLocal(el, "smartTag")) {
18755
+ result.push(...effectiveChildElements(el));
18557
18756
  } else {
18558
18757
  result.push(el);
18559
18758
  }
@@ -18618,6 +18817,21 @@ function parseStyles(xml) {
18618
18817
  }
18619
18818
  styles.set(styleId, { name, basedOn, outlineLevel });
18620
18819
  }
18820
+ for (const [styleId, info] of styles) {
18821
+ if (info.outlineLevel !== void 0) continue;
18822
+ const seen = /* @__PURE__ */ new Set([styleId]);
18823
+ let cur = info.basedOn;
18824
+ while (cur && !seen.has(cur)) {
18825
+ seen.add(cur);
18826
+ const parent = styles.get(cur);
18827
+ if (!parent) break;
18828
+ if (parent.outlineLevel !== void 0) {
18829
+ info.outlineLevel = parent.outlineLevel;
18830
+ break;
18831
+ }
18832
+ cur = parent.basedOn;
18833
+ }
18834
+ }
18621
18835
  return styles;
18622
18836
  }
18623
18837
  function parseNumbering(xml) {
@@ -18705,8 +18919,12 @@ function collectOmmlRoots(p) {
18705
18919
  return out;
18706
18920
  }
18707
18921
  function extractRun(r) {
18708
- const tElements = getChildElements(r, "t");
18709
- const text = tElements.map((t) => t.textContent ?? "").join("");
18922
+ let text = "";
18923
+ for (const el of effectiveChildElements(r)) {
18924
+ if (matchesLocal(el, "t")) text += el.textContent ?? "";
18925
+ else if (matchesLocal(el, "br") || matchesLocal(el, "cr")) text += "\n";
18926
+ else if (matchesLocal(el, "tab")) text += " ";
18927
+ }
18710
18928
  let bold = false;
18711
18929
  let italic = false;
18712
18930
  const rPrEls = getChildElements(r, "rPr");
@@ -18861,7 +19079,7 @@ function collectTextboxParagraphs(node, inTxbx = false, out = [], depth = 0) {
18861
19079
  }
18862
19080
  return out;
18863
19081
  }
18864
- function parseTable(tbl2, styles, numbering, footnotes, rels) {
19082
+ function parseTable(tbl2, styles, numbering, footnotes, rels, keepEmptyCols) {
18865
19083
  const trElements = getChildElements(tbl2, "tr");
18866
19084
  if (trElements.length === 0) return null;
18867
19085
  const rawRows = [];
@@ -18923,7 +19141,7 @@ ${cell2.text}` : cell2.text;
18923
19141
  return { text: cell2.text, colSpan: cell2.colSpan, rowSpan, colAddr: cell2.col, rowAddr: r };
18924
19142
  })
18925
19143
  );
18926
- const table2 = buildTable(cellRows);
19144
+ const table2 = buildTable(cellRows, { keepAnchoredEmptyCols: keepEmptyCols });
18927
19145
  if (table2.rows === 0 || table2.cols === 0) return null;
18928
19146
  return { type: "table", table: table2 };
18929
19147
  }
@@ -19025,7 +19243,7 @@ function emitParagraphImages(p, imageMap, linked, out) {
19025
19243
  }
19026
19244
  }
19027
19245
  async function parseDocxDocument(buffer, options) {
19028
- precheckZipSize(buffer, MAX_DECOMPRESS_SIZE4);
19246
+ precheckZipSize(buffer, MAX_DECOMPRESS_SIZE5);
19029
19247
  const zip = await JSZip4.loadAsync(buffer);
19030
19248
  const warnings = [];
19031
19249
  const docFile = zip.file("word/document.xml");
@@ -19096,7 +19314,7 @@ async function parseDocxDocument(buffer, options) {
19096
19314
  if (imageMap.size > 0) emitParagraphImages(tp, imageMap, linkedImages, blocks);
19097
19315
  }
19098
19316
  } else if (localName2 === "tbl") {
19099
- const block = parseTable(el, styles, numbering, footnotes, rels);
19317
+ const block = parseTable(el, styles, numbering, footnotes, rels, options?.keepTrailingEmptyCols);
19100
19318
  if (block) blocks.push(block);
19101
19319
  }
19102
19320
  }
@@ -19188,7 +19406,7 @@ function parseHwpmlDocument(buffer, options) {
19188
19406
  if (localName(el) !== "SECTION") continue;
19189
19407
  sectionIdx++;
19190
19408
  if (pageFilter && !pageFilter.has(sectionIdx)) continue;
19191
- parseSection2(el, blocks, paraShapeMap, sectionIdx, warnings);
19409
+ parseSection2(el, blocks, paraShapeMap, sectionIdx, warnings, options?.keepTrailingEmptyCols ?? false);
19192
19410
  }
19193
19411
  const outline = blocks.filter((b) => b.type === "heading" && b.text).map((b) => ({ level: b.level ?? 1, text: b.text, pageNumber: b.pageNumber }));
19194
19412
  const markdown = blocksToMarkdown(blocks);
@@ -19224,10 +19442,10 @@ function buildParaShapeMap(root) {
19224
19442
  }
19225
19443
  return map;
19226
19444
  }
19227
- function parseSection2(section, blocks, paraShapeMap, sectionNum, warnings) {
19228
- walkContent(section, blocks, paraShapeMap, sectionNum, warnings, false);
19445
+ function parseSection2(section, blocks, paraShapeMap, sectionNum, warnings, keep) {
19446
+ walkContent(section, blocks, paraShapeMap, sectionNum, warnings, false, keep);
19229
19447
  }
19230
- function walkContent(node, blocks, paraShapeMap, sectionNum, warnings, inHeaderFooter, depth = 0) {
19448
+ function walkContent(node, blocks, paraShapeMap, sectionNum, warnings, inHeaderFooter, keep, depth = 0) {
19231
19449
  if (depth > MAX_XML_DEPTH2) return;
19232
19450
  const children = node.childNodes;
19233
19451
  for (let i = 0; i < children.length; i++) {
@@ -19240,24 +19458,24 @@ function walkContent(node, blocks, paraShapeMap, sectionNum, warnings, inHeaderF
19240
19458
  if (tag === "P") {
19241
19459
  if (!inHeaderFooter) {
19242
19460
  parseParagraph3(el, blocks, paraShapeMap, sectionNum);
19243
- walkTablesInP(el, blocks, paraShapeMap, sectionNum, warnings);
19461
+ walkTablesInP(el, blocks, paraShapeMap, sectionNum, warnings, keep);
19244
19462
  }
19245
19463
  continue;
19246
19464
  }
19247
19465
  if (tag === "TABLE") {
19248
19466
  if (!inHeaderFooter) {
19249
- parseTable2(el, blocks, paraShapeMap, sectionNum, warnings);
19467
+ parseTable2(el, blocks, paraShapeMap, sectionNum, warnings, keep);
19250
19468
  }
19251
19469
  continue;
19252
19470
  }
19253
19471
  if (tag === "PARALIST" || tag === "SECTION" || tag === "COLDEF") {
19254
- walkContent(el, blocks, paraShapeMap, sectionNum, warnings, inHeaderFooter, depth + 1);
19472
+ walkContent(el, blocks, paraShapeMap, sectionNum, warnings, inHeaderFooter, keep, depth + 1);
19255
19473
  continue;
19256
19474
  }
19257
- walkContent(el, blocks, paraShapeMap, sectionNum, warnings, inHeaderFooter, depth + 1);
19475
+ walkContent(el, blocks, paraShapeMap, sectionNum, warnings, inHeaderFooter, keep, depth + 1);
19258
19476
  }
19259
19477
  }
19260
- function walkTablesInP(node, blocks, paraShapeMap, sectionNum, warnings, depth = 0) {
19478
+ function walkTablesInP(node, blocks, paraShapeMap, sectionNum, warnings, keep, depth = 0) {
19261
19479
  if (depth > MAX_XML_DEPTH2) return;
19262
19480
  const children = node.childNodes;
19263
19481
  for (let i = 0; i < children.length; i++) {
@@ -19265,11 +19483,11 @@ function walkTablesInP(node, blocks, paraShapeMap, sectionNum, warnings, depth =
19265
19483
  if (el.nodeType !== 1) continue;
19266
19484
  const tag = localName(el);
19267
19485
  if (tag === "TABLE") {
19268
- parseTable2(el, blocks, paraShapeMap, sectionNum, warnings);
19486
+ parseTable2(el, blocks, paraShapeMap, sectionNum, warnings, keep);
19269
19487
  continue;
19270
19488
  }
19271
19489
  if (tag === "FOOTNOTE" || tag === "ENDNOTE" || tag === "HEADER" || tag === "FOOTER") continue;
19272
- walkTablesInP(el, blocks, paraShapeMap, sectionNum, warnings, depth + 1);
19490
+ walkTablesInP(el, blocks, paraShapeMap, sectionNum, warnings, keep, depth + 1);
19273
19491
  }
19274
19492
  }
19275
19493
  function parseParagraph3(el, blocks, paraShapeMap, sectionNum) {
@@ -19305,7 +19523,7 @@ function collectCharText(node, parts, depth = 0) {
19305
19523
  }
19306
19524
  }
19307
19525
  }
19308
- function parseTable2(el, blocks, paraShapeMap, sectionNum, warnings) {
19526
+ function parseTable2(el, blocks, paraShapeMap, sectionNum, warnings, keep) {
19309
19527
  const cells = [];
19310
19528
  const rowCount = parseInt(el.getAttribute("RowCount") ?? "0", 10);
19311
19529
  const colCount = parseInt(el.getAttribute("ColCount") ?? "0", 10);
@@ -19349,7 +19567,7 @@ function parseTable2(el, blocks, paraShapeMap, sectionNum, warnings) {
19349
19567
  const cellRows = grid.map(
19350
19568
  (row) => row.map((cell2) => cell2 ?? { text: "", colSpan: 1, rowSpan: 1 })
19351
19569
  );
19352
- const table2 = buildTable(cellRows);
19570
+ const table2 = buildTable(cellRows, { keepAnchoredEmptyCols: keep });
19353
19571
  const caption = extractShapeCaption(el);
19354
19572
  if (caption.text && caption.before) {
19355
19573
  blocks.push({ type: "paragraph", text: caption.text, pageNumber: sectionNum });
@@ -19585,7 +19803,7 @@ function findMatchingKey(cellLabel, values) {
19585
19803
  bestKey = key;
19586
19804
  }
19587
19805
  } else if (key.startsWith(cellLabel)) {
19588
- if (cellLabel.length >= key.length * 0.6 && cellLabel.length > bestLen) {
19806
+ if (cellLabel.length >= key.length * 0.75 && cellLabel.length > bestLen) {
19589
19807
  bestLen = cellLabel.length;
19590
19808
  bestKey = key;
19591
19809
  }
@@ -19625,7 +19843,7 @@ function fillInCellPatterns(cellText, values, matchedLabels, blockedLabels) {
19625
19843
  const matchKey = values.available(normalizedKw) ? normalizedKw : void 0;
19626
19844
  if (matchKey === void 0) return match;
19627
19845
  const val = values.peek(matchKey);
19628
- const isTruthy = ["\u2611", "\u2713", "\u2714", "v", "V", "true", "1", "yes", "o", "O"].includes(val.trim()) || val.trim() === "";
19846
+ const isTruthy = ["\u2611", "\u2713", "\u2714", "v", "V", "true", "1", "yes", "o", "O"].includes(val.trim());
19629
19847
  if (!isTruthy) return match;
19630
19848
  values.consume(matchKey);
19631
19849
  matchedLabels.add(matchKey);
@@ -19647,14 +19865,20 @@ function fillInCellPatterns(cellText, values, matchedLabels, blockedLabels) {
19647
19865
  );
19648
19866
  return matches.length > 0 ? { text, matches } : null;
19649
19867
  }
19650
- var INLINE_LABEL_RE = /([가-힣A-Za-z]{2,10})\s*[::]/g;
19868
+ var INLINE_LABEL_RE = /((?:[가-힣A-Za-z]{1,10} )?)([가-힣A-Za-z]{2,10})\s*[::]/g;
19651
19869
  function scanInlineSegments(text) {
19652
19870
  const labels = [];
19653
19871
  INLINE_LABEL_RE.lastIndex = 0;
19654
19872
  let m;
19655
19873
  while ((m = INLINE_LABEL_RE.exec(text)) !== null) {
19656
19874
  if (text[INLINE_LABEL_RE.lastIndex] === "/") continue;
19657
- labels.push({ label: m[1], start: m.index, end: INLINE_LABEL_RE.lastIndex });
19875
+ labels.push({
19876
+ label: m[2],
19877
+ ext: m[1] ? m[1] + m[2] : void 0,
19878
+ extStart: m[1] ? m.index : void 0,
19879
+ start: m.index + m[1].length,
19880
+ end: INLINE_LABEL_RE.lastIndex
19881
+ });
19658
19882
  }
19659
19883
  const segments = [];
19660
19884
  for (let i = 0; i < labels.length; i++) {
@@ -19672,21 +19896,44 @@ function scanInlineSegments(text) {
19672
19896
  labelStart: cur.start,
19673
19897
  valueStart: vs,
19674
19898
  valueEnd: ve,
19675
- value: text.slice(vs, ve)
19899
+ value: text.slice(vs, ve),
19900
+ ...cur.ext !== void 0 ? { extLabel: cur.ext, extStart: cur.extStart } : {}
19676
19901
  });
19677
19902
  }
19678
19903
  return segments;
19679
19904
  }
19905
+ function matchInlineSegment(seg, values, blockedLabels) {
19906
+ const nlabel = normalizeLabel(seg.label);
19907
+ if (seg.extLabel !== void 0) {
19908
+ const nExt = normalizeLabel(seg.extLabel);
19909
+ if (nExt !== nlabel && !blockedLabels?.has(nExt) && values.has(nExt)) {
19910
+ return { key: nExt, label: seg.extLabel, viaExt: true };
19911
+ }
19912
+ }
19913
+ if (blockedLabels?.has(nlabel)) return void 0;
19914
+ const key = findMatchingKey(nlabel, values);
19915
+ return key !== void 0 ? { key, label: seg.label, viaExt: false } : void 0;
19916
+ }
19917
+ function clampSegmentEnd(text, seg, next, nextViaExt) {
19918
+ let ve = seg.valueEnd;
19919
+ if (nextViaExt && next?.extStart !== void 0 && next.extStart < ve) ve = next.extStart;
19920
+ while (ve > seg.valueStart && /\s/.test(text[ve - 1])) ve--;
19921
+ return ve;
19922
+ }
19680
19923
  function padInsertion(text, pos, value) {
19681
19924
  const lead = pos > 0 && !/\s/.test(text[pos - 1]) ? " " : "";
19682
19925
  const trail = pos < text.length && !/\s/.test(text[pos]) ? " " : "";
19683
19926
  return lead + value + trail;
19684
19927
  }
19685
- function normalizeValues(values) {
19928
+ function normalizeValues(values, warnings) {
19686
19929
  const map = /* @__PURE__ */ new Map();
19687
19930
  for (const [label, raw] of Object.entries(values)) {
19688
19931
  const { value, format } = typeof raw === "object" && !Array.isArray(raw) ? raw : { value: raw, format: void 0 };
19689
- map.set(normalizeLabel(label), Array.isArray(value) ? value.map((v) => formatFillValue(v, format)) : formatFillValue(value, format));
19932
+ const key = normalizeLabel(label);
19933
+ if (map.has(key)) {
19934
+ warnings?.push(`\uC785\uB825 \uB77C\uBCA8 "${label}"\uC774 \uC815\uADDC\uD654 \uD0A4 "${key}"\uC5D0\uC11C \uB2E4\uB978 \uB77C\uBCA8\uACFC \uCDA9\uB3CC \u2014 \uB4A4 \uAC12\uC73C\uB85C \uB36E\uC5B4\uC500`);
19935
+ }
19936
+ map.set(key, Array.isArray(value) ? value.map((v) => formatFillValue(v, format)) : formatFillValue(value, format));
19690
19937
  }
19691
19938
  return map;
19692
19939
  }
@@ -19875,10 +20122,10 @@ function extractFromTable(table2) {
19875
20122
  }
19876
20123
  if (fields.length === 0 && table2.rows >= 2 && table2.cols >= 2) {
19877
20124
  const headerRow = table2.cells[0];
19878
- const allLabels = headerRow.every((cell2) => {
19879
- const t = cell2.text.trim();
20125
+ const allLabels = headerRow?.every((cell2) => {
20126
+ const t = cell2?.text.trim() ?? "";
19880
20127
  return t.length > 0 && t.length <= 20;
19881
- });
20128
+ }) ?? false;
19882
20129
  if (allLabels) {
19883
20130
  for (let r = 1; r < table2.rows; r++) {
19884
20131
  for (let c = 0; c < table2.cols; c++) {
@@ -19969,7 +20216,8 @@ function fillFormFields(blocks, values, blockedLabels) {
19969
20216
  const cloned = structuredClone(blocks);
19970
20217
  const filled = [];
19971
20218
  const matchedLabels = /* @__PURE__ */ new Set();
19972
- const normalizedValues = normalizeValues(values);
20219
+ const warnings = [];
20220
+ const normalizedValues = normalizeValues(values, warnings);
19973
20221
  const cursor = new ValueCursor(normalizedValues);
19974
20222
  const allTables = collectIRTables(cloned, 0);
19975
20223
  const patternFilledCells = /* @__PURE__ */ new Set();
@@ -19998,7 +20246,7 @@ function fillFormFields(blocks, values, blockedLabels) {
19998
20246
  if (newText !== block.text) block.text = newText;
19999
20247
  }
20000
20248
  const unmatched = resolveUnmatched(normalizedValues, matchedLabels, values);
20001
- return { blocks: cloned, filled, unmatched };
20249
+ return { blocks: cloned, filled, unmatched, ...warnings.length > 0 ? { warnings } : {} };
20002
20250
  }
20003
20251
  function collectIRTables(blocks, depth) {
20004
20252
  if (depth > 16) return [];
@@ -20032,13 +20280,31 @@ function coveredPositions(table2) {
20032
20280
  }
20033
20281
  return covered;
20034
20282
  }
20283
+ function isHeaderDataTable(table2, covered) {
20284
+ if (table2.rows < 2) return false;
20285
+ const headerRow = table2.cells[0];
20286
+ if (!headerRow?.length) return false;
20287
+ const allLabels = headerRow.every((cell2) => {
20288
+ const t = cell2?.text.trim() ?? "";
20289
+ return t.length > 0 && t.length <= 20 && isLabelCell(t);
20290
+ });
20291
+ if (!allLabels) return false;
20292
+ for (let c = 0; c < table2.cols; c++) {
20293
+ if (covered.has(`1,${c}`)) continue;
20294
+ const cell2 = table2.cells[1]?.[c];
20295
+ if (!cell2) break;
20296
+ return !isLabelCell(cell2.text);
20297
+ }
20298
+ return true;
20299
+ }
20035
20300
  function fillTable(table2, values, filled, matchedLabels, patternFilledCells, blockedLabels) {
20036
20301
  if (table2.cols < 2) return;
20037
20302
  const covered = coveredPositions(table2);
20038
- for (let r = 0; r < table2.rows; r++) {
20303
+ const skipHeaderRow = isHeaderDataTable(table2, covered);
20304
+ for (let r = skipHeaderRow ? 1 : 0; r < table2.rows; r++) {
20039
20305
  for (let c = 0; c < table2.cols; c++) {
20040
20306
  if (covered.has(`${r},${c}`)) continue;
20041
- const labelCell = table2.cells[r][c];
20307
+ const labelCell = table2.cells[r]?.[c];
20042
20308
  if (!labelCell) continue;
20043
20309
  if (!isLabelCell(labelCell.text)) continue;
20044
20310
  let vc = c + labelCell.colSpan;
@@ -20071,10 +20337,10 @@ function fillTable(table2, values, filled, matchedLabels, patternFilledCells, bl
20071
20337
  }
20072
20338
  if (table2.rows >= 2 && table2.cols >= 2) {
20073
20339
  const headerRow = table2.cells[0];
20074
- const allLabels = headerRow.every((cell2) => {
20075
- const t = cell2.text.trim();
20340
+ const allLabels = headerRow?.every((cell2) => {
20341
+ const t = cell2?.text.trim() ?? "";
20076
20342
  return t.length > 0 && t.length <= 20 && isLabelCell(t);
20077
- });
20343
+ }) ?? false;
20078
20344
  if (!allLabels) return;
20079
20345
  for (let r = 1; r < table2.rows; r++) {
20080
20346
  for (let c = 0; c < table2.cols; c++) {
@@ -20089,7 +20355,11 @@ function fillTable(table2, values, filled, matchedLabels, patternFilledCells, bl
20089
20355
  if (!values.isArray(matchKey) && matchedLabels.has(matchKey)) continue;
20090
20356
  const newValue = values.consume(matchKey);
20091
20357
  if (newValue === void 0) continue;
20092
- valueCell.text = newValue;
20358
+ if (patternFilledCells?.has(valueCell)) {
20359
+ valueCell.text = newValue + " " + valueCell.text;
20360
+ } else {
20361
+ valueCell.text = newValue;
20362
+ }
20093
20363
  matchedLabels.add(matchKey);
20094
20364
  filled.push({
20095
20365
  label: headerCell.text.trim(),
@@ -20105,20 +20375,22 @@ function fillTable(table2, values, filled, matchedLabels, patternFilledCells, bl
20105
20375
  function fillInlineFields(text, values, filled, matchedLabels, blockedLabels) {
20106
20376
  const segments = scanInlineSegments(text);
20107
20377
  if (segments.length === 0) return text;
20378
+ const matches = segments.map((seg) => matchInlineSegment(seg, values, blockedLabels));
20108
20379
  let out = "";
20109
20380
  let pos = 0;
20110
- for (const seg of segments) {
20111
- const nlabel = normalizeLabel(seg.label);
20112
- if (blockedLabels?.has(nlabel)) continue;
20113
- const matchKey = findMatchingKey(nlabel, values);
20114
- if (matchKey === void 0) continue;
20381
+ for (let i = 0; i < segments.length; i++) {
20382
+ const seg = segments[i];
20383
+ const matched = matches[i];
20384
+ if (matched === void 0) continue;
20385
+ const matchKey = matched.key;
20115
20386
  const newValue = values.consume(matchKey);
20116
20387
  if (newValue === void 0) continue;
20117
20388
  matchedLabels.add(matchKey);
20118
- filled.push({ label: seg.label.trim(), value: newValue, row: -1, col: -1, key: matchKey });
20389
+ filled.push({ label: matched.label.trim(), value: newValue, row: -1, col: -1, key: matchKey });
20390
+ const ve = clampSegmentEnd(text, seg, segments[i + 1], matches[i + 1]?.viaExt ?? false);
20119
20391
  out += text.slice(pos, seg.valueStart);
20120
- out += seg.valueStart === seg.valueEnd ? padInsertion(text, seg.valueStart, newValue) : newValue;
20121
- pos = seg.valueEnd;
20392
+ out += seg.valueStart === ve ? padInsertion(text, seg.valueStart, newValue) : newValue;
20393
+ pos = ve;
20122
20394
  }
20123
20395
  out += text.slice(pos);
20124
20396
  return out;
@@ -20133,7 +20405,8 @@ async function fillHwpx(hwpxBuffer, values, blockedLabels) {
20133
20405
  if (sectionPaths.length === 0) {
20134
20406
  throw new KordocError("HWPX\uC5D0\uC11C \uC139\uC158 \uD30C\uC77C\uC744 \uCC3E\uC744 \uC218 \uC5C6\uC2B5\uB2C8\uB2E4");
20135
20407
  }
20136
- const normalizedValues = normalizeValues(values);
20408
+ const warnings = [];
20409
+ const normalizedValues = normalizeValues(values, warnings);
20137
20410
  const cursor = new ValueCursor(normalizedValues);
20138
20411
  const matchedLabels = /* @__PURE__ */ new Set();
20139
20412
  const filled = [];
@@ -20195,7 +20468,17 @@ async function fillHwpx(hwpxBuffer, values, blockedLabels) {
20195
20468
  }
20196
20469
  }
20197
20470
  for (const table2 of allTables) {
20198
- for (let rowIdx = 0; rowIdx < table2.rows.length; rowIdx++) {
20471
+ const skipHeaderRow = table2.rows.length >= 2 && (() => {
20472
+ const first = table2.rows[0];
20473
+ const allLabels = first.length > 0 && first.every((cell2) => {
20474
+ const t = cellLabelText(cell2).trim();
20475
+ return t.length > 0 && t.length <= 20 && isLabelCell(t);
20476
+ });
20477
+ if (!allLabels) return false;
20478
+ const d0 = table2.rows[1][0];
20479
+ return d0 === void 0 || !isLabelCell(cellLabelText(d0));
20480
+ })();
20481
+ for (let rowIdx = skipHeaderRow ? 1 : 0; rowIdx < table2.rows.length; rowIdx++) {
20199
20482
  const cells = table2.rows[rowIdx];
20200
20483
  for (let colIdx = 0; colIdx < cells.length - 1; colIdx++) {
20201
20484
  const labelText = cellLabelText(cells[colIdx]);
@@ -20266,9 +20549,30 @@ async function fillHwpx(hwpxBuffer, values, blockedLabels) {
20266
20549
  const matchKey = findMatchingKey(headerLabel, cursor);
20267
20550
  if (matchKey === void 0) continue;
20268
20551
  if (!cursor.isArray(matchKey) && matchedLabels.has(matchKey)) continue;
20552
+ const dataCell = dataCells[colIdx];
20553
+ if (patternApplied.has(dataCell)) {
20554
+ const target = dataCell.paragraphs.find((p) => p.tRanges.length > 0) ?? dataCell.paragraphs[0];
20555
+ if (!target) continue;
20556
+ const l = led(target);
20557
+ if (l.fullText !== void 0) continue;
20558
+ const newValue2 = cursor.consume(matchKey);
20559
+ if (newValue2 === void 0) continue;
20560
+ l.ranges.push({ start: 0, end: 0, replacement: newValue2 + " " });
20561
+ l.filledIdx.push(filled.length);
20562
+ l.matchKeys.push(matchKey);
20563
+ matchedLabels.add(matchKey);
20564
+ filled.push({
20565
+ label: cellLabelText(headerCells[colIdx]).trim(),
20566
+ value: newValue2,
20567
+ row: rowIdx,
20568
+ col: colIdx,
20569
+ key: matchKey
20570
+ });
20571
+ continue;
20572
+ }
20269
20573
  const newValue = cursor.consume(matchKey);
20270
20574
  if (newValue === void 0) continue;
20271
- const paras = dataCells[colIdx].paragraphs;
20575
+ const paras = dataCell.paragraphs;
20272
20576
  if (paras.length === 0) continue;
20273
20577
  const l0 = led(paras[0]);
20274
20578
  l0.fullText = newValue;
@@ -20297,20 +20601,23 @@ async function fillHwpx(hwpxBuffer, values, blockedLabels) {
20297
20601
  const existing = ledger.get(para2);
20298
20602
  if (existing?.fullText !== void 0) continue;
20299
20603
  const text = matchText(para2);
20300
- for (const seg of scanInlineSegments(text)) {
20301
- const nlabel = normalizeLabel(seg.label);
20302
- if (blockedLabels?.has(nlabel)) continue;
20303
- const matchKey = findMatchingKey(nlabel, cursor);
20304
- if (matchKey === void 0) continue;
20604
+ const segments = scanInlineSegments(text);
20605
+ const matches = segments.map((seg) => matchInlineSegment(seg, cursor, blockedLabels));
20606
+ for (let i = 0; i < segments.length; i++) {
20607
+ const seg = segments[i];
20608
+ const matched = matches[i];
20609
+ if (matched === void 0) continue;
20610
+ const matchKey = matched.key;
20305
20611
  const newValue = cursor.consume(matchKey);
20306
20612
  if (newValue === void 0) continue;
20307
- const replacement = seg.valueStart === seg.valueEnd ? padInsertion(text, seg.valueStart, newValue) : newValue;
20613
+ const ve = clampSegmentEnd(text, seg, segments[i + 1], matches[i + 1]?.viaExt ?? false);
20614
+ const replacement = seg.valueStart === ve ? padInsertion(text, seg.valueStart, newValue) : newValue;
20308
20615
  const l = led(para2);
20309
- l.ranges.push({ start: seg.valueStart, end: seg.valueEnd, replacement });
20616
+ l.ranges.push({ start: seg.valueStart, end: ve, replacement });
20310
20617
  matchedLabels.add(matchKey);
20311
20618
  l.filledIdx.push(filled.length);
20312
20619
  l.matchKeys.push(matchKey);
20313
- filled.push({ label: seg.label.trim(), value: newValue, row: -1, col: -1, key: matchKey });
20620
+ filled.push({ label: matched.label.trim(), value: newValue, row: -1, col: -1, key: matchKey });
20314
20621
  }
20315
20622
  }
20316
20623
  const splices = [];
@@ -20363,7 +20670,7 @@ async function fillHwpx(hwpxBuffer, values, blockedLabels) {
20363
20670
  }
20364
20671
  }
20365
20672
  for (const k of failedKeys) {
20366
- if (!succeededKeys.has(k)) matchedLabels.delete(k);
20673
+ if (!succeededKeys.has(k) || cursor.isArray(k)) matchedLabels.delete(k);
20367
20674
  }
20368
20675
  const cleanFilled = filled.filter((f) => f !== null);
20369
20676
  const unmatched = resolveUnmatched(normalizedValues, matchedLabels, values);
@@ -20371,7 +20678,8 @@ async function fillHwpx(hwpxBuffer, values, blockedLabels) {
20371
20678
  return {
20372
20679
  buffer: out.buffer.slice(out.byteOffset, out.byteOffset + out.byteLength),
20373
20680
  filled: cleanFilled,
20374
- unmatched
20681
+ unmatched,
20682
+ ...warnings.length > 0 ? { warnings } : {}
20375
20683
  };
20376
20684
  }
20377
20685
 
@@ -20648,27 +20956,39 @@ function parseMarkdownToBlocks(md2) {
20648
20956
  if (/^<table[\s>]/i.test(line.trimStart())) {
20649
20957
  const htmlLines = [];
20650
20958
  let depth = 0;
20651
- while (i < lines.length) {
20652
- const l = lines[i];
20959
+ let closed = false;
20960
+ let j = i;
20961
+ while (j < lines.length) {
20962
+ const l = lines[j];
20653
20963
  htmlLines.push(l);
20654
20964
  depth += (l.match(/<table[\s>]/gi) ?? []).length;
20655
20965
  depth -= (l.match(/<\/table>/gi) ?? []).length;
20656
- i++;
20657
- if (depth <= 0) break;
20966
+ j++;
20967
+ if (depth <= 0) {
20968
+ closed = true;
20969
+ break;
20970
+ }
20971
+ }
20972
+ if (closed) {
20973
+ blocks.push({ type: "html_table", text: htmlLines.join("\n") });
20974
+ i = j;
20975
+ continue;
20658
20976
  }
20659
- blocks.push({ type: "html_table", text: htmlLines.join("\n") });
20660
- continue;
20661
20977
  }
20662
20978
  if (line.trimStart().startsWith("|")) {
20663
20979
  const tableRows = [];
20980
+ let sepSeen = false;
20664
20981
  while (i < lines.length && lines[i].trimStart().startsWith("|")) {
20665
20982
  const row = lines[i];
20666
- const sepCells = row.trim().replace(/^\|/, "").replace(/\|$/, "").split("|");
20667
- if (sepCells.every((c) => /^\s*:?-+:?\s*$/.test(c))) {
20668
- i++;
20669
- continue;
20983
+ if (tableRows.length === 1 && !sepSeen) {
20984
+ const sepCells = row.trim().replace(/^\|/, "").replace(/\|$/, "").split("|");
20985
+ if (sepCells.every((c) => /^\s*:?-+:?\s*$/.test(c))) {
20986
+ sepSeen = true;
20987
+ i++;
20988
+ continue;
20989
+ }
20670
20990
  }
20671
- const cells = row.split("|").slice(1, -1).map((c) => c.trim());
20991
+ const cells = row.split(/(?<!\\)\|/).slice(1, -1).map((c) => c.trim().replace(/\\\|/g, "|"));
20672
20992
  if (cells.length > 0) tableRows.push(cells);
20673
20993
  i++;
20674
20994
  }
@@ -21617,13 +21937,26 @@ function normalize(s) {
21617
21937
  return s.replace(/\s+/g, " ").trim();
21618
21938
  }
21619
21939
  var MAX_LEVENSHTEIN_LEN = 1e4;
21940
+ function approxDistance(a, b) {
21941
+ const bigramCounts = (s) => {
21942
+ const m = /* @__PURE__ */ new Map();
21943
+ for (let i = 0; i < s.length - 1; i++) {
21944
+ const g = s.slice(i, i + 2);
21945
+ m.set(g, (m.get(g) ?? 0) + 1);
21946
+ }
21947
+ return m;
21948
+ };
21949
+ const ca = bigramCounts(a);
21950
+ const cb = bigramCounts(b);
21951
+ let inter = 0;
21952
+ for (const [g, n] of ca) inter += Math.min(n, cb.get(g) ?? 0);
21953
+ const total = Math.max(a.length - 1, 0) + Math.max(b.length - 1, 0);
21954
+ const dice = total > 0 ? 2 * inter / total : 1;
21955
+ return Math.round(Math.max(a.length, b.length) * (1 - dice));
21956
+ }
21620
21957
  function levenshtein(a, b) {
21621
21958
  if (a.length + b.length > MAX_LEVENSHTEIN_LEN) {
21622
- const sampleLen = Math.min(500, a.length, b.length);
21623
- let diffs = 0;
21624
- for (let i = 0; i < sampleLen; i++) if (a[i] !== b[i]) diffs++;
21625
- const sampleRate = sampleLen > 0 ? diffs / sampleLen : 1;
21626
- return Math.abs(a.length - b.length) + Math.round(Math.min(a.length, b.length) * sampleRate);
21959
+ return approxDistance(a, b);
21627
21960
  }
21628
21961
  if (a.length > b.length) [a, b] = [b, a];
21629
21962
  const m = a.length;
@@ -21775,9 +22108,16 @@ function bestSimInRange(arr, from, to, target) {
21775
22108
  return best;
21776
22109
  }
21777
22110
  function escapeGfm(text) {
21778
- return text.replace(/([~*])/g, "\\$1");
22111
+ const NUL = String.fromCharCode(0);
22112
+ const spans = [];
22113
+ const masked = text.replace(/!\[[^\]]*\]\([^)\n]*\)|\]\((?:https?:|mailto:|tel:|#)[^)\n]*\)/gi, (m) => {
22114
+ spans.push(m);
22115
+ return NUL + (spans.length - 1) + NUL;
22116
+ });
22117
+ const escaped = masked.replace(/([~*_`])/g, "\\$1");
22118
+ return escaped.replace(new RegExp(NUL + "(\\d+)" + NUL, "g"), (_, n) => spans[Number(n)]);
21779
22119
  }
21780
- var HWP_SHAPE_ALT_TEXT_RE = /(?:모서리가 둥근 |둥근 )?(?:사각형|직사각형|정사각형|원|타원|삼각형|이등변 삼각형|직각 삼각형|선|직선|곡선|화살표|굵은 화살표|이중 화살표|오각형|육각형|팔각형|별|[4-8]점별|십자|십자형|구름|구름형|마름모|도넛|평행사변형|사다리꼴|부채꼴|호|반원|물결|번개|하트|빗금|블록 화살표|수식|표|그림|개체|그리기\s?개체|묶음\s?개체|글상자|수식\s?개체|OLE\s?개체)\s?입니다\.?/g;
22120
+ var HWP_SHAPE_ALT_TEXT_RE = /^(?:모서리가 둥근 |둥근 )?(?:사각형|직사각형|정사각형|원|타원|삼각형|이등변 삼각형|직각 삼각형|선|직선|곡선|화살표|굵은 화살표|이중 화살표|오각형|육각형|팔각형|별|[4-8]점별|십자|십자형|구름|구름형|마름모|도넛|평행사변형|사다리꼴|부채꼴|호|반원|물결|번개|하트|빗금|블록 화살표|수식|표|그림|개체|그리기\s?개체|묶음\s?개체|글상자|수식\s?개체|OLE\s?개체)\s?입니다\.?$/gm;
21781
22121
  function sanitizeText(text) {
21782
22122
  let result = mapPuaText(text).replace(/[\u{F0000}-\u{FFFFD}]/gu, "").replace(HWP_SHAPE_ALT_TEXT_RE, "").replace(/ +/g, " ").trim();
21783
22123
  if (result.length <= 30 && result.includes(" ")) {
@@ -21793,7 +22133,7 @@ function normForMatch(text) {
21793
22133
  return sanitizeText(text).replace(/\s+/g, " ").trim();
21794
22134
  }
21795
22135
  function unescapeGfm(text) {
21796
- return text.replace(/\\([~*])/g, "$1");
22136
+ return text.replace(/\\([~*_`])/g, "$1");
21797
22137
  }
21798
22138
  function summarize(text) {
21799
22139
  const t = text.replace(/\s+/g, " ").trim();
@@ -21860,7 +22200,7 @@ function parseGfmTable(lines) {
21860
22200
  return rows;
21861
22201
  }
21862
22202
  function unescapeGfmCell(text) {
21863
- return text.replace(/<br\s*\/?>/gi, "\n").replace(/\\\|/g, "|").replace(/\\([~*])/g, "$1");
22203
+ return text.replace(/<br\s*\/?>/gi, "\n").replace(/\\\|/g, "|").replace(/\\([~*_`])/g, "$1");
21864
22204
  }
21865
22205
  function replicateCellInnerHtml(cell2) {
21866
22206
  if (cell2.blocks?.length) {
@@ -21948,10 +22288,10 @@ function parseHtmlTable(raw) {
21948
22288
  }
21949
22289
  } else {
21950
22290
  if (!isClose) {
21951
- const cs = parseInt(attrs.match(/colspan\s*=\s*"(\d+)"/i)?.[1] || "1", 10);
21952
- const rs = parseInt(attrs.match(/rowspan\s*=\s*"(\d+)"/i)?.[1] || "1", 10);
22291
+ const cs = parseInt(attrs.match(/colspan\s*=\s*["']?(\d+)/i)?.[1] || "1", 10);
22292
+ const rs = parseInt(attrs.match(/rowspan\s*=\s*["']?(\d+)/i)?.[1] || "1", 10);
21953
22293
  cellStart = m.index + m[0].length;
21954
- cellInfo = { colSpan: isNaN(cs) ? 1 : cs, rowSpan: isNaN(rs) ? 1 : rs };
22294
+ cellInfo = { colSpan: clampSpan(isNaN(cs) ? 1 : cs, MAX_COLS), rowSpan: clampSpan(isNaN(rs) ? 1 : rs, MAX_ROWS) };
21955
22295
  } else if (cellStart >= 0 && cellInfo && currentRow) {
21956
22296
  currentRow.push({ inner: raw.slice(cellStart, m.index), colSpan: cellInfo.colSpan, rowSpan: cellInfo.rowSpan });
21957
22297
  cellStart = -1;
@@ -22595,8 +22935,8 @@ function layoutHtmlRows(rows) {
22595
22935
  let c = 0;
22596
22936
  for (const cell2 of rows[r].cells) {
22597
22937
  while (occupied.has(`${r},${c}`)) c++;
22598
- const colSpan = Math.max(1, cell2.colSpan);
22599
- const rowSpan = Math.max(1, cell2.rowSpan);
22938
+ const colSpan = clampSpan(cell2.colSpan, MAX_COLS);
22939
+ const rowSpan = clampSpan(cell2.rowSpan, MAX_ROWS);
22600
22940
  placed.push({ r, c, colSpan, rowSpan, inner: cell2.inner, isHeader: rows[r].tag === "th" });
22601
22941
  for (let dr = 0; dr < rowSpan; dr++) {
22602
22942
  for (let dc = 0; dc < colSpan; dc++) occupied.add(`${r + dr},${c + dc}`);
@@ -23755,12 +24095,20 @@ function diffBlocks(blocksA, blocksB) {
23755
24095
  function alignBlocks(a, b) {
23756
24096
  const m = a.length, n = b.length;
23757
24097
  if (m * n > 1e7) return fallbackAlign(a, b);
24098
+ const lenOf = (blk) => {
24099
+ const t = blk.text !== void 0 ? blk.text : blk.type === "table" && blk.table ? blk.table.cells.flat().map((c) => c?.text ?? "").join(" ") : "";
24100
+ return t.replace(/\s+/g, " ").trim().length;
24101
+ };
24102
+ const aLen = a.map(lenOf);
24103
+ const bLen = b.map(lenOf);
23758
24104
  const simCache = /* @__PURE__ */ new Map();
23759
24105
  const getSim = (i2, j2) => {
23760
24106
  const key = `${i2},${j2}`;
23761
24107
  let v = simCache.get(key);
23762
24108
  if (v === void 0) {
23763
- v = blockSimilarity(a[i2], b[j2]);
24109
+ const mx = Math.max(aLen[i2], bLen[j2]);
24110
+ const cut = a[i2].type === "table" || b[j2].type === "table" ? 6 / 7 : 1 - SIMILARITY_THRESHOLD;
24111
+ v = mx > 0 && (mx - Math.min(aLen[i2], bLen[j2])) / mx > cut ? 0 : blockSimilarity(a[i2], b[j2]);
23764
24112
  simCache.set(key, v);
23765
24113
  }
23766
24114
  return v;
@@ -23821,8 +24169,8 @@ function blockSimilarity(a, b) {
23821
24169
  }
23822
24170
  function tableSimilarity(a, b) {
23823
24171
  const dimSim = 1 - Math.abs(a.rows * a.cols - b.rows * b.cols) / Math.max(a.rows * a.cols, b.rows * b.cols, 1);
23824
- const textsA = a.cells.flat().map((c) => c.text).join(" ");
23825
- const textsB = b.cells.flat().map((c) => c.text).join(" ");
24172
+ const textsA = a.cells.flat().map((c) => c?.text ?? "").join(" ");
24173
+ const textsB = b.cells.flat().map((c) => c?.text ?? "").join(" ");
23826
24174
  const contentSim = normalizedSimilarity(textsA, textsB);
23827
24175
  return dimSim * 0.3 + contentSim * 0.7;
23828
24176
  }
@@ -23833,8 +24181,8 @@ function diffTableCells(a, b) {
23833
24181
  for (let r = 0; r < maxRows; r++) {
23834
24182
  const row = [];
23835
24183
  for (let c = 0; c < maxCols; c++) {
23836
- const cellA = r < a.rows && c < a.cols ? a.cells[r][c].text : void 0;
23837
- const cellB = r < b.rows && c < b.cols ? b.cells[r][c].text : void 0;
24184
+ const cellA = r < a.rows && c < a.cols ? a.cells[r]?.[c]?.text : void 0;
24185
+ const cellB = r < b.rows && c < b.cols ? b.cells[r]?.[c]?.text : void 0;
23838
24186
  let type;
23839
24187
  if (cellA === void 0) type = "added";
23840
24188
  else if (cellB === void 0) type = "removed";
@@ -25371,6 +25719,9 @@ var Surgeon = class {
25371
25719
  miniFatSectors = [];
25372
25720
  dirSectors = [];
25373
25721
  entries = [];
25722
+ /** replace()가 FREESECT로 해제한 섹터 — finish()에서 재할당 안 된 것만 0으로 지움 (데이터 잔존 방지) */
25723
+ freedSectors = [];
25724
+ freedMiniSectors = [];
25374
25725
  constructor(file) {
25375
25726
  if (file.length < SECTOR || file.readUInt32LE(0) !== 3759263696) {
25376
25727
  throw new OleSurgeonError("OLE \uC2DC\uADF8\uB2C8\uCC98\uAC00 \uC544\uB2D9\uB2C8\uB2E4");
@@ -25494,6 +25845,8 @@ var Surgeon = class {
25494
25845
  if (this.fat[i] !== FREESECT) continue;
25495
25846
  if (SECTOR + (i + 1) * SECTOR > this.buf.length) continue;
25496
25847
  this.fat[i] = ENDOFCHAIN;
25848
+ const off = this.sectorOffset(i);
25849
+ this.buf.fill(0, off, off + SECTOR);
25497
25850
  out.push(i);
25498
25851
  }
25499
25852
  while (out.length < n) {
@@ -25593,9 +25946,15 @@ var Surgeon = class {
25593
25946
  const entry = this.findEntry(path);
25594
25947
  if (entry.size > 0 && entry.start !== ENDOFCHAIN) {
25595
25948
  if (entry.size < MINI_CUTOFF) {
25596
- for (const s of this.miniChain(entry.start)) this.miniFat[s] = FREESECT;
25949
+ for (const s of this.miniChain(entry.start)) {
25950
+ this.miniFat[s] = FREESECT;
25951
+ this.freedMiniSectors.push(s);
25952
+ }
25597
25953
  } else {
25598
- for (const s of this.chain(entry.start)) this.fat[s] = FREESECT;
25954
+ for (const s of this.chain(entry.start)) {
25955
+ this.fat[s] = FREESECT;
25956
+ this.freedSectors.push(s);
25957
+ }
25599
25958
  }
25600
25959
  }
25601
25960
  if (newData.length < MINI_CUTOFF) {
@@ -25624,9 +25983,27 @@ var Surgeon = class {
25624
25983
  this.writeDirEntry(entry);
25625
25984
  }
25626
25985
  finish() {
25986
+ this.wipeFreedSectors();
25627
25987
  this.flushFat();
25628
25988
  return this.buf;
25629
25989
  }
25990
+ /** 해제 후 재할당되지 않고 남은 FREESECT 섹터의 바이트를 0으로 채움 (데이터 remanence 제거) */
25991
+ wipeFreedSectors() {
25992
+ for (const s of this.freedSectors) {
25993
+ if (this.fat[s] !== FREESECT) continue;
25994
+ const off = this.sectorOffset(s);
25995
+ this.buf.fill(0, off, off + SECTOR);
25996
+ }
25997
+ if (this.freedMiniSectors.length > 0) {
25998
+ const root = this.rootEntry();
25999
+ const rootChain = root.start === ENDOFCHAIN || root.size === 0 ? [] : this.chain(root.start);
26000
+ for (const s of this.freedMiniSectors) {
26001
+ if (this.miniFat[s] !== FREESECT) continue;
26002
+ const off = this.miniOffset(s, rootChain);
26003
+ this.buf.fill(0, off, off + MINI_SECTOR);
26004
+ }
26005
+ }
26006
+ }
25630
26007
  };
25631
26008
 
25632
26009
  // src/roundtrip/hwp5-patch.ts
@@ -27057,7 +27434,7 @@ async function parseHwp3(buffer, options) {
27057
27434
  const { markdown, blocks, metadata, outline, warnings } = parseHwp3Document(buffer, options);
27058
27435
  return { success: true, fileType: "hwp3", markdown, blocks, metadata, outline, warnings };
27059
27436
  } catch (err) {
27060
- return { success: false, fileType: "hwp3", error: err instanceof Error ? err.message : "HWP3 \uD30C\uC2F1 \uC2E4\uD328", code: classifyError(err) };
27437
+ return { success: false, fileType: "hwp3", error: sanitizeError(err), code: classifyError(err) };
27061
27438
  }
27062
27439
  }
27063
27440
  async function parseHwpx(buffer, options) {
@@ -27065,7 +27442,7 @@ async function parseHwpx(buffer, options) {
27065
27442
  const { markdown, blocks, metadata, outline, warnings, images } = await parseHwpxDocument(buffer, options);
27066
27443
  return { success: true, fileType: "hwpx", markdown, blocks, metadata, outline, warnings, images: images?.length ? images : void 0 };
27067
27444
  } catch (err) {
27068
- return { success: false, fileType: "hwpx", error: err instanceof Error ? err.message : "HWPX \uD30C\uC2F1 \uC2E4\uD328", code: classifyError(err) };
27445
+ return { success: false, fileType: "hwpx", error: sanitizeError(err), code: classifyError(err) };
27069
27446
  }
27070
27447
  }
27071
27448
  async function parseHwp(buffer, options) {
@@ -27090,13 +27467,13 @@ async function parseHwp(buffer, options) {
27090
27467
  }
27091
27468
  return { success: true, fileType: "hwp", markdown, blocks, metadata, outline, warnings, images: images?.length ? images : void 0 };
27092
27469
  } catch (err) {
27093
- return { success: false, fileType: "hwp", error: err instanceof Error ? err.message : "HWP \uD30C\uC2F1 \uC2E4\uD328", code: classifyError(err) };
27470
+ return { success: false, fileType: "hwp", error: sanitizeError(err), code: classifyError(err) };
27094
27471
  }
27095
27472
  }
27096
27473
  async function parsePdf(buffer, options) {
27097
27474
  let parsePdfDocument;
27098
27475
  try {
27099
- const mod = await import("./parser-TCTCSBYZ.js");
27476
+ const mod = await import("./parser-PEOYBXBO.js");
27100
27477
  parsePdfDocument = mod.parsePdfDocument;
27101
27478
  } catch {
27102
27479
  return {
@@ -27111,7 +27488,7 @@ async function parsePdf(buffer, options) {
27111
27488
  return { success: true, fileType: "pdf", markdown, blocks, metadata, outline, warnings, isImageBased, pageQuality, qualitySummary, images };
27112
27489
  } catch (err) {
27113
27490
  const isImageBased = err instanceof Error && "isImageBased" in err ? true : void 0;
27114
- return { success: false, fileType: "pdf", error: err instanceof Error ? err.message : "PDF \uD30C\uC2F1 \uC2E4\uD328", code: classifyError(err), isImageBased };
27491
+ return { success: false, fileType: "pdf", error: sanitizeError(err), code: classifyError(err), isImageBased };
27115
27492
  }
27116
27493
  }
27117
27494
  async function parseXlsx(buffer, options) {
@@ -27119,7 +27496,7 @@ async function parseXlsx(buffer, options) {
27119
27496
  const { markdown, blocks, metadata, warnings } = await parseXlsxDocument(buffer, options);
27120
27497
  return { success: true, fileType: "xlsx", markdown, blocks, metadata, warnings };
27121
27498
  } catch (err) {
27122
- return { success: false, fileType: "xlsx", error: err instanceof Error ? err.message : "XLSX \uD30C\uC2F1 \uC2E4\uD328", code: classifyError(err) };
27499
+ return { success: false, fileType: "xlsx", error: sanitizeError(err), code: classifyError(err) };
27123
27500
  }
27124
27501
  }
27125
27502
  async function parseXls(buffer, options) {
@@ -27127,7 +27504,7 @@ async function parseXls(buffer, options) {
27127
27504
  const { markdown, blocks, metadata, warnings } = await parseXlsDocument(buffer, options);
27128
27505
  return { success: true, fileType: "xls", markdown, blocks, metadata, warnings };
27129
27506
  } catch (err) {
27130
- return { success: false, fileType: "xls", error: err instanceof Error ? err.message : "XLS \uD30C\uC2F1 \uC2E4\uD328", code: classifyError(err) };
27507
+ return { success: false, fileType: "xls", error: sanitizeError(err), code: classifyError(err) };
27131
27508
  }
27132
27509
  }
27133
27510
  async function parseDocx(buffer, options) {
@@ -27135,7 +27512,7 @@ async function parseDocx(buffer, options) {
27135
27512
  const { markdown, blocks, metadata, outline, warnings, images } = await parseDocxDocument(buffer, options);
27136
27513
  return { success: true, fileType: "docx", markdown, blocks, metadata, outline, warnings, images: images?.length ? images : void 0 };
27137
27514
  } catch (err) {
27138
- return { success: false, fileType: "docx", error: err instanceof Error ? err.message : "DOCX \uD30C\uC2F1 \uC2E4\uD328", code: classifyError(err) };
27515
+ return { success: false, fileType: "docx", error: sanitizeError(err), code: classifyError(err) };
27139
27516
  }
27140
27517
  }
27141
27518
  async function parseHwpml(buffer, options) {
@@ -27143,7 +27520,7 @@ async function parseHwpml(buffer, options) {
27143
27520
  const { markdown, blocks, metadata, outline, warnings } = parseHwpmlDocument(buffer, options);
27144
27521
  return { success: true, fileType: "hwpml", markdown, blocks, metadata, outline, warnings };
27145
27522
  } catch (err) {
27146
- return { success: false, fileType: "hwpml", error: err instanceof Error ? err.message : "HWPML \uD30C\uC2F1 \uC2E4\uD328", code: classifyError(err) };
27523
+ return { success: false, fileType: "hwpml", error: sanitizeError(err), code: classifyError(err) };
27147
27524
  }
27148
27525
  }
27149
27526
  async function fillForm(input, values, outputFormat = "markdown") {
@@ -27229,4 +27606,4 @@ export {
27229
27606
  parseHwpml,
27230
27607
  fillForm
27231
27608
  };
27232
- //# sourceMappingURL=chunk-7S3M4N4E.js.map
27609
+ //# sourceMappingURL=chunk-L2XQQTOY.js.map