kordoc 4.0.8 → 4.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (88) hide show
  1. package/README.md +25 -11
  2. package/dist/{-LE4RLXVA.js → -AS4IABL2.js} +23 -10
  3. package/dist/{chunk-2DZQF6YJ.js → chunk-37HJHCPA.js} +24 -15
  4. package/dist/chunk-37HJHCPA.js.map +1 -0
  5. package/dist/{chunk-6XAAYAWL.cjs → chunk-37VKMUST.cjs} +47 -37
  6. package/dist/chunk-37VKMUST.cjs.map +1 -0
  7. package/dist/{chunk-OQPILS7B.js → chunk-4KI23VHA.js} +33 -23
  8. package/dist/chunk-4KI23VHA.js.map +1 -0
  9. package/dist/{chunk-D35VBACN.js → chunk-AX2R5Q2N.js} +7 -4
  10. package/dist/chunk-AX2R5Q2N.js.map +1 -0
  11. package/dist/chunk-DZIXKL7E.js +145 -0
  12. package/dist/chunk-DZIXKL7E.js.map +1 -0
  13. package/dist/chunk-F2ZU3IZG.js +82 -0
  14. package/dist/chunk-F2ZU3IZG.js.map +1 -0
  15. package/dist/{chunk-BNU5QGIZ.js → chunk-GQVSHGG4.js} +12 -4
  16. package/dist/chunk-GQVSHGG4.js.map +1 -0
  17. package/dist/{chunk-7S3M4N4E.js → chunk-L2XQQTOY.js} +566 -189
  18. package/dist/chunk-L2XQQTOY.js.map +1 -0
  19. package/dist/{chunk-HMUEU7D7.js → chunk-O5VS7RIR.js} +2 -2
  20. package/dist/{chunk-UEIYETCQ.js → chunk-SSTK6IKK.js} +18 -160
  21. package/dist/chunk-SSTK6IKK.js.map +1 -0
  22. package/dist/chunk-UOO7LSDG.js +152 -0
  23. package/dist/chunk-UOO7LSDG.js.map +1 -0
  24. package/dist/{chunk-RROH5WHO.js → chunk-WO52HVQJ.js} +2 -2
  25. package/dist/{chunk-MEPHGCPQ.js → chunk-YOO6ET6M.js} +7 -3
  26. package/dist/chunk-YOO6ET6M.js.map +1 -0
  27. package/dist/chunks-WVMGF7JL.js +10 -0
  28. package/dist/cli.js +170 -35
  29. package/dist/cli.js.map +1 -1
  30. package/dist/{detect-RI2MQ33K.js → detect-YEZX2XCF.js} +2 -2
  31. package/dist/index.cjs +1142 -531
  32. package/dist/index.cjs.map +1 -1
  33. package/dist/index.d.cts +105 -1
  34. package/dist/index.d.ts +105 -1
  35. package/dist/index.js +802 -191
  36. package/dist/index.js.map +1 -1
  37. package/dist/mcp.js +273 -65
  38. package/dist/mcp.js.map +1 -1
  39. package/dist/{parser-ITMU2WN5.cjs → parser-N4A7UGDK.cjs} +127 -45
  40. package/dist/parser-N4A7UGDK.cjs.map +1 -0
  41. package/dist/{parser-TCTCSBYZ.js → parser-PEOYBXBO.js} +118 -34
  42. package/dist/parser-PEOYBXBO.js.map +1 -0
  43. package/dist/{parser-KMMUS73Q.js → parser-XX4QTDFQ.js} +114 -32
  44. package/dist/parser-XX4QTDFQ.js.map +1 -0
  45. package/dist/{profile-io-SLN4P76T.js → profile-io-PLDQFPJA.js} +3 -3
  46. package/dist/{provider-G4C2V2PD.cjs → provider-4FAYHJ6N.cjs} +3 -3
  47. package/dist/provider-4FAYHJ6N.cjs.map +1 -0
  48. package/dist/{provider-4ZJKV3DC.js → provider-C6IOGIS5.js} +3 -3
  49. package/dist/provider-C6IOGIS5.js.map +1 -0
  50. package/dist/{provider-AKROB7WQ.js → provider-UIBW2MIH.js} +3 -3
  51. package/dist/provider-UIBW2MIH.js.map +1 -0
  52. package/dist/rasterize-WWNQLKYW.js +40 -0
  53. package/dist/rasterize-WWNQLKYW.js.map +1 -0
  54. package/dist/redact-7ILEAUTS.js +12 -0
  55. package/dist/render-T7S6H6AN.js +10 -0
  56. package/dist/render-T7S6H6AN.js.map +1 -0
  57. package/dist/seal-7PTL6MXH.js +10 -0
  58. package/dist/seal-7PTL6MXH.js.map +1 -0
  59. package/dist/{setup-57FB3LSP.js → setup-Q2PRE7UA.js} +5 -3
  60. package/dist/setup-Q2PRE7UA.js.map +1 -0
  61. package/dist/{watch-OHQWSVPE.js → watch-SFHXZALC.js} +77 -23
  62. package/dist/watch-SFHXZALC.js.map +1 -0
  63. package/package.json +1 -1
  64. package/dist/chunk-2DZQF6YJ.js.map +0 -1
  65. package/dist/chunk-6XAAYAWL.cjs.map +0 -1
  66. package/dist/chunk-7S3M4N4E.js.map +0 -1
  67. package/dist/chunk-BNU5QGIZ.js.map +0 -1
  68. package/dist/chunk-D35VBACN.js.map +0 -1
  69. package/dist/chunk-MEPHGCPQ.js.map +0 -1
  70. package/dist/chunk-OQPILS7B.js.map +0 -1
  71. package/dist/chunk-UEIYETCQ.js.map +0 -1
  72. package/dist/parser-ITMU2WN5.cjs.map +0 -1
  73. package/dist/parser-KMMUS73Q.js.map +0 -1
  74. package/dist/parser-TCTCSBYZ.js.map +0 -1
  75. package/dist/provider-4ZJKV3DC.js.map +0 -1
  76. package/dist/provider-AKROB7WQ.js.map +0 -1
  77. package/dist/provider-G4C2V2PD.cjs.map +0 -1
  78. package/dist/render-25BT623I.js +0 -10
  79. package/dist/seal-R4TETA4C.js +0 -10
  80. package/dist/setup-57FB3LSP.js.map +0 -1
  81. package/dist/watch-OHQWSVPE.js.map +0 -1
  82. /package/dist/{-LE4RLXVA.js.map → -AS4IABL2.js.map} +0 -0
  83. /package/dist/{chunk-HMUEU7D7.js.map → chunk-O5VS7RIR.js.map} +0 -0
  84. /package/dist/{chunk-RROH5WHO.js.map → chunk-WO52HVQJ.js.map} +0 -0
  85. /package/dist/{detect-RI2MQ33K.js.map → chunks-WVMGF7JL.js.map} +0 -0
  86. /package/dist/{profile-io-SLN4P76T.js.map → detect-YEZX2XCF.js.map} +0 -0
  87. /package/dist/{render-25BT623I.js.map → profile-io-PLDQFPJA.js.map} +0 -0
  88. /package/dist/{seal-R4TETA4C.js.map → redact-7ILEAUTS.js.map} +0 -0
package/dist/index.js CHANGED
@@ -18,10 +18,11 @@ import {
18
18
  mapPuaText,
19
19
  normalizeSectionHref,
20
20
  precheckZipSize,
21
+ sanitizeError,
21
22
  sanitizeHref,
22
23
  stripDtd,
23
24
  toArrayBuffer
24
- } from "./chunk-OQPILS7B.js";
25
+ } from "./chunk-4KI23VHA.js";
25
26
  import {
26
27
  parsePageRange
27
28
  } from "./chunk-GE43BE46.js";
@@ -48,8 +49,11 @@ function parseLenientCfb(data) {
48
49
  const miniSectorSizeShift = data.readUInt16LE(32);
49
50
  if (miniSectorSizeShift > 16) throw new Error("\uC720\uD6A8\uD558\uC9C0 \uC54A\uC740 \uBBF8\uB2C8 \uC139\uD130 \uD06C\uAE30 \uC2DC\uD504\uD2B8: " + miniSectorSizeShift);
50
51
  const miniSectorSize = 1 << miniSectorSizeShift;
51
- const fatSectorCount = data.readUInt32LE(44);
52
- if (fatSectorCount > 1e4) throw new Error("FAT \uC139\uD130 \uC218\uAC00 \uB108\uBB34 \uB9CE\uC2B5\uB2C8\uB2E4: " + fatSectorCount);
52
+ const headerFatSectorCount = data.readUInt32LE(44);
53
+ if (headerFatSectorCount > 1e4) throw new Error("FAT \uC139\uD130 \uC218\uAC00 \uB108\uBB34 \uB9CE\uC2B5\uB2C8\uB2E4: " + headerFatSectorCount);
54
+ const maxRealSectors = Math.ceil(Math.max(0, data.length - 512) / sectorSize);
55
+ const maxFatSectors = Math.max(1, Math.ceil(maxRealSectors / (sectorSize / 4)));
56
+ const fatSectorCount = Math.min(headerFatSectorCount, maxFatSectors);
53
57
  const firstDirSector = data.readUInt32LE(48);
54
58
  const miniStreamCutoff = data.readUInt32LE(56);
55
59
  const firstMiniFatSector = data.readUInt32LE(60);
@@ -76,6 +80,7 @@ function parseLenientCfb(data) {
76
80
  if (visitedDifat.has(difatSector)) break;
77
81
  visitedDifat.add(difatSector);
78
82
  const buf = readSectorData(difatSector);
83
+ if (buf.length < sectorSize) break;
79
84
  const entriesPerSector = sectorSize / 4 - 1;
80
85
  for (let i = 0; i < entriesPerSector && fatSectors.length < fatSectorCount; i++) {
81
86
  const sid = buf.readUInt32LE(i * 4);
@@ -394,6 +399,12 @@ function comResultToParseResult(pages, pageCount, warnings) {
394
399
  import { DOMParser } from "@xmldom/xmldom";
395
400
  var MAX_DECOMPRESS_SIZE = 100 * 1024 * 1024;
396
401
  var MAX_ZIP_ENTRIES = 500;
402
+ var ZipBombError = class extends KordocError {
403
+ constructor(message) {
404
+ super(message);
405
+ this.name = "ZipBombError";
406
+ }
407
+ };
397
408
  function clampSpan(val, max) {
398
409
  return Math.max(1, Math.min(val, max));
399
410
  }
@@ -429,14 +440,15 @@ function findChildByLocalName(parent, name) {
429
440
  }
430
441
  return null;
431
442
  }
432
- function extractTextFromNode(node) {
443
+ function extractTextFromNode(node, depth = 0) {
433
444
  let result = "";
445
+ if (depth > MAX_XML_DEPTH) return result;
434
446
  const children = node.childNodes;
435
447
  if (!children) return result;
436
448
  for (let i = 0; i < children.length; i++) {
437
449
  const child = children[i];
438
450
  if (child.nodeType === 3) result += child.textContent || "";
439
- else if (child.nodeType === 1) result += extractTextFromNode(child);
451
+ else if (child.nodeType === 1) result += extractTextFromNode(child, depth + 1);
440
452
  }
441
453
  return result.trim();
442
454
  }
@@ -1320,7 +1332,7 @@ function simulateWrap(text, firstWidth, contWidth, height, ratioPct, mode = "kee
1320
1332
  for (const ch of text.slice(from, to)) w += charW2(ch);
1321
1333
  return w;
1322
1334
  };
1323
- const units = text.match(mode === "keep" ? / +|[^ ]+/g : / +|[^ ]/g) ?? [];
1335
+ const units = text.match(mode === "keep" ? / +|[^ ]+/gu : / +|[^ ]/gu) ?? [];
1324
1336
  const starts = [0];
1325
1337
  let lineW = 0;
1326
1338
  let avail = firstWidth;
@@ -1789,6 +1801,7 @@ function toRoman(n) {
1789
1801
  return out;
1790
1802
  }
1791
1803
  function formatHeadNumber(n, numFormat) {
1804
+ if (n === 0 && numFormat === "DIGIT") return "0";
1792
1805
  if (n <= 0) n = 1;
1793
1806
  switch (numFormat) {
1794
1807
  case "DIGIT":
@@ -1846,25 +1859,25 @@ function resolveParaHeading(paraEl, ctx) {
1846
1859
  if (!numDef) return headingLevel ? { headingLevel } : null;
1847
1860
  let counters = ctx.shared.numState.get(numId);
1848
1861
  if (!counters) {
1849
- counters = new Array(11).fill(0);
1862
+ counters = new Array(11).fill(-1);
1850
1863
  ctx.shared.numState.set(numId, counters);
1851
1864
  }
1852
1865
  const head = numDef.heads.get(level);
1853
- counters[level] = counters[level] === 0 ? head?.start ?? 1 : counters[level] + 1;
1854
- for (let l = level + 1; l <= 10; l++) counters[l] = 0;
1866
+ counters[level] = counters[level] < 0 ? head?.start ?? 1 : counters[level] + 1;
1867
+ for (let l = level + 1; l <= 10; l++) counters[l] = -1;
1855
1868
  const fmtText = head ? head.text.trim() : `^${level}.`;
1856
1869
  const prefix = fmtText.replace(/\^(10|[1-9])/g, (_, d) => {
1857
1870
  const lv = parseInt(d, 10);
1858
1871
  const refHead = numDef.heads.get(lv);
1859
- const n = counters[lv] || refHead?.start || 1;
1872
+ const n = counters[lv] >= 0 ? counters[lv] : refHead?.start ?? 1;
1860
1873
  return formatHeadNumber(n, refHead?.numFormat || "DIGIT");
1861
1874
  });
1862
1875
  return { prefix: prefix || void 0, headingLevel };
1863
1876
  }
1864
1877
 
1865
1878
  // src/hwpx/table-build.ts
1866
- function buildTableWithCellMeta(state) {
1867
- const table2 = buildTable(state.rows);
1879
+ function buildTableWithCellMeta(state, keepAnchoredEmptyCols) {
1880
+ const table2 = buildTable(state.rows, { keepAnchoredEmptyCols });
1868
1881
  if (state.caption) table2.caption = state.caption;
1869
1882
  const anchors = [];
1870
1883
  {
@@ -1928,7 +1941,7 @@ function completeTable(newTable, tableStack, blocks, ctx) {
1928
1941
  if (newTable.caption) blocks.push({ type: "paragraph", text: newTable.caption, pageNumber: ctx.sectionNum });
1929
1942
  return parentTable;
1930
1943
  }
1931
- const ir = buildTableWithCellMeta(newTable);
1944
+ const ir = buildTableWithCellMeta(newTable, ctx.shared.keepTrailingEmptyCols);
1932
1945
  const block = { type: "table", table: ir, pageNumber: ctx.sectionNum };
1933
1946
  if (parentTable?.cell) {
1934
1947
  const cell2 = parentTable.cell;
@@ -1962,7 +1975,8 @@ function parseSectionXml(xml, styleMap, warnings, sectionNum, shared) {
1962
1975
  walkSection(doc.documentElement, blocks, null, [], ctx);
1963
1976
  return blocks;
1964
1977
  }
1965
- function extractImageRef(el) {
1978
+ function extractImageRef(el, depth = 0) {
1979
+ if (depth > MAX_XML_DEPTH) return null;
1966
1980
  const children = el.childNodes;
1967
1981
  if (!children) return null;
1968
1982
  for (let i = 0; i < children.length; i++) {
@@ -1973,7 +1987,7 @@ function extractImageRef(el) {
1973
1987
  const ref = child.getAttribute("binaryItemIDRef") || child.getAttribute("href") || "";
1974
1988
  if (ref) return ref;
1975
1989
  }
1976
- const nested = extractImageRef(child);
1990
+ const nested = extractImageRef(child, depth + 1);
1977
1991
  if (nested) return nested;
1978
1992
  }
1979
1993
  const directRef = el.getAttribute("binaryItemIDRef") || "";
@@ -2080,7 +2094,7 @@ function walkSection(node, blocks, tableCtx, tableStack, ctx, depth = 0) {
2080
2094
  ;
2081
2095
  (cell2.blocks ??= []).push(cellBlock);
2082
2096
  } else if (!tableCtx) {
2083
- if (/^─{10,}$/.test(text)) {
2097
+ if (ctx.shared.kordocLayout && /^─{10,}$/.test(text)) {
2084
2098
  blocks.push({ type: "separator", pageNumber: ctx.sectionNum });
2085
2099
  tableCtx = walkParagraphChildren(el, blocks, tableCtx, tableStack, ctx, depth + 1);
2086
2100
  break;
@@ -2356,8 +2370,8 @@ function extractDrawTextBlocks(drawTextNode, blocks, ctx) {
2356
2370
  } else {
2357
2371
  const info = extractParagraphInfo(child, ctx.styleMap, ctx);
2358
2372
  let text = info.text.trim();
2373
+ const ph = resolveParaHeading(child, ctx);
2359
2374
  if (text) {
2360
- const ph = resolveParaHeading(child, ctx);
2361
2375
  if (ph?.prefix) text = ph.prefix + " " + text;
2362
2376
  const block = { type: "paragraph", text, style: info.style ?? void 0, pageNumber: ctx.sectionNum };
2363
2377
  if (info.href) block.href = info.href;
@@ -2470,7 +2484,8 @@ function extractParagraphInfo(para2, styleMap, ctx) {
2470
2484
  }
2471
2485
  }
2472
2486
  };
2473
- const walk = (node) => {
2487
+ const walk = (node, depth = 0) => {
2488
+ if (depth > MAX_XML_DEPTH) return;
2474
2489
  const children = node.childNodes;
2475
2490
  if (!children) return;
2476
2491
  for (let i = 0; i < children.length; i++) {
@@ -2491,7 +2506,7 @@ function extractParagraphInfo(para2, styleMap, ctx) {
2491
2506
  const tag = (child.tagName || child.localName || "").replace(/^[^:]+:/, "");
2492
2507
  switch (tag) {
2493
2508
  case "t":
2494
- walk(child);
2509
+ walk(child, depth + 1);
2495
2510
  break;
2496
2511
  // 자식 순회 (tab 등 하위 요소 처리)
2497
2512
  case "tab": {
@@ -2524,7 +2539,7 @@ function extractParagraphInfo(para2, styleMap, ctx) {
2524
2539
  const safe = sanitizeHref(url);
2525
2540
  if (safe) href = safe;
2526
2541
  }
2527
- walk(child);
2542
+ walk(child, depth + 1);
2528
2543
  break;
2529
2544
  }
2530
2545
  // 각주/미주
@@ -2595,11 +2610,11 @@ function extractParagraphInfo(para2, styleMap, ctx) {
2595
2610
  case "r": {
2596
2611
  const runCharPr = child.getAttribute("charPrIDRef");
2597
2612
  if (runCharPr && !charPrId) charPrId = runCharPr;
2598
- walk(child);
2613
+ walk(child, depth + 1);
2599
2614
  break;
2600
2615
  }
2601
2616
  default:
2602
- walk(child);
2617
+ walk(child, depth + 1);
2603
2618
  break;
2604
2619
  }
2605
2620
  }
@@ -2610,7 +2625,7 @@ function extractParagraphInfo(para2, styleMap, ctx) {
2610
2625
  let cleanText = text.replace(/[ \t]+/g, " ").trim();
2611
2626
  if (/^그림입니다\.?\s*원본\s*그림의\s*(이름|크기)/.test(cleanText)) cleanText = "";
2612
2627
  cleanText = cleanText.replace(/그림입니다\.?\s*원본\s*그림의\s*(이름|크기)[^\n]*(\n[^\n]*원본\s*그림의\s*(이름|크기)[^\n]*)*/g, "").trim();
2613
- cleanText = cleanText.replace(/(?:모서리가 둥근 |둥근 )?(?:사각형|직사각형|정사각형|원|타원|삼각형|선|직선|곡선|화살표|오각형|육각형|팔각형|별|십자|구름|마름모|도넛|평행사변형|사다리꼴|개체|그리기\s?개체|묶음\s?개체|글상자|표|그림|OLE\s?개체)\s?입니다\.?/g, "").trim();
2628
+ cleanText = cleanText.replace(/^(?:모서리가 둥근 |둥근 )?(?:사각형|직사각형|정사각형|원|타원|삼각형|선|직선|곡선|화살표|오각형|육각형|팔각형|별|십자|구름|마름모|도넛|평행사변형|사다리꼴|개체|그리기\s?개체|묶음\s?개체|글상자|표|그림|OLE\s?개체)\s?입니다\.?$/gm, "").trim();
2614
2629
  let style;
2615
2630
  if (styleMap && charPrId) {
2616
2631
  const charProp = styleMap.charProperties.get(charPrId);
@@ -2734,10 +2749,9 @@ var CHAR_LINE = 0;
2734
2749
  var CHAR_SECTION_BREAK = 10;
2735
2750
  var CHAR_PARA = 13;
2736
2751
  var CHAR_TAB = 9;
2737
- var CHAR_HYPHEN = 30;
2738
- var CHAR_NBSP = 31;
2739
- var CHAR_FIXED_NBSP = 24;
2740
- var CHAR_FIXED_WIDTH = 25;
2752
+ var CHAR_HYPHEN = 24;
2753
+ var CHAR_NBSP = 30;
2754
+ var CHAR_FIXED_WIDTH = 31;
2741
2755
  var FLAG_COMPRESSED = 1 << 0;
2742
2756
  var FLAG_ENCRYPTED = 1 << 1;
2743
2757
  var FLAG_DISTRIBUTION = 1 << 2;
@@ -2936,9 +2950,6 @@ function appendParaText(state, data, resolveControl) {
2936
2950
  result += "-";
2937
2951
  break;
2938
2952
  case CHAR_NBSP:
2939
- result += " ";
2940
- break;
2941
- case CHAR_FIXED_NBSP:
2942
2953
  result += "\xA0";
2943
2954
  break;
2944
2955
  // 진짜 NBSP
@@ -3237,8 +3248,8 @@ async function extractImagesFromZip(zip, blocks, decompressed, warnings, sweepUn
3237
3248
  const ext = path.includes(".") ? path.split(".").pop() || "png" : "png";
3238
3249
  const mimeType = imageExtToMime(ext);
3239
3250
  imageIndex++;
3240
- const filename = `image_${String(imageIndex).padStart(3, "0")}.${mimeToExt(mimeType)}`;
3241
- img = { filename, data, mimeType };
3251
+ const filename2 = `image_${String(imageIndex).padStart(3, "0")}.${mimeToExt(mimeType)}`;
3252
+ img = { filename: filename2, data, mimeType };
3242
3253
  images.push(img);
3243
3254
  usedPaths.add(path);
3244
3255
  break;
@@ -3252,12 +3263,13 @@ async function extractImagesFromZip(zip, blocks, decompressed, warnings, sweepUn
3252
3263
  if (!img) {
3253
3264
  block.type = "paragraph";
3254
3265
  block.text = `[\uC774\uBBF8\uC9C0: ${ref}]`;
3255
- if (ownerCell) ownerCell.text = ownerCell.text.replace(`![image](${ref})`, `[\uC774\uBBF8\uC9C0: ${ref}]`);
3266
+ if (ownerCell) ownerCell.text = ownerCell.text.replace(`![image](${ref})`, () => `[\uC774\uBBF8\uC9C0: ${ref}]`);
3256
3267
  continue;
3257
3268
  }
3258
- block.text = img.filename;
3269
+ const filename = img.filename;
3270
+ block.text = filename;
3259
3271
  block.imageData = { data: img.data, mimeType: img.mimeType, filename: ref };
3260
- if (ownerCell) ownerCell.text = ownerCell.text.replace(`![image](${ref})`, `![image](${img.filename})`);
3272
+ if (ownerCell) ownerCell.text = ownerCell.text.replace(`![image](${ref})`, () => `![image](${filename})`);
3261
3273
  }
3262
3274
  if (sweepUnreferenced) {
3263
3275
  const binEntries = zip.file(/(?:^|\/)BinData\//i);
@@ -3490,6 +3502,7 @@ async function parseHwpxDocument(buffer, options) {
3490
3502
  const blocks = [];
3491
3503
  const shared = createSectionShared();
3492
3504
  shared.kordocLayout = await readKordocLayout(zip);
3505
+ shared.keepTrailingEmptyCols = options?.keepTrailingEmptyCols;
3493
3506
  let parsedSections = 0;
3494
3507
  for (let si = 0; si < sectionPaths.length; si++) {
3495
3508
  if (pageFilter && !pageFilter.has(si + 1)) continue;
@@ -3498,12 +3511,12 @@ async function parseHwpxDocument(buffer, options) {
3498
3511
  try {
3499
3512
  const xml = await file.async("text");
3500
3513
  decompressed.total += xml.length * 2;
3501
- if (decompressed.total > MAX_DECOMPRESS_SIZE) throw new KordocError("ZIP \uC555\uCD95 \uD574\uC81C \uD06C\uAE30 \uCD08\uACFC (ZIP bomb \uC758\uC2EC)");
3514
+ if (decompressed.total > MAX_DECOMPRESS_SIZE) throw new ZipBombError("ZIP \uC555\uCD95 \uD574\uC81C \uD06C\uAE30 \uCD08\uACFC (ZIP bomb \uC758\uC2EC)");
3502
3515
  blocks.push(...parseSectionXml(xml, styleMap, warnings, si + 1, shared));
3503
3516
  parsedSections++;
3504
3517
  options?.onProgress?.(parsedSections, totalTarget);
3505
3518
  } catch (secErr) {
3506
- if (secErr instanceof KordocError) throw secErr;
3519
+ if (secErr instanceof ZipBombError) throw secErr;
3507
3520
  warnings.push({ page: si + 1, message: `\uC139\uC158 ${si + 1} \uD30C\uC2F1 \uC2E4\uD328: ${secErr instanceof Error ? secErr.message : "\uC54C \uC218 \uC5C6\uB294 \uC624\uB958"}`, code: "PARTIAL_PARSE" });
3508
3521
  }
3509
3522
  }
@@ -4720,6 +4733,7 @@ function parseHwp5Document(buffer, options) {
4720
4733
  const totalTarget = pageFilter ? pageFilter.size : sections.length;
4721
4734
  const bodyBlocks = [];
4722
4735
  const doc = createHwp5DocState();
4736
+ doc.keepTrailingEmptyCols = options?.keepTrailingEmptyCols;
4723
4737
  let totalDecompressed = 0;
4724
4738
  let parsedSections = 0;
4725
4739
  for (let si = 0; si < sections.length; si++) {
@@ -5321,7 +5335,7 @@ function parseTableControl(ctrl, records, ctx) {
5321
5335
  return table3;
5322
5336
  }
5323
5337
  const cellRows = arrangeCells(rows, cols, cells);
5324
- const table2 = buildTable(cellRows);
5338
+ const table2 = buildTable(cellRows, { keepAnchoredEmptyCols: ctx.doc.keepTrailingEmptyCols });
5325
5339
  if (caption && table2.rows > 0) table2.caption = caption;
5326
5340
  return table2.rows > 0 ? table2 : null;
5327
5341
  }
@@ -17515,6 +17529,7 @@ function readHeader(reader) {
17515
17529
  }
17516
17530
 
17517
17531
  // src/hwp3/parser.ts
17532
+ var MAX_DECOMPRESS_SIZE3 = 100 * 1024 * 1024;
17518
17533
  var PARA_SHAPE_SIZE = 187;
17519
17534
  var LINE_INFO_SIZE = 14;
17520
17535
  var INLINE_CHAR_SHAPE_SIZE = 31;
@@ -17550,8 +17565,11 @@ function parseHwp3Document(buffer, _options) {
17550
17565
  const warnings = [];
17551
17566
  if (header.compressed !== 0) {
17552
17567
  try {
17553
- body = inflateRawSync3(tail);
17568
+ body = inflateRawSync3(tail, { maxOutputLength: MAX_DECOMPRESS_SIZE3 });
17554
17569
  } catch (err) {
17570
+ if (err?.code === "ERR_BUFFER_TOO_LARGE") {
17571
+ throw new Error(`HWP3 \uC555\uCD95 \uD574\uC81C \uACB0\uACFC\uAC00 \uCD5C\uB300 \uD5C8\uC6A9 \uD06C\uAE30(${MAX_DECOMPRESS_SIZE3 / 1024 / 1024}MB)\uB97C \uCD08\uACFC\uD588\uC2B5\uB2C8\uB2E4`);
17572
+ }
17555
17573
  const msg2 = err instanceof Error ? err.message : String(err);
17556
17574
  throw new Error(`HWP3 \uC555\uCD95 \uD574\uC81C \uC2E4\uD328: ${msg2}`);
17557
17575
  }
@@ -17748,7 +17766,7 @@ function isDistributionSentinel(markdown) {
17748
17766
  import JSZip4 from "jszip";
17749
17767
  import { DOMParser as DOMParser2 } from "@xmldom/xmldom";
17750
17768
  var MAX_SHEETS = 100;
17751
- var MAX_DECOMPRESS_SIZE3 = 100 * 1024 * 1024;
17769
+ var MAX_DECOMPRESS_SIZE4 = 100 * 1024 * 1024;
17752
17770
  var MAX_ROWS2 = 1e4;
17753
17771
  var MAX_COLS2 = 200;
17754
17772
  function cleanNumericValue(raw) {
@@ -17788,16 +17806,87 @@ function getTextContent(el) {
17788
17806
  function parseXml(text) {
17789
17807
  return new DOMParser2().parseFromString(stripDtd(text), "text/xml");
17790
17808
  }
17809
+ function collectRichText(root) {
17810
+ let out = "";
17811
+ const walk = (node) => {
17812
+ const children = node.childNodes;
17813
+ for (let i = 0; i < children.length; i++) {
17814
+ if (children[i].nodeType !== 1) continue;
17815
+ const el = children[i];
17816
+ const local = el.localName || el.tagName?.replace(/^[^:]+:/, "") || "";
17817
+ if (local === "rPh") continue;
17818
+ if (local === "t") out += el.textContent ?? "";
17819
+ else walk(el);
17820
+ }
17821
+ };
17822
+ walk(root);
17823
+ return out;
17824
+ }
17791
17825
  function parseSharedStrings(xml) {
17792
17826
  const doc = parseXml(xml);
17793
17827
  const strings = [];
17794
17828
  const siList = getElements(doc.documentElement, "si");
17795
17829
  for (const si of siList) {
17796
- const tElements = getElements(si, "t");
17797
- strings.push(tElements.map((t) => t.textContent ?? "").join(""));
17830
+ strings.push(collectRichText(si));
17798
17831
  }
17799
17832
  return strings;
17800
17833
  }
17834
+ var BUILTIN_DATE_FMT = /* @__PURE__ */ new Map([
17835
+ [14, "date"],
17836
+ [15, "date"],
17837
+ [16, "date"],
17838
+ [17, "date"],
17839
+ [18, "datetime"],
17840
+ [19, "datetime"],
17841
+ [20, "datetime"],
17842
+ [21, "datetime"],
17843
+ [22, "datetime"],
17844
+ [45, "datetime"],
17845
+ [46, "datetime"],
17846
+ [47, "datetime"]
17847
+ ]);
17848
+ function classifyDateFormat(code) {
17849
+ const stripped = code.replace(/"[^"]*"/g, "").replace(/\[[^\]]*\]/g, "").replace(/\\./g, "");
17850
+ if (!/[ymdh]/i.test(stripped)) return null;
17851
+ return /[hs]/i.test(stripped) ? "datetime" : "date";
17852
+ }
17853
+ function dateKindOfFmt(fmtId, customFormats) {
17854
+ const builtin = BUILTIN_DATE_FMT.get(fmtId);
17855
+ if (builtin) return builtin;
17856
+ const code = customFormats.get(fmtId);
17857
+ return code !== void 0 ? classifyDateFormat(code) : null;
17858
+ }
17859
+ function dateSerialToIso(serial, date1904, kind) {
17860
+ if (!isFinite(serial) || serial < 0) return null;
17861
+ let days = serial;
17862
+ if (date1904) days += 1462;
17863
+ else if (days < 60) days += 1;
17864
+ const ms = Math.round((days - 25569) * 864e5);
17865
+ const d = new Date(ms);
17866
+ if (isNaN(d.getTime()) || d.getUTCFullYear() > 9999) return null;
17867
+ const iso = d.toISOString();
17868
+ return kind === "datetime" ? iso.slice(0, 19) : iso.slice(0, 10);
17869
+ }
17870
+ function parseStyleDateXfs(xml) {
17871
+ const doc = parseXml(xml);
17872
+ const customFormats = /* @__PURE__ */ new Map();
17873
+ for (const el of getElements(doc.documentElement, "numFmt")) {
17874
+ const id = parseInt(el.getAttribute("numFmtId") ?? "", 10);
17875
+ if (isNaN(id)) continue;
17876
+ customFormats.set(id, el.getAttribute("formatCode") ?? "");
17877
+ }
17878
+ const dateXfs = /* @__PURE__ */ new Map();
17879
+ const cellXfsEls = getElements(doc.documentElement, "cellXfs");
17880
+ if (cellXfsEls.length === 0) return dateXfs;
17881
+ const xfs = getElements(cellXfsEls[0], "xf");
17882
+ for (let i = 0; i < xfs.length; i++) {
17883
+ const fmtId = parseInt(xfs[i].getAttribute("numFmtId") ?? "", 10);
17884
+ if (isNaN(fmtId)) continue;
17885
+ const kind = dateKindOfFmt(fmtId, customFormats);
17886
+ if (kind) dateXfs.set(i, kind);
17887
+ }
17888
+ return dateXfs;
17889
+ }
17801
17890
  function parseWorkbook(xml) {
17802
17891
  const doc = parseXml(xml);
17803
17892
  const sheets = [];
@@ -17809,7 +17898,9 @@ function parseWorkbook(xml) {
17809
17898
  rId: el.getAttribute("r:id") ?? ""
17810
17899
  });
17811
17900
  }
17812
- return sheets;
17901
+ const prEls = getElements(doc.documentElement, "workbookPr");
17902
+ const d1904 = prEls.length > 0 ? prEls[0].getAttribute("date1904") : null;
17903
+ return { sheets, date1904: d1904 === "1" || d1904 === "true" };
17813
17904
  }
17814
17905
  function parseRels(xml) {
17815
17906
  const doc = parseXml(xml);
@@ -17822,21 +17913,25 @@ function parseRels(xml) {
17822
17913
  }
17823
17914
  return map;
17824
17915
  }
17825
- function parseWorksheet(xml, sharedStrings) {
17916
+ function parseWorksheet(xml, sharedStrings, dateXfs, date1904) {
17826
17917
  const doc = parseXml(xml);
17827
17918
  const grid = [];
17828
17919
  let maxRow = 0;
17829
17920
  let maxCol = 0;
17830
17921
  const rows = getElements(doc.documentElement, "row");
17922
+ let prevRow = -1;
17831
17923
  for (const rowEl of rows) {
17832
- const rowNum = parseInt(rowEl.getAttribute("r") ?? "0", 10) - 1;
17924
+ const rAttr = rowEl.getAttribute("r");
17925
+ const rowNum = rAttr !== null ? parseInt(rAttr, 10) - 1 : prevRow + 1;
17833
17926
  if (rowNum < 0 || rowNum >= MAX_ROWS2) continue;
17927
+ if (Number.isFinite(rowNum)) prevRow = rowNum;
17834
17928
  const cells = getElements(rowEl, "c");
17929
+ let prevCol = -1;
17835
17930
  for (const cellEl of cells) {
17836
17931
  const ref = cellEl.getAttribute("r");
17837
- if (!ref) continue;
17838
- const pos = parseCellRef(ref);
17839
- if (!pos || pos.col >= MAX_COLS2) continue;
17932
+ const pos = ref !== null ? parseCellRef(ref) : { col: prevCol + 1, row: rowNum };
17933
+ if (!pos || !Number.isFinite(pos.row) || pos.row < 0 || pos.row >= MAX_ROWS2 || pos.col >= MAX_COLS2) continue;
17934
+ prevCol = pos.col;
17840
17935
  const type = cellEl.getAttribute("t");
17841
17936
  const vElements = getElements(cellEl, "v");
17842
17937
  const fElements = getElements(cellEl, "f");
@@ -17850,12 +17945,19 @@ function parseWorksheet(xml, sharedStrings) {
17850
17945
  value = raw === "1" ? "TRUE" : "FALSE";
17851
17946
  } else {
17852
17947
  value = cleanNumericValue(raw);
17948
+ if (type === null || type === "n") {
17949
+ const sAttr = cellEl.getAttribute("s");
17950
+ const kind = sAttr !== null ? dateXfs.get(parseInt(sAttr, 10)) : void 0;
17951
+ if (kind) {
17952
+ const iso = dateSerialToIso(parseFloat(raw), date1904, kind);
17953
+ if (iso) value = iso;
17954
+ }
17955
+ }
17853
17956
  }
17854
17957
  } else if (type === "inlineStr") {
17855
17958
  const isEl = getElements(cellEl, "is");
17856
17959
  if (isEl.length > 0) {
17857
- const tElements = getElements(isEl[0], "t");
17858
- value = tElements.map((t) => t.textContent ?? "").join("");
17960
+ value = collectRichText(isEl[0]);
17859
17961
  }
17860
17962
  }
17861
17963
  if (!value && fElements.length > 0) {
@@ -17874,7 +17976,14 @@ function parseWorksheet(xml, sharedStrings) {
17874
17976
  const ref = el.getAttribute("ref");
17875
17977
  if (!ref) continue;
17876
17978
  const m = parseMergeRef(ref);
17877
- if (m) merges.push(m);
17979
+ if (m) {
17980
+ merges.push({
17981
+ startCol: Math.min(m.startCol, MAX_COLS2 - 1),
17982
+ startRow: Math.min(m.startRow, MAX_ROWS2 - 1),
17983
+ endCol: Math.min(m.endCol, MAX_COLS2 - 1),
17984
+ endRow: Math.min(m.endRow, MAX_ROWS2 - 1)
17985
+ });
17986
+ }
17878
17987
  }
17879
17988
  return { grid, merges, maxRow, maxCol };
17880
17989
  }
@@ -17938,7 +18047,7 @@ function sheetToBlocks(sheetName, grid, merges, maxRow, maxCol, sheetIndex) {
17938
18047
  return blocks;
17939
18048
  }
17940
18049
  async function parseXlsxDocument(buffer, options) {
17941
- precheckZipSize(buffer, MAX_DECOMPRESS_SIZE3);
18050
+ precheckZipSize(buffer, MAX_DECOMPRESS_SIZE4);
17942
18051
  const zip = await JSZip4.loadAsync(buffer);
17943
18052
  const warnings = [];
17944
18053
  const workbookFile = zip.file("xl/workbook.xml");
@@ -17950,10 +18059,18 @@ async function parseXlsxDocument(buffer, options) {
17950
18059
  if (ssFile) {
17951
18060
  sharedStrings = parseSharedStrings(await ssFile.async("text"));
17952
18061
  }
17953
- const sheets = parseWorkbook(await workbookFile.async("text"));
18062
+ const { sheets, date1904 } = parseWorkbook(await workbookFile.async("text"));
17954
18063
  if (sheets.length === 0) {
17955
18064
  throw new KordocError("XLSX \uD30C\uC77C\uC5D0 \uC2DC\uD2B8\uAC00 \uC5C6\uC2B5\uB2C8\uB2E4");
17956
18065
  }
18066
+ let dateXfs = /* @__PURE__ */ new Map();
18067
+ const stylesFile = zip.file("xl/styles.xml");
18068
+ if (stylesFile) {
18069
+ try {
18070
+ dateXfs = parseStyleDateXfs(await stylesFile.async("text"));
18071
+ } catch {
18072
+ }
18073
+ }
17957
18074
  let relsMap = /* @__PURE__ */ new Map();
17958
18075
  const relsFile = zip.file("xl/_rels/workbook.xml.rels");
17959
18076
  if (relsFile) {
@@ -17991,7 +18108,7 @@ async function parseXlsxDocument(buffer, options) {
17991
18108
  }
17992
18109
  try {
17993
18110
  const sheetXml = await sheetFile.async("text");
17994
- const { grid, merges, maxRow, maxCol } = parseWorksheet(sheetXml, sharedStrings);
18111
+ const { grid, merges, maxRow, maxCol } = parseWorksheet(sheetXml, sharedStrings, dateXfs, date1904);
17995
18112
  const sheetBlocks = sheetToBlocks(sheet.name, grid, merges, maxRow, maxCol, i);
17996
18113
  blocks.push(...sheetBlocks);
17997
18114
  } catch (err) {
@@ -18035,7 +18152,10 @@ var OP_CONTINUE = 60;
18035
18152
  var OP_BOUNDSHEET8 = 133;
18036
18153
  var OP_SST = 252;
18037
18154
  var OP_CODEPAGE = 66;
18155
+ var OP_DATE1904 = 34;
18038
18156
  var OP_FILEPASS = 47;
18157
+ var OP_FORMAT = 1054;
18158
+ var OP_XF = 224;
18039
18159
  var OP_NUMBER = 515;
18040
18160
  var OP_RK = 638;
18041
18161
  var OP_MULRK = 189;
@@ -18047,6 +18167,8 @@ var OP_BOOLERR = 517;
18047
18167
  var OP_BLANK = 513;
18048
18168
  var OP_MULBLANK = 190;
18049
18169
  var OP_MERGECELLS = 229;
18170
+ var OP_SHRFMLA = 1212;
18171
+ var OP_ARRAY = 545;
18050
18172
  var DT_GLOBALS = 5;
18051
18173
  var DT_WORKSHEET = 16;
18052
18174
  var MAX_RECORDS2 = 1e6;
@@ -18142,7 +18264,7 @@ function decodeUtf16Le(buf) {
18142
18264
  }
18143
18265
 
18144
18266
  // src/xls/sst.ts
18145
- function parseString(buf, offset, segments) {
18267
+ function parseString(buf, offset, segments, segCursor) {
18146
18268
  if (offset + 3 > buf.length) return null;
18147
18269
  const cch = buf.readUInt16LE(offset);
18148
18270
  let flags = buf.readUInt8(offset + 2);
@@ -18165,7 +18287,15 @@ function parseString(buf, offset, segments) {
18165
18287
  const charBytes = [];
18166
18288
  let charsRead = 0;
18167
18289
  while (charsRead < cch) {
18168
- const nextBoundary = segments.find((s) => s > off) ?? buf.length;
18290
+ while (segCursor.idx < segments.length && segments[segCursor.idx] < off) segCursor.idx++;
18291
+ if (segCursor.idx < segments.length && segments[segCursor.idx] === off) {
18292
+ if (off >= buf.length) return null;
18293
+ flags = buf.readUInt8(off);
18294
+ highByte = (flags & 1) !== 0;
18295
+ off += 1;
18296
+ segCursor.idx++;
18297
+ }
18298
+ const nextBoundary = segCursor.idx < segments.length ? segments[segCursor.idx] : buf.length;
18169
18299
  const remainChars = cch - charsRead;
18170
18300
  const bytesPerChar = highByte ? 2 : 1;
18171
18301
  const bytesAvail = nextBoundary - off;
@@ -18176,12 +18306,10 @@ function parseString(buf, offset, segments) {
18176
18306
  charBytes.push(highByte ? slice : padToUtf16(slice));
18177
18307
  off += bytesToRead;
18178
18308
  charsRead += charsInThisRun;
18179
- }
18180
- if (charsRead < cch) {
18181
- if (off >= buf.length) return null;
18182
- flags = buf.readUInt8(off);
18183
- highByte = (flags & 1) !== 0;
18184
- off += 1;
18309
+ } else if (nextBoundary < buf.length) {
18310
+ off = nextBoundary;
18311
+ } else {
18312
+ return null;
18185
18313
  }
18186
18314
  }
18187
18315
  const text = decodeUtf16Le(Buffer.concat(charBytes));
@@ -18206,8 +18334,9 @@ function decodeSST(records) {
18206
18334
  const cstUnique = combined.readUInt32LE(4);
18207
18335
  const strings = [];
18208
18336
  let off = 8;
18337
+ const segCursor = { idx: 0 };
18209
18338
  for (let i = 0; i < cstUnique && off < combined.length; i++) {
18210
- const r = parseString(combined, off, segments);
18339
+ const r = parseString(combined, off, segments, segCursor);
18211
18340
  if (!r) break;
18212
18341
  strings.push(r.text);
18213
18342
  off += r.consumed;
@@ -18282,7 +18411,7 @@ function decodeFormulaStringRecord(data) {
18282
18411
  return decodeUtf16Le(padded);
18283
18412
  }
18284
18413
  }
18285
- function extractSheetCells(records, bofIndex, sst) {
18414
+ function extractSheetCells(records, bofIndex, sst, convertNum) {
18286
18415
  const cells = [];
18287
18416
  const merges = [];
18288
18417
  const bofOffset = records[bofIndex].offset;
@@ -18300,14 +18429,16 @@ function extractSheetCells(records, bofIndex, sst) {
18300
18429
  case OP_NUMBER: {
18301
18430
  const h = readCellHeader(rec.data);
18302
18431
  if (h && rec.data.length >= 14) {
18303
- cells.push({ row: h.row, col: h.col, value: rec.data.readDoubleLE(6) });
18432
+ const n = rec.data.readDoubleLE(6);
18433
+ cells.push({ row: h.row, col: h.col, value: convertNum ? convertNum(n, h.ixfe) : n });
18304
18434
  }
18305
18435
  break;
18306
18436
  }
18307
18437
  case OP_RK: {
18308
18438
  const h = readCellHeader(rec.data);
18309
18439
  if (h && rec.data.length >= 10) {
18310
- cells.push({ row: h.row, col: h.col, value: decodeRk(rec.data.readInt32LE(6)) });
18440
+ const n = decodeRk(rec.data.readInt32LE(6));
18441
+ cells.push({ row: h.row, col: h.col, value: convertNum ? convertNum(n, h.ixfe) : n });
18311
18442
  }
18312
18443
  break;
18313
18444
  }
@@ -18315,7 +18446,7 @@ function extractSheetCells(records, bofIndex, sst) {
18315
18446
  const m = decodeMulRk(rec.data);
18316
18447
  if (m) {
18317
18448
  for (const c of m.cells) {
18318
- cells.push({ row: m.row, col: c.col, value: c.value });
18449
+ cells.push({ row: m.row, col: c.col, value: convertNum ? convertNum(c.value, c.ixfe) : c.value });
18319
18450
  }
18320
18451
  }
18321
18452
  break;
@@ -18340,19 +18471,26 @@ function extractSheetCells(records, bofIndex, sst) {
18340
18471
  if (h && rec.data.length >= 14) {
18341
18472
  const result = decodeFormulaResult(rec.data.subarray(6, 14));
18342
18473
  if (result.kind === "stringRef") {
18343
- const next = records[i + 1];
18474
+ let j = i + 1;
18475
+ while (j < records.length && (records[j].opcode === OP_SHRFMLA || records[j].opcode === OP_ARRAY)) j++;
18476
+ const next = records[j];
18344
18477
  if (next && next.opcode === OP_STRING) {
18345
18478
  cells.push({
18346
18479
  row: h.row,
18347
18480
  col: h.col,
18348
18481
  value: decodeFormulaStringRecord(next.data)
18349
18482
  });
18350
- i++;
18483
+ i = j;
18351
18484
  } else {
18352
18485
  cells.push({ row: h.row, col: h.col, value: "" });
18353
18486
  }
18354
18487
  } else {
18355
- cells.push({ row: h.row, col: h.col, value: result.value });
18488
+ const v = result.value;
18489
+ cells.push({
18490
+ row: h.row,
18491
+ col: h.col,
18492
+ value: convertNum && typeof v === "number" ? convertNum(v, h.ixfe) : v
18493
+ });
18356
18494
  }
18357
18495
  }
18358
18496
  break;
@@ -18402,7 +18540,7 @@ function extractSheetCells(records, bofIndex, sst) {
18402
18540
 
18403
18541
  // src/xls/parser.ts
18404
18542
  var MAX_SHEETS2 = 100;
18405
- var MAX_ROWS3 = 1e5;
18543
+ var MAX_ROWS3 = 65536;
18406
18544
  var MAX_COLS3 = 1e3;
18407
18545
  function decodeBoundSheet(data) {
18408
18546
  if (data.length < 8) return null;
@@ -18425,10 +18563,33 @@ function decodeBoundSheet(data) {
18425
18563
  }
18426
18564
  return { name, lbPlyPos, dt };
18427
18565
  }
18566
+ function decodeFormatRecord(data) {
18567
+ if (data.length < 5) return null;
18568
+ const ifmt = data.readUInt16LE(0);
18569
+ const cch = data.readUInt16LE(2);
18570
+ const flags = data.readUInt8(4);
18571
+ const highByte = (flags & 1) !== 0;
18572
+ const start = 5;
18573
+ let code;
18574
+ if (highByte) {
18575
+ const end = Math.min(start + cch * 2, data.length);
18576
+ code = decodeUtf16Le(data.subarray(start, end));
18577
+ } else {
18578
+ const end = Math.min(start + cch, data.length);
18579
+ const slice = data.subarray(start, end);
18580
+ const padded = Buffer.alloc(slice.length * 2);
18581
+ for (let i = 0; i < slice.length; i++) padded[i * 2] = slice[i];
18582
+ code = decodeUtf16Le(padded);
18583
+ }
18584
+ return { ifmt, code };
18585
+ }
18428
18586
  function processGlobals(records) {
18429
18587
  const sheets = [];
18430
18588
  let codePage = 1200;
18431
18589
  let encrypted = false;
18590
+ let date1904 = false;
18591
+ const customFormats = /* @__PURE__ */ new Map();
18592
+ const xfFmtIds = [];
18432
18593
  const firstBof = records[0];
18433
18594
  if (!firstBof || firstBof.opcode !== OP_BOF) {
18434
18595
  throw new KordocError("XLS: \uCCAB \uB808\uCF54\uB4DC\uAC00 BOF\uAC00 \uC544\uB2D8");
@@ -18451,21 +18612,41 @@ function processGlobals(records) {
18451
18612
  codePage = r.data.readUInt16LE(0);
18452
18613
  } else if (r.opcode === OP_FILEPASS) {
18453
18614
  encrypted = true;
18615
+ } else if (r.opcode === OP_DATE1904 && r.data.length >= 2) {
18616
+ date1904 = r.data.readUInt16LE(0) === 1;
18617
+ } else if (r.opcode === OP_FORMAT) {
18618
+ const f = decodeFormatRecord(r.data);
18619
+ if (f) customFormats.set(f.ifmt, f.code);
18620
+ } else if (r.opcode === OP_XF && r.data.length >= 4) {
18621
+ xfFmtIds.push(r.data.readUInt16LE(2));
18454
18622
  }
18455
18623
  i++;
18456
18624
  }
18625
+ const dateXfs = /* @__PURE__ */ new Map();
18626
+ for (let k = 0; k < xfFmtIds.length; k++) {
18627
+ const kind = dateKindOfFmt(xfFmtIds[k], customFormats);
18628
+ if (kind) dateXfs.set(k, kind);
18629
+ }
18457
18630
  const globalsRecords = records.slice(0, i);
18458
18631
  const sst = decodeSST(globalsRecords);
18459
- return { sheets, sst, codePage, encrypted, endIndex: i };
18632
+ return { sheets, sst, codePage, encrypted, dateXfs, date1904, endIndex: i };
18460
18633
  }
18461
18634
  function findSheetBofIndex(records, lbPlyPos) {
18462
18635
  const exact = records.findIndex(
18463
18636
  (r) => r.opcode === OP_BOF && r.offset === lbPlyPos
18464
18637
  );
18465
18638
  if (exact >= 0) return exact;
18466
- const bofIndices = records.map((r, idx) => r.opcode === OP_BOF ? idx : -1).filter((idx) => idx >= 0);
18467
- if (bofIndices.length === 0) return -1;
18468
- return bofIndices.length > 1 ? bofIndices[1] : -1;
18639
+ let best = -1;
18640
+ let bestOffset = Infinity;
18641
+ for (let idx = 1; idx < records.length; idx++) {
18642
+ const r = records[idx];
18643
+ if (r.opcode !== OP_BOF) continue;
18644
+ if (r.offset >= lbPlyPos && r.offset < bestOffset) {
18645
+ best = idx;
18646
+ bestOffset = r.offset;
18647
+ }
18648
+ }
18649
+ return best;
18469
18650
  }
18470
18651
  function cellValueToText(v) {
18471
18652
  if (v === null || v === void 0) return "";
@@ -18596,6 +18777,14 @@ async function parseXlsDocument(buffer, options) {
18596
18777
  ]
18597
18778
  };
18598
18779
  }
18780
+ const convertNum = globals.dateXfs.size > 0 ? (n, ixfe) => {
18781
+ const kind = globals.dateXfs.get(ixfe);
18782
+ if (kind) {
18783
+ const iso = dateSerialToIso(n, globals.date1904, kind);
18784
+ if (iso) return iso;
18785
+ }
18786
+ return n;
18787
+ } : void 0;
18599
18788
  const totalSheets = Math.min(globals.sheets.length, MAX_SHEETS2);
18600
18789
  let pageFilter = null;
18601
18790
  if (options?.pages) {
@@ -18622,7 +18811,7 @@ async function parseXlsDocument(buffer, options) {
18622
18811
  continue;
18623
18812
  }
18624
18813
  try {
18625
- const { sheet } = extractSheetCells(records, bofIdx, globals.sst);
18814
+ const { sheet } = extractSheetCells(records, bofIdx, globals.sst, convertNum);
18626
18815
  const blocks = sheetToBlocks2(meta.name, sheet, i);
18627
18816
  allBlocks.push(...blocks);
18628
18817
  } catch (e) {
@@ -18750,6 +18939,9 @@ var NARY_MAP = {
18750
18939
  "\u2A02": "\\bigotimes",
18751
18940
  "\u2A00": "\\bigodot"
18752
18941
  };
18942
+ function onOffVal(v) {
18943
+ return v !== "0" && v !== "false" && v !== "off";
18944
+ }
18753
18945
  function mapDelim(ch, isLeft) {
18754
18946
  const l = {
18755
18947
  "(": "(",
@@ -18776,10 +18968,23 @@ function mapDelim(ch, isLeft) {
18776
18968
  const map = isLeft ? l : r;
18777
18969
  return map[ch] ?? ch;
18778
18970
  }
18971
+ function isBalancedWrap(s) {
18972
+ let depth = 0;
18973
+ for (let i = 0; i < s.length; i++) {
18974
+ const ch = s[i];
18975
+ if (ch === "{" && s[i - 1] !== "\\") depth++;
18976
+ else if (ch === "}" && s[i - 1] !== "\\") {
18977
+ depth--;
18978
+ if (depth === 0) return i === s.length - 1;
18979
+ if (depth < 0) return false;
18980
+ }
18981
+ }
18982
+ return false;
18983
+ }
18779
18984
  function grp(body) {
18780
18985
  const s = body.trim();
18781
18986
  if (s.length === 0) return "{}";
18782
- if (s.startsWith("{") && s.endsWith("}")) return s;
18987
+ if (s.startsWith("{") && s.endsWith("}") && isBalancedWrap(s)) return s;
18783
18988
  return "{" + s + "}";
18784
18989
  }
18785
18990
  function childrenToLatex(parent) {
@@ -18873,8 +19078,8 @@ function nodeToLatex(el) {
18873
19078
  }
18874
19079
  const sh = firstKid(naryPr, "subHide");
18875
19080
  const ph = firstKid(naryPr, "supHide");
18876
- if (sh) subHide = (sh.getAttribute("m:val") ?? sh.getAttribute("val")) !== "0";
18877
- if (ph) supHide = (ph.getAttribute("m:val") ?? ph.getAttribute("val")) !== "0";
19081
+ if (sh) subHide = onOffVal(sh.getAttribute("m:val") ?? sh.getAttribute("val"));
19082
+ if (ph) supHide = onOffVal(ph.getAttribute("m:val") ?? ph.getAttribute("val"));
18878
19083
  const ll = firstKid(naryPr, "limLoc");
18879
19084
  if (ll) limLoc = ll.getAttribute("m:val") ?? ll.getAttribute("val") ?? "";
18880
19085
  }
@@ -19021,7 +19226,7 @@ function isDisplayMath(el) {
19021
19226
  }
19022
19227
 
19023
19228
  // src/docx/parser.ts
19024
- var MAX_DECOMPRESS_SIZE4 = 100 * 1024 * 1024;
19229
+ var MAX_DECOMPRESS_SIZE5 = 100 * 1024 * 1024;
19025
19230
  function matchesLocal(el, localName2) {
19026
19231
  return el.localName === localName2 || (el.tagName?.endsWith(`:${localName2}`) ?? false);
19027
19232
  }
@@ -19039,6 +19244,8 @@ function effectiveChildElements(parent) {
19039
19244
  result.push(...effectiveChildElements(c));
19040
19245
  }
19041
19246
  }
19247
+ } else if (matchesLocal(el, "ins") || matchesLocal(el, "smartTag")) {
19248
+ result.push(...effectiveChildElements(el));
19042
19249
  } else {
19043
19250
  result.push(el);
19044
19251
  }
@@ -19103,6 +19310,21 @@ function parseStyles(xml) {
19103
19310
  }
19104
19311
  styles.set(styleId, { name, basedOn, outlineLevel });
19105
19312
  }
19313
+ for (const [styleId, info] of styles) {
19314
+ if (info.outlineLevel !== void 0) continue;
19315
+ const seen = /* @__PURE__ */ new Set([styleId]);
19316
+ let cur = info.basedOn;
19317
+ while (cur && !seen.has(cur)) {
19318
+ seen.add(cur);
19319
+ const parent = styles.get(cur);
19320
+ if (!parent) break;
19321
+ if (parent.outlineLevel !== void 0) {
19322
+ info.outlineLevel = parent.outlineLevel;
19323
+ break;
19324
+ }
19325
+ cur = parent.basedOn;
19326
+ }
19327
+ }
19106
19328
  return styles;
19107
19329
  }
19108
19330
  function parseNumbering(xml) {
@@ -19190,8 +19412,12 @@ function collectOmmlRoots(p) {
19190
19412
  return out;
19191
19413
  }
19192
19414
  function extractRun(r) {
19193
- const tElements = getChildElements(r, "t");
19194
- const text = tElements.map((t) => t.textContent ?? "").join("");
19415
+ let text = "";
19416
+ for (const el of effectiveChildElements(r)) {
19417
+ if (matchesLocal(el, "t")) text += el.textContent ?? "";
19418
+ else if (matchesLocal(el, "br") || matchesLocal(el, "cr")) text += "\n";
19419
+ else if (matchesLocal(el, "tab")) text += " ";
19420
+ }
19195
19421
  let bold = false;
19196
19422
  let italic = false;
19197
19423
  const rPrEls = getChildElements(r, "rPr");
@@ -19346,7 +19572,7 @@ function collectTextboxParagraphs(node, inTxbx = false, out = [], depth = 0) {
19346
19572
  }
19347
19573
  return out;
19348
19574
  }
19349
- function parseTable(tbl2, styles, numbering, footnotes, rels) {
19575
+ function parseTable(tbl2, styles, numbering, footnotes, rels, keepEmptyCols) {
19350
19576
  const trElements = getChildElements(tbl2, "tr");
19351
19577
  if (trElements.length === 0) return null;
19352
19578
  const rawRows = [];
@@ -19408,7 +19634,7 @@ ${cell2.text}` : cell2.text;
19408
19634
  return { text: cell2.text, colSpan: cell2.colSpan, rowSpan, colAddr: cell2.col, rowAddr: r };
19409
19635
  })
19410
19636
  );
19411
- const table2 = buildTable(cellRows);
19637
+ const table2 = buildTable(cellRows, { keepAnchoredEmptyCols: keepEmptyCols });
19412
19638
  if (table2.rows === 0 || table2.cols === 0) return null;
19413
19639
  return { type: "table", table: table2 };
19414
19640
  }
@@ -19510,7 +19736,7 @@ function emitParagraphImages(p, imageMap, linked, out) {
19510
19736
  }
19511
19737
  }
19512
19738
  async function parseDocxDocument(buffer, options) {
19513
- precheckZipSize(buffer, MAX_DECOMPRESS_SIZE4);
19739
+ precheckZipSize(buffer, MAX_DECOMPRESS_SIZE5);
19514
19740
  const zip = await JSZip5.loadAsync(buffer);
19515
19741
  const warnings = [];
19516
19742
  const docFile = zip.file("word/document.xml");
@@ -19581,7 +19807,7 @@ async function parseDocxDocument(buffer, options) {
19581
19807
  if (imageMap.size > 0) emitParagraphImages(tp, imageMap, linkedImages, blocks);
19582
19808
  }
19583
19809
  } else if (localName2 === "tbl") {
19584
- const block = parseTable(el, styles, numbering, footnotes, rels);
19810
+ const block = parseTable(el, styles, numbering, footnotes, rels, options?.keepTrailingEmptyCols);
19585
19811
  if (block) blocks.push(block);
19586
19812
  }
19587
19813
  }
@@ -19673,7 +19899,7 @@ function parseHwpmlDocument(buffer, options) {
19673
19899
  if (localName(el) !== "SECTION") continue;
19674
19900
  sectionIdx++;
19675
19901
  if (pageFilter && !pageFilter.has(sectionIdx)) continue;
19676
- parseSection2(el, blocks, paraShapeMap, sectionIdx, warnings);
19902
+ parseSection2(el, blocks, paraShapeMap, sectionIdx, warnings, options?.keepTrailingEmptyCols ?? false);
19677
19903
  }
19678
19904
  const outline = blocks.filter((b) => b.type === "heading" && b.text).map((b) => ({ level: b.level ?? 1, text: b.text, pageNumber: b.pageNumber }));
19679
19905
  const markdown = blocksToMarkdown(blocks);
@@ -19709,10 +19935,10 @@ function buildParaShapeMap(root) {
19709
19935
  }
19710
19936
  return map;
19711
19937
  }
19712
- function parseSection2(section, blocks, paraShapeMap, sectionNum, warnings) {
19713
- walkContent(section, blocks, paraShapeMap, sectionNum, warnings, false);
19938
+ function parseSection2(section, blocks, paraShapeMap, sectionNum, warnings, keep) {
19939
+ walkContent(section, blocks, paraShapeMap, sectionNum, warnings, false, keep);
19714
19940
  }
19715
- function walkContent(node, blocks, paraShapeMap, sectionNum, warnings, inHeaderFooter, depth = 0) {
19941
+ function walkContent(node, blocks, paraShapeMap, sectionNum, warnings, inHeaderFooter, keep, depth = 0) {
19716
19942
  if (depth > MAX_XML_DEPTH2) return;
19717
19943
  const children = node.childNodes;
19718
19944
  for (let i = 0; i < children.length; i++) {
@@ -19725,24 +19951,24 @@ function walkContent(node, blocks, paraShapeMap, sectionNum, warnings, inHeaderF
19725
19951
  if (tag === "P") {
19726
19952
  if (!inHeaderFooter) {
19727
19953
  parseParagraph3(el, blocks, paraShapeMap, sectionNum);
19728
- walkTablesInP(el, blocks, paraShapeMap, sectionNum, warnings);
19954
+ walkTablesInP(el, blocks, paraShapeMap, sectionNum, warnings, keep);
19729
19955
  }
19730
19956
  continue;
19731
19957
  }
19732
19958
  if (tag === "TABLE") {
19733
19959
  if (!inHeaderFooter) {
19734
- parseTable2(el, blocks, paraShapeMap, sectionNum, warnings);
19960
+ parseTable2(el, blocks, paraShapeMap, sectionNum, warnings, keep);
19735
19961
  }
19736
19962
  continue;
19737
19963
  }
19738
19964
  if (tag === "PARALIST" || tag === "SECTION" || tag === "COLDEF") {
19739
- walkContent(el, blocks, paraShapeMap, sectionNum, warnings, inHeaderFooter, depth + 1);
19965
+ walkContent(el, blocks, paraShapeMap, sectionNum, warnings, inHeaderFooter, keep, depth + 1);
19740
19966
  continue;
19741
19967
  }
19742
- walkContent(el, blocks, paraShapeMap, sectionNum, warnings, inHeaderFooter, depth + 1);
19968
+ walkContent(el, blocks, paraShapeMap, sectionNum, warnings, inHeaderFooter, keep, depth + 1);
19743
19969
  }
19744
19970
  }
19745
- function walkTablesInP(node, blocks, paraShapeMap, sectionNum, warnings, depth = 0) {
19971
+ function walkTablesInP(node, blocks, paraShapeMap, sectionNum, warnings, keep, depth = 0) {
19746
19972
  if (depth > MAX_XML_DEPTH2) return;
19747
19973
  const children = node.childNodes;
19748
19974
  for (let i = 0; i < children.length; i++) {
@@ -19750,11 +19976,11 @@ function walkTablesInP(node, blocks, paraShapeMap, sectionNum, warnings, depth =
19750
19976
  if (el.nodeType !== 1) continue;
19751
19977
  const tag = localName(el);
19752
19978
  if (tag === "TABLE") {
19753
- parseTable2(el, blocks, paraShapeMap, sectionNum, warnings);
19979
+ parseTable2(el, blocks, paraShapeMap, sectionNum, warnings, keep);
19754
19980
  continue;
19755
19981
  }
19756
19982
  if (tag === "FOOTNOTE" || tag === "ENDNOTE" || tag === "HEADER" || tag === "FOOTER") continue;
19757
- walkTablesInP(el, blocks, paraShapeMap, sectionNum, warnings, depth + 1);
19983
+ walkTablesInP(el, blocks, paraShapeMap, sectionNum, warnings, keep, depth + 1);
19758
19984
  }
19759
19985
  }
19760
19986
  function parseParagraph3(el, blocks, paraShapeMap, sectionNum) {
@@ -19790,7 +20016,7 @@ function collectCharText(node, parts, depth = 0) {
19790
20016
  }
19791
20017
  }
19792
20018
  }
19793
- function parseTable2(el, blocks, paraShapeMap, sectionNum, warnings) {
20019
+ function parseTable2(el, blocks, paraShapeMap, sectionNum, warnings, keep) {
19794
20020
  const cells = [];
19795
20021
  const rowCount = parseInt(el.getAttribute("RowCount") ?? "0", 10);
19796
20022
  const colCount = parseInt(el.getAttribute("ColCount") ?? "0", 10);
@@ -19834,7 +20060,7 @@ function parseTable2(el, blocks, paraShapeMap, sectionNum, warnings) {
19834
20060
  const cellRows = grid.map(
19835
20061
  (row) => row.map((cell2) => cell2 ?? { text: "", colSpan: 1, rowSpan: 1 })
19836
20062
  );
19837
- const table2 = buildTable(cellRows);
20063
+ const table2 = buildTable(cellRows, { keepAnchoredEmptyCols: keep });
19838
20064
  const caption = extractShapeCaption(el);
19839
20065
  if (caption.text && caption.before) {
19840
20066
  blocks.push({ type: "paragraph", text: caption.text, pageNumber: sectionNum });
@@ -20070,7 +20296,7 @@ function findMatchingKey(cellLabel, values) {
20070
20296
  bestKey = key;
20071
20297
  }
20072
20298
  } else if (key.startsWith(cellLabel)) {
20073
- if (cellLabel.length >= key.length * 0.6 && cellLabel.length > bestLen) {
20299
+ if (cellLabel.length >= key.length * 0.75 && cellLabel.length > bestLen) {
20074
20300
  bestLen = cellLabel.length;
20075
20301
  bestKey = key;
20076
20302
  }
@@ -20110,7 +20336,7 @@ function fillInCellPatterns(cellText, values, matchedLabels, blockedLabels) {
20110
20336
  const matchKey = values.available(normalizedKw) ? normalizedKw : void 0;
20111
20337
  if (matchKey === void 0) return match;
20112
20338
  const val = values.peek(matchKey);
20113
- const isTruthy = ["\u2611", "\u2713", "\u2714", "v", "V", "true", "1", "yes", "o", "O"].includes(val.trim()) || val.trim() === "";
20339
+ const isTruthy = ["\u2611", "\u2713", "\u2714", "v", "V", "true", "1", "yes", "o", "O"].includes(val.trim());
20114
20340
  if (!isTruthy) return match;
20115
20341
  values.consume(matchKey);
20116
20342
  matchedLabels.add(matchKey);
@@ -20132,14 +20358,20 @@ function fillInCellPatterns(cellText, values, matchedLabels, blockedLabels) {
20132
20358
  );
20133
20359
  return matches.length > 0 ? { text, matches } : null;
20134
20360
  }
20135
- var INLINE_LABEL_RE = /([가-힣A-Za-z]{2,10})\s*[::]/g;
20361
+ var INLINE_LABEL_RE = /((?:[가-힣A-Za-z]{1,10} )?)([가-힣A-Za-z]{2,10})\s*[::]/g;
20136
20362
  function scanInlineSegments(text) {
20137
20363
  const labels = [];
20138
20364
  INLINE_LABEL_RE.lastIndex = 0;
20139
20365
  let m;
20140
20366
  while ((m = INLINE_LABEL_RE.exec(text)) !== null) {
20141
20367
  if (text[INLINE_LABEL_RE.lastIndex] === "/") continue;
20142
- labels.push({ label: m[1], start: m.index, end: INLINE_LABEL_RE.lastIndex });
20368
+ labels.push({
20369
+ label: m[2],
20370
+ ext: m[1] ? m[1] + m[2] : void 0,
20371
+ extStart: m[1] ? m.index : void 0,
20372
+ start: m.index + m[1].length,
20373
+ end: INLINE_LABEL_RE.lastIndex
20374
+ });
20143
20375
  }
20144
20376
  const segments = [];
20145
20377
  for (let i = 0; i < labels.length; i++) {
@@ -20157,21 +20389,44 @@ function scanInlineSegments(text) {
20157
20389
  labelStart: cur.start,
20158
20390
  valueStart: vs,
20159
20391
  valueEnd: ve,
20160
- value: text.slice(vs, ve)
20392
+ value: text.slice(vs, ve),
20393
+ ...cur.ext !== void 0 ? { extLabel: cur.ext, extStart: cur.extStart } : {}
20161
20394
  });
20162
20395
  }
20163
20396
  return segments;
20164
20397
  }
20398
+ function matchInlineSegment(seg, values, blockedLabels) {
20399
+ const nlabel = normalizeLabel(seg.label);
20400
+ if (seg.extLabel !== void 0) {
20401
+ const nExt = normalizeLabel(seg.extLabel);
20402
+ if (nExt !== nlabel && !blockedLabels?.has(nExt) && values.has(nExt)) {
20403
+ return { key: nExt, label: seg.extLabel, viaExt: true };
20404
+ }
20405
+ }
20406
+ if (blockedLabels?.has(nlabel)) return void 0;
20407
+ const key = findMatchingKey(nlabel, values);
20408
+ return key !== void 0 ? { key, label: seg.label, viaExt: false } : void 0;
20409
+ }
20410
+ function clampSegmentEnd(text, seg, next, nextViaExt) {
20411
+ let ve = seg.valueEnd;
20412
+ if (nextViaExt && next?.extStart !== void 0 && next.extStart < ve) ve = next.extStart;
20413
+ while (ve > seg.valueStart && /\s/.test(text[ve - 1])) ve--;
20414
+ return ve;
20415
+ }
20165
20416
  function padInsertion(text, pos, value) {
20166
20417
  const lead = pos > 0 && !/\s/.test(text[pos - 1]) ? " " : "";
20167
20418
  const trail = pos < text.length && !/\s/.test(text[pos]) ? " " : "";
20168
20419
  return lead + value + trail;
20169
20420
  }
20170
- function normalizeValues(values) {
20421
+ function normalizeValues(values, warnings) {
20171
20422
  const map = /* @__PURE__ */ new Map();
20172
20423
  for (const [label, raw] of Object.entries(values)) {
20173
20424
  const { value, format } = typeof raw === "object" && !Array.isArray(raw) ? raw : { value: raw, format: void 0 };
20174
- map.set(normalizeLabel(label), Array.isArray(value) ? value.map((v) => formatFillValue(v, format)) : formatFillValue(value, format));
20425
+ const key = normalizeLabel(label);
20426
+ if (map.has(key)) {
20427
+ warnings?.push(`\uC785\uB825 \uB77C\uBCA8 "${label}"\uC774 \uC815\uADDC\uD654 \uD0A4 "${key}"\uC5D0\uC11C \uB2E4\uB978 \uB77C\uBCA8\uACFC \uCDA9\uB3CC \u2014 \uB4A4 \uAC12\uC73C\uB85C \uB36E\uC5B4\uC500`);
20428
+ }
20429
+ map.set(key, Array.isArray(value) ? value.map((v) => formatFillValue(v, format)) : formatFillValue(value, format));
20175
20430
  }
20176
20431
  return map;
20177
20432
  }
@@ -20360,10 +20615,10 @@ function extractFromTable(table2) {
20360
20615
  }
20361
20616
  if (fields.length === 0 && table2.rows >= 2 && table2.cols >= 2) {
20362
20617
  const headerRow = table2.cells[0];
20363
- const allLabels = headerRow.every((cell2) => {
20364
- const t = cell2.text.trim();
20618
+ const allLabels = headerRow?.every((cell2) => {
20619
+ const t = cell2?.text.trim() ?? "";
20365
20620
  return t.length > 0 && t.length <= 20;
20366
- });
20621
+ }) ?? false;
20367
20622
  if (allLabels) {
20368
20623
  for (let r = 1; r < table2.rows; r++) {
20369
20624
  for (let c = 0; c < table2.cols; c++) {
@@ -20454,7 +20709,8 @@ function fillFormFields(blocks, values, blockedLabels) {
20454
20709
  const cloned = structuredClone(blocks);
20455
20710
  const filled = [];
20456
20711
  const matchedLabels = /* @__PURE__ */ new Set();
20457
- const normalizedValues = normalizeValues(values);
20712
+ const warnings = [];
20713
+ const normalizedValues = normalizeValues(values, warnings);
20458
20714
  const cursor = new ValueCursor(normalizedValues);
20459
20715
  const allTables = collectIRTables(cloned, 0);
20460
20716
  const patternFilledCells = /* @__PURE__ */ new Set();
@@ -20483,7 +20739,7 @@ function fillFormFields(blocks, values, blockedLabels) {
20483
20739
  if (newText !== block.text) block.text = newText;
20484
20740
  }
20485
20741
  const unmatched = resolveUnmatched(normalizedValues, matchedLabels, values);
20486
- return { blocks: cloned, filled, unmatched };
20742
+ return { blocks: cloned, filled, unmatched, ...warnings.length > 0 ? { warnings } : {} };
20487
20743
  }
20488
20744
  function collectIRTables(blocks, depth) {
20489
20745
  if (depth > 16) return [];
@@ -20517,13 +20773,31 @@ function coveredPositions(table2) {
20517
20773
  }
20518
20774
  return covered;
20519
20775
  }
20776
+ function isHeaderDataTable(table2, covered) {
20777
+ if (table2.rows < 2) return false;
20778
+ const headerRow = table2.cells[0];
20779
+ if (!headerRow?.length) return false;
20780
+ const allLabels = headerRow.every((cell2) => {
20781
+ const t = cell2?.text.trim() ?? "";
20782
+ return t.length > 0 && t.length <= 20 && isLabelCell(t);
20783
+ });
20784
+ if (!allLabels) return false;
20785
+ for (let c = 0; c < table2.cols; c++) {
20786
+ if (covered.has(`1,${c}`)) continue;
20787
+ const cell2 = table2.cells[1]?.[c];
20788
+ if (!cell2) break;
20789
+ return !isLabelCell(cell2.text);
20790
+ }
20791
+ return true;
20792
+ }
20520
20793
  function fillTable(table2, values, filled, matchedLabels, patternFilledCells, blockedLabels) {
20521
20794
  if (table2.cols < 2) return;
20522
20795
  const covered = coveredPositions(table2);
20523
- for (let r = 0; r < table2.rows; r++) {
20796
+ const skipHeaderRow = isHeaderDataTable(table2, covered);
20797
+ for (let r = skipHeaderRow ? 1 : 0; r < table2.rows; r++) {
20524
20798
  for (let c = 0; c < table2.cols; c++) {
20525
20799
  if (covered.has(`${r},${c}`)) continue;
20526
- const labelCell = table2.cells[r][c];
20800
+ const labelCell = table2.cells[r]?.[c];
20527
20801
  if (!labelCell) continue;
20528
20802
  if (!isLabelCell(labelCell.text)) continue;
20529
20803
  let vc = c + labelCell.colSpan;
@@ -20556,10 +20830,10 @@ function fillTable(table2, values, filled, matchedLabels, patternFilledCells, bl
20556
20830
  }
20557
20831
  if (table2.rows >= 2 && table2.cols >= 2) {
20558
20832
  const headerRow = table2.cells[0];
20559
- const allLabels = headerRow.every((cell2) => {
20560
- const t = cell2.text.trim();
20833
+ const allLabels = headerRow?.every((cell2) => {
20834
+ const t = cell2?.text.trim() ?? "";
20561
20835
  return t.length > 0 && t.length <= 20 && isLabelCell(t);
20562
- });
20836
+ }) ?? false;
20563
20837
  if (!allLabels) return;
20564
20838
  for (let r = 1; r < table2.rows; r++) {
20565
20839
  for (let c = 0; c < table2.cols; c++) {
@@ -20574,7 +20848,11 @@ function fillTable(table2, values, filled, matchedLabels, patternFilledCells, bl
20574
20848
  if (!values.isArray(matchKey) && matchedLabels.has(matchKey)) continue;
20575
20849
  const newValue = values.consume(matchKey);
20576
20850
  if (newValue === void 0) continue;
20577
- valueCell.text = newValue;
20851
+ if (patternFilledCells?.has(valueCell)) {
20852
+ valueCell.text = newValue + " " + valueCell.text;
20853
+ } else {
20854
+ valueCell.text = newValue;
20855
+ }
20578
20856
  matchedLabels.add(matchKey);
20579
20857
  filled.push({
20580
20858
  label: headerCell.text.trim(),
@@ -20590,20 +20868,22 @@ function fillTable(table2, values, filled, matchedLabels, patternFilledCells, bl
20590
20868
  function fillInlineFields(text, values, filled, matchedLabels, blockedLabels) {
20591
20869
  const segments = scanInlineSegments(text);
20592
20870
  if (segments.length === 0) return text;
20871
+ const matches = segments.map((seg) => matchInlineSegment(seg, values, blockedLabels));
20593
20872
  let out = "";
20594
20873
  let pos = 0;
20595
- for (const seg of segments) {
20596
- const nlabel = normalizeLabel(seg.label);
20597
- if (blockedLabels?.has(nlabel)) continue;
20598
- const matchKey = findMatchingKey(nlabel, values);
20599
- if (matchKey === void 0) continue;
20874
+ for (let i = 0; i < segments.length; i++) {
20875
+ const seg = segments[i];
20876
+ const matched = matches[i];
20877
+ if (matched === void 0) continue;
20878
+ const matchKey = matched.key;
20600
20879
  const newValue = values.consume(matchKey);
20601
20880
  if (newValue === void 0) continue;
20602
20881
  matchedLabels.add(matchKey);
20603
- filled.push({ label: seg.label.trim(), value: newValue, row: -1, col: -1, key: matchKey });
20882
+ filled.push({ label: matched.label.trim(), value: newValue, row: -1, col: -1, key: matchKey });
20883
+ const ve = clampSegmentEnd(text, seg, segments[i + 1], matches[i + 1]?.viaExt ?? false);
20604
20884
  out += text.slice(pos, seg.valueStart);
20605
- out += seg.valueStart === seg.valueEnd ? padInsertion(text, seg.valueStart, newValue) : newValue;
20606
- pos = seg.valueEnd;
20885
+ out += seg.valueStart === ve ? padInsertion(text, seg.valueStart, newValue) : newValue;
20886
+ pos = ve;
20607
20887
  }
20608
20888
  out += text.slice(pos);
20609
20889
  return out;
@@ -20614,7 +20894,7 @@ import JSZip6 from "jszip";
20614
20894
 
20615
20895
  // src/roundtrip/source-map.ts
20616
20896
  function escapeXmlText(text) {
20617
- return text.replace(/&/g, "&amp;").replace(/</g, "&lt;").replace(/>/g, "&gt;");
20897
+ return text.replace(/[\x00-\x08\x0B\x0C\x0E-\x1F]/g, "").replace(/&/g, "&amp;").replace(/</g, "&lt;").replace(/>/g, "&gt;");
20618
20898
  }
20619
20899
  function decodeXmlEntities(text) {
20620
20900
  return text.replace(/&(lt|gt|amp|quot|apos|#x?[0-9a-fA-F]+);/g, (m, ent) => {
@@ -21244,6 +21524,9 @@ function patchZipEntries(original, replacements, additions) {
21244
21524
  const header = copyBytes(original, e.localOffset, e.localOffset + headerLen);
21245
21525
  const hview = new DataView(header.buffer, header.byteOffset, header.byteLength);
21246
21526
  const method = e.method;
21527
+ if (method !== 0 && method !== 8) {
21528
+ throw new KordocError(`\uC9C0\uC6D0\uD558\uC9C0 \uC54A\uB294 ZIP \uC555\uCD95 \uBC29\uC2DD(method=${method}): ${e.name} \u2014 STORE(0)/DEFLATE(8)\uB9CC \uAD50\uCCB4 \uAC00\uB2A5`);
21529
+ }
21247
21530
  const compData = method === 0 ? newData : new Uint8Array(deflateRawSync(newData));
21248
21531
  const crc = crc32(newData);
21249
21532
  const flags = e.flags & ~8;
@@ -21345,7 +21628,8 @@ async function fillHwpx(hwpxBuffer, values, blockedLabels) {
21345
21628
  if (sectionPaths.length === 0) {
21346
21629
  throw new KordocError("HWPX\uC5D0\uC11C \uC139\uC158 \uD30C\uC77C\uC744 \uCC3E\uC744 \uC218 \uC5C6\uC2B5\uB2C8\uB2E4");
21347
21630
  }
21348
- const normalizedValues = normalizeValues(values);
21631
+ const warnings = [];
21632
+ const normalizedValues = normalizeValues(values, warnings);
21349
21633
  const cursor = new ValueCursor(normalizedValues);
21350
21634
  const matchedLabels = /* @__PURE__ */ new Set();
21351
21635
  const filled = [];
@@ -21407,7 +21691,17 @@ async function fillHwpx(hwpxBuffer, values, blockedLabels) {
21407
21691
  }
21408
21692
  }
21409
21693
  for (const table2 of allTables) {
21410
- for (let rowIdx = 0; rowIdx < table2.rows.length; rowIdx++) {
21694
+ const skipHeaderRow = table2.rows.length >= 2 && (() => {
21695
+ const first = table2.rows[0];
21696
+ const allLabels = first.length > 0 && first.every((cell2) => {
21697
+ const t = cellLabelText(cell2).trim();
21698
+ return t.length > 0 && t.length <= 20 && isLabelCell(t);
21699
+ });
21700
+ if (!allLabels) return false;
21701
+ const d0 = table2.rows[1][0];
21702
+ return d0 === void 0 || !isLabelCell(cellLabelText(d0));
21703
+ })();
21704
+ for (let rowIdx = skipHeaderRow ? 1 : 0; rowIdx < table2.rows.length; rowIdx++) {
21411
21705
  const cells = table2.rows[rowIdx];
21412
21706
  for (let colIdx = 0; colIdx < cells.length - 1; colIdx++) {
21413
21707
  const labelText = cellLabelText(cells[colIdx]);
@@ -21478,9 +21772,30 @@ async function fillHwpx(hwpxBuffer, values, blockedLabels) {
21478
21772
  const matchKey = findMatchingKey(headerLabel, cursor);
21479
21773
  if (matchKey === void 0) continue;
21480
21774
  if (!cursor.isArray(matchKey) && matchedLabels.has(matchKey)) continue;
21775
+ const dataCell = dataCells[colIdx];
21776
+ if (patternApplied.has(dataCell)) {
21777
+ const target = dataCell.paragraphs.find((p) => p.tRanges.length > 0) ?? dataCell.paragraphs[0];
21778
+ if (!target) continue;
21779
+ const l = led(target);
21780
+ if (l.fullText !== void 0) continue;
21781
+ const newValue2 = cursor.consume(matchKey);
21782
+ if (newValue2 === void 0) continue;
21783
+ l.ranges.push({ start: 0, end: 0, replacement: newValue2 + " " });
21784
+ l.filledIdx.push(filled.length);
21785
+ l.matchKeys.push(matchKey);
21786
+ matchedLabels.add(matchKey);
21787
+ filled.push({
21788
+ label: cellLabelText(headerCells[colIdx]).trim(),
21789
+ value: newValue2,
21790
+ row: rowIdx,
21791
+ col: colIdx,
21792
+ key: matchKey
21793
+ });
21794
+ continue;
21795
+ }
21481
21796
  const newValue = cursor.consume(matchKey);
21482
21797
  if (newValue === void 0) continue;
21483
- const paras = dataCells[colIdx].paragraphs;
21798
+ const paras = dataCell.paragraphs;
21484
21799
  if (paras.length === 0) continue;
21485
21800
  const l0 = led(paras[0]);
21486
21801
  l0.fullText = newValue;
@@ -21509,20 +21824,23 @@ async function fillHwpx(hwpxBuffer, values, blockedLabels) {
21509
21824
  const existing = ledger.get(para2);
21510
21825
  if (existing?.fullText !== void 0) continue;
21511
21826
  const text = matchText(para2);
21512
- for (const seg of scanInlineSegments(text)) {
21513
- const nlabel = normalizeLabel(seg.label);
21514
- if (blockedLabels?.has(nlabel)) continue;
21515
- const matchKey = findMatchingKey(nlabel, cursor);
21516
- if (matchKey === void 0) continue;
21827
+ const segments = scanInlineSegments(text);
21828
+ const matches = segments.map((seg) => matchInlineSegment(seg, cursor, blockedLabels));
21829
+ for (let i = 0; i < segments.length; i++) {
21830
+ const seg = segments[i];
21831
+ const matched = matches[i];
21832
+ if (matched === void 0) continue;
21833
+ const matchKey = matched.key;
21517
21834
  const newValue = cursor.consume(matchKey);
21518
21835
  if (newValue === void 0) continue;
21519
- const replacement = seg.valueStart === seg.valueEnd ? padInsertion(text, seg.valueStart, newValue) : newValue;
21836
+ const ve = clampSegmentEnd(text, seg, segments[i + 1], matches[i + 1]?.viaExt ?? false);
21837
+ const replacement = seg.valueStart === ve ? padInsertion(text, seg.valueStart, newValue) : newValue;
21520
21838
  const l = led(para2);
21521
- l.ranges.push({ start: seg.valueStart, end: seg.valueEnd, replacement });
21839
+ l.ranges.push({ start: seg.valueStart, end: ve, replacement });
21522
21840
  matchedLabels.add(matchKey);
21523
21841
  l.filledIdx.push(filled.length);
21524
21842
  l.matchKeys.push(matchKey);
21525
- filled.push({ label: seg.label.trim(), value: newValue, row: -1, col: -1, key: matchKey });
21843
+ filled.push({ label: matched.label.trim(), value: newValue, row: -1, col: -1, key: matchKey });
21526
21844
  }
21527
21845
  }
21528
21846
  const splices = [];
@@ -21575,7 +21893,7 @@ async function fillHwpx(hwpxBuffer, values, blockedLabels) {
21575
21893
  }
21576
21894
  }
21577
21895
  for (const k of failedKeys) {
21578
- if (!succeededKeys.has(k)) matchedLabels.delete(k);
21896
+ if (!succeededKeys.has(k) || cursor.isArray(k)) matchedLabels.delete(k);
21579
21897
  }
21580
21898
  const cleanFilled = filled.filter((f) => f !== null);
21581
21899
  const unmatched = resolveUnmatched(normalizedValues, matchedLabels, values);
@@ -21583,7 +21901,8 @@ async function fillHwpx(hwpxBuffer, values, blockedLabels) {
21583
21901
  return {
21584
21902
  buffer: out.buffer.slice(out.byteOffset, out.byteOffset + out.byteLength),
21585
21903
  filled: cleanFilled,
21586
- unmatched
21904
+ unmatched,
21905
+ ...warnings.length > 0 ? { warnings } : {}
21587
21906
  };
21588
21907
  }
21589
21908
 
@@ -21860,27 +22179,39 @@ function parseMarkdownToBlocks(md2) {
21860
22179
  if (/^<table[\s>]/i.test(line.trimStart())) {
21861
22180
  const htmlLines = [];
21862
22181
  let depth = 0;
21863
- while (i < lines.length) {
21864
- const l = lines[i];
22182
+ let closed = false;
22183
+ let j = i;
22184
+ while (j < lines.length) {
22185
+ const l = lines[j];
21865
22186
  htmlLines.push(l);
21866
22187
  depth += (l.match(/<table[\s>]/gi) ?? []).length;
21867
22188
  depth -= (l.match(/<\/table>/gi) ?? []).length;
21868
- i++;
21869
- if (depth <= 0) break;
22189
+ j++;
22190
+ if (depth <= 0) {
22191
+ closed = true;
22192
+ break;
22193
+ }
22194
+ }
22195
+ if (closed) {
22196
+ blocks.push({ type: "html_table", text: htmlLines.join("\n") });
22197
+ i = j;
22198
+ continue;
21870
22199
  }
21871
- blocks.push({ type: "html_table", text: htmlLines.join("\n") });
21872
- continue;
21873
22200
  }
21874
22201
  if (line.trimStart().startsWith("|")) {
21875
22202
  const tableRows = [];
22203
+ let sepSeen = false;
21876
22204
  while (i < lines.length && lines[i].trimStart().startsWith("|")) {
21877
22205
  const row = lines[i];
21878
- const sepCells = row.trim().replace(/^\|/, "").replace(/\|$/, "").split("|");
21879
- if (sepCells.every((c) => /^\s*:?-+:?\s*$/.test(c))) {
21880
- i++;
21881
- continue;
22206
+ if (tableRows.length === 1 && !sepSeen) {
22207
+ const sepCells = row.trim().replace(/^\|/, "").replace(/\|$/, "").split("|");
22208
+ if (sepCells.every((c) => /^\s*:?-+:?\s*$/.test(c))) {
22209
+ sepSeen = true;
22210
+ i++;
22211
+ continue;
22212
+ }
21882
22213
  }
21883
- const cells = row.split("|").slice(1, -1).map((c) => c.trim());
22214
+ const cells = row.split(/(?<!\\)\|/).slice(1, -1).map((c) => c.trim().replace(/\\\|/g, "|"));
21884
22215
  if (cells.length > 0) tableRows.push(cells);
21885
22216
  i++;
21886
22217
  }
@@ -22829,13 +23160,26 @@ function normalize(s) {
22829
23160
  return s.replace(/\s+/g, " ").trim();
22830
23161
  }
22831
23162
  var MAX_LEVENSHTEIN_LEN = 1e4;
23163
+ function approxDistance(a, b) {
23164
+ const bigramCounts = (s) => {
23165
+ const m = /* @__PURE__ */ new Map();
23166
+ for (let i = 0; i < s.length - 1; i++) {
23167
+ const g = s.slice(i, i + 2);
23168
+ m.set(g, (m.get(g) ?? 0) + 1);
23169
+ }
23170
+ return m;
23171
+ };
23172
+ const ca = bigramCounts(a);
23173
+ const cb = bigramCounts(b);
23174
+ let inter = 0;
23175
+ for (const [g, n] of ca) inter += Math.min(n, cb.get(g) ?? 0);
23176
+ const total = Math.max(a.length - 1, 0) + Math.max(b.length - 1, 0);
23177
+ const dice = total > 0 ? 2 * inter / total : 1;
23178
+ return Math.round(Math.max(a.length, b.length) * (1 - dice));
23179
+ }
22832
23180
  function levenshtein(a, b) {
22833
23181
  if (a.length + b.length > MAX_LEVENSHTEIN_LEN) {
22834
- const sampleLen = Math.min(500, a.length, b.length);
22835
- let diffs = 0;
22836
- for (let i = 0; i < sampleLen; i++) if (a[i] !== b[i]) diffs++;
22837
- const sampleRate = sampleLen > 0 ? diffs / sampleLen : 1;
22838
- return Math.abs(a.length - b.length) + Math.round(Math.min(a.length, b.length) * sampleRate);
23182
+ return approxDistance(a, b);
22839
23183
  }
22840
23184
  if (a.length > b.length) [a, b] = [b, a];
22841
23185
  const m = a.length;
@@ -22987,9 +23331,16 @@ function bestSimInRange(arr, from, to, target) {
22987
23331
  return best;
22988
23332
  }
22989
23333
  function escapeGfm(text) {
22990
- return text.replace(/([~*])/g, "\\$1");
23334
+ const NUL = String.fromCharCode(0);
23335
+ const spans = [];
23336
+ const masked = text.replace(/!\[[^\]]*\]\([^)\n]*\)|\]\((?:https?:|mailto:|tel:|#)[^)\n]*\)/gi, (m) => {
23337
+ spans.push(m);
23338
+ return NUL + (spans.length - 1) + NUL;
23339
+ });
23340
+ const escaped = masked.replace(/([~*_`])/g, "\\$1");
23341
+ return escaped.replace(new RegExp(NUL + "(\\d+)" + NUL, "g"), (_, n) => spans[Number(n)]);
22991
23342
  }
22992
- var HWP_SHAPE_ALT_TEXT_RE = /(?:모서리가 둥근 |둥근 )?(?:사각형|직사각형|정사각형|원|타원|삼각형|이등변 삼각형|직각 삼각형|선|직선|곡선|화살표|굵은 화살표|이중 화살표|오각형|육각형|팔각형|별|[4-8]점별|십자|십자형|구름|구름형|마름모|도넛|평행사변형|사다리꼴|부채꼴|호|반원|물결|번개|하트|빗금|블록 화살표|수식|표|그림|개체|그리기\s?개체|묶음\s?개체|글상자|수식\s?개체|OLE\s?개체)\s?입니다\.?/g;
23343
+ var HWP_SHAPE_ALT_TEXT_RE = /^(?:모서리가 둥근 |둥근 )?(?:사각형|직사각형|정사각형|원|타원|삼각형|이등변 삼각형|직각 삼각형|선|직선|곡선|화살표|굵은 화살표|이중 화살표|오각형|육각형|팔각형|별|[4-8]점별|십자|십자형|구름|구름형|마름모|도넛|평행사변형|사다리꼴|부채꼴|호|반원|물결|번개|하트|빗금|블록 화살표|수식|표|그림|개체|그리기\s?개체|묶음\s?개체|글상자|수식\s?개체|OLE\s?개체)\s?입니다\.?$/gm;
22993
23344
  function sanitizeText(text) {
22994
23345
  let result = mapPuaText(text).replace(/[\u{F0000}-\u{FFFFD}]/gu, "").replace(HWP_SHAPE_ALT_TEXT_RE, "").replace(/ +/g, " ").trim();
22995
23346
  if (result.length <= 30 && result.includes(" ")) {
@@ -23005,7 +23356,7 @@ function normForMatch(text) {
23005
23356
  return sanitizeText(text).replace(/\s+/g, " ").trim();
23006
23357
  }
23007
23358
  function unescapeGfm(text) {
23008
- return text.replace(/\\([~*])/g, "$1");
23359
+ return text.replace(/\\([~*_`])/g, "$1");
23009
23360
  }
23010
23361
  function summarize(text) {
23011
23362
  const t = text.replace(/\s+/g, " ").trim();
@@ -23072,7 +23423,7 @@ function parseGfmTable(lines) {
23072
23423
  return rows;
23073
23424
  }
23074
23425
  function unescapeGfmCell(text) {
23075
- return text.replace(/<br\s*\/?>/gi, "\n").replace(/\\\|/g, "|").replace(/\\([~*])/g, "$1");
23426
+ return text.replace(/<br\s*\/?>/gi, "\n").replace(/\\\|/g, "|").replace(/\\([~*_`])/g, "$1");
23076
23427
  }
23077
23428
  function replicateCellInnerHtml(cell2) {
23078
23429
  if (cell2.blocks?.length) {
@@ -23160,10 +23511,10 @@ function parseHtmlTable(raw) {
23160
23511
  }
23161
23512
  } else {
23162
23513
  if (!isClose) {
23163
- const cs = parseInt(attrs.match(/colspan\s*=\s*"(\d+)"/i)?.[1] || "1", 10);
23164
- const rs = parseInt(attrs.match(/rowspan\s*=\s*"(\d+)"/i)?.[1] || "1", 10);
23514
+ const cs = parseInt(attrs.match(/colspan\s*=\s*["']?(\d+)/i)?.[1] || "1", 10);
23515
+ const rs = parseInt(attrs.match(/rowspan\s*=\s*["']?(\d+)/i)?.[1] || "1", 10);
23165
23516
  cellStart = m.index + m[0].length;
23166
- cellInfo = { colSpan: isNaN(cs) ? 1 : cs, rowSpan: isNaN(rs) ? 1 : rs };
23517
+ cellInfo = { colSpan: clampSpan(isNaN(cs) ? 1 : cs, MAX_COLS), rowSpan: clampSpan(isNaN(rs) ? 1 : rs, MAX_ROWS) };
23167
23518
  } else if (cellStart >= 0 && cellInfo && currentRow) {
23168
23519
  currentRow.push({ inner: raw.slice(cellStart, m.index), colSpan: cellInfo.colSpan, rowSpan: cellInfo.rowSpan });
23169
23520
  cellStart = -1;
@@ -23807,8 +24158,8 @@ function layoutHtmlRows(rows) {
23807
24158
  let c = 0;
23808
24159
  for (const cell2 of rows[r].cells) {
23809
24160
  while (occupied.has(`${r},${c}`)) c++;
23810
- const colSpan = Math.max(1, cell2.colSpan);
23811
- const rowSpan = Math.max(1, cell2.rowSpan);
24161
+ const colSpan = clampSpan(cell2.colSpan, MAX_COLS);
24162
+ const rowSpan = clampSpan(cell2.rowSpan, MAX_ROWS);
23812
24163
  placed.push({ r, c, colSpan, rowSpan, inner: cell2.inner, isHeader: rows[r].tag === "th" });
23813
24164
  for (let dr = 0; dr < rowSpan; dr++) {
23814
24165
  for (let dc = 0; dc < colSpan; dc++) occupied.add(`${r + dr},${c + dc}`);
@@ -24967,12 +25318,20 @@ function diffBlocks(blocksA, blocksB) {
24967
25318
  function alignBlocks(a, b) {
24968
25319
  const m = a.length, n = b.length;
24969
25320
  if (m * n > 1e7) return fallbackAlign(a, b);
25321
+ const lenOf = (blk) => {
25322
+ const t = blk.text !== void 0 ? blk.text : blk.type === "table" && blk.table ? blk.table.cells.flat().map((c) => c?.text ?? "").join(" ") : "";
25323
+ return t.replace(/\s+/g, " ").trim().length;
25324
+ };
25325
+ const aLen = a.map(lenOf);
25326
+ const bLen = b.map(lenOf);
24970
25327
  const simCache = /* @__PURE__ */ new Map();
24971
25328
  const getSim = (i2, j2) => {
24972
25329
  const key = `${i2},${j2}`;
24973
25330
  let v = simCache.get(key);
24974
25331
  if (v === void 0) {
24975
- v = blockSimilarity(a[i2], b[j2]);
25332
+ const mx = Math.max(aLen[i2], bLen[j2]);
25333
+ const cut = a[i2].type === "table" || b[j2].type === "table" ? 6 / 7 : 1 - SIMILARITY_THRESHOLD;
25334
+ v = mx > 0 && (mx - Math.min(aLen[i2], bLen[j2])) / mx > cut ? 0 : blockSimilarity(a[i2], b[j2]);
24976
25335
  simCache.set(key, v);
24977
25336
  }
24978
25337
  return v;
@@ -25033,8 +25392,8 @@ function blockSimilarity(a, b) {
25033
25392
  }
25034
25393
  function tableSimilarity(a, b) {
25035
25394
  const dimSim = 1 - Math.abs(a.rows * a.cols - b.rows * b.cols) / Math.max(a.rows * a.cols, b.rows * b.cols, 1);
25036
- const textsA = a.cells.flat().map((c) => c.text).join(" ");
25037
- const textsB = b.cells.flat().map((c) => c.text).join(" ");
25395
+ const textsA = a.cells.flat().map((c) => c?.text ?? "").join(" ");
25396
+ const textsB = b.cells.flat().map((c) => c?.text ?? "").join(" ");
25038
25397
  const contentSim = normalizedSimilarity(textsA, textsB);
25039
25398
  return dimSim * 0.3 + contentSim * 0.7;
25040
25399
  }
@@ -25045,8 +25404,8 @@ function diffTableCells(a, b) {
25045
25404
  for (let r = 0; r < maxRows; r++) {
25046
25405
  const row = [];
25047
25406
  for (let c = 0; c < maxCols; c++) {
25048
- const cellA = r < a.rows && c < a.cols ? a.cells[r][c].text : void 0;
25049
- const cellB = r < b.rows && c < b.cols ? b.cells[r][c].text : void 0;
25407
+ const cellA = r < a.rows && c < a.cols ? a.cells[r]?.[c]?.text : void 0;
25408
+ const cellB = r < b.rows && c < b.cols ? b.cells[r]?.[c]?.text : void 0;
25050
25409
  let type;
25051
25410
  if (cellA === void 0) type = "added";
25052
25411
  else if (cellB === void 0) type = "removed";
@@ -27081,6 +27440,9 @@ var Surgeon = class {
27081
27440
  miniFatSectors = [];
27082
27441
  dirSectors = [];
27083
27442
  entries = [];
27443
+ /** replace()가 FREESECT로 해제한 섹터 — finish()에서 재할당 안 된 것만 0으로 지움 (데이터 잔존 방지) */
27444
+ freedSectors = [];
27445
+ freedMiniSectors = [];
27084
27446
  constructor(file) {
27085
27447
  if (file.length < SECTOR || file.readUInt32LE(0) !== 3759263696) {
27086
27448
  throw new OleSurgeonError("OLE \uC2DC\uADF8\uB2C8\uCC98\uAC00 \uC544\uB2D9\uB2C8\uB2E4");
@@ -27204,6 +27566,8 @@ var Surgeon = class {
27204
27566
  if (this.fat[i] !== FREESECT) continue;
27205
27567
  if (SECTOR + (i + 1) * SECTOR > this.buf.length) continue;
27206
27568
  this.fat[i] = ENDOFCHAIN;
27569
+ const off = this.sectorOffset(i);
27570
+ this.buf.fill(0, off, off + SECTOR);
27207
27571
  out.push(i);
27208
27572
  }
27209
27573
  while (out.length < n) {
@@ -27303,9 +27667,15 @@ var Surgeon = class {
27303
27667
  const entry = this.findEntry(path);
27304
27668
  if (entry.size > 0 && entry.start !== ENDOFCHAIN) {
27305
27669
  if (entry.size < MINI_CUTOFF) {
27306
- for (const s of this.miniChain(entry.start)) this.miniFat[s] = FREESECT;
27670
+ for (const s of this.miniChain(entry.start)) {
27671
+ this.miniFat[s] = FREESECT;
27672
+ this.freedMiniSectors.push(s);
27673
+ }
27307
27674
  } else {
27308
- for (const s of this.chain(entry.start)) this.fat[s] = FREESECT;
27675
+ for (const s of this.chain(entry.start)) {
27676
+ this.fat[s] = FREESECT;
27677
+ this.freedSectors.push(s);
27678
+ }
27309
27679
  }
27310
27680
  }
27311
27681
  if (newData.length < MINI_CUTOFF) {
@@ -27334,9 +27704,27 @@ var Surgeon = class {
27334
27704
  this.writeDirEntry(entry);
27335
27705
  }
27336
27706
  finish() {
27707
+ this.wipeFreedSectors();
27337
27708
  this.flushFat();
27338
27709
  return this.buf;
27339
27710
  }
27711
+ /** 해제 후 재할당되지 않고 남은 FREESECT 섹터의 바이트를 0으로 채움 (데이터 remanence 제거) */
27712
+ wipeFreedSectors() {
27713
+ for (const s of this.freedSectors) {
27714
+ if (this.fat[s] !== FREESECT) continue;
27715
+ const off = this.sectorOffset(s);
27716
+ this.buf.fill(0, off, off + SECTOR);
27717
+ }
27718
+ if (this.freedMiniSectors.length > 0) {
27719
+ const root = this.rootEntry();
27720
+ const rootChain = root.start === ENDOFCHAIN || root.size === 0 ? [] : this.chain(root.start);
27721
+ for (const s of this.freedMiniSectors) {
27722
+ if (this.miniFat[s] !== FREESECT) continue;
27723
+ const off = this.miniOffset(s, rootChain);
27724
+ this.buf.fill(0, off, off + MINI_SECTOR);
27725
+ }
27726
+ }
27727
+ }
27340
27728
  };
27341
27729
 
27342
27730
  // src/roundtrip/hwp5-patch.ts
@@ -28221,6 +28609,216 @@ async function validateHwpx(buffer) {
28221
28609
  return { ok: issues.length === 0, issues, entryCount: names.length };
28222
28610
  }
28223
28611
 
28612
+ // src/redact.ts
28613
+ var DEFAULT_REDACT_RULES = [
28614
+ "rrn",
28615
+ "phone",
28616
+ "email",
28617
+ "card",
28618
+ "account"
28619
+ ];
28620
+ var RULE_PRIORITY = [
28621
+ "rrn",
28622
+ "email",
28623
+ "card",
28624
+ "phone",
28625
+ "driver",
28626
+ "account",
28627
+ "passport"
28628
+ ];
28629
+ function luhnValid(digits) {
28630
+ let sum = 0;
28631
+ for (let i = 0; i < digits.length; i++) {
28632
+ let d = digits.charCodeAt(digits.length - 1 - i) - 48;
28633
+ if (i % 2 === 1) {
28634
+ d *= 2;
28635
+ if (d > 9) d -= 9;
28636
+ }
28637
+ sum += d;
28638
+ }
28639
+ return sum % 10 === 0;
28640
+ }
28641
+ function birthdateValid(front6) {
28642
+ const mm = Number(front6.slice(2, 4));
28643
+ const dd = Number(front6.slice(4, 6));
28644
+ return mm >= 1 && mm <= 12 && dd >= 1 && dd <= 31;
28645
+ }
28646
+ var RULES2 = {
28647
+ // 주민/외국인등록번호 — 뒷자리 첫 숫자 1-8 + 생년월일 유효성으로 오탐 축소.
28648
+ // 유니코드 대시 변형(‐ ‑ – —)은 rrn만 허용. 앞 6자리 유지, 뒤 7자리 전부 마스크.
28649
+ rrn: {
28650
+ pattern: /(?<!\d)(\d{6})([-‐‑–—])([1-8]\d{6})(?!\d)/g,
28651
+ validate: (m) => birthdateValid(m[1]),
28652
+ mask: (m, mc) => m[1] + m[2] + mc.repeat(7)
28653
+ },
28654
+ // 이메일 — 로컬파트 첫 글자만 남기고 마스크, 도메인 유지
28655
+ email: {
28656
+ pattern: /[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\.[A-Za-z]{2,}/g,
28657
+ mask: (m, mc) => {
28658
+ const at = m[0].indexOf("@");
28659
+ return m[0][0] + mc.repeat(at - 1) + m[0].slice(at);
28660
+ }
28661
+ },
28662
+ // 카드번호 — 구분자 필수(무구분 16자리는 오탐 높아 제외), 동일 구분자 강제(\2),
28663
+ // Luhn 체크. 가운데 8자리 마스크.
28664
+ card: {
28665
+ pattern: /(?<!\d)(\d{4})([- ])(\d{4})\2(\d{4})\2(\d{4})(?!\d)/g,
28666
+ validate: (m) => luhnValid(m[1] + m[3] + m[4] + m[5]),
28667
+ mask: (m, mc) => m[1] + m[2] + mc.repeat(4) + m[2] + mc.repeat(4) + m[2] + m[5]
28668
+ },
28669
+ // 전화번호 — 휴대폰(01[016789])·서울(02)·지역(0[3-6]\d)·인터넷(070)은 구분자
28670
+ // -·.·공백 또는 무구분(동일 구분자 강제), 대표번호(15xx/16xx/18xx)는 구분자 필수.
28671
+ // 가운데 자리만 마스크 (대표번호는 뒤 4자리). 선행 [\d-] 금지 — 계좌 부분매치 방지.
28672
+ phone: {
28673
+ pattern: /(?<![\d-])(?:(01[016789]|070|02|0[3-6]\d)([-. ]?)(\d{3,4})\2(\d{4})|(1[568]\d{2})([-. ])(\d{4}))(?!\d)/g,
28674
+ mask: (m, mc) => m[1] !== void 0 ? m[1] + m[2] + mc.repeat(m[3].length) + m[2] + m[4] : m[5] + m[6] + mc.repeat(4)
28675
+ },
28676
+ // 운전면허 (기본 OFF) — 신형 12자리만 (지역명 2글자 선행 구버전은 스킵). 뒷 8자리 마스크.
28677
+ driver: {
28678
+ pattern: /(?<![\d-])(\d{2})-(\d{2})-\d{6}-\d{2}(?!-?\d)/g,
28679
+ mask: (m, mc) => m[1] + "-" + m[2] + "-" + mc.repeat(6) + "-" + mc.repeat(2)
28680
+ },
28681
+ // 계좌번호 — 3~4그룹 + 총 자릿수 10~16. rrn·card·phone과 겹치면 그쪽 우선.
28682
+ // 마지막 그룹 빼고 전부 마스크. 사업자등록번호(3-2-5, 10자리)도 걸린다 — 계약상 의도
28683
+ // (테스트로 명시). 날짜(2026-07-16)는 8자리라 총자릿수 검증에서 탈락.
28684
+ account: {
28685
+ pattern: /(?<!\d)(?<!\d-)\d{2,6}(?:-\d{2,6}){1,2}-\d{2,8}(?!-?\d)/g,
28686
+ validate: (m) => {
28687
+ const digits = m[0].replace(/-/g, "").length;
28688
+ return digits >= 10 && digits <= 16;
28689
+ },
28690
+ mask: (m, mc) => {
28691
+ const parts = m[0].split("-");
28692
+ return parts.map((p, i) => i === parts.length - 1 ? p : mc.repeat(p.length)).join("-");
28693
+ }
28694
+ },
28695
+ // 여권번호 (기본 OFF) — 단어 경계, 첫 글자만 남기고 전부 마스크
28696
+ passport: {
28697
+ pattern: /\b([MSRODG])\d{8}(?![0-9A-Za-z])/g,
28698
+ mask: (m, mc) => m[1] + mc.repeat(8)
28699
+ }
28700
+ };
28701
+ function redactText(text, options) {
28702
+ const maskChar = options?.maskChar ?? "\u25CF";
28703
+ if (maskChar.length !== 1 || /[0-9A-Za-z]/.test(maskChar)) {
28704
+ throw new Error(`maskChar\uB294 \uC601\uC22B\uC790\uAC00 \uC544\uB2CC 1\uAE00\uC790\uC5EC\uC57C \uD568: ${JSON.stringify(maskChar)}`);
28705
+ }
28706
+ const enabled = options?.rules ?? DEFAULT_REDACT_RULES;
28707
+ const hits = [];
28708
+ if (text === "" || enabled.length === 0) return { text, hits };
28709
+ const occupied = [];
28710
+ for (const rule of RULE_PRIORITY) {
28711
+ if (!enabled.includes(rule)) continue;
28712
+ const def = RULES2[rule];
28713
+ for (const m of text.matchAll(def.pattern)) {
28714
+ if (def.validate && !def.validate(m)) continue;
28715
+ const start = m.index;
28716
+ const end = start + m[0].length;
28717
+ if (occupied.some((o) => start < o.end && end > o.start)) continue;
28718
+ occupied.push({ start, end });
28719
+ hits.push({ rule, masked: def.mask(m, maskChar), index: start, length: m[0].length });
28720
+ }
28721
+ }
28722
+ hits.sort((a, b) => a.index - b.index);
28723
+ let out = "";
28724
+ let cursor = 0;
28725
+ for (const h of hits) {
28726
+ out += text.slice(cursor, h.index) + h.masked;
28727
+ cursor = h.index + h.length;
28728
+ }
28729
+ out += text.slice(cursor);
28730
+ return { text: out, hits };
28731
+ }
28732
+ function redactMarkdown(markdown, options) {
28733
+ const lines = markdown.split("\n");
28734
+ const hits = [];
28735
+ let offset = 0;
28736
+ const outLines = lines.map((line) => {
28737
+ if (line.includes("data:image/")) {
28738
+ offset += line.length + 1;
28739
+ return line;
28740
+ }
28741
+ const r = redactText(line, options);
28742
+ for (const h of r.hits) hits.push({ ...h, index: h.index + offset });
28743
+ offset += line.length + 1;
28744
+ return r.text;
28745
+ });
28746
+ return { text: outLines.join("\n"), hits };
28747
+ }
28748
+
28749
+ // src/chunks.ts
28750
+ var LIST_MARKER_RE = /^(?:[□■◇◆○◎●◦ㅇ•▪▸▶※-]|\d{1,3}[.)]|[가나다라마바사아자차카타파하][.)]|\([가나다라마바사아자차카타파하0-9]{1,3}\)|[①-⑳㉮-㉻㈎-㈛])\s/;
28751
+ function listDepthOf(block) {
28752
+ if (block.listDepth !== void 0) return block.listDepth;
28753
+ if (block.type === "list") return 0;
28754
+ if (block.type === "paragraph" && block.text && LIST_MARKER_RE.test(block.text.trim())) return 0;
28755
+ return void 0;
28756
+ }
28757
+ function sameBreadcrumb(a, b) {
28758
+ return a.length === b.length && a.every((v, i) => v === b[i]);
28759
+ }
28760
+ function blocksToChunks(blocks, options) {
28761
+ const granularity = options?.granularity ?? "section";
28762
+ const includeCells = options?.includeTableCells ?? false;
28763
+ const chunks = [];
28764
+ const headingStack = [];
28765
+ const listStack = [];
28766
+ const crumb = () => [...headingStack.map((h) => h.text), ...listStack.map((l) => l.text)];
28767
+ const push = (type, breadcrumb, text, blockRange, page, table2) => {
28768
+ const chunk = { id: "c" + String(chunks.length + 1).padStart(4, "0"), type, breadcrumb, text, blockRange };
28769
+ if (page !== void 0) chunk.page = page;
28770
+ if (table2) chunk.table = table2;
28771
+ chunks.push(chunk);
28772
+ };
28773
+ let run = null;
28774
+ const flushRun = () => {
28775
+ if (!run) return;
28776
+ push("text", run.breadcrumb, blocksToMarkdown(run.blocks), [run.start, run.end], run.page);
28777
+ run = null;
28778
+ };
28779
+ for (let i = 0; i < blocks.length; i++) {
28780
+ const block = blocks[i];
28781
+ const md2 = blocksToMarkdown([block]);
28782
+ if (!md2) continue;
28783
+ if (block.type === "heading") {
28784
+ flushRun();
28785
+ const level = Math.min(block.level || 2, 6);
28786
+ while (headingStack.length && headingStack[headingStack.length - 1].level >= level) headingStack.pop();
28787
+ listStack.length = 0;
28788
+ push("heading", crumb(), md2, [i, i], block.pageNumber);
28789
+ headingStack.push({ level, text: (block.text ?? "").trim() });
28790
+ continue;
28791
+ }
28792
+ if (block.type === "table" && block.table) {
28793
+ flushRun();
28794
+ const summary = { rows: block.table.rows, cols: block.table.cols };
28795
+ if (includeCells) summary.cells = block.table.cells.map((row) => row.map((c) => c.text));
28796
+ push("table", crumb(), md2, [i, i], block.pageNumber, summary);
28797
+ continue;
28798
+ }
28799
+ const depth = listDepthOf(block);
28800
+ if (depth !== void 0) {
28801
+ while (listStack.length && listStack[listStack.length - 1].depth >= depth) listStack.pop();
28802
+ }
28803
+ const breadcrumb = crumb();
28804
+ if (depth !== void 0) listStack.push({ depth, text: (block.text ?? "").trim() });
28805
+ if (granularity === "block") {
28806
+ push("text", breadcrumb, md2, [i, i], block.pageNumber);
28807
+ continue;
28808
+ }
28809
+ if (run && sameBreadcrumb(run.breadcrumb, breadcrumb)) {
28810
+ run.blocks.push(block);
28811
+ run.end = i;
28812
+ if (run.page === void 0) run.page = block.pageNumber;
28813
+ } else {
28814
+ flushRun();
28815
+ run = { breadcrumb, blocks: [block], start: i, end: i, page: block.pageNumber };
28816
+ }
28817
+ }
28818
+ flushRun();
28819
+ return chunks;
28820
+ }
28821
+
28224
28822
  // src/roundtrip/session.ts
28225
28823
  import JSZip12 from "jszip";
28226
28824
  async function buildState(bytes) {
@@ -29196,7 +29794,7 @@ function drawPara(p, ox, oy, areaW, ctx, depth, segPages) {
29196
29794
  }
29197
29795
  if (text.trim().length > 0) {
29198
29796
  const attrs = [`x="${pt(cx)}"`, `y="${pt(y)}"`, `font-size="${pt(st.height)}"`];
29199
- if (st.fontFamily) attrs.push(`font-family="${st.fontFamily}"`);
29797
+ if (st.fontFamily) attrs.push(`font-family="${escapeXml3(st.fontFamily)}"`);
29200
29798
  if ([...text].length > 1 && sw > 50) {
29201
29799
  attrs.push(`textLength="${pt(sw)}"`, `lengthAdjust="${plan.scale < 1 ? "spacingAndGlyphs" : "spacing"}"`);
29202
29800
  }
@@ -29363,8 +29961,10 @@ function drawShape(o, x, y, ctx, depth) {
29363
29961
  for (const p of elements2(sub)) if (ln2(p) === "p") drawPara(p, x, y, w, ctx, depth + 1);
29364
29962
  }
29365
29963
  }
29366
- function cellContentExtent(cell2) {
29964
+ function cellContentExtent(cell2, memo) {
29367
29965
  if (!cell2.sub) return 0;
29966
+ const hit = memo?.cell.get(cell2.el);
29967
+ if (hit !== void 0) return hit;
29368
29968
  let ext = 0;
29369
29969
  for (const p of elements2(cell2.sub)) {
29370
29970
  if (ln2(p) !== "p") continue;
@@ -29373,7 +29973,7 @@ function cellContentExtent(cell2) {
29373
29973
  const baseV = m.segs[0]?.vertpos ?? 0;
29374
29974
  for (const o of m.objs) {
29375
29975
  if (o.inline) {
29376
- const h = o.tag === "tbl" ? Math.max(o.height, measureTableHeight(o.el)) : o.height;
29976
+ const h = o.tag === "tbl" ? Math.max(o.height, measureTableHeight(o.el, memo)) : o.height;
29377
29977
  ext = Math.max(ext, baseV + h);
29378
29978
  continue;
29379
29979
  }
@@ -29385,6 +29985,7 @@ function cellContentExtent(cell2) {
29385
29985
  ext = Math.max(ext, anchor + num3(om, "top") + num3(pos, "vertOffset") + o.height);
29386
29986
  }
29387
29987
  }
29988
+ memo?.cell.set(cell2.el, ext);
29388
29989
  return ext;
29389
29990
  }
29390
29991
  function edgeLine(x1, y1, x2, y2, e) {
@@ -29407,8 +30008,9 @@ function collectCells(tbl2) {
29407
30008
  if (!addr || !csz) continue;
29408
30009
  cells.push({
29409
30010
  el: tc2,
29410
- ca: num3(addr, "colAddr"),
29411
- ra: num3(addr, "rowAddr"),
30011
+ // 음수 주소(uint32 역변환·손상 입력) 방어 — colX/rowY 인덱스 이탈로 NaN 좌표 방지
30012
+ ca: Math.max(0, num3(addr, "colAddr")),
30013
+ ra: Math.max(0, num3(addr, "rowAddr")),
29412
30014
  cs: Math.max(1, num3(span, "colSpan", 1)),
29413
30015
  rs: Math.max(1, num3(span, "rowSpan", 1)),
29414
30016
  w: num3(csz, "width"),
@@ -29424,16 +30026,19 @@ function collectCells(tbl2) {
29424
30026
  }
29425
30027
  return cells;
29426
30028
  }
29427
- function measureTableHeight(tbl2) {
30029
+ function measureTableHeight(tbl2, memo) {
30030
+ const hit = memo?.table.get(tbl2);
30031
+ if (hit !== void 0) return hit;
29428
30032
  const cells = collectCells(tbl2);
29429
30033
  if (cells.length === 0 || cells.length > 4096) return 0;
29430
30034
  const nRows = Math.max(...cells.map((c) => c.ra + c.rs));
29431
30035
  const rowH = solveRowHeights(
29432
- cells.map((c) => ({ rowAddr: c.ra, rowSpan: c.rs, height: c.h, contentH: c.rs === 1 ? cellContentExtent(c) : void 0 })),
30036
+ cells.map((c) => ({ rowAddr: c.ra, rowSpan: c.rs, height: c.h, contentH: c.rs === 1 ? cellContentExtent(c, memo) : void 0 })),
29433
30037
  nRows
29434
30038
  );
29435
30039
  let sum = 0;
29436
30040
  for (const h of rowH) sum += h;
30041
+ memo?.table.set(tbl2, sum);
29437
30042
  return sum;
29438
30043
  }
29439
30044
  function drawTable(tbl2, tx, ty, ctx, depth) {
@@ -29450,7 +30055,7 @@ function drawTable(tbl2, tx, ty, ctx, depth) {
29450
30055
  const colCons = cells.map((c) => ({ a: c.ca, b: c.ca + c.cs, size: c.w }));
29451
30056
  const colX = solveBoundaries(colCons, nCols, num3(tblSz, "width") || void 0);
29452
30057
  const rowH = solveRowHeights(
29453
- cells.map((c) => ({ rowAddr: c.ra, rowSpan: c.rs, height: c.h, contentH: c.rs === 1 ? cellContentExtent(c) : void 0 })),
30058
+ cells.map((c) => ({ rowAddr: c.ra, rowSpan: c.rs, height: c.h, contentH: c.rs === 1 ? cellContentExtent(c, ctx.extentMemo) : void 0 })),
29454
30059
  nRows
29455
30060
  );
29456
30061
  const rowY = [0];
@@ -29470,7 +30075,7 @@ function drawTable(tbl2, tx, ty, ctx, depth) {
29470
30075
  const { c } = g;
29471
30076
  if (!c.sub) continue;
29472
30077
  const innerH = g.h - c.marginT - c.marginB;
29473
- const extent = cellContentExtent(c);
30078
+ const extent = cellContentExtent(c, ctx.extentMemo);
29474
30079
  const va = c.sub.getAttribute("vertAlign") ?? "TOP";
29475
30080
  let yoff = 0;
29476
30081
  if (va === "CENTER") yoff = Math.max(0, (innerH - extent) / 2);
@@ -29533,8 +30138,9 @@ function sniffMime(name, bytes) {
29533
30138
  const lower = name.toLowerCase();
29534
30139
  if (lower.endsWith(".png") || bytes.length > 4 && bytes[0] === 137 && bytes[1] === 80) return "image/png";
29535
30140
  if (lower.endsWith(".bmp") || bytes.length > 2 && bytes[0] === 66 && bytes[1] === 77) return "image/bmp";
29536
- if (lower.endsWith(".gif")) return "image/gif";
30141
+ if (lower.endsWith(".gif") || bytes.length > 3 && bytes[0] === 71 && bytes[1] === 73 && bytes[2] === 70) return "image/gif";
29537
30142
  if (lower.endsWith(".svg")) return "image/svg+xml";
30143
+ if (bytes.length > 3 && bytes[0] === 255 && bytes[1] === 216 && bytes[2] === 255) return "image/jpeg";
29538
30144
  return "image/jpeg";
29539
30145
  }
29540
30146
  function readSectionGeom(root) {
@@ -29665,7 +30271,8 @@ async function renderHwpxToSvg(input, options) {
29665
30271
  highlights: (options?.highlights ?? []).map((s) => s.trim().toLowerCase()).filter((s) => s.length > 0),
29666
30272
  warnings,
29667
30273
  warned: /* @__PURE__ */ new Set(),
29668
- stats: { texts: 0, images: 0, tables: 0 }
30274
+ stats: { texts: 0, images: 0, tables: 0 },
30275
+ extentMemo: { cell: /* @__PURE__ */ new WeakMap(), table: /* @__PURE__ */ new WeakMap() }
29669
30276
  };
29670
30277
  const rendered = [];
29671
30278
  let noCacheSkipped = false;
@@ -29768,7 +30375,7 @@ async function parseHwp3(buffer, options) {
29768
30375
  const { markdown, blocks, metadata, outline, warnings } = parseHwp3Document(buffer, options);
29769
30376
  return { success: true, fileType: "hwp3", markdown, blocks, metadata, outline, warnings };
29770
30377
  } catch (err) {
29771
- return { success: false, fileType: "hwp3", error: err instanceof Error ? err.message : "HWP3 \uD30C\uC2F1 \uC2E4\uD328", code: classifyError(err) };
30378
+ return { success: false, fileType: "hwp3", error: sanitizeError(err), code: classifyError(err) };
29772
30379
  }
29773
30380
  }
29774
30381
  async function parseHwpx(buffer, options) {
@@ -29776,7 +30383,7 @@ async function parseHwpx(buffer, options) {
29776
30383
  const { markdown, blocks, metadata, outline, warnings, images } = await parseHwpxDocument(buffer, options);
29777
30384
  return { success: true, fileType: "hwpx", markdown, blocks, metadata, outline, warnings, images: images?.length ? images : void 0 };
29778
30385
  } catch (err) {
29779
- return { success: false, fileType: "hwpx", error: err instanceof Error ? err.message : "HWPX \uD30C\uC2F1 \uC2E4\uD328", code: classifyError(err) };
30386
+ return { success: false, fileType: "hwpx", error: sanitizeError(err), code: classifyError(err) };
29780
30387
  }
29781
30388
  }
29782
30389
  async function parseHwp(buffer, options) {
@@ -29801,13 +30408,13 @@ async function parseHwp(buffer, options) {
29801
30408
  }
29802
30409
  return { success: true, fileType: "hwp", markdown, blocks, metadata, outline, warnings, images: images?.length ? images : void 0 };
29803
30410
  } catch (err) {
29804
- return { success: false, fileType: "hwp", error: err instanceof Error ? err.message : "HWP \uD30C\uC2F1 \uC2E4\uD328", code: classifyError(err) };
30411
+ return { success: false, fileType: "hwp", error: sanitizeError(err), code: classifyError(err) };
29805
30412
  }
29806
30413
  }
29807
30414
  async function parsePdf(buffer, options) {
29808
30415
  let parsePdfDocument;
29809
30416
  try {
29810
- const mod = await import("./parser-KMMUS73Q.js");
30417
+ const mod = await import("./parser-XX4QTDFQ.js");
29811
30418
  parsePdfDocument = mod.parsePdfDocument;
29812
30419
  } catch {
29813
30420
  return {
@@ -29822,7 +30429,7 @@ async function parsePdf(buffer, options) {
29822
30429
  return { success: true, fileType: "pdf", markdown, blocks, metadata, outline, warnings, isImageBased, pageQuality, qualitySummary, images };
29823
30430
  } catch (err) {
29824
30431
  const isImageBased = err instanceof Error && "isImageBased" in err ? true : void 0;
29825
- return { success: false, fileType: "pdf", error: err instanceof Error ? err.message : "PDF \uD30C\uC2F1 \uC2E4\uD328", code: classifyError(err), isImageBased };
30432
+ return { success: false, fileType: "pdf", error: sanitizeError(err), code: classifyError(err), isImageBased };
29826
30433
  }
29827
30434
  }
29828
30435
  async function parseXlsx(buffer, options) {
@@ -29830,7 +30437,7 @@ async function parseXlsx(buffer, options) {
29830
30437
  const { markdown, blocks, metadata, warnings } = await parseXlsxDocument(buffer, options);
29831
30438
  return { success: true, fileType: "xlsx", markdown, blocks, metadata, warnings };
29832
30439
  } catch (err) {
29833
- return { success: false, fileType: "xlsx", error: err instanceof Error ? err.message : "XLSX \uD30C\uC2F1 \uC2E4\uD328", code: classifyError(err) };
30440
+ return { success: false, fileType: "xlsx", error: sanitizeError(err), code: classifyError(err) };
29834
30441
  }
29835
30442
  }
29836
30443
  async function parseXls(buffer, options) {
@@ -29838,7 +30445,7 @@ async function parseXls(buffer, options) {
29838
30445
  const { markdown, blocks, metadata, warnings } = await parseXlsDocument(buffer, options);
29839
30446
  return { success: true, fileType: "xls", markdown, blocks, metadata, warnings };
29840
30447
  } catch (err) {
29841
- return { success: false, fileType: "xls", error: err instanceof Error ? err.message : "XLS \uD30C\uC2F1 \uC2E4\uD328", code: classifyError(err) };
30448
+ return { success: false, fileType: "xls", error: sanitizeError(err), code: classifyError(err) };
29842
30449
  }
29843
30450
  }
29844
30451
  async function parseDocx(buffer, options) {
@@ -29846,7 +30453,7 @@ async function parseDocx(buffer, options) {
29846
30453
  const { markdown, blocks, metadata, outline, warnings, images } = await parseDocxDocument(buffer, options);
29847
30454
  return { success: true, fileType: "docx", markdown, blocks, metadata, outline, warnings, images: images?.length ? images : void 0 };
29848
30455
  } catch (err) {
29849
- return { success: false, fileType: "docx", error: err instanceof Error ? err.message : "DOCX \uD30C\uC2F1 \uC2E4\uD328", code: classifyError(err) };
30456
+ return { success: false, fileType: "docx", error: sanitizeError(err), code: classifyError(err) };
29850
30457
  }
29851
30458
  }
29852
30459
  async function parseHwpml(buffer, options) {
@@ -29854,7 +30461,7 @@ async function parseHwpml(buffer, options) {
29854
30461
  const { markdown, blocks, metadata, outline, warnings } = parseHwpmlDocument(buffer, options);
29855
30462
  return { success: true, fileType: "hwpml", markdown, blocks, metadata, outline, warnings };
29856
30463
  } catch (err) {
29857
- return { success: false, fileType: "hwpml", error: err instanceof Error ? err.message : "HWPML \uD30C\uC2F1 \uC2E4\uD328", code: classifyError(err) };
30464
+ return { success: false, fileType: "hwpml", error: sanitizeError(err), code: classifyError(err) };
29858
30465
  }
29859
30466
  }
29860
30467
  async function fillForm(input, values, outputFormat = "markdown") {
@@ -29897,6 +30504,7 @@ async function fillForm(input, values, outputFormat = "markdown") {
29897
30504
  return { output: markdown, format: "markdown", fill };
29898
30505
  }
29899
30506
  export {
30507
+ DEFAULT_REDACT_RULES,
29900
30508
  HwpxSession,
29901
30509
  PRESET_ALIAS,
29902
30510
  SPACE_EM_FIXED,
@@ -29904,6 +30512,7 @@ export {
29904
30512
  VERSION,
29905
30513
  ValueCursor,
29906
30514
  applySplices,
30515
+ blocksToChunks,
29907
30516
  blocksToMarkdown,
29908
30517
  blocksToPdf,
29909
30518
  buildParagraphSplices,
@@ -29951,6 +30560,8 @@ export {
29951
30560
  patchHwpx,
29952
30561
  patchHwpxBlocks,
29953
30562
  placeSealHwpx,
30563
+ redactMarkdown,
30564
+ redactText,
29954
30565
  renderHtml,
29955
30566
  renderHwpxToSvg,
29956
30567
  scanSectionXml,