kordoc 4.2.1 → 4.2.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (97) hide show
  1. package/README.md +19 -2
  2. package/dist/{-BSO3CDQQ.js → -GJYDKVQD.js} +9 -9
  3. package/dist/{chunk-LSXXGUWA.cjs → chunk-5AXGFJQO.cjs} +8 -6
  4. package/dist/chunk-5AXGFJQO.cjs.map +1 -0
  5. package/dist/{chunk-R22KDUKW.cjs → chunk-AA6UPPZF.cjs} +9 -9
  6. package/dist/{chunk-R22KDUKW.cjs.map → chunk-AA6UPPZF.cjs.map} +1 -1
  7. package/dist/{chunk-WJSGYVFD.js → chunk-CGFWS3BQ.js} +3 -3
  8. package/dist/{chunk-RBUBZ2TA.cjs → chunk-CWP56BRC.cjs} +3 -3
  9. package/dist/chunk-CWP56BRC.cjs.map +1 -0
  10. package/dist/{chunk-QSB2RTX5.js → chunk-DE6FORIP.js} +6 -4
  11. package/dist/chunk-DE6FORIP.js.map +1 -0
  12. package/dist/{chunk-F5ECB53I.js → chunk-ELRVXOMP.js} +2 -2
  13. package/dist/{chunk-35U3MVA2.cjs → chunk-FF5QNEKZ.cjs} +3 -3
  14. package/dist/{chunk-35U3MVA2.cjs.map → chunk-FF5QNEKZ.cjs.map} +1 -1
  15. package/dist/{chunk-54WVWWT4.js → chunk-GJO3ORFW.js} +1 -1
  16. package/dist/chunk-GJO3ORFW.js.map +1 -0
  17. package/dist/{chunk-7IMNZ2BL.js → chunk-K5LGSH5T.js} +6 -4
  18. package/dist/chunk-K5LGSH5T.js.map +1 -0
  19. package/dist/{chunk-BKBYTPSL.js → chunk-LMNSIPWZ.js} +2 -2
  20. package/dist/{chunk-EZHWPKCH.js → chunk-LPWP4FWU.js} +2 -2
  21. package/dist/{chunk-QOC7S5VE.js → chunk-NIPPOAT6.js} +2 -2
  22. package/dist/{chunk-KRAYUNFM.js → chunk-NOMNOG74.js} +2 -2
  23. package/dist/{chunk-HTCKGSSW.js → chunk-QNVZAKUQ.js} +3 -3
  24. package/dist/chunk-QNVZAKUQ.js.map +1 -0
  25. package/dist/{chunk-Z6LPEWIX.js → chunk-RAS3HM4X.js} +3 -3
  26. package/dist/chunk-RAS3HM4X.js.map +1 -0
  27. package/dist/{chunk-L5JWKIOT.js → chunk-U6XYOV32.js} +3 -3
  28. package/dist/{chunk-PLTG6ZRF.js → chunk-WTJIRGPN.js} +3 -3
  29. package/dist/{chunk-WWYAYZ5U.js → chunk-WX75LYAS.js} +128 -35
  30. package/dist/chunk-WX75LYAS.js.map +1 -0
  31. package/dist/{chunk-BHILWIE6.js → chunk-X634MXVJ.js} +3 -3
  32. package/dist/chunk-X634MXVJ.js.map +1 -0
  33. package/dist/chunks-JOGD2KUK.js +10 -0
  34. package/dist/cli.js +18 -18
  35. package/dist/{image-ocr-SLUEM3FA.js → image-ocr-O3YQ3KEN.js} +5 -5
  36. package/dist/{image-ocr-DALT7PFQ.js → image-ocr-PLTV4NOK.js} +4 -4
  37. package/dist/{image-ocr-N5Z32R4U.cjs → image-ocr-TCJLNMTG.cjs} +9 -9
  38. package/dist/{image-ocr-N5Z32R4U.cjs.map → image-ocr-TCJLNMTG.cjs.map} +1 -1
  39. package/dist/index.cjs +545 -452
  40. package/dist/index.cjs.map +1 -1
  41. package/dist/index.d.cts +10 -2
  42. package/dist/index.d.ts +10 -2
  43. package/dist/index.js +124 -31
  44. package/dist/index.js.map +1 -1
  45. package/dist/mcp.js +15 -15
  46. package/dist/{parser-F5KY3YSD.js → parser-UX5T6KTZ.js} +6 -6
  47. package/dist/{parser-R4CXIXRN.cjs → parser-VQIJPCI5.cjs} +24 -24
  48. package/dist/{parser-R4CXIXRN.cjs.map → parser-VQIJPCI5.cjs.map} +1 -1
  49. package/dist/{parser-33TK5SMH.js → parser-Y2X4JMAK.js} +5 -5
  50. package/dist/{pdf-ocr-BXQZKU3N.js → pdf-ocr-DOPZP6QS.js} +5 -5
  51. package/dist/pdf-ocr-I3Y6YF5K.js +12 -0
  52. package/dist/pdf-ocr-XMDJ2IF5.cjs +13 -0
  53. package/dist/{pdf-ocr-WGWZDHMV.cjs.map → pdf-ocr-XMDJ2IF5.cjs.map} +1 -1
  54. package/dist/{profile-io-CBYHKLFU.js → profile-io-EGVJAZDG.js} +3 -3
  55. package/dist/{rasterize-AWMM2CSW.js → rasterize-TH26U5E2.js} +2 -2
  56. package/dist/render-C4R3OJ45.js +10 -0
  57. package/dist/seal-JF2J4S52.js +10 -0
  58. package/dist/{setup-Q2PRE7UA.js → setup-GUQT5ELJ.js} +23 -3
  59. package/dist/setup-GUQT5ELJ.js.map +1 -0
  60. package/dist/{watch-HKXCISVO.js → watch-4SP7V6SV.js} +9 -9
  61. package/package.json +1 -1
  62. package/dist/chunk-54WVWWT4.js.map +0 -1
  63. package/dist/chunk-7IMNZ2BL.js.map +0 -1
  64. package/dist/chunk-BHILWIE6.js.map +0 -1
  65. package/dist/chunk-HTCKGSSW.js.map +0 -1
  66. package/dist/chunk-LSXXGUWA.cjs.map +0 -1
  67. package/dist/chunk-QSB2RTX5.js.map +0 -1
  68. package/dist/chunk-RBUBZ2TA.cjs.map +0 -1
  69. package/dist/chunk-WWYAYZ5U.js.map +0 -1
  70. package/dist/chunk-Z6LPEWIX.js.map +0 -1
  71. package/dist/chunks-LRFS5NJE.js +0 -10
  72. package/dist/pdf-ocr-34F5AQKK.js +0 -12
  73. package/dist/pdf-ocr-WGWZDHMV.cjs +0 -13
  74. package/dist/render-Y3IE2FIK.js +0 -10
  75. package/dist/seal-7RXFCRMT.js +0 -10
  76. package/dist/setup-Q2PRE7UA.js.map +0 -1
  77. /package/dist/{-BSO3CDQQ.js.map → -GJYDKVQD.js.map} +0 -0
  78. /package/dist/{chunk-WJSGYVFD.js.map → chunk-CGFWS3BQ.js.map} +0 -0
  79. /package/dist/{chunk-F5ECB53I.js.map → chunk-ELRVXOMP.js.map} +0 -0
  80. /package/dist/{chunk-BKBYTPSL.js.map → chunk-LMNSIPWZ.js.map} +0 -0
  81. /package/dist/{chunk-EZHWPKCH.js.map → chunk-LPWP4FWU.js.map} +0 -0
  82. /package/dist/{chunk-QOC7S5VE.js.map → chunk-NIPPOAT6.js.map} +0 -0
  83. /package/dist/{chunk-KRAYUNFM.js.map → chunk-NOMNOG74.js.map} +0 -0
  84. /package/dist/{chunk-L5JWKIOT.js.map → chunk-U6XYOV32.js.map} +0 -0
  85. /package/dist/{chunk-PLTG6ZRF.js.map → chunk-WTJIRGPN.js.map} +0 -0
  86. /package/dist/{chunks-LRFS5NJE.js.map → chunks-JOGD2KUK.js.map} +0 -0
  87. /package/dist/{image-ocr-SLUEM3FA.js.map → image-ocr-O3YQ3KEN.js.map} +0 -0
  88. /package/dist/{image-ocr-DALT7PFQ.js.map → image-ocr-PLTV4NOK.js.map} +0 -0
  89. /package/dist/{parser-F5KY3YSD.js.map → parser-UX5T6KTZ.js.map} +0 -0
  90. /package/dist/{parser-33TK5SMH.js.map → parser-Y2X4JMAK.js.map} +0 -0
  91. /package/dist/{pdf-ocr-34F5AQKK.js.map → pdf-ocr-DOPZP6QS.js.map} +0 -0
  92. /package/dist/{pdf-ocr-BXQZKU3N.js.map → pdf-ocr-I3Y6YF5K.js.map} +0 -0
  93. /package/dist/{profile-io-CBYHKLFU.js.map → profile-io-EGVJAZDG.js.map} +0 -0
  94. /package/dist/{rasterize-AWMM2CSW.js.map → rasterize-TH26U5E2.js.map} +0 -0
  95. /package/dist/{render-Y3IE2FIK.js.map → render-C4R3OJ45.js.map} +0 -0
  96. /package/dist/{seal-7RXFCRMT.js.map → seal-JF2J4S52.js.map} +0 -0
  97. /package/dist/{watch-HKXCISVO.js.map → watch-4SP7V6SV.js.map} +0 -0
package/dist/index.d.cts CHANGED
@@ -15,6 +15,7 @@ interface IRSpan {
15
15
  text: string;
16
16
  bold?: boolean;
17
17
  italic?: boolean;
18
+ strike?: boolean;
18
19
  code?: boolean;
19
20
  }
20
21
  interface IRBlock {
@@ -78,6 +79,8 @@ interface BoundingBox {
78
79
  interface InlineStyle {
79
80
  bold?: boolean;
80
81
  italic?: boolean;
82
+ /** 취소선 — 법령 개정문 등의 삭제 표시. 판정은 취소선 모양 whitelist (비트만 믿으면 오탐) */
83
+ strike?: boolean;
81
84
  fontSize?: number;
82
85
  fontName?: string;
83
86
  }
@@ -96,7 +99,8 @@ interface IRCell {
96
99
  rowSpan: number;
97
100
  /**
98
101
  * 셀 내부 블록 콘텐츠 — v3.0.
99
- * 중첩 표·이미지·다중 문단을 구조 그대로 보존한다.
102
+ * 중첩 표·이미지·다중 문단을 구조 그대로, 문서(원문) 순서대로 보존한다.
103
+ * 표와 텍스트가 한 줄에 번갈아 놓인 셀도 배치 순서를 따른다 (v4.2.3, #49).
100
104
  * blocks가 있으면 text는 blocks의 평탄화 텍스트(하위 호환용)다.
101
105
  */
102
106
  blocks?: IRBlock[];
@@ -215,7 +219,11 @@ interface ParseSuccess extends ParseResultBase {
215
219
  success: true;
216
220
  /** 추출된 마크다운 텍스트 */
217
221
  markdown: string;
218
- /** 중간 표현 블록 (구조화된 데이터 접근용) */
222
+ /**
223
+ * 중간 표현 블록 (구조화된 데이터 접근용).
224
+ * 블록은 문서(원문) 읽기 순서를 따른다 — 한 문단 안에 글자취급(treatAsChar)
225
+ * 표와 텍스트가 섞여 있어도 배치 순서대로 방출된다 (v4.2.3, #50).
226
+ */
219
227
  blocks: IRBlock[];
220
228
  /** 문서 메타데이터 */
221
229
  metadata?: DocumentMetadata;
package/dist/index.d.ts CHANGED
@@ -15,6 +15,7 @@ interface IRSpan {
15
15
  text: string;
16
16
  bold?: boolean;
17
17
  italic?: boolean;
18
+ strike?: boolean;
18
19
  code?: boolean;
19
20
  }
20
21
  interface IRBlock {
@@ -78,6 +79,8 @@ interface BoundingBox {
78
79
  interface InlineStyle {
79
80
  bold?: boolean;
80
81
  italic?: boolean;
82
+ /** 취소선 — 법령 개정문 등의 삭제 표시. 판정은 취소선 모양 whitelist (비트만 믿으면 오탐) */
83
+ strike?: boolean;
81
84
  fontSize?: number;
82
85
  fontName?: string;
83
86
  }
@@ -96,7 +99,8 @@ interface IRCell {
96
99
  rowSpan: number;
97
100
  /**
98
101
  * 셀 내부 블록 콘텐츠 — v3.0.
99
- * 중첩 표·이미지·다중 문단을 구조 그대로 보존한다.
102
+ * 중첩 표·이미지·다중 문단을 구조 그대로, 문서(원문) 순서대로 보존한다.
103
+ * 표와 텍스트가 한 줄에 번갈아 놓인 셀도 배치 순서를 따른다 (v4.2.3, #49).
100
104
  * blocks가 있으면 text는 blocks의 평탄화 텍스트(하위 호환용)다.
101
105
  */
102
106
  blocks?: IRBlock[];
@@ -215,7 +219,11 @@ interface ParseSuccess extends ParseResultBase {
215
219
  success: true;
216
220
  /** 추출된 마크다운 텍스트 */
217
221
  markdown: string;
218
- /** 중간 표현 블록 (구조화된 데이터 접근용) */
222
+ /**
223
+ * 중간 표현 블록 (구조화된 데이터 접근용).
224
+ * 블록은 문서(원문) 읽기 순서를 따른다 — 한 문단 안에 글자취급(treatAsChar)
225
+ * 표와 텍스트가 섞여 있어도 배치 순서대로 방출된다 (v4.2.3, #50).
226
+ */
219
227
  blocks: IRBlock[];
220
228
  /** 문서 메타데이터 */
221
229
  metadata?: DocumentMetadata;
package/dist/index.js CHANGED
@@ -8,7 +8,7 @@ import {
8
8
  flattenLayoutTables,
9
9
  inlineImagesIntoMarkdown,
10
10
  mapPuaText
11
- } from "./chunk-7IMNZ2BL.js";
11
+ } from "./chunk-K5LGSH5T.js";
12
12
  import {
13
13
  parsePageRange
14
14
  } from "./chunk-GE43BE46.js";
@@ -27,7 +27,7 @@ import {
27
27
  sanitizeHref,
28
28
  stripDtd,
29
29
  toArrayBuffer
30
- } from "./chunk-HTCKGSSW.js";
30
+ } from "./chunk-QNVZAKUQ.js";
31
31
 
32
32
  // src/index.ts
33
33
  import { readFile } from "fs/promises";
@@ -408,7 +408,7 @@ function comResultToParseResult(pages, pageCount, warnings) {
408
408
 
409
409
  // src/hwpx/parser-shared.ts
410
410
  import { DOMParser } from "@xmldom/xmldom";
411
- var MAX_DECOMPRESS_SIZE = 100 * 1024 * 1024;
411
+ var MAX_DECOMPRESS_SIZE = 256 * 1024 * 1024;
412
412
  var MAX_ZIP_ENTRIES = 500;
413
413
  var ZipBombError = class extends KordocError {
414
414
  constructor(message) {
@@ -465,6 +465,26 @@ function extractTextFromNode(node, depth = 0) {
465
465
  }
466
466
 
467
467
  // src/hwpx/styles.ts
468
+ function isRealStrikeShape(shape) {
469
+ switch (shape) {
470
+ case "SOLID":
471
+ case "DASH":
472
+ case "DOT":
473
+ case "DASH_DOT":
474
+ case "DASH_DOT_DOT":
475
+ case "LONG_DASH":
476
+ case "CIRCLE":
477
+ case "DOUBLE_SLIM":
478
+ case "SLIM_THICK":
479
+ case "THICK_SLIM":
480
+ case "SLIM_THICK_SLIM":
481
+ case "WAVE":
482
+ case "DOUBLE_WAVE":
483
+ return true;
484
+ default:
485
+ return false;
486
+ }
487
+ }
468
488
  async function extractHwpxStyles(zip, decompressed) {
469
489
  const result = {
470
490
  charProperties: /* @__PURE__ */ new Map(),
@@ -529,6 +549,10 @@ function parseCharProperties(doc, map) {
529
549
  const localTag = (k.tagName || "").replace(/^[^:]+:/, "");
530
550
  if (localTag === "bold") prop.bold = true;
531
551
  else if (localTag === "italic") prop.italic = true;
552
+ else if (localTag === "strikeout") {
553
+ const shape = k.getAttribute("shape") || "";
554
+ if (isRealStrikeShape(shape)) prop.strike = true;
555
+ }
532
556
  }
533
557
  const fontFaces = el.getElementsByTagName("*");
534
558
  for (let j = 0; j < fontFaces.length; j++) {
@@ -2081,7 +2105,7 @@ function walkSection(node, blocks, tableCtx, tableStack, ctx, depth = 0) {
2081
2105
  }
2082
2106
  break;
2083
2107
  case "p": {
2084
- const { text: rawText, href, footnote, style } = extractParagraphInfo(el, ctx.styleMap, ctx);
2108
+ const { text: rawText, href, footnote, style, segments } = extractParagraphInfo(el, ctx.styleMap, ctx);
2085
2109
  let text = rawText;
2086
2110
  let headingLevel;
2087
2111
  const ph = resolveParaHeading(el, ctx);
@@ -2089,6 +2113,33 @@ function walkSection(node, blocks, tableCtx, tableStack, ctx, depth = 0) {
2089
2113
  if (ph?.prefix) text = ph.prefix + " " + text;
2090
2114
  headingLevel = ph?.headingLevel;
2091
2115
  }
2116
+ if (segments) {
2117
+ const cell2 = tableCtx?.cell ?? null;
2118
+ if (text && cell2) cell2.text += (cell2.text ? "\n" : "") + (footnote ? `${text} (\uC8FC: ${footnote})` : text);
2119
+ const segs = [...segments];
2120
+ if (ph?.prefix) {
2121
+ const fi = segs.findIndex((s) => s);
2122
+ if (fi >= 0) segs[fi] = ph.prefix + " " + segs[fi];
2123
+ }
2124
+ let segIdx = 0;
2125
+ let first = true;
2126
+ const flush = () => {
2127
+ const s = segs[segIdx++];
2128
+ if (!s) return;
2129
+ const block = { type: "paragraph", text: s, pageNumber: ctx.sectionNum };
2130
+ if (first && !cell2) {
2131
+ first = false;
2132
+ if (style) block.style = style;
2133
+ if (href) block.href = href;
2134
+ if (footnote) block.footnoteText = footnote;
2135
+ }
2136
+ if (cell2) (cell2.blocks ??= []).push(block);
2137
+ else blocks.push(block);
2138
+ };
2139
+ tableCtx = walkParagraphChildren(el, blocks, tableCtx, tableStack, ctx, depth + 1, flush);
2140
+ while (segIdx < segs.length) flush();
2141
+ break;
2142
+ }
2092
2143
  if (text) {
2093
2144
  if (tableCtx?.cell) {
2094
2145
  const cell2 = tableCtx.cell;
@@ -2307,7 +2358,7 @@ function findTopLevelTbls(el, out, depth = 0) {
2307
2358
  findTopLevelTbls(ch, out, depth + 1);
2308
2359
  }
2309
2360
  }
2310
- function walkParagraphChildren(node, blocks, tableCtx, tableStack, ctx, depth = 0) {
2361
+ function walkParagraphChildren(node, blocks, tableCtx, tableStack, ctx, depth = 0, onTbl) {
2311
2362
  if (depth > MAX_XML_DEPTH) return tableCtx;
2312
2363
  const children = node.childNodes;
2313
2364
  if (!children) return tableCtx;
@@ -2321,6 +2372,7 @@ function walkParagraphChildren(node, blocks, tableCtx, tableStack, ctx, depth =
2321
2372
  const tag = el.tagName || el.localName || "";
2322
2373
  const localTag = tag.replace(/^[^:]+:/, "");
2323
2374
  if (localTag === "tbl") {
2375
+ if (onTbl && isInlineTbl(el)) onTbl();
2324
2376
  if (!tableCtx) {
2325
2377
  const chan = kordocTableChannel(el, ctx);
2326
2378
  if (chan) {
@@ -2367,6 +2419,9 @@ function walkParagraphChildren(node, blocks, tableCtx, tableStack, ctx, depth =
2367
2419
  walkChildren(node, depth);
2368
2420
  return tableCtx;
2369
2421
  }
2422
+ function isInlineTbl(tbl2) {
2423
+ return findChildByLocalName(tbl2, "pos")?.getAttribute("treatAsChar") === "1";
2424
+ }
2370
2425
  function findDescendant(node, targetTag, depth = 0) {
2371
2426
  if (depth > 5) return null;
2372
2427
  const children = node.childNodes;
@@ -2540,7 +2595,7 @@ function extractParagraphInfo(para2, styleMap, ctx) {
2540
2595
  ctx.warnings?.push({ page: ctx.sectionNum, message: "\uBCC0\uACBD\uCD94\uC801 \uC0AD\uC81C \uD14D\uC2A4\uD2B8 \uCD9C\uB825 \uC81C\uC678", code: "HIDDEN_TEXT_FILTERED" });
2541
2596
  }
2542
2597
  } else {
2543
- text += t;
2598
+ text += t.replace(/\x1E/g, "");
2544
2599
  }
2545
2600
  continue;
2546
2601
  }
@@ -2571,9 +2626,12 @@ function extractParagraphInfo(para2, styleMap, ctx) {
2571
2626
  case "hwSpace":
2572
2627
  text += " ";
2573
2628
  break;
2629
+ // 테이블 자체는 walkSection에서 처리 — 글자취급(inline) 표만 경계 마커를 남겨
2630
+ // 표 앞뒤 텍스트를 문서 순서대로 분할 방출할 수 있게 한다 (#49/#50).
2631
+ // float·페이지 앵커 표는 텍스트 흐름 불참(reflow 모델 정합) — 종전대로 텍스트 뒤 방출
2574
2632
  case "tbl":
2633
+ if (isInlineTbl(child)) text += "";
2575
2634
  break;
2576
- // 테이블은 walkSection에서 처리
2577
2635
  // 하이퍼링크
2578
2636
  case "hyperlink": {
2579
2637
  const url = child.getAttribute("url") || child.getAttribute("href") || "";
@@ -2668,7 +2726,7 @@ function extractParagraphInfo(para2, styleMap, ctx) {
2668
2726
  for (const r of closed) {
2669
2727
  if (applied.some(([s, e]) => r.start < e && r.end > s)) continue;
2670
2728
  const anchor = text.slice(r.start, r.end);
2671
- if (!anchor.trim() || /[\n\x1F\[\]]/.test(anchor)) continue;
2729
+ if (!anchor.trim() || /[\n\x1F\x1E\[\]]/.test(anchor)) continue;
2672
2730
  text = text.slice(0, r.start) + `[${anchor}](${r.url})` + text.slice(r.end);
2673
2731
  applied.push([r.start, r.end]);
2674
2732
  }
@@ -2676,10 +2734,18 @@ function extractParagraphInfo(para2, styleMap, ctx) {
2676
2734
  }
2677
2735
  const leaderIdx = text.indexOf("");
2678
2736
  if (leaderIdx >= 0) text = text.substring(0, leaderIdx);
2679
- let cleanText = text.replace(/[ \t]+/g, " ").trim();
2680
- if (/^그림입니다\.?\s*원본\s*그림의\s*(이름|크기)/.test(cleanText)) cleanText = "";
2681
- cleanText = cleanText.replace(/그림입니다\.?\s*원본\s*그림의\s*(이름|크기)[^\n]*(\n[^\n]*원본\s*그림의\s*(이름|크기)[^\n]*)*/g, "").trim();
2682
- cleanText = cleanText.replace(/^(?:모서리가 둥근 |둥근 )?(?:사각형|직사각형|정사각형|원|타원|삼각형|선|직선|곡선|화살표|오각형|육각형|팔각형|별|십자|구름|마름모|도넛|평행사변형|사다리꼴|개체|그리기\s?개체|묶음\s?개체|글상자|표|그림|OLE\s?개체)\s?입니다\.?$/gm, "").trim();
2737
+ const cleanParaText = (raw) => {
2738
+ let t = raw.replace(/[ \t]+/g, " ").trim();
2739
+ if (/^그림입니다\.?\s*원본\s*그림의\s*(이름|크기)/.test(t)) t = "";
2740
+ t = t.replace(/그림입니다\.?\s*원본\s*그림의\s*(이름|크기)[^\n]*(\n[^\n]*원본\s*그림의\s*(이름|크기)[^\n]*)*/g, "").trim();
2741
+ return t.replace(/^(?:모서리가 둥근 |둥근 )?(?:사각형|직사각형|정사각형|원|타원|삼각형|선|직선|곡선|화살표|오각형|육각형|팔각형|별|십자|구름|마름모|도넛|평행사변형|사다리꼴|개체|그리기\s?개체|묶음\s?개체|글상자|표|그림|OLE\s?개체)\s?입니다\.?$/gm, "").trim();
2742
+ };
2743
+ let segments;
2744
+ if (text.includes("")) {
2745
+ segments = text.split("").map(cleanParaText);
2746
+ text = text.replace(/\x1E/g, "");
2747
+ }
2748
+ const cleanText = cleanParaText(text);
2683
2749
  let style;
2684
2750
  if (styleMap && charPrId) {
2685
2751
  const charProp = styleMap.charProperties.get(charPrId);
@@ -2692,7 +2758,7 @@ function extractParagraphInfo(para2, styleMap, ctx) {
2692
2758
  if (!style.fontSize && !style.bold && !style.italic) style = void 0;
2693
2759
  }
2694
2760
  }
2695
- return { text: cleanText, href, footnote, style };
2761
+ return { text: cleanText, href, footnote, style, segments };
2696
2762
  }
2697
2763
  var KORDOC_CHAR_CODE = "4";
2698
2764
  var KORDOC_PARA_QUOTE = "6";
@@ -2743,21 +2809,22 @@ function extractRunSpans(para2, ctx, mode, requireMixed) {
2743
2809
  else {
2744
2810
  if (cp?.bold) span.bold = true;
2745
2811
  if (cp?.italic) span.italic = true;
2812
+ if (cp?.strike) span.strike = true;
2746
2813
  }
2747
- if (span.bold || span.italic || span.code) styled = true;
2814
+ if (span.bold || span.italic || span.strike || span.code) styled = true;
2748
2815
  spans.push(span);
2749
2816
  }
2750
2817
  if (!styled || spans.length === 0) return null;
2751
2818
  const merged = [];
2752
2819
  for (const s of spans) {
2753
2820
  const last = merged[merged.length - 1];
2754
- if (last && !!last.bold === !!s.bold && !!last.italic === !!s.italic && !!last.code === !!s.code) {
2821
+ if (last && !!last.bold === !!s.bold && !!last.italic === !!s.italic && !!last.strike === !!s.strike && !!last.code === !!s.code) {
2755
2822
  last.text += s.text;
2756
2823
  } else {
2757
2824
  merged.push(s);
2758
2825
  }
2759
2826
  }
2760
- if (requireMixed && !merged.some((s) => !(s.bold || s.italic || s.code))) return null;
2827
+ if (requireMixed && !merged.some((s) => s.strike) && !merged.some((s) => !(s.bold || s.italic || s.code))) return null;
2761
2828
  return merged;
2762
2829
  }
2763
2830
  function gongmunDepthFromIndent(para2, left, ctx) {
@@ -5149,7 +5216,10 @@ function parseParagraph(records, start, end, ctx) {
5149
5216
  if (headingLevel > 0) block.level = headingLevel;
5150
5217
  if (ctx.docInfo && charShapeIds.length > 0) {
5151
5218
  const style = resolveCharStyle(charShapeIds, ctx.docInfo);
5152
- if (style) block.style = style;
5219
+ if (style) {
5220
+ block.style = style;
5221
+ if (style.strike) block.text = headMarker ? `${headMarker} ~~${trimmed}~~` : `~~${trimmed}~~`;
5222
+ }
5153
5223
  }
5154
5224
  if (footnotes.length > 0) block.footnoteText = footnotes.join("; ");
5155
5225
  blocks.push(block);
@@ -5540,7 +5610,13 @@ function resolveCharStyle(charShapeIds, docInfo) {
5540
5610
  if (cs.fontSize > 0) style.fontSize = cs.fontSize / 10;
5541
5611
  if (cs.attrFlags & 1) style.italic = true;
5542
5612
  if (cs.attrFlags & 2) style.bold = true;
5543
- return style.fontSize || style.bold || style.italic ? style : void 0;
5613
+ if (hasRealStrike(cs.attrFlags)) style.strike = true;
5614
+ return style.fontSize || style.bold || style.italic || style.strike ? style : void 0;
5615
+ }
5616
+ function hasRealStrike(attrFlags) {
5617
+ const strikeBits = attrFlags >> 18 & 7;
5618
+ const shapeId = attrFlags >> 26 & 15;
5619
+ return strikeBits !== 0 && shapeId <= 12;
5544
5620
  }
5545
5621
 
5546
5622
  // src/hwp3/parser.ts
@@ -23635,6 +23711,18 @@ function htmlCellInnerToLines(inner) {
23635
23711
  const lines = work.split(/<br\s*\/?>/gi).map((s) => s.trim()).filter((s) => s.length > 0);
23636
23712
  return { lines, hadNonText, imgSrcs };
23637
23713
  }
23714
+ function splitCellByTopLevelTables(html) {
23715
+ const tables = extractTopLevelTables(html);
23716
+ const texts = [];
23717
+ let rest = html;
23718
+ for (const t of tables) {
23719
+ const i = rest.indexOf(t);
23720
+ texts.push(rest.slice(0, i));
23721
+ rest = rest.slice(i + t.length);
23722
+ }
23723
+ texts.push(rest);
23724
+ return { texts, tables };
23725
+ }
23638
23726
  function extractTopLevelTables(html) {
23639
23727
  const result = [];
23640
23728
  let depth = 0;
@@ -24313,17 +24401,18 @@ function generateHtmlTableXml(rawHtml, theme, totalWidth = 44e3, style = null, r
24313
24401
  const rowSpan = Math.min(cell2.rowSpan, rowCnt - cell2.r);
24314
24402
  const lines = cellLines[i];
24315
24403
  let nestedH = 0;
24316
- const nestedXmls = [];
24317
- for (const nested of extractTopLevelTables(cell2.inner)) {
24404
+ const { texts: segTexts, tables: nestedTables } = splitCellByTopLevelTables(cell2.inner);
24405
+ const nestedXmls = nestedTables.map((nested) => {
24318
24406
  const sw = spanW(cell2);
24319
24407
  const nestedW = Math.max(Math.min(Math.max(sw - 1020, 4e3), sw - 282), 500);
24320
24408
  const nestedXml = generateHtmlTableXml(nested, theme, nestedW, style ? { ...style, totalWidth: nestedW } : null);
24321
24409
  if (nestedXml) {
24322
- nestedXmls.push(nestedXml);
24323
24410
  const szH = nestedXml.match(/<hp:sz [^>]*height="(\d+)"/)?.[1];
24324
24411
  nestedH += (szH ? Number(szH) : (nested.match(/<tr[\s>]/gi) ?? []).length * cellH) + 300;
24325
24412
  }
24326
- }
24413
+ return nestedXml;
24414
+ });
24415
+ const segLines = segTexts.map((t) => htmlCellInnerToLines(t).lines);
24327
24416
  const usable = Math.max(spanW(cell2) - CELL_PAD, 1e3);
24328
24417
  let wrapLines = 0;
24329
24418
  for (const line of lines) wrapLines += Math.max(1, Math.ceil(measureTextWidth(unescapeHtml(line).trim(), measureH, 100) / usable));
@@ -24332,9 +24421,9 @@ function generateHtmlTableXml(rawHtml, theme, totalWidth = 44e3, style = null, r
24332
24421
  const cellHeight = Math.max(prof?.cellH.get(`${cell2.r},${cell2.c}`) ?? 0, contentH);
24333
24422
  const perRow = Math.ceil(cellHeight / rowSpan);
24334
24423
  for (let r = cell2.r; r < cell2.r + rowSpan; r++) tableRowHeights[r] = Math.max(tableRowHeights[r], perRow);
24335
- return { cell: cell2, rowSpan, lines, nestedXmls, imgSrcs: cellParsed[i].imgSrcs };
24424
+ return { cell: cell2, rowSpan, lines, nestedXmls, segLines, imgSrcs: cellParsed[i].imgSrcs };
24336
24425
  });
24337
- const tcXmls = meta.map(({ cell: cell2, rowSpan, lines, nestedXmls, imgSrcs }) => {
24426
+ const tcXmls = meta.map(({ cell: cell2, rowSpan, nestedXmls, segLines, imgSrcs }) => {
24338
24427
  const k = `${cell2.r},${cell2.c}`;
24339
24428
  const isHeader = cell2.isHeader;
24340
24429
  const baseCharPr = style ? style.charPr : CHAR_NORMAL;
@@ -24345,7 +24434,7 @@ function generateHtmlTableXml(rawHtml, theme, totalWidth = 44e3, style = null, r
24345
24434
  const paraPrId = style ? centered ? reg ? style.tblCenterParaPr ?? style.centerParaPr : style.centerParaPr : reg ? style.tblLeftParaPr ?? PARA_NORMAL : PARA_NORMAL : PARA_NORMAL;
24346
24435
  const picUrls = images ? [...imgSrcs] : [];
24347
24436
  const paras = [];
24348
- for (const line of lines) {
24437
+ const pushTextLine = (line) => {
24349
24438
  let text = unescapeHtml(line);
24350
24439
  if (images) {
24351
24440
  const { text: rest, urls } = splitImageRefs(text);
@@ -24357,7 +24446,14 @@ function generateHtmlTableXml(rawHtml, theme, totalWidth = 44e3, style = null, r
24357
24446
  if (!images || text.trim() || picUrls.length === 0) {
24358
24447
  paras.push(`<hp:p paraPrIDRef="${paraPrId}" styleIDRef="0"><hp:run charPrIDRef="${charPrId}"><hp:t>${escapeXml(text)}</hp:t></hp:run></hp:p>`);
24359
24448
  }
24360
- }
24449
+ };
24450
+ segLines.forEach((seg, si) => {
24451
+ for (const line of seg) pushTextLine(line);
24452
+ const nestedXml = nestedXmls[si];
24453
+ if (nestedXml) {
24454
+ paras.push(`<hp:p paraPrIDRef="0" styleIDRef="0"><hp:run charPrIDRef="0">${nestedXml}</hp:run></hp:p>`);
24455
+ }
24456
+ });
24361
24457
  if (images && picUrls.length > 0) {
24362
24458
  const pics = picUrls.map((u) => {
24363
24459
  const part = images.take(u);
@@ -24367,9 +24463,6 @@ function generateHtmlTableXml(rawHtml, theme, totalWidth = 44e3, style = null, r
24367
24463
  paras.push(`<hp:p paraPrIDRef="${paraPrId}" styleIDRef="0"><hp:run charPrIDRef="${charPrId}">${pics.join("")}</hp:run></hp:p>`);
24368
24464
  }
24369
24465
  }
24370
- for (const nestedXml of nestedXmls) {
24371
- paras.push(`<hp:p paraPrIDRef="0" styleIDRef="0"><hp:run charPrIDRef="0">${nestedXml}</hp:run></hp:p>`);
24372
- }
24373
24466
  if (paras.length === 0) {
24374
24467
  paras.push(`<hp:p paraPrIDRef="${paraPrId}" styleIDRef="0"><hp:run charPrIDRef="${charPrId}"><hp:t></hp:t></hp:run></hp:p>`);
24375
24468
  }
@@ -30463,7 +30556,7 @@ async function parse(input, options) {
30463
30556
  }
30464
30557
  async function parseImage(buffer, options) {
30465
30558
  try {
30466
- const { parseImageDocument } = await import("./image-ocr-DALT7PFQ.js");
30559
+ const { parseImageDocument } = await import("./image-ocr-PLTV4NOK.js");
30467
30560
  const { blocks, warnings } = await parseImageDocument(buffer, options);
30468
30561
  return {
30469
30562
  success: true,
@@ -30521,7 +30614,7 @@ async function parseHwp(buffer, options) {
30521
30614
  async function parsePdf(buffer, options) {
30522
30615
  let parsePdfDocument;
30523
30616
  try {
30524
- const mod = await import("./parser-33TK5SMH.js");
30617
+ const mod = await import("./parser-Y2X4JMAK.js");
30525
30618
  parsePdfDocument = mod.parsePdfDocument;
30526
30619
  } catch {
30527
30620
  return {