kordoc 4.2.2 → 4.2.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +14 -2
- package/dist/{-NFHVGTVM.js → -GJYDKVQD.js} +9 -9
- package/dist/{chunk-WMU2KJHF.cjs → chunk-5AXGFJQO.cjs} +4 -4
- package/dist/{chunk-WMU2KJHF.cjs.map → chunk-5AXGFJQO.cjs.map} +1 -1
- package/dist/{chunk-EFF2JPPA.cjs → chunk-AA6UPPZF.cjs} +9 -9
- package/dist/{chunk-EFF2JPPA.cjs.map → chunk-AA6UPPZF.cjs.map} +1 -1
- package/dist/{chunk-LUJN3VFB.js → chunk-CGFWS3BQ.js} +3 -3
- package/dist/{chunk-KKZATKQW.cjs → chunk-CWP56BRC.cjs} +2 -2
- package/dist/{chunk-KKZATKQW.cjs.map → chunk-CWP56BRC.cjs.map} +1 -1
- package/dist/{chunk-DSL6EVFJ.js → chunk-DE6FORIP.js} +2 -2
- package/dist/{chunk-T4AWY45Q.js → chunk-ELRVXOMP.js} +2 -2
- package/dist/{chunk-DLFZAU4X.cjs → chunk-FF5QNEKZ.cjs} +3 -3
- package/dist/{chunk-DLFZAU4X.cjs.map → chunk-FF5QNEKZ.cjs.map} +1 -1
- package/dist/{chunk-ZF7QL4IY.js → chunk-GJO3ORFW.js} +1 -1
- package/dist/chunk-GJO3ORFW.js.map +1 -0
- package/dist/{chunk-IXHBRMDB.js → chunk-K5LGSH5T.js} +2 -2
- package/dist/{chunk-MYD3QLBL.js → chunk-LMNSIPWZ.js} +2 -2
- package/dist/{chunk-X77U36TV.js → chunk-LPWP4FWU.js} +2 -2
- package/dist/{chunk-L5B7PQOP.js → chunk-NIPPOAT6.js} +2 -2
- package/dist/{chunk-MOOVF6CT.js → chunk-NOMNOG74.js} +2 -2
- package/dist/{chunk-GUYIXVV7.js → chunk-QNVZAKUQ.js} +2 -2
- package/dist/chunk-QNVZAKUQ.js.map +1 -0
- package/dist/{chunk-2L7MIKOM.js → chunk-RAS3HM4X.js} +2 -2
- package/dist/{chunk-OLDIDBJ2.js → chunk-U6XYOV32.js} +3 -3
- package/dist/{chunk-GHMVNMUN.js → chunk-WTJIRGPN.js} +3 -3
- package/dist/{chunk-TXYQJXOV.js → chunk-WX75LYAS.js} +89 -30
- package/dist/chunk-WX75LYAS.js.map +1 -0
- package/dist/{chunk-FHYLLKIP.js → chunk-X634MXVJ.js} +2 -2
- package/dist/chunks-JOGD2KUK.js +10 -0
- package/dist/cli.js +18 -18
- package/dist/{image-ocr-S4REURLR.js → image-ocr-O3YQ3KEN.js} +5 -5
- package/dist/{image-ocr-TGHAMIQL.js → image-ocr-PLTV4NOK.js} +4 -4
- package/dist/{image-ocr-3AFBDDJI.cjs → image-ocr-TCJLNMTG.cjs} +9 -9
- package/dist/{image-ocr-3AFBDDJI.cjs.map → image-ocr-TCJLNMTG.cjs.map} +1 -1
- package/dist/index.cjs +506 -447
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +7 -2
- package/dist/index.d.ts +7 -2
- package/dist/index.js +84 -25
- package/dist/index.js.map +1 -1
- package/dist/mcp.js +15 -15
- package/dist/{parser-C3APVZGT.js → parser-UX5T6KTZ.js} +6 -6
- package/dist/{parser-MRM6WICO.cjs → parser-VQIJPCI5.cjs} +24 -24
- package/dist/{parser-MRM6WICO.cjs.map → parser-VQIJPCI5.cjs.map} +1 -1
- package/dist/{parser-KMC4YOZH.js → parser-Y2X4JMAK.js} +5 -5
- package/dist/{pdf-ocr-UURMV6FD.js → pdf-ocr-DOPZP6QS.js} +5 -5
- package/dist/pdf-ocr-I3Y6YF5K.js +12 -0
- package/dist/pdf-ocr-XMDJ2IF5.cjs +13 -0
- package/dist/{pdf-ocr-5BRMIJAT.cjs.map → pdf-ocr-XMDJ2IF5.cjs.map} +1 -1
- package/dist/{profile-io-RLQ62KS5.js → profile-io-EGVJAZDG.js} +3 -3
- package/dist/{rasterize-LBABF2WI.js → rasterize-TH26U5E2.js} +2 -2
- package/dist/render-C4R3OJ45.js +10 -0
- package/dist/seal-JF2J4S52.js +10 -0
- package/dist/{setup-Q2PRE7UA.js → setup-GUQT5ELJ.js} +23 -3
- package/dist/setup-GUQT5ELJ.js.map +1 -0
- package/dist/{watch-XEUEGOXW.js → watch-4SP7V6SV.js} +9 -9
- package/package.json +1 -1
- package/dist/chunk-GUYIXVV7.js.map +0 -1
- package/dist/chunk-TXYQJXOV.js.map +0 -1
- package/dist/chunk-ZF7QL4IY.js.map +0 -1
- package/dist/chunks-N4CWHTJI.js +0 -10
- package/dist/pdf-ocr-5BRMIJAT.cjs +0 -13
- package/dist/pdf-ocr-UDTEQEWA.js +0 -12
- package/dist/render-CYXQEHS7.js +0 -10
- package/dist/seal-SV6PFFYU.js +0 -10
- package/dist/setup-Q2PRE7UA.js.map +0 -1
- /package/dist/{-NFHVGTVM.js.map → -GJYDKVQD.js.map} +0 -0
- /package/dist/{chunk-LUJN3VFB.js.map → chunk-CGFWS3BQ.js.map} +0 -0
- /package/dist/{chunk-DSL6EVFJ.js.map → chunk-DE6FORIP.js.map} +0 -0
- /package/dist/{chunk-T4AWY45Q.js.map → chunk-ELRVXOMP.js.map} +0 -0
- /package/dist/{chunk-IXHBRMDB.js.map → chunk-K5LGSH5T.js.map} +0 -0
- /package/dist/{chunk-MYD3QLBL.js.map → chunk-LMNSIPWZ.js.map} +0 -0
- /package/dist/{chunk-X77U36TV.js.map → chunk-LPWP4FWU.js.map} +0 -0
- /package/dist/{chunk-L5B7PQOP.js.map → chunk-NIPPOAT6.js.map} +0 -0
- /package/dist/{chunk-MOOVF6CT.js.map → chunk-NOMNOG74.js.map} +0 -0
- /package/dist/{chunk-2L7MIKOM.js.map → chunk-RAS3HM4X.js.map} +0 -0
- /package/dist/{chunk-OLDIDBJ2.js.map → chunk-U6XYOV32.js.map} +0 -0
- /package/dist/{chunk-GHMVNMUN.js.map → chunk-WTJIRGPN.js.map} +0 -0
- /package/dist/{chunk-FHYLLKIP.js.map → chunk-X634MXVJ.js.map} +0 -0
- /package/dist/{chunks-N4CWHTJI.js.map → chunks-JOGD2KUK.js.map} +0 -0
- /package/dist/{image-ocr-S4REURLR.js.map → image-ocr-O3YQ3KEN.js.map} +0 -0
- /package/dist/{image-ocr-TGHAMIQL.js.map → image-ocr-PLTV4NOK.js.map} +0 -0
- /package/dist/{parser-C3APVZGT.js.map → parser-UX5T6KTZ.js.map} +0 -0
- /package/dist/{parser-KMC4YOZH.js.map → parser-Y2X4JMAK.js.map} +0 -0
- /package/dist/{pdf-ocr-UDTEQEWA.js.map → pdf-ocr-DOPZP6QS.js.map} +0 -0
- /package/dist/{pdf-ocr-UURMV6FD.js.map → pdf-ocr-I3Y6YF5K.js.map} +0 -0
- /package/dist/{profile-io-RLQ62KS5.js.map → profile-io-EGVJAZDG.js.map} +0 -0
- /package/dist/{rasterize-LBABF2WI.js.map → rasterize-TH26U5E2.js.map} +0 -0
- /package/dist/{render-CYXQEHS7.js.map → render-C4R3OJ45.js.map} +0 -0
- /package/dist/{seal-SV6PFFYU.js.map → seal-JF2J4S52.js.map} +0 -0
- /package/dist/{watch-XEUEGOXW.js.map → watch-4SP7V6SV.js.map} +0 -0
package/dist/index.d.cts
CHANGED
|
@@ -99,7 +99,8 @@ interface IRCell {
|
|
|
99
99
|
rowSpan: number;
|
|
100
100
|
/**
|
|
101
101
|
* 셀 내부 블록 콘텐츠 — v3.0.
|
|
102
|
-
* 중첩 표·이미지·다중 문단을 구조
|
|
102
|
+
* 중첩 표·이미지·다중 문단을 구조 그대로, 문서(원문) 순서대로 보존한다.
|
|
103
|
+
* 표와 텍스트가 한 줄에 번갈아 놓인 셀도 배치 순서를 따른다 (v4.2.3, #49).
|
|
103
104
|
* blocks가 있으면 text는 blocks의 평탄화 텍스트(하위 호환용)다.
|
|
104
105
|
*/
|
|
105
106
|
blocks?: IRBlock[];
|
|
@@ -218,7 +219,11 @@ interface ParseSuccess extends ParseResultBase {
|
|
|
218
219
|
success: true;
|
|
219
220
|
/** 추출된 마크다운 텍스트 */
|
|
220
221
|
markdown: string;
|
|
221
|
-
/**
|
|
222
|
+
/**
|
|
223
|
+
* 중간 표현 블록 (구조화된 데이터 접근용).
|
|
224
|
+
* 블록은 문서(원문) 읽기 순서를 따른다 — 한 문단 안에 글자취급(treatAsChar)
|
|
225
|
+
* 표와 텍스트가 섞여 있어도 배치 순서대로 방출된다 (v4.2.3, #50).
|
|
226
|
+
*/
|
|
222
227
|
blocks: IRBlock[];
|
|
223
228
|
/** 문서 메타데이터 */
|
|
224
229
|
metadata?: DocumentMetadata;
|
package/dist/index.d.ts
CHANGED
|
@@ -99,7 +99,8 @@ interface IRCell {
|
|
|
99
99
|
rowSpan: number;
|
|
100
100
|
/**
|
|
101
101
|
* 셀 내부 블록 콘텐츠 — v3.0.
|
|
102
|
-
* 중첩 표·이미지·다중 문단을 구조
|
|
102
|
+
* 중첩 표·이미지·다중 문단을 구조 그대로, 문서(원문) 순서대로 보존한다.
|
|
103
|
+
* 표와 텍스트가 한 줄에 번갈아 놓인 셀도 배치 순서를 따른다 (v4.2.3, #49).
|
|
103
104
|
* blocks가 있으면 text는 blocks의 평탄화 텍스트(하위 호환용)다.
|
|
104
105
|
*/
|
|
105
106
|
blocks?: IRBlock[];
|
|
@@ -218,7 +219,11 @@ interface ParseSuccess extends ParseResultBase {
|
|
|
218
219
|
success: true;
|
|
219
220
|
/** 추출된 마크다운 텍스트 */
|
|
220
221
|
markdown: string;
|
|
221
|
-
/**
|
|
222
|
+
/**
|
|
223
|
+
* 중간 표현 블록 (구조화된 데이터 접근용).
|
|
224
|
+
* 블록은 문서(원문) 읽기 순서를 따른다 — 한 문단 안에 글자취급(treatAsChar)
|
|
225
|
+
* 표와 텍스트가 섞여 있어도 배치 순서대로 방출된다 (v4.2.3, #50).
|
|
226
|
+
*/
|
|
222
227
|
blocks: IRBlock[];
|
|
223
228
|
/** 문서 메타데이터 */
|
|
224
229
|
metadata?: DocumentMetadata;
|
package/dist/index.js
CHANGED
|
@@ -8,7 +8,7 @@ import {
|
|
|
8
8
|
flattenLayoutTables,
|
|
9
9
|
inlineImagesIntoMarkdown,
|
|
10
10
|
mapPuaText
|
|
11
|
-
} from "./chunk-
|
|
11
|
+
} from "./chunk-K5LGSH5T.js";
|
|
12
12
|
import {
|
|
13
13
|
parsePageRange
|
|
14
14
|
} from "./chunk-GE43BE46.js";
|
|
@@ -27,7 +27,7 @@ import {
|
|
|
27
27
|
sanitizeHref,
|
|
28
28
|
stripDtd,
|
|
29
29
|
toArrayBuffer
|
|
30
|
-
} from "./chunk-
|
|
30
|
+
} from "./chunk-QNVZAKUQ.js";
|
|
31
31
|
|
|
32
32
|
// src/index.ts
|
|
33
33
|
import { readFile } from "fs/promises";
|
|
@@ -2105,7 +2105,7 @@ function walkSection(node, blocks, tableCtx, tableStack, ctx, depth = 0) {
|
|
|
2105
2105
|
}
|
|
2106
2106
|
break;
|
|
2107
2107
|
case "p": {
|
|
2108
|
-
const { text: rawText, href, footnote, style } = extractParagraphInfo(el, ctx.styleMap, ctx);
|
|
2108
|
+
const { text: rawText, href, footnote, style, segments } = extractParagraphInfo(el, ctx.styleMap, ctx);
|
|
2109
2109
|
let text = rawText;
|
|
2110
2110
|
let headingLevel;
|
|
2111
2111
|
const ph = resolveParaHeading(el, ctx);
|
|
@@ -2113,6 +2113,33 @@ function walkSection(node, blocks, tableCtx, tableStack, ctx, depth = 0) {
|
|
|
2113
2113
|
if (ph?.prefix) text = ph.prefix + " " + text;
|
|
2114
2114
|
headingLevel = ph?.headingLevel;
|
|
2115
2115
|
}
|
|
2116
|
+
if (segments) {
|
|
2117
|
+
const cell2 = tableCtx?.cell ?? null;
|
|
2118
|
+
if (text && cell2) cell2.text += (cell2.text ? "\n" : "") + (footnote ? `${text} (\uC8FC: ${footnote})` : text);
|
|
2119
|
+
const segs = [...segments];
|
|
2120
|
+
if (ph?.prefix) {
|
|
2121
|
+
const fi = segs.findIndex((s) => s);
|
|
2122
|
+
if (fi >= 0) segs[fi] = ph.prefix + " " + segs[fi];
|
|
2123
|
+
}
|
|
2124
|
+
let segIdx = 0;
|
|
2125
|
+
let first = true;
|
|
2126
|
+
const flush = () => {
|
|
2127
|
+
const s = segs[segIdx++];
|
|
2128
|
+
if (!s) return;
|
|
2129
|
+
const block = { type: "paragraph", text: s, pageNumber: ctx.sectionNum };
|
|
2130
|
+
if (first && !cell2) {
|
|
2131
|
+
first = false;
|
|
2132
|
+
if (style) block.style = style;
|
|
2133
|
+
if (href) block.href = href;
|
|
2134
|
+
if (footnote) block.footnoteText = footnote;
|
|
2135
|
+
}
|
|
2136
|
+
if (cell2) (cell2.blocks ??= []).push(block);
|
|
2137
|
+
else blocks.push(block);
|
|
2138
|
+
};
|
|
2139
|
+
tableCtx = walkParagraphChildren(el, blocks, tableCtx, tableStack, ctx, depth + 1, flush);
|
|
2140
|
+
while (segIdx < segs.length) flush();
|
|
2141
|
+
break;
|
|
2142
|
+
}
|
|
2116
2143
|
if (text) {
|
|
2117
2144
|
if (tableCtx?.cell) {
|
|
2118
2145
|
const cell2 = tableCtx.cell;
|
|
@@ -2331,7 +2358,7 @@ function findTopLevelTbls(el, out, depth = 0) {
|
|
|
2331
2358
|
findTopLevelTbls(ch, out, depth + 1);
|
|
2332
2359
|
}
|
|
2333
2360
|
}
|
|
2334
|
-
function walkParagraphChildren(node, blocks, tableCtx, tableStack, ctx, depth = 0) {
|
|
2361
|
+
function walkParagraphChildren(node, blocks, tableCtx, tableStack, ctx, depth = 0, onTbl) {
|
|
2335
2362
|
if (depth > MAX_XML_DEPTH) return tableCtx;
|
|
2336
2363
|
const children = node.childNodes;
|
|
2337
2364
|
if (!children) return tableCtx;
|
|
@@ -2345,6 +2372,7 @@ function walkParagraphChildren(node, blocks, tableCtx, tableStack, ctx, depth =
|
|
|
2345
2372
|
const tag = el.tagName || el.localName || "";
|
|
2346
2373
|
const localTag = tag.replace(/^[^:]+:/, "");
|
|
2347
2374
|
if (localTag === "tbl") {
|
|
2375
|
+
if (onTbl && isInlineTbl(el)) onTbl();
|
|
2348
2376
|
if (!tableCtx) {
|
|
2349
2377
|
const chan = kordocTableChannel(el, ctx);
|
|
2350
2378
|
if (chan) {
|
|
@@ -2391,6 +2419,9 @@ function walkParagraphChildren(node, blocks, tableCtx, tableStack, ctx, depth =
|
|
|
2391
2419
|
walkChildren(node, depth);
|
|
2392
2420
|
return tableCtx;
|
|
2393
2421
|
}
|
|
2422
|
+
function isInlineTbl(tbl2) {
|
|
2423
|
+
return findChildByLocalName(tbl2, "pos")?.getAttribute("treatAsChar") === "1";
|
|
2424
|
+
}
|
|
2394
2425
|
function findDescendant(node, targetTag, depth = 0) {
|
|
2395
2426
|
if (depth > 5) return null;
|
|
2396
2427
|
const children = node.childNodes;
|
|
@@ -2564,7 +2595,7 @@ function extractParagraphInfo(para2, styleMap, ctx) {
|
|
|
2564
2595
|
ctx.warnings?.push({ page: ctx.sectionNum, message: "\uBCC0\uACBD\uCD94\uC801 \uC0AD\uC81C \uD14D\uC2A4\uD2B8 \uCD9C\uB825 \uC81C\uC678", code: "HIDDEN_TEXT_FILTERED" });
|
|
2565
2596
|
}
|
|
2566
2597
|
} else {
|
|
2567
|
-
text += t;
|
|
2598
|
+
text += t.replace(/\x1E/g, "");
|
|
2568
2599
|
}
|
|
2569
2600
|
continue;
|
|
2570
2601
|
}
|
|
@@ -2595,9 +2626,12 @@ function extractParagraphInfo(para2, styleMap, ctx) {
|
|
|
2595
2626
|
case "hwSpace":
|
|
2596
2627
|
text += " ";
|
|
2597
2628
|
break;
|
|
2629
|
+
// 테이블 자체는 walkSection에서 처리 — 글자취급(inline) 표만 경계 마커를 남겨
|
|
2630
|
+
// 표 앞뒤 텍스트를 문서 순서대로 분할 방출할 수 있게 한다 (#49/#50).
|
|
2631
|
+
// float·페이지 앵커 표는 텍스트 흐름 불참(reflow 모델 정합) — 종전대로 텍스트 뒤 방출
|
|
2598
2632
|
case "tbl":
|
|
2633
|
+
if (isInlineTbl(child)) text += "";
|
|
2599
2634
|
break;
|
|
2600
|
-
// 테이블은 walkSection에서 처리
|
|
2601
2635
|
// 하이퍼링크
|
|
2602
2636
|
case "hyperlink": {
|
|
2603
2637
|
const url = child.getAttribute("url") || child.getAttribute("href") || "";
|
|
@@ -2692,7 +2726,7 @@ function extractParagraphInfo(para2, styleMap, ctx) {
|
|
|
2692
2726
|
for (const r of closed) {
|
|
2693
2727
|
if (applied.some(([s, e]) => r.start < e && r.end > s)) continue;
|
|
2694
2728
|
const anchor = text.slice(r.start, r.end);
|
|
2695
|
-
if (!anchor.trim() || /[\n\x1F\[\]]/.test(anchor)) continue;
|
|
2729
|
+
if (!anchor.trim() || /[\n\x1F\x1E\[\]]/.test(anchor)) continue;
|
|
2696
2730
|
text = text.slice(0, r.start) + `[${anchor}](${r.url})` + text.slice(r.end);
|
|
2697
2731
|
applied.push([r.start, r.end]);
|
|
2698
2732
|
}
|
|
@@ -2700,10 +2734,18 @@ function extractParagraphInfo(para2, styleMap, ctx) {
|
|
|
2700
2734
|
}
|
|
2701
2735
|
const leaderIdx = text.indexOf("");
|
|
2702
2736
|
if (leaderIdx >= 0) text = text.substring(0, leaderIdx);
|
|
2703
|
-
|
|
2704
|
-
|
|
2705
|
-
|
|
2706
|
-
|
|
2737
|
+
const cleanParaText = (raw) => {
|
|
2738
|
+
let t = raw.replace(/[ \t]+/g, " ").trim();
|
|
2739
|
+
if (/^그림입니다\.?\s*원본\s*그림의\s*(이름|크기)/.test(t)) t = "";
|
|
2740
|
+
t = t.replace(/그림입니다\.?\s*원본\s*그림의\s*(이름|크기)[^\n]*(\n[^\n]*원본\s*그림의\s*(이름|크기)[^\n]*)*/g, "").trim();
|
|
2741
|
+
return t.replace(/^(?:모서리가 둥근 |둥근 )?(?:사각형|직사각형|정사각형|원|타원|삼각형|선|직선|곡선|화살표|오각형|육각형|팔각형|별|십자|구름|마름모|도넛|평행사변형|사다리꼴|개체|그리기\s?개체|묶음\s?개체|글상자|표|그림|OLE\s?개체)\s?입니다\.?$/gm, "").trim();
|
|
2742
|
+
};
|
|
2743
|
+
let segments;
|
|
2744
|
+
if (text.includes("")) {
|
|
2745
|
+
segments = text.split("").map(cleanParaText);
|
|
2746
|
+
text = text.replace(/\x1E/g, "");
|
|
2747
|
+
}
|
|
2748
|
+
const cleanText = cleanParaText(text);
|
|
2707
2749
|
let style;
|
|
2708
2750
|
if (styleMap && charPrId) {
|
|
2709
2751
|
const charProp = styleMap.charProperties.get(charPrId);
|
|
@@ -2716,7 +2758,7 @@ function extractParagraphInfo(para2, styleMap, ctx) {
|
|
|
2716
2758
|
if (!style.fontSize && !style.bold && !style.italic) style = void 0;
|
|
2717
2759
|
}
|
|
2718
2760
|
}
|
|
2719
|
-
return { text: cleanText, href, footnote, style };
|
|
2761
|
+
return { text: cleanText, href, footnote, style, segments };
|
|
2720
2762
|
}
|
|
2721
2763
|
var KORDOC_CHAR_CODE = "4";
|
|
2722
2764
|
var KORDOC_PARA_QUOTE = "6";
|
|
@@ -23669,6 +23711,18 @@ function htmlCellInnerToLines(inner) {
|
|
|
23669
23711
|
const lines = work.split(/<br\s*\/?>/gi).map((s) => s.trim()).filter((s) => s.length > 0);
|
|
23670
23712
|
return { lines, hadNonText, imgSrcs };
|
|
23671
23713
|
}
|
|
23714
|
+
function splitCellByTopLevelTables(html) {
|
|
23715
|
+
const tables = extractTopLevelTables(html);
|
|
23716
|
+
const texts = [];
|
|
23717
|
+
let rest = html;
|
|
23718
|
+
for (const t of tables) {
|
|
23719
|
+
const i = rest.indexOf(t);
|
|
23720
|
+
texts.push(rest.slice(0, i));
|
|
23721
|
+
rest = rest.slice(i + t.length);
|
|
23722
|
+
}
|
|
23723
|
+
texts.push(rest);
|
|
23724
|
+
return { texts, tables };
|
|
23725
|
+
}
|
|
23672
23726
|
function extractTopLevelTables(html) {
|
|
23673
23727
|
const result = [];
|
|
23674
23728
|
let depth = 0;
|
|
@@ -24347,17 +24401,18 @@ function generateHtmlTableXml(rawHtml, theme, totalWidth = 44e3, style = null, r
|
|
|
24347
24401
|
const rowSpan = Math.min(cell2.rowSpan, rowCnt - cell2.r);
|
|
24348
24402
|
const lines = cellLines[i];
|
|
24349
24403
|
let nestedH = 0;
|
|
24350
|
-
const
|
|
24351
|
-
|
|
24404
|
+
const { texts: segTexts, tables: nestedTables } = splitCellByTopLevelTables(cell2.inner);
|
|
24405
|
+
const nestedXmls = nestedTables.map((nested) => {
|
|
24352
24406
|
const sw = spanW(cell2);
|
|
24353
24407
|
const nestedW = Math.max(Math.min(Math.max(sw - 1020, 4e3), sw - 282), 500);
|
|
24354
24408
|
const nestedXml = generateHtmlTableXml(nested, theme, nestedW, style ? { ...style, totalWidth: nestedW } : null);
|
|
24355
24409
|
if (nestedXml) {
|
|
24356
|
-
nestedXmls.push(nestedXml);
|
|
24357
24410
|
const szH = nestedXml.match(/<hp:sz [^>]*height="(\d+)"/)?.[1];
|
|
24358
24411
|
nestedH += (szH ? Number(szH) : (nested.match(/<tr[\s>]/gi) ?? []).length * cellH) + 300;
|
|
24359
24412
|
}
|
|
24360
|
-
|
|
24413
|
+
return nestedXml;
|
|
24414
|
+
});
|
|
24415
|
+
const segLines = segTexts.map((t) => htmlCellInnerToLines(t).lines);
|
|
24361
24416
|
const usable = Math.max(spanW(cell2) - CELL_PAD, 1e3);
|
|
24362
24417
|
let wrapLines = 0;
|
|
24363
24418
|
for (const line of lines) wrapLines += Math.max(1, Math.ceil(measureTextWidth(unescapeHtml(line).trim(), measureH, 100) / usable));
|
|
@@ -24366,9 +24421,9 @@ function generateHtmlTableXml(rawHtml, theme, totalWidth = 44e3, style = null, r
|
|
|
24366
24421
|
const cellHeight = Math.max(prof?.cellH.get(`${cell2.r},${cell2.c}`) ?? 0, contentH);
|
|
24367
24422
|
const perRow = Math.ceil(cellHeight / rowSpan);
|
|
24368
24423
|
for (let r = cell2.r; r < cell2.r + rowSpan; r++) tableRowHeights[r] = Math.max(tableRowHeights[r], perRow);
|
|
24369
|
-
return { cell: cell2, rowSpan, lines, nestedXmls, imgSrcs: cellParsed[i].imgSrcs };
|
|
24424
|
+
return { cell: cell2, rowSpan, lines, nestedXmls, segLines, imgSrcs: cellParsed[i].imgSrcs };
|
|
24370
24425
|
});
|
|
24371
|
-
const tcXmls = meta.map(({ cell: cell2, rowSpan,
|
|
24426
|
+
const tcXmls = meta.map(({ cell: cell2, rowSpan, nestedXmls, segLines, imgSrcs }) => {
|
|
24372
24427
|
const k = `${cell2.r},${cell2.c}`;
|
|
24373
24428
|
const isHeader = cell2.isHeader;
|
|
24374
24429
|
const baseCharPr = style ? style.charPr : CHAR_NORMAL;
|
|
@@ -24379,7 +24434,7 @@ function generateHtmlTableXml(rawHtml, theme, totalWidth = 44e3, style = null, r
|
|
|
24379
24434
|
const paraPrId = style ? centered ? reg ? style.tblCenterParaPr ?? style.centerParaPr : style.centerParaPr : reg ? style.tblLeftParaPr ?? PARA_NORMAL : PARA_NORMAL : PARA_NORMAL;
|
|
24380
24435
|
const picUrls = images ? [...imgSrcs] : [];
|
|
24381
24436
|
const paras = [];
|
|
24382
|
-
|
|
24437
|
+
const pushTextLine = (line) => {
|
|
24383
24438
|
let text = unescapeHtml(line);
|
|
24384
24439
|
if (images) {
|
|
24385
24440
|
const { text: rest, urls } = splitImageRefs(text);
|
|
@@ -24391,7 +24446,14 @@ function generateHtmlTableXml(rawHtml, theme, totalWidth = 44e3, style = null, r
|
|
|
24391
24446
|
if (!images || text.trim() || picUrls.length === 0) {
|
|
24392
24447
|
paras.push(`<hp:p paraPrIDRef="${paraPrId}" styleIDRef="0"><hp:run charPrIDRef="${charPrId}"><hp:t>${escapeXml(text)}</hp:t></hp:run></hp:p>`);
|
|
24393
24448
|
}
|
|
24394
|
-
}
|
|
24449
|
+
};
|
|
24450
|
+
segLines.forEach((seg, si) => {
|
|
24451
|
+
for (const line of seg) pushTextLine(line);
|
|
24452
|
+
const nestedXml = nestedXmls[si];
|
|
24453
|
+
if (nestedXml) {
|
|
24454
|
+
paras.push(`<hp:p paraPrIDRef="0" styleIDRef="0"><hp:run charPrIDRef="0">${nestedXml}</hp:run></hp:p>`);
|
|
24455
|
+
}
|
|
24456
|
+
});
|
|
24395
24457
|
if (images && picUrls.length > 0) {
|
|
24396
24458
|
const pics = picUrls.map((u) => {
|
|
24397
24459
|
const part = images.take(u);
|
|
@@ -24401,9 +24463,6 @@ function generateHtmlTableXml(rawHtml, theme, totalWidth = 44e3, style = null, r
|
|
|
24401
24463
|
paras.push(`<hp:p paraPrIDRef="${paraPrId}" styleIDRef="0"><hp:run charPrIDRef="${charPrId}">${pics.join("")}</hp:run></hp:p>`);
|
|
24402
24464
|
}
|
|
24403
24465
|
}
|
|
24404
|
-
for (const nestedXml of nestedXmls) {
|
|
24405
|
-
paras.push(`<hp:p paraPrIDRef="0" styleIDRef="0"><hp:run charPrIDRef="0">${nestedXml}</hp:run></hp:p>`);
|
|
24406
|
-
}
|
|
24407
24466
|
if (paras.length === 0) {
|
|
24408
24467
|
paras.push(`<hp:p paraPrIDRef="${paraPrId}" styleIDRef="0"><hp:run charPrIDRef="${charPrId}"><hp:t></hp:t></hp:run></hp:p>`);
|
|
24409
24468
|
}
|
|
@@ -30497,7 +30556,7 @@ async function parse(input, options) {
|
|
|
30497
30556
|
}
|
|
30498
30557
|
async function parseImage(buffer, options) {
|
|
30499
30558
|
try {
|
|
30500
|
-
const { parseImageDocument } = await import("./image-ocr-
|
|
30559
|
+
const { parseImageDocument } = await import("./image-ocr-PLTV4NOK.js");
|
|
30501
30560
|
const { blocks, warnings } = await parseImageDocument(buffer, options);
|
|
30502
30561
|
return {
|
|
30503
30562
|
success: true,
|
|
@@ -30555,7 +30614,7 @@ async function parseHwp(buffer, options) {
|
|
|
30555
30614
|
async function parsePdf(buffer, options) {
|
|
30556
30615
|
let parsePdfDocument;
|
|
30557
30616
|
try {
|
|
30558
|
-
const mod = await import("./parser-
|
|
30617
|
+
const mod = await import("./parser-Y2X4JMAK.js");
|
|
30559
30618
|
parsePdfDocument = mod.parsePdfDocument;
|
|
30560
30619
|
} catch {
|
|
30561
30620
|
return {
|