@sofereditor/import-docx 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +661 -0
- package/README.md +20 -0
- package/dist/index.cjs +665 -0
- package/dist/index.cjs.map +1 -0
- package/dist/index.d.cts +17 -0
- package/dist/index.d.ts +17 -0
- package/dist/index.js +627 -0
- package/dist/index.js.map +1 -0
- package/package.json +55 -0
package/dist/index.cjs
ADDED
|
@@ -0,0 +1,665 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
var __create = Object.create;
|
|
3
|
+
var __defProp = Object.defineProperty;
|
|
4
|
+
var __getOwnPropDesc = Object.getOwnPropertyDescriptor;
|
|
5
|
+
var __getOwnPropNames = Object.getOwnPropertyNames;
|
|
6
|
+
var __getProtoOf = Object.getPrototypeOf;
|
|
7
|
+
var __hasOwnProp = Object.prototype.hasOwnProperty;
|
|
8
|
+
var __export = (target, all) => {
|
|
9
|
+
for (var name in all)
|
|
10
|
+
__defProp(target, name, { get: all[name], enumerable: true });
|
|
11
|
+
};
|
|
12
|
+
var __copyProps = (to, from, except, desc) => {
|
|
13
|
+
if (from && typeof from === "object" || typeof from === "function") {
|
|
14
|
+
for (let key of __getOwnPropNames(from))
|
|
15
|
+
if (!__hasOwnProp.call(to, key) && key !== except)
|
|
16
|
+
__defProp(to, key, { get: () => from[key], enumerable: !(desc = __getOwnPropDesc(from, key)) || desc.enumerable });
|
|
17
|
+
}
|
|
18
|
+
return to;
|
|
19
|
+
};
|
|
20
|
+
var __toESM = (mod, isNodeMode, target) => (target = mod != null ? __create(__getProtoOf(mod)) : {}, __copyProps(
|
|
21
|
+
// If the importer is in node compatibility mode or this is not an ESM
|
|
22
|
+
// file that has been converted to a CommonJS file using a Babel-
|
|
23
|
+
// compatible transform (i.e. "__esModule" has not been set), then set
|
|
24
|
+
// "default" to the CommonJS "module.exports" for node compatibility.
|
|
25
|
+
isNodeMode || !mod || !mod.__esModule ? __defProp(target, "default", { value: mod, enumerable: true }) : target,
|
|
26
|
+
mod
|
|
27
|
+
));
|
|
28
|
+
var __toCommonJS = (mod) => __copyProps(__defProp({}, "__esModule", { value: true }), mod);
|
|
29
|
+
|
|
30
|
+
// src/index.ts
|
|
31
|
+
var index_exports = {};
|
|
32
|
+
__export(index_exports, {
|
|
33
|
+
docxBlobToDocument: () => docxBlobToDocument,
|
|
34
|
+
docxBlobToEditorDocument: () => docxBlobToEditorDocument
|
|
35
|
+
});
|
|
36
|
+
module.exports = __toCommonJS(index_exports);
|
|
37
|
+
|
|
38
|
+
// src/docx.ts
|
|
39
|
+
var import_core2 = require("@sofereditor/core");
|
|
40
|
+
|
|
41
|
+
// src/parse-xml.ts
|
|
42
|
+
var import_jszip = __toESM(require("jszip"), 1);
|
|
43
|
+
var import_fast_xml_parser = require("fast-xml-parser");
|
|
44
|
+
var parser = new import_fast_xml_parser.XMLParser({
|
|
45
|
+
ignoreAttributes: false,
|
|
46
|
+
attributeNamePrefix: "@_",
|
|
47
|
+
preserveOrder: true,
|
|
48
|
+
trimValues: false,
|
|
49
|
+
// keep whitespace inside <w:t>
|
|
50
|
+
parseAttributeValue: false,
|
|
51
|
+
parseTagValue: false
|
|
52
|
+
});
|
|
53
|
+
async function readDocx(input) {
|
|
54
|
+
const data = input instanceof Uint8Array ? input : input instanceof ArrayBuffer ? new Uint8Array(input) : new Uint8Array(await input.arrayBuffer());
|
|
55
|
+
const zip = await import_jszip.default.loadAsync(data);
|
|
56
|
+
const documentXml = await readXml(zip, "word/document.xml");
|
|
57
|
+
if (!documentXml) throw new Error("Invalid .docx: missing word/document.xml");
|
|
58
|
+
const numberingXml = await readXml(zip, "word/numbering.xml");
|
|
59
|
+
const relsXml = await readXml(zip, "word/_rels/document.xml.rels");
|
|
60
|
+
const relationships = parseRelationships(relsXml);
|
|
61
|
+
const mediaBytes = /* @__PURE__ */ new Map();
|
|
62
|
+
const mediaFolder = zip.folder("word/media");
|
|
63
|
+
if (mediaFolder) {
|
|
64
|
+
const promises = [];
|
|
65
|
+
mediaFolder.forEach((relativePath, file) => {
|
|
66
|
+
if (file.dir) return;
|
|
67
|
+
promises.push(
|
|
68
|
+
file.async("uint8array").then((bytes) => {
|
|
69
|
+
mediaBytes.set(`media/${relativePath}`, bytes);
|
|
70
|
+
})
|
|
71
|
+
);
|
|
72
|
+
});
|
|
73
|
+
await Promise.all(promises);
|
|
74
|
+
}
|
|
75
|
+
return { documentXml, numberingXml, relationships, mediaBytes };
|
|
76
|
+
}
|
|
77
|
+
async function readXml(zip, path) {
|
|
78
|
+
const entry = zip.file(path);
|
|
79
|
+
if (!entry) return void 0;
|
|
80
|
+
const text = await entry.async("string");
|
|
81
|
+
return parser.parse(text);
|
|
82
|
+
}
|
|
83
|
+
function parseRelationships(nodes) {
|
|
84
|
+
const out = /* @__PURE__ */ new Map();
|
|
85
|
+
if (!nodes) return out;
|
|
86
|
+
for (const top of nodes) {
|
|
87
|
+
const children = top["Relationships"];
|
|
88
|
+
if (!Array.isArray(children)) continue;
|
|
89
|
+
for (const child of children) {
|
|
90
|
+
if (!("Relationship" in child)) continue;
|
|
91
|
+
const attrs = child[":@"] ?? {};
|
|
92
|
+
const id = attrs["@_Id"];
|
|
93
|
+
const target = attrs["@_Target"];
|
|
94
|
+
if (typeof id === "string" && typeof target === "string") {
|
|
95
|
+
out.set(id, target);
|
|
96
|
+
}
|
|
97
|
+
}
|
|
98
|
+
}
|
|
99
|
+
return out;
|
|
100
|
+
}
|
|
101
|
+
function tagOf(node) {
|
|
102
|
+
for (const key of Object.keys(node)) {
|
|
103
|
+
if (key === ":@") continue;
|
|
104
|
+
return key;
|
|
105
|
+
}
|
|
106
|
+
return null;
|
|
107
|
+
}
|
|
108
|
+
function childrenOf(node) {
|
|
109
|
+
const tag = tagOf(node);
|
|
110
|
+
if (!tag) return [];
|
|
111
|
+
const v = node[tag];
|
|
112
|
+
return Array.isArray(v) ? v : [];
|
|
113
|
+
}
|
|
114
|
+
function attrsOf(node) {
|
|
115
|
+
return node[":@"] ?? {} ?? {};
|
|
116
|
+
}
|
|
117
|
+
function attr(node, key) {
|
|
118
|
+
const v = attrsOf(node)[`@_${key}`];
|
|
119
|
+
return typeof v === "string" ? v : void 0;
|
|
120
|
+
}
|
|
121
|
+
function findChild(node, tag) {
|
|
122
|
+
for (const child of childrenOf(node)) {
|
|
123
|
+
if (tagOf(child) === tag) return child;
|
|
124
|
+
}
|
|
125
|
+
return void 0;
|
|
126
|
+
}
|
|
127
|
+
function findChildren(node, tag) {
|
|
128
|
+
const out = [];
|
|
129
|
+
for (const child of childrenOf(node)) {
|
|
130
|
+
if (tagOf(child) === tag) out.push(child);
|
|
131
|
+
}
|
|
132
|
+
return out;
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
// src/units.ts
|
|
136
|
+
function halfPointsToCssPt(halfPoints) {
|
|
137
|
+
if (!Number.isFinite(halfPoints) || halfPoints <= 0) return void 0;
|
|
138
|
+
const pt = halfPoints / 2;
|
|
139
|
+
return `${stripTrailingZeros(pt)}pt`;
|
|
140
|
+
}
|
|
141
|
+
function docxHexToCssColor(hex) {
|
|
142
|
+
if (!hex) return void 0;
|
|
143
|
+
const m = /^[0-9a-f]{6}$/i.exec(hex.trim());
|
|
144
|
+
if (!m) return void 0;
|
|
145
|
+
return `#${hex.toLowerCase()}`;
|
|
146
|
+
}
|
|
147
|
+
function parseAlign(val) {
|
|
148
|
+
if (!val) return void 0;
|
|
149
|
+
switch (val.toLowerCase()) {
|
|
150
|
+
case "left":
|
|
151
|
+
case "start":
|
|
152
|
+
return "left";
|
|
153
|
+
case "center":
|
|
154
|
+
return "center";
|
|
155
|
+
case "right":
|
|
156
|
+
case "end":
|
|
157
|
+
return "right";
|
|
158
|
+
case "both":
|
|
159
|
+
case "distribute":
|
|
160
|
+
case "justify":
|
|
161
|
+
return "justify";
|
|
162
|
+
default:
|
|
163
|
+
return void 0;
|
|
164
|
+
}
|
|
165
|
+
}
|
|
166
|
+
function parseIntAttr(val, fallback = 0) {
|
|
167
|
+
if (typeof val === "number" && Number.isFinite(val)) return Math.trunc(val);
|
|
168
|
+
if (typeof val !== "string") return fallback;
|
|
169
|
+
const n = Number.parseInt(val, 10);
|
|
170
|
+
return Number.isFinite(n) ? n : fallback;
|
|
171
|
+
}
|
|
172
|
+
function stripTrailingZeros(n) {
|
|
173
|
+
return Number.isInteger(n) ? n.toString() : n.toFixed(2).replace(/\.?0+$/, "");
|
|
174
|
+
}
|
|
175
|
+
|
|
176
|
+
// src/images.ts
|
|
177
|
+
var EMU_PER_PX = 9525;
|
|
178
|
+
var MIME_BY_EXT = {
|
|
179
|
+
png: "image/png",
|
|
180
|
+
jpg: "image/jpeg",
|
|
181
|
+
jpeg: "image/jpeg",
|
|
182
|
+
gif: "image/gif",
|
|
183
|
+
bmp: "image/bmp",
|
|
184
|
+
svg: "image/svg+xml"
|
|
185
|
+
};
|
|
186
|
+
function extractImageFromDrawing(drawing, ctx) {
|
|
187
|
+
let embedId;
|
|
188
|
+
let cx;
|
|
189
|
+
let cy;
|
|
190
|
+
walk(drawing, (n) => {
|
|
191
|
+
const tag = tagOf(n);
|
|
192
|
+
if (!tag) return;
|
|
193
|
+
if (tag === "a:blip") {
|
|
194
|
+
const id = attr(n, "r:embed") ?? attr(n, "r:link");
|
|
195
|
+
if (id) embedId = id;
|
|
196
|
+
} else if (tag === "wp:extent") {
|
|
197
|
+
cx = parseIntAttr(attr(n, "cx"));
|
|
198
|
+
cy = parseIntAttr(attr(n, "cy"));
|
|
199
|
+
}
|
|
200
|
+
});
|
|
201
|
+
if (!embedId) return null;
|
|
202
|
+
const target = ctx.relationships.get(embedId);
|
|
203
|
+
if (!target) return null;
|
|
204
|
+
const bytes = ctx.mediaBytes.get(target);
|
|
205
|
+
if (!bytes) return null;
|
|
206
|
+
const ext = extensionOf(target);
|
|
207
|
+
const mime = MIME_BY_EXT[ext] ?? "image/png";
|
|
208
|
+
const src = `data:${mime};base64,${bytesToBase64(bytes)}`;
|
|
209
|
+
const widthPx = cx != null ? Math.max(1, Math.round(cx / EMU_PER_PX)) : 100;
|
|
210
|
+
const heightPx = cy != null ? Math.max(1, Math.round(cy / EMU_PER_PX)) : 100;
|
|
211
|
+
return { type: "image", src, width: widthPx, height: heightPx };
|
|
212
|
+
}
|
|
213
|
+
function walk(node, visit) {
|
|
214
|
+
visit(node);
|
|
215
|
+
for (const child of childrenOf(node)) {
|
|
216
|
+
if ("#text" in child) continue;
|
|
217
|
+
walk(child, visit);
|
|
218
|
+
}
|
|
219
|
+
}
|
|
220
|
+
function extensionOf(path) {
|
|
221
|
+
const m = /\.([a-z0-9]+)$/i.exec(path);
|
|
222
|
+
return m ? m[1].toLowerCase() : "";
|
|
223
|
+
}
|
|
224
|
+
function bytesToBase64(bytes) {
|
|
225
|
+
if (typeof btoa === "function") {
|
|
226
|
+
let bin = "";
|
|
227
|
+
const CHUNK = 32768;
|
|
228
|
+
for (let i = 0; i < bytes.length; i += CHUNK) {
|
|
229
|
+
bin += String.fromCharCode.apply(null, Array.from(bytes.subarray(i, i + CHUNK)));
|
|
230
|
+
}
|
|
231
|
+
return btoa(bin);
|
|
232
|
+
}
|
|
233
|
+
return Buffer.from(bytes).toString("base64");
|
|
234
|
+
}
|
|
235
|
+
|
|
236
|
+
// src/runs.ts
|
|
237
|
+
function paragraphChildrenToDelta(children, ctx) {
|
|
238
|
+
const ops = [];
|
|
239
|
+
for (const child of children) {
|
|
240
|
+
const tag = tagOf(child);
|
|
241
|
+
if (!tag) continue;
|
|
242
|
+
if (tag === "w:pPr") continue;
|
|
243
|
+
if (tag === "w:r") {
|
|
244
|
+
collectRun(child, ctx, void 0, ops);
|
|
245
|
+
} else if (tag === "w:hyperlink") {
|
|
246
|
+
const rId = attr(child, "r:id");
|
|
247
|
+
const href = rId ? ctx.relationships.get(rId) : void 0;
|
|
248
|
+
const linkMark = href ? { link: { href } } : void 0;
|
|
249
|
+
for (const sub of childrenOf(child)) {
|
|
250
|
+
if (tagOf(sub) === "w:r") collectRun(sub, ctx, linkMark, ops);
|
|
251
|
+
}
|
|
252
|
+
}
|
|
253
|
+
}
|
|
254
|
+
return mergeAdjacent(ops);
|
|
255
|
+
}
|
|
256
|
+
function collectRun(run, ctx, wrappingMarks, out) {
|
|
257
|
+
const marks = mergeMarks(wrappingMarks, runMarks(findChild(run, "w:rPr")));
|
|
258
|
+
for (const child of childrenOf(run)) {
|
|
259
|
+
const tag = tagOf(child);
|
|
260
|
+
if (!tag) continue;
|
|
261
|
+
if (tag === "w:t") {
|
|
262
|
+
const text = textContent(child);
|
|
263
|
+
if (text.length > 0) push(out, text, marks);
|
|
264
|
+
} else if (tag === "w:tab") {
|
|
265
|
+
push(out, " ", marks);
|
|
266
|
+
} else if (tag === "w:br") {
|
|
267
|
+
push(out, "\n", marks);
|
|
268
|
+
} else if (tag === "w:drawing") {
|
|
269
|
+
const embed = extractImageFromDrawing(child, ctx);
|
|
270
|
+
if (embed) out.push({ insert: embed });
|
|
271
|
+
}
|
|
272
|
+
}
|
|
273
|
+
}
|
|
274
|
+
function runMarks(rPr) {
|
|
275
|
+
const m = {};
|
|
276
|
+
if (!rPr) return m;
|
|
277
|
+
for (const child of childrenOf(rPr)) {
|
|
278
|
+
const tag = tagOf(child);
|
|
279
|
+
if (!tag) continue;
|
|
280
|
+
switch (tag) {
|
|
281
|
+
case "w:b":
|
|
282
|
+
if (!isFalse(attr(child, "w:val"))) m.bold = true;
|
|
283
|
+
break;
|
|
284
|
+
case "w:i":
|
|
285
|
+
if (!isFalse(attr(child, "w:val"))) m.italic = true;
|
|
286
|
+
break;
|
|
287
|
+
case "w:u": {
|
|
288
|
+
const val = attr(child, "w:val");
|
|
289
|
+
if (val && val.toLowerCase() !== "none") m.underline = true;
|
|
290
|
+
break;
|
|
291
|
+
}
|
|
292
|
+
case "w:strike":
|
|
293
|
+
if (!isFalse(attr(child, "w:val"))) m.strike = true;
|
|
294
|
+
break;
|
|
295
|
+
case "w:color": {
|
|
296
|
+
const hex = attr(child, "w:val");
|
|
297
|
+
const css = docxHexToCssColor(hex);
|
|
298
|
+
if (css) m.color = css;
|
|
299
|
+
break;
|
|
300
|
+
}
|
|
301
|
+
// w:rFonts intentionally ignored: school standard is Arial for all
|
|
302
|
+
// exams. Any font declared in the .docx is dropped on import so the
|
|
303
|
+
// imported doc inherits Arial from `.ed-root` / PDF base stylesheet.
|
|
304
|
+
case "w:sz": {
|
|
305
|
+
const halfPts = Number.parseFloat(attr(child, "w:val") ?? "");
|
|
306
|
+
const css = halfPointsToCssPt(halfPts);
|
|
307
|
+
if (css) m.fontSize = css;
|
|
308
|
+
break;
|
|
309
|
+
}
|
|
310
|
+
}
|
|
311
|
+
}
|
|
312
|
+
return m;
|
|
313
|
+
}
|
|
314
|
+
function mergeMarks(a, b) {
|
|
315
|
+
if (!a) return b;
|
|
316
|
+
return { ...a, ...b };
|
|
317
|
+
}
|
|
318
|
+
function textContent(node) {
|
|
319
|
+
let out = "";
|
|
320
|
+
for (const child of childrenOf(node)) {
|
|
321
|
+
if ("#text" in child) {
|
|
322
|
+
const t = child["#text"];
|
|
323
|
+
if (typeof t === "string") out += t;
|
|
324
|
+
}
|
|
325
|
+
}
|
|
326
|
+
return out;
|
|
327
|
+
}
|
|
328
|
+
function isFalse(val) {
|
|
329
|
+
if (val == null) return false;
|
|
330
|
+
const v = val.toLowerCase();
|
|
331
|
+
return v === "0" || v === "false" || v === "none";
|
|
332
|
+
}
|
|
333
|
+
function push(out, text, marks) {
|
|
334
|
+
const op = Object.keys(marks).length === 0 ? { insert: text } : { insert: text, attributes: { ...marks } };
|
|
335
|
+
out.push(op);
|
|
336
|
+
}
|
|
337
|
+
function mergeAdjacent(ops) {
|
|
338
|
+
const out = [];
|
|
339
|
+
for (const op of ops) {
|
|
340
|
+
const prev = out[out.length - 1];
|
|
341
|
+
if (prev && typeof prev.insert === "string" && typeof op.insert === "string" && sameMarks(prev.attributes, op.attributes)) {
|
|
342
|
+
prev.insert = prev.insert + op.insert;
|
|
343
|
+
continue;
|
|
344
|
+
}
|
|
345
|
+
out.push(op);
|
|
346
|
+
}
|
|
347
|
+
return out;
|
|
348
|
+
}
|
|
349
|
+
function sameMarks(a, b) {
|
|
350
|
+
const ak = a ? Object.keys(a).sort() : [];
|
|
351
|
+
const bk = b ? Object.keys(b).sort() : [];
|
|
352
|
+
if (ak.length !== bk.length) return false;
|
|
353
|
+
for (let i = 0; i < ak.length; i++) {
|
|
354
|
+
if (ak[i] !== bk[i]) return false;
|
|
355
|
+
const av = a[ak[i]];
|
|
356
|
+
const bv = b[bk[i]];
|
|
357
|
+
if (typeof av === "object" && av !== null && typeof bv === "object" && bv !== null) {
|
|
358
|
+
if (JSON.stringify(av) !== JSON.stringify(bv)) return false;
|
|
359
|
+
} else if (av !== bv) {
|
|
360
|
+
return false;
|
|
361
|
+
}
|
|
362
|
+
}
|
|
363
|
+
return true;
|
|
364
|
+
}
|
|
365
|
+
|
|
366
|
+
// src/paragraphs.ts
|
|
367
|
+
function paragraphToBlock(p, ctx, numbering) {
|
|
368
|
+
const pPr = findChild(p, "w:pPr");
|
|
369
|
+
const numPr = pPr ? findChild(pPr, "w:numPr") : void 0;
|
|
370
|
+
const styleId = pPr ? attr(findChild(pPr, "w:pStyle") ?? {}, "w:val") : void 0;
|
|
371
|
+
const delta = paragraphChildrenToDelta(childrenOf(p), ctx);
|
|
372
|
+
const text = collectText(delta);
|
|
373
|
+
const align = parseAlign(pPr ? attr(findChild(pPr, "w:jc") ?? {}, "w:val") : void 0);
|
|
374
|
+
const isRtl = pPr ? !!findChild(pPr, "w:bidi") : false;
|
|
375
|
+
if (numPr) {
|
|
376
|
+
const resolved = numbering.resolve(numPr);
|
|
377
|
+
const attrs2 = {
|
|
378
|
+
listKind: resolved?.listKind ?? "bullet",
|
|
379
|
+
listLevel: resolved?.listLevel ?? 0
|
|
380
|
+
};
|
|
381
|
+
if (align) attrs2.align = align;
|
|
382
|
+
if (isRtl) attrs2.dir = "rtl";
|
|
383
|
+
return { type: "listItem", text, delta, attrs: attrs2 };
|
|
384
|
+
}
|
|
385
|
+
const headingLevel = styleId ? headingFromStyleId(styleId) : null;
|
|
386
|
+
if (headingLevel != null) {
|
|
387
|
+
const attrs2 = { level: headingLevel };
|
|
388
|
+
if (align) attrs2.align = align;
|
|
389
|
+
if (isRtl) attrs2.dir = "rtl";
|
|
390
|
+
return { type: "heading", text, delta, attrs: attrs2 };
|
|
391
|
+
}
|
|
392
|
+
if (pPr && hasLeftBorder(pPr)) {
|
|
393
|
+
const attrs2 = {};
|
|
394
|
+
if (align) attrs2.align = align;
|
|
395
|
+
if (isRtl) attrs2.dir = "rtl";
|
|
396
|
+
return { type: "blockquote", text, delta, attrs: attrs2 };
|
|
397
|
+
}
|
|
398
|
+
if (pPr && hasCodeBlockShading(pPr)) {
|
|
399
|
+
const attrs2 = {};
|
|
400
|
+
if (align) attrs2.align = align;
|
|
401
|
+
if (isRtl) attrs2.dir = "rtl";
|
|
402
|
+
return { type: "codeBlock", text, delta, attrs: attrs2 };
|
|
403
|
+
}
|
|
404
|
+
const attrs = {};
|
|
405
|
+
if (align) attrs.align = align;
|
|
406
|
+
if (isRtl) attrs.dir = "rtl";
|
|
407
|
+
return { type: "paragraph", text, delta, attrs };
|
|
408
|
+
}
|
|
409
|
+
function collectText(delta) {
|
|
410
|
+
let out = "";
|
|
411
|
+
for (const op of delta) {
|
|
412
|
+
if (typeof op.insert === "string") out += op.insert;
|
|
413
|
+
}
|
|
414
|
+
return out;
|
|
415
|
+
}
|
|
416
|
+
function headingFromStyleId(id) {
|
|
417
|
+
const m = /^Heading\s*([1-6])$/i.exec(id);
|
|
418
|
+
if (!m) return null;
|
|
419
|
+
const n = Number.parseInt(m[1], 10);
|
|
420
|
+
return n ?? null;
|
|
421
|
+
}
|
|
422
|
+
function hasLeftBorder(pPr) {
|
|
423
|
+
const pBdr = findChild(pPr, "w:pBdr");
|
|
424
|
+
if (!pBdr) return false;
|
|
425
|
+
return !!findChild(pBdr, "w:left");
|
|
426
|
+
}
|
|
427
|
+
function hasCodeBlockShading(pPr) {
|
|
428
|
+
const shd = findChild(pPr, "w:shd");
|
|
429
|
+
if (!shd) return false;
|
|
430
|
+
const fill = attr(shd, "w:fill");
|
|
431
|
+
if (!fill) return false;
|
|
432
|
+
return fill.toLowerCase() === "f1f5f9";
|
|
433
|
+
}
|
|
434
|
+
|
|
435
|
+
// src/tables.ts
|
|
436
|
+
function tableToBlock(tbl, ctx) {
|
|
437
|
+
const tblGrid = findChild(tbl, "w:tblGrid");
|
|
438
|
+
const gridCols = tblGrid ? findChildren(tblGrid, "w:gridCol") : [];
|
|
439
|
+
const trs = findChildren(tbl, "w:tr");
|
|
440
|
+
let cols = gridCols.length;
|
|
441
|
+
for (const tr of trs) {
|
|
442
|
+
const tcs = findChildren(tr, "w:tc");
|
|
443
|
+
let count = 0;
|
|
444
|
+
for (const tc of tcs) count += gridSpanOf(tc);
|
|
445
|
+
if (count > cols) cols = count;
|
|
446
|
+
}
|
|
447
|
+
if (cols < 1) cols = 1;
|
|
448
|
+
const rows = trs.length;
|
|
449
|
+
const slots = new Array(rows * cols).fill(null);
|
|
450
|
+
const vMergeAnchorByCol = /* @__PURE__ */ new Map();
|
|
451
|
+
for (let r = 0; r < rows; r++) {
|
|
452
|
+
const tr = trs[r];
|
|
453
|
+
let cursor = 0;
|
|
454
|
+
for (const tc of findChildren(tr, "w:tc")) {
|
|
455
|
+
while (cursor < cols && slots[r * cols + cursor] !== null) cursor++;
|
|
456
|
+
if (cursor >= cols) break;
|
|
457
|
+
const tcPr = findChild(tc, "w:tcPr");
|
|
458
|
+
const gridSpan = gridSpanOf(tc);
|
|
459
|
+
const vMergeNode = tcPr ? findChild(tcPr, "w:vMerge") : void 0;
|
|
460
|
+
const vMergeVal = vMergeNode ? attr(vMergeNode, "w:val") ?? "continue" : null;
|
|
461
|
+
if (vMergeNode && vMergeVal !== "restart") {
|
|
462
|
+
const anchor = vMergeAnchorByCol.get(cursor);
|
|
463
|
+
if (anchor) {
|
|
464
|
+
const anchorIdx = anchor.row * cols + anchor.col;
|
|
465
|
+
const a = slots[anchorIdx];
|
|
466
|
+
if (a) {
|
|
467
|
+
a.attrs = { ...a.attrs ?? {}, rowspan: (a.attrs?.rowspan ?? 1) + 1 };
|
|
468
|
+
}
|
|
469
|
+
}
|
|
470
|
+
for (let c = 0; c < gridSpan && cursor + c < cols; c++) {
|
|
471
|
+
slots[r * cols + cursor + c] = coveredCell();
|
|
472
|
+
}
|
|
473
|
+
cursor += gridSpan;
|
|
474
|
+
continue;
|
|
475
|
+
}
|
|
476
|
+
const delta = cellChildrenToDelta(tc, ctx);
|
|
477
|
+
const text = textOfDelta(delta);
|
|
478
|
+
const cellAttrs = {};
|
|
479
|
+
if (gridSpan > 1) cellAttrs.colspan = gridSpan;
|
|
480
|
+
const realCell = { text, delta, attrs: cellAttrs };
|
|
481
|
+
slots[r * cols + cursor] = realCell;
|
|
482
|
+
for (let c = 1; c < gridSpan && cursor + c < cols; c++) {
|
|
483
|
+
slots[r * cols + cursor + c] = coveredCell();
|
|
484
|
+
}
|
|
485
|
+
if (vMergeVal === "restart") {
|
|
486
|
+
vMergeAnchorByCol.set(cursor, { row: r, col: cursor });
|
|
487
|
+
} else {
|
|
488
|
+
vMergeAnchorByCol.delete(cursor);
|
|
489
|
+
}
|
|
490
|
+
cursor += gridSpan;
|
|
491
|
+
}
|
|
492
|
+
for (let c = 0; c < cols; c++) {
|
|
493
|
+
if (slots[r * cols + c] === null) {
|
|
494
|
+
slots[r * cols + c] = emptyCell();
|
|
495
|
+
}
|
|
496
|
+
}
|
|
497
|
+
}
|
|
498
|
+
const cells = slots.map((s) => s ?? emptyCell());
|
|
499
|
+
const attrs = { rows, cols };
|
|
500
|
+
return { type: "table", text: "", delta: [], attrs, cells };
|
|
501
|
+
}
|
|
502
|
+
function gridSpanOf(tc) {
|
|
503
|
+
const tcPr = findChild(tc, "w:tcPr");
|
|
504
|
+
if (!tcPr) return 1;
|
|
505
|
+
const gs = findChild(tcPr, "w:gridSpan");
|
|
506
|
+
if (!gs) return 1;
|
|
507
|
+
const n = parseIntAttr(attr(gs, "w:val"), 1);
|
|
508
|
+
return Math.max(1, n);
|
|
509
|
+
}
|
|
510
|
+
function cellChildrenToDelta(tc, ctx) {
|
|
511
|
+
const out = [];
|
|
512
|
+
let first = true;
|
|
513
|
+
for (const child of childrenOf(tc)) {
|
|
514
|
+
if (tagOf(child) !== "w:p") continue;
|
|
515
|
+
if (!first) out.push({ insert: "\n" });
|
|
516
|
+
first = false;
|
|
517
|
+
out.push(...paragraphChildrenToDelta(childrenOf(child), ctx));
|
|
518
|
+
}
|
|
519
|
+
return out;
|
|
520
|
+
}
|
|
521
|
+
function textOfDelta(delta) {
|
|
522
|
+
let out = "";
|
|
523
|
+
for (const op of delta) {
|
|
524
|
+
if (typeof op.insert === "string") out += op.insert;
|
|
525
|
+
}
|
|
526
|
+
return out;
|
|
527
|
+
}
|
|
528
|
+
function coveredCell() {
|
|
529
|
+
return { text: "", delta: [], attrs: { covered: true } };
|
|
530
|
+
}
|
|
531
|
+
function emptyCell() {
|
|
532
|
+
return { text: "", delta: [], attrs: {} };
|
|
533
|
+
}
|
|
534
|
+
|
|
535
|
+
// src/numbering.ts
|
|
536
|
+
var NumberingResolver = class {
|
|
537
|
+
numIdToAbstract = /* @__PURE__ */ new Map();
|
|
538
|
+
abstractLevelFmt = /* @__PURE__ */ new Map();
|
|
539
|
+
constructor(numberingXml) {
|
|
540
|
+
if (!numberingXml) return;
|
|
541
|
+
for (const top of numberingXml) {
|
|
542
|
+
if (tagOf(top) !== "w:numbering") continue;
|
|
543
|
+
for (const child of childrenOf(top)) {
|
|
544
|
+
const tag = tagOf(child);
|
|
545
|
+
if (tag === "w:abstractNum") {
|
|
546
|
+
const abstractId = attr(child, "w:abstractNumId");
|
|
547
|
+
if (!abstractId) continue;
|
|
548
|
+
const levels = /* @__PURE__ */ new Map();
|
|
549
|
+
for (const lvl of findChildren(child, "w:lvl")) {
|
|
550
|
+
const ilvl = parseIntAttr(attr(lvl, "w:ilvl"), 0);
|
|
551
|
+
const fmt = attr(findChild(lvl, "w:numFmt") ?? {}, "w:val");
|
|
552
|
+
if (fmt) levels.set(ilvl, fmt);
|
|
553
|
+
}
|
|
554
|
+
this.abstractLevelFmt.set(abstractId, levels);
|
|
555
|
+
} else if (tag === "w:num") {
|
|
556
|
+
const numId = attr(child, "w:numId");
|
|
557
|
+
const abstractRef = findChild(child, "w:abstractNumId");
|
|
558
|
+
const abstractId = abstractRef ? attr(abstractRef, "w:val") : void 0;
|
|
559
|
+
if (numId && abstractId) this.numIdToAbstract.set(numId, abstractId);
|
|
560
|
+
}
|
|
561
|
+
}
|
|
562
|
+
}
|
|
563
|
+
}
|
|
564
|
+
/** Returns `{listKind, listLevel}` for a given paragraph's `<w:numPr>` node, or null. */
|
|
565
|
+
resolve(numPr) {
|
|
566
|
+
if (!numPr) return null;
|
|
567
|
+
const ilvlRef = findChild(numPr, "w:ilvl");
|
|
568
|
+
const numIdRef = findChild(numPr, "w:numId");
|
|
569
|
+
if (!numIdRef) return null;
|
|
570
|
+
const numId = attr(numIdRef, "w:val");
|
|
571
|
+
if (!numId) return null;
|
|
572
|
+
const listLevel = clampLevel(parseIntAttr(ilvlRef ? attr(ilvlRef, "w:val") : void 0, 0));
|
|
573
|
+
const abstractId = this.numIdToAbstract.get(numId);
|
|
574
|
+
if (abstractId == null) {
|
|
575
|
+
return { listKind: "bullet", listLevel };
|
|
576
|
+
}
|
|
577
|
+
const levels = this.abstractLevelFmt.get(abstractId);
|
|
578
|
+
const fmt = (levels?.get(listLevel) ?? "bullet").toLowerCase();
|
|
579
|
+
const listKind = fmt === "bullet" || fmt === "none" ? "bullet" : "ordered";
|
|
580
|
+
return { listKind, listLevel };
|
|
581
|
+
}
|
|
582
|
+
};
|
|
583
|
+
function clampLevel(n) {
|
|
584
|
+
if (!Number.isFinite(n)) return 0;
|
|
585
|
+
return Math.max(0, Math.min(5, Math.trunc(n)));
|
|
586
|
+
}
|
|
587
|
+
|
|
588
|
+
// src/page-settings.ts
|
|
589
|
+
var import_core = require("@sofereditor/core");
|
|
590
|
+
var MM_PER_TWIP = 25.4 / 1440;
|
|
591
|
+
function twipsToPx(twips) {
|
|
592
|
+
return (0, import_core.mmToPx)(twips * MM_PER_TWIP);
|
|
593
|
+
}
|
|
594
|
+
function sectPrToPageSettings(sectPr) {
|
|
595
|
+
if (!sectPr) return void 0;
|
|
596
|
+
const pgSz = findChild(sectPr, "w:pgSz");
|
|
597
|
+
const pgMar = findChild(sectPr, "w:pgMar");
|
|
598
|
+
if (!pgSz && !pgMar) return void 0;
|
|
599
|
+
const widthTwips = pgSz ? parseIntAttr(attr(pgSz, "w:w"), 0) : 0;
|
|
600
|
+
const heightTwips = pgSz ? parseIntAttr(attr(pgSz, "w:h"), 0) : 0;
|
|
601
|
+
const topTwips = pgMar ? parseIntAttr(attr(pgMar, "w:top"), 0) : 0;
|
|
602
|
+
const bottomTwips = pgMar ? parseIntAttr(attr(pgMar, "w:bottom"), 0) : 0;
|
|
603
|
+
const leftTwips = pgMar ? parseIntAttr(attr(pgMar, "w:left"), 0) : 0;
|
|
604
|
+
const rightTwips = pgMar ? parseIntAttr(attr(pgMar, "w:right"), 0) : 0;
|
|
605
|
+
if (widthTwips <= 0 && heightTwips <= 0 && topTwips <= 0 && leftTwips <= 0) {
|
|
606
|
+
return void 0;
|
|
607
|
+
}
|
|
608
|
+
const width = widthTwips > 0 ? twipsToPx(widthTwips) : 794;
|
|
609
|
+
const height = heightTwips > 0 ? twipsToPx(heightTwips) : 1123;
|
|
610
|
+
const settings = {
|
|
611
|
+
width,
|
|
612
|
+
height,
|
|
613
|
+
marginTop: twipsToPx(topTwips || 0),
|
|
614
|
+
marginBottom: twipsToPx(bottomTwips || 0),
|
|
615
|
+
marginLeft: twipsToPx(leftTwips || 0),
|
|
616
|
+
marginRight: twipsToPx(rightTwips || 0)
|
|
617
|
+
};
|
|
618
|
+
settings.preset = (0, import_core.detectPreset)(settings.width, settings.height);
|
|
619
|
+
return settings;
|
|
620
|
+
}
|
|
621
|
+
|
|
622
|
+
// src/docx.ts
|
|
623
|
+
async function docxBlobToDocument(input) {
|
|
624
|
+
const file = await readDocx(input);
|
|
625
|
+
const numbering = new NumberingResolver(file.numberingXml);
|
|
626
|
+
const ctx = {
|
|
627
|
+
relationships: file.relationships,
|
|
628
|
+
mediaBytes: file.mediaBytes
|
|
629
|
+
};
|
|
630
|
+
const body = findBody(file.documentXml);
|
|
631
|
+
if (!body) return { blocks: [{ type: "paragraph", text: "", delta: [], attrs: {} }] };
|
|
632
|
+
const blocks = [];
|
|
633
|
+
for (const child of childrenOf(body)) {
|
|
634
|
+
const tag = tagOf(child);
|
|
635
|
+
if (tag === "w:p") {
|
|
636
|
+
blocks.push(paragraphToBlock(child, ctx, numbering));
|
|
637
|
+
} else if (tag === "w:tbl") {
|
|
638
|
+
blocks.push(tableToBlock(child, ctx));
|
|
639
|
+
}
|
|
640
|
+
}
|
|
641
|
+
if (blocks.length === 0) {
|
|
642
|
+
blocks.push({ type: "paragraph", text: "", delta: [], attrs: {} });
|
|
643
|
+
}
|
|
644
|
+
const pageSettings = sectPrToPageSettings(findChild(body, "w:sectPr"));
|
|
645
|
+
return pageSettings ? { blocks, pageSettings } : { blocks };
|
|
646
|
+
}
|
|
647
|
+
async function docxBlobToEditorDocument(input) {
|
|
648
|
+
const serialized = await docxBlobToDocument(input);
|
|
649
|
+
return import_core2.EditorDocument.fromJSON(serialized);
|
|
650
|
+
}
|
|
651
|
+
function findBody(documentXml) {
|
|
652
|
+
for (const top of documentXml) {
|
|
653
|
+
if (tagOf(top) === "w:document") {
|
|
654
|
+
const body = findChild(top, "w:body");
|
|
655
|
+
if (body) return body;
|
|
656
|
+
}
|
|
657
|
+
}
|
|
658
|
+
return void 0;
|
|
659
|
+
}
|
|
660
|
+
// Annotate the CommonJS export names for ESM import in node:
|
|
661
|
+
0 && (module.exports = {
|
|
662
|
+
docxBlobToDocument,
|
|
663
|
+
docxBlobToEditorDocument
|
|
664
|
+
});
|
|
665
|
+
//# sourceMappingURL=index.cjs.map
|