@stll/folio-core 0.54.0 → 0.54.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/ai-edits/apply.d.ts +1 -1
- package/dist/ai-edits/apply.js +43 -31
- package/dist/ai-edits/headless.d.ts +1 -2
- package/dist/ai-edits/headless.js +12 -27
- package/dist/ai-edits/read.d.ts +1 -0
- package/dist/ai-edits/read.js +7 -2
- package/dist/ai-edits/table-mutation-plan.d.ts +2 -1
- package/dist/ai-edits/table-mutation-plan.js +39 -1
- package/dist/ai-suggestions/apply.d.ts +1 -1
- package/dist/ai-suggestions/apply.js +3 -2
- package/dist/compare/compare.js +2 -1
- package/dist/docx/bookmarkIds.d.ts +11 -0
- package/dist/docx/bookmarkIds.js +33 -0
- package/dist/docx/documentParser.d.ts +14 -1
- package/dist/docx/documentParser.js +17 -4
- package/dist/docx/drawingIdNormalization.js +23 -8
- package/dist/docx/ensureParaIds.d.ts +5 -0
- package/dist/docx/ensureParaIds.js +10 -2
- package/dist/docx/listNumberingInstances.js +11 -9
- package/dist/docx/noteIds.d.ts +4 -0
- package/dist/docx/noteIds.js +13 -0
- package/dist/docx/numberingIds.d.ts +14 -0
- package/dist/docx/numberingIds.js +16 -0
- package/dist/docx/numericIdAllocator.d.ts +12 -0
- package/dist/docx/numericIdAllocator.js +33 -0
- package/dist/docx/numericIdNormalization.d.ts +16 -0
- package/dist/docx/numericIdNormalization.js +174 -0
- package/dist/docx/paragraphPropertySource.js +1 -1
- package/dist/docx/paragraphTextBoxEnrichment.js +2 -1
- package/dist/docx/parser.js +17 -5
- package/dist/docx/rasterMime.d.ts +64 -0
- package/dist/docx/rasterMime.js +116 -0
- package/dist/docx/replyToComment.js +13 -6
- package/dist/docx/rezip.js +14 -7
- package/dist/docx/selectiveSave.js +10 -0
- package/dist/docx/serializer/partNamespaces.js +4 -2
- package/dist/docx/serializer/runSerializer.js +5 -4
- package/dist/docx/serializer/trackedChangeAttributes.js +7 -1
- package/dist/docx/server/build.js +14 -3
- package/dist/docx/server/createBilingualDocument.js +10 -9
- package/dist/docx/streamingXmlParser.d.ts +10 -1
- package/dist/docx/streamingXmlParser.js +88 -19
- package/dist/docx/unzip.d.ts +3 -1
- package/dist/docx/unzip.js +36 -16
- package/dist/docx/verticalMergeProjection.d.ts +17 -0
- package/dist/docx/verticalMergeProjection.js +120 -0
- package/dist/layout-bridge/convert/fixedTableColumnWidths.d.ts +6 -0
- package/dist/layout-bridge/convert/fixedTableColumnWidths.js +23 -0
- package/dist/layout-bridge/convert/tableConversion.js +4 -1
- package/dist/layout-painter/index.js +2 -5
- package/dist/paged-layout/editorScrollRoot.d.ts +15 -0
- package/dist/paged-layout/editorScrollRoot.js +72 -0
- package/dist/paged-layout/scrollToPmPosition.js +8 -22
- package/dist/prosemirror/commands/clearParagraphIndent.d.ts +6 -0
- package/dist/prosemirror/commands/clearParagraphIndent.js +25 -0
- package/dist/prosemirror/commands/comments.js +6 -1
- package/dist/prosemirror/commands/propertyChangeScope.js +4 -1
- package/dist/prosemirror/commands/resolveAllTableChanges.js +18 -24
- package/dist/prosemirror/commands/resolveParagraphProperties.js +13 -2
- package/dist/prosemirror/commands/tableCellMergeResolution.d.ts +8 -9
- package/dist/prosemirror/commands/tableCellMergeResolution.js +28 -25
- package/dist/prosemirror/commands/tableMergeFoldDecisions.d.ts +61 -0
- package/dist/prosemirror/commands/tableMergeFoldDecisions.js +79 -0
- package/dist/prosemirror/commentIdAllocator.d.ts +15 -22
- package/dist/prosemirror/commentIdAllocator.js +50 -25
- package/dist/prosemirror/conversion/toProseDoc.d.ts +1 -6
- package/dist/prosemirror/conversion/toProseDoc.js +2 -117
- package/dist/prosemirror/extensions/core/HistoryExtension.js +97 -5
- package/dist/prosemirror/extensions/core/ParagraphExtension.js +3 -1
- package/dist/prosemirror/extensions/features/BaseKeymapExtension.js +22 -28
- package/dist/prosemirror/extensions/nodes/ShapeExtension.js +8 -6
- package/dist/prosemirror/listNumbering.js +14 -14
- package/dist/prosemirror/listRendering.d.ts +4 -2
- package/dist/prosemirror/listRendering.js +3 -4
- package/dist/prosemirror/paragraphPropertyCarry.js +2 -5
- package/dist/prosemirror/plugins/revisionIds.d.ts +23 -19
- package/dist/prosemirror/plugins/revisionIds.js +110 -29
- package/dist/prosemirror/plugins/suggestionMode.d.ts +2 -2
- package/dist/prosemirror/plugins/suggestionMode.js +197 -45
- package/dist/prosemirror/plugins/templateDirectives.d.ts +6 -0
- package/dist/prosemirror/plugins/templateDirectives.js +21 -7
- package/dist/prosemirror/plugins/templatePreviewValues.js +14 -21
- package/dist/prosemirror/storyListNumbering.js +11 -9
- package/dist/prosemirror/textInput.d.ts +6 -1
- package/dist/prosemirror/textInput.js +9 -2
- package/dist/prosemirror/utils/visualLineNavigation.js +6 -8
- package/dist/utils/mergeDocumentContent.d.ts +2 -2
- package/dist/utils/mergeDocumentContent.js +11 -8
- package/package.json +3 -3
|
@@ -15,6 +15,11 @@ type EnsureParaIdsResult = {
|
|
|
15
15
|
deduplicated: number;
|
|
16
16
|
/** True when the input already had full, unique coverage. */
|
|
17
17
|
alreadyComplete: boolean;
|
|
18
|
+
/**
|
|
19
|
+
* Every id this pass wrote (assigned or deduplicated), in scan order. An id
|
|
20
|
+
* absent from this list was already the package's own.
|
|
21
|
+
*/
|
|
22
|
+
mintedParaIds: readonly string[];
|
|
18
23
|
};
|
|
19
24
|
/** Controls mutation of package metadata that has security implications. */
|
|
20
25
|
type EnsureParaIdsOptions = {
|
|
@@ -233,6 +233,7 @@ const mintParaId = (context, partPath, ordinal) => {
|
|
|
233
233
|
*/
|
|
234
234
|
const scanPart = (xml, partPath, spelling, context, seen) => {
|
|
235
235
|
const edits = [];
|
|
236
|
+
const minted = [];
|
|
236
237
|
const w14 = spelling.w14Prefix;
|
|
237
238
|
let assigned = 0;
|
|
238
239
|
let deduplicated = 0;
|
|
@@ -261,6 +262,7 @@ const scanPart = (xml, partPath, spelling, context, seen) => {
|
|
|
261
262
|
text: ` ${w14}:paraId="${id}" ${w14}:textId="${id}"`
|
|
262
263
|
});
|
|
263
264
|
assigned += 1;
|
|
265
|
+
minted.push(id);
|
|
264
266
|
seen.add(id);
|
|
265
267
|
continue;
|
|
266
268
|
}
|
|
@@ -287,10 +289,12 @@ const scanPart = (xml, partPath, spelling, context, seen) => {
|
|
|
287
289
|
});
|
|
288
290
|
if (unassigned) assigned += 1;
|
|
289
291
|
else deduplicated += 1;
|
|
292
|
+
minted.push(id);
|
|
290
293
|
seen.add(id);
|
|
291
294
|
}
|
|
292
295
|
return {
|
|
293
296
|
edits,
|
|
297
|
+
minted,
|
|
294
298
|
assigned,
|
|
295
299
|
deduplicated,
|
|
296
300
|
paragraphs: ordinal
|
|
@@ -404,6 +408,7 @@ const ensureParaIdsInternal = async (docx, options) => {
|
|
|
404
408
|
collectFallbackParaIds(xml, partPath, spelling, seen);
|
|
405
409
|
} else collectExistingParaIds(xml, seen);
|
|
406
410
|
const updates = /* @__PURE__ */ new Map();
|
|
411
|
+
const mintedParaIds = [];
|
|
407
412
|
let assigned = 0;
|
|
408
413
|
let deduplicated = 0;
|
|
409
414
|
for (const partPath of targetParts) {
|
|
@@ -414,13 +419,15 @@ const ensureParaIdsInternal = async (docx, options) => {
|
|
|
414
419
|
if (scan.edits.length === 0) continue;
|
|
415
420
|
assigned += scan.assigned;
|
|
416
421
|
deduplicated += scan.deduplicated;
|
|
422
|
+
for (const id of scan.minted) mintedParaIds.push(id);
|
|
417
423
|
updates.set(partPath, applySplices(xml, [...scan.edits, ...ensureRootNamespaces(xml, partPath, spelling)], partPath));
|
|
418
424
|
}
|
|
419
425
|
if (updates.size === 0) return {
|
|
420
426
|
docx: toUint8Array(docx),
|
|
421
427
|
assigned: 0,
|
|
422
428
|
deduplicated: 0,
|
|
423
|
-
alreadyComplete: true
|
|
429
|
+
alreadyComplete: true,
|
|
430
|
+
mintedParaIds: []
|
|
424
431
|
};
|
|
425
432
|
const zip = await JSZip.loadAsync(docx);
|
|
426
433
|
if (hasDigitalSignatureParts(zip) && options.allowSignedPackageMutation !== true) throw createEnsureParaIdsError("Refusing to normalize a digitally signed package because rewriting OOXML invalidates its signatures. Warn the user and pass allowSignedPackageMutation only if invalidation is acceptable.");
|
|
@@ -441,7 +448,8 @@ const ensureParaIdsInternal = async (docx, options) => {
|
|
|
441
448
|
}),
|
|
442
449
|
assigned,
|
|
443
450
|
deduplicated,
|
|
444
|
-
alreadyComplete: false
|
|
451
|
+
alreadyComplete: false,
|
|
452
|
+
mintedParaIds
|
|
445
453
|
};
|
|
446
454
|
};
|
|
447
455
|
/**
|
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { createNumberingIdAllocator, mintNumberingId } from "./numberingIds.js";
|
|
1
2
|
import { NUMBER_FORMATS } from "@stll/docx-core/model";
|
|
2
3
|
//#region src/docx/listNumberingInstances.ts
|
|
3
4
|
/**
|
|
@@ -115,15 +116,16 @@ const EMPTY_DEFINITIONS = {
|
|
|
115
116
|
abstractNums: [],
|
|
116
117
|
nums: []
|
|
117
118
|
};
|
|
118
|
-
const nextId = (ids, floor) => {
|
|
119
|
-
let next = floor;
|
|
120
|
-
for (const id of ids) next = Math.max(next, id + 1);
|
|
121
|
-
return next;
|
|
122
|
-
};
|
|
123
119
|
/** A fresh `w:numId` for `definitions`; zero is reserved for "no numbering". */
|
|
124
|
-
const nextNumId = (definitions) =>
|
|
120
|
+
const nextNumId = (definitions) => mintNumberingId({
|
|
121
|
+
kind: "num",
|
|
122
|
+
existingIds: (definitions?.nums ?? []).map(({ numId }) => numId)
|
|
123
|
+
});
|
|
125
124
|
/** A fresh `w:abstractNumId` for `definitions`. */
|
|
126
|
-
const nextAbstractNumId = (definitions) =>
|
|
125
|
+
const nextAbstractNumId = (definitions) => mintNumberingId({
|
|
126
|
+
kind: "abstract",
|
|
127
|
+
existingIds: (definitions?.abstractNums ?? []).map(({ abstractNumId }) => abstractNumId)
|
|
128
|
+
});
|
|
127
129
|
/**
|
|
128
130
|
* Define a new list: a new `w:abstractNum` of the requested kind and the
|
|
129
131
|
* `w:num` naming it. `definitions` must already hold every instance the
|
|
@@ -218,11 +220,11 @@ const completeListNumbering = (definitions, references) => {
|
|
|
218
220
|
const current = definitions ?? EMPTY_DEFINITIONS;
|
|
219
221
|
const abstractIds = new Set(current.abstractNums.map(({ abstractNumId }) => abstractNumId));
|
|
220
222
|
const statedAbstractIds = [...missing.values()].flatMap((group) => group.flatMap(({ rendering }) => rendering.abstractNumId === void 0 ? [] : [rendering.abstractNumId]));
|
|
221
|
-
|
|
223
|
+
const missingAbstractIds = createNumberingIdAllocator("abstract", [...abstractIds, ...statedAbstractIds]);
|
|
222
224
|
const abstractNums = [];
|
|
223
225
|
const nums = [];
|
|
224
226
|
for (const [numId, group] of missing) {
|
|
225
|
-
const abstractNumId = group.find(({ rendering }) => rendering.abstractNumId !== void 0)?.rendering.abstractNumId ??
|
|
227
|
+
const abstractNumId = group.find(({ rendering }) => rendering.abstractNumId !== void 0)?.rendering.abstractNumId ?? missingAbstractIds.next();
|
|
226
228
|
if (!abstractIds.has(abstractNumId)) {
|
|
227
229
|
abstractIds.add(abstractNumId);
|
|
228
230
|
const kind = group.some(({ rendering }) => rendering.isBullet) ? "bullet" : "numbered";
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
import { createNumericIdAllocator } from "./numericIdAllocator.js";
|
|
2
|
+
//#region src/docx/noteIds.ts
|
|
3
|
+
let endnoteIds;
|
|
4
|
+
const mintEndnoteId = (existingIds) => {
|
|
5
|
+
const allocator = endnoteIds ??= createNumericIdAllocator({
|
|
6
|
+
space: "endnote",
|
|
7
|
+
firstId: 1
|
|
8
|
+
});
|
|
9
|
+
allocator.reserve(existingIds);
|
|
10
|
+
return allocator.next();
|
|
11
|
+
};
|
|
12
|
+
//#endregion
|
|
13
|
+
export { mintEndnoteId };
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
//#region src/docx/numberingIds.d.ts
|
|
2
|
+
type NumberingKind = "num" | "abstract";
|
|
3
|
+
type MintNumberingIdOptions = {
|
|
4
|
+
kind: NumberingKind;
|
|
5
|
+
existingIds: Iterable<number>;
|
|
6
|
+
};
|
|
7
|
+
/** Independent offline packages can allocate deterministically without retaining realm state. */
|
|
8
|
+
declare const createNumberingIdAllocator: (kind: NumberingKind, existingIds: Iterable<number>) => {
|
|
9
|
+
reserve: (ids: Iterable<number>) => void;
|
|
10
|
+
next: () => number;
|
|
11
|
+
};
|
|
12
|
+
declare const mintNumberingId: ({ kind, existingIds }: MintNumberingIdOptions) => number;
|
|
13
|
+
//#endregion
|
|
14
|
+
export { createNumberingIdAllocator, mintNumberingId };
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
import { createNumericIdAllocator } from "./numericIdAllocator.js";
|
|
2
|
+
//#region src/docx/numberingIds.ts
|
|
3
|
+
/** Independent offline packages can allocate deterministically without retaining realm state. */
|
|
4
|
+
const createNumberingIdAllocator = (kind, existingIds) => {
|
|
5
|
+
const allocator = createNumericIdAllocator({
|
|
6
|
+
space: kind === "num" ? "numbering instance" : "abstract numbering",
|
|
7
|
+
firstId: kind === "num" ? 1 : 0
|
|
8
|
+
});
|
|
9
|
+
allocator.reserve(existingIds);
|
|
10
|
+
return allocator;
|
|
11
|
+
};
|
|
12
|
+
const mintNumberingId = ({ kind, existingIds }) => {
|
|
13
|
+
return createNumberingIdAllocator(kind, existingIds).next();
|
|
14
|
+
};
|
|
15
|
+
//#endregion
|
|
16
|
+
export { createNumberingIdAllocator, mintNumberingId };
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
//#region src/docx/numericIdAllocator.d.ts
|
|
2
|
+
type NumericIdAllocatorOptions = {
|
|
3
|
+
space: string;
|
|
4
|
+
firstId: 0 | 1;
|
|
5
|
+
};
|
|
6
|
+
/** Each OOXML id space owns one instance, retaining loaded and minted ids through rollover. */
|
|
7
|
+
declare const createNumericIdAllocator: ({ space, firstId }: NumericIdAllocatorOptions) => {
|
|
8
|
+
reserve: (ids: Iterable<number>) => void;
|
|
9
|
+
next: () => number;
|
|
10
|
+
};
|
|
11
|
+
//#endregion
|
|
12
|
+
export { createNumericIdAllocator };
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
import { TaggedError } from "better-result";
|
|
2
|
+
import { MAX_REVISION_ID } from "@stll/docx-core/model";
|
|
3
|
+
//#region src/docx/numericIdAllocator.ts
|
|
4
|
+
var OoxmlIdSpaceExhaustedError = class extends TaggedError("OoxmlIdSpaceExhaustedError") {};
|
|
5
|
+
/** Each OOXML id space owns one instance, retaining loaded and minted ids through rollover. */
|
|
6
|
+
const createNumericIdAllocator = ({ space, firstId }) => {
|
|
7
|
+
const reserved = /* @__PURE__ */ new Set();
|
|
8
|
+
let nextId = firstId;
|
|
9
|
+
return {
|
|
10
|
+
reserve: (ids) => {
|
|
11
|
+
let max = -1;
|
|
12
|
+
for (const id of ids) {
|
|
13
|
+
if (!Number.isInteger(id) || id < firstId || id > MAX_REVISION_ID) continue;
|
|
14
|
+
reserved.add(id);
|
|
15
|
+
max = Math.max(max, id);
|
|
16
|
+
}
|
|
17
|
+
if (max >= nextId) nextId = max === MAX_REVISION_ID ? firstId : max + 1;
|
|
18
|
+
},
|
|
19
|
+
next: () => {
|
|
20
|
+
if (reserved.size >= MAX_REVISION_ID - firstId + 1) throw new OoxmlIdSpaceExhaustedError({
|
|
21
|
+
message: `OOXML ${space} id space is exhausted`,
|
|
22
|
+
space
|
|
23
|
+
});
|
|
24
|
+
while (reserved.has(nextId)) nextId = nextId === MAX_REVISION_ID ? firstId : nextId + 1;
|
|
25
|
+
const id = nextId;
|
|
26
|
+
reserved.add(id);
|
|
27
|
+
nextId = id === MAX_REVISION_ID ? firstId : id + 1;
|
|
28
|
+
return id;
|
|
29
|
+
}
|
|
30
|
+
};
|
|
31
|
+
};
|
|
32
|
+
//#endregion
|
|
33
|
+
export { createNumericIdAllocator };
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
import { RawDocxContent } from "./unzip.js";
|
|
2
|
+
import { XmlElement } from "./xmlParser.js";
|
|
3
|
+
//#region src/docx/numericIdNormalization.d.ts
|
|
4
|
+
type NumericIdNormalizationOptions = {
|
|
5
|
+
onParsedDocument?: (document: XmlElement) => void;
|
|
6
|
+
};
|
|
7
|
+
/**
|
|
8
|
+
* Repair out-of-range imported integers before any model or opaque XML is captured.
|
|
9
|
+
* Reserve the complete package first, then remap each old value once per space.
|
|
10
|
+
* Source splices preserve unrelated XML bytes; a valid package takes the fast path.
|
|
11
|
+
*/
|
|
12
|
+
declare const normalizeImportedNumericIds: (parts: ReadonlyMap<string, string>, options?: NumericIdNormalizationOptions) => ReadonlyMap<string, string>;
|
|
13
|
+
/** Parser boundary: all captures and selective-save bytes see the same identities. */
|
|
14
|
+
declare const normalizeRawDocxNumericIds: (raw: RawDocxContent) => Promise<XmlElement | undefined>;
|
|
15
|
+
//#endregion
|
|
16
|
+
export { normalizeImportedNumericIds, normalizeRawDocxNumericIds };
|
|
@@ -0,0 +1,174 @@
|
|
|
1
|
+
import { REVISION_ELEMENT_NAMES } from "./revisionIdNormalization.js";
|
|
2
|
+
import { parseStreamingXmlWithIdentityVisitor, scanStreamingXmlNumericIdAttributes } from "./streamingXmlParser.js";
|
|
3
|
+
import { replaceRawDocxXmlParts } from "./unzip.js";
|
|
4
|
+
import { getLocalName, getNamespaceUri, resolveAttributeNamespaceUri } from "./xmlParser.js";
|
|
5
|
+
import { XmlResourceLimitError } from "./xmlResourceLimits.js";
|
|
6
|
+
import { panic } from "better-result";
|
|
7
|
+
import { assertValidOoxmlNumericId, isOoxmlNumericIdAttributeName, isValidOoxmlNumericId, mayContainInvalidOoxmlNumericIds, mayContainOoxmlNumericIds, ooxmlNumericIdDomain } from "@stll/docx-core";
|
|
8
|
+
//#region src/docx/numericIdNormalization.ts
|
|
9
|
+
const identityKey = (value) => {
|
|
10
|
+
const spelling = String(value).trim();
|
|
11
|
+
if (!/^[+-]?\d+$/u.test(spelling)) return spelling;
|
|
12
|
+
const digits = spelling.replace(/^[+-]?0*/u, "");
|
|
13
|
+
return digits === "" ? "0" : `${spelling.startsWith("-") ? "-" : ""}${digits}`;
|
|
14
|
+
};
|
|
15
|
+
const identitySpace = ({ elementName, attributeName, domain }) => {
|
|
16
|
+
if (domain === "unsigned32") return "drawing";
|
|
17
|
+
const element = getLocalName(elementName);
|
|
18
|
+
const attribute = getLocalName(attributeName);
|
|
19
|
+
switch (attribute === "val" ? element : attribute) {
|
|
20
|
+
case "numId": return "numbering-instance";
|
|
21
|
+
case "abstractNumId": return "numbering-abstract";
|
|
22
|
+
case "numPicBulletId":
|
|
23
|
+
case "lvlPicBulletId": return "numbering-picture";
|
|
24
|
+
default: break;
|
|
25
|
+
}
|
|
26
|
+
switch (element) {
|
|
27
|
+
case "footnote":
|
|
28
|
+
case "footnoteReference": return "footnote";
|
|
29
|
+
case "endnote":
|
|
30
|
+
case "endnoteReference": return "endnote";
|
|
31
|
+
case "id": return "content-control";
|
|
32
|
+
default: return `annotation:${REVISION_ELEMENT_NAMES.has(element) ? "revision" : element.replace(/(?:Range)?(?:Start|End)$/u, "").replace(/^comment.*$/u, "comment")}`;
|
|
33
|
+
}
|
|
34
|
+
};
|
|
35
|
+
const identifiedAttributes = (element) => {
|
|
36
|
+
const attributes = [];
|
|
37
|
+
for (const [name, value] of Object.entries(element.attributes ?? {})) {
|
|
38
|
+
if (!isOoxmlNumericIdAttributeName({
|
|
39
|
+
elementName: element.name ?? "",
|
|
40
|
+
attributeName: name
|
|
41
|
+
})) continue;
|
|
42
|
+
if (value === void 0 || !/^[+-]?\d+$/u.test(String(value).trim())) continue;
|
|
43
|
+
const domain = ooxmlNumericIdDomain({
|
|
44
|
+
elementName: element.name ?? "",
|
|
45
|
+
elementNamespace: getNamespaceUri(element),
|
|
46
|
+
attributeName: name,
|
|
47
|
+
attributeNamespace: resolveAttributeNamespaceUri(element, name)
|
|
48
|
+
});
|
|
49
|
+
if (domain === void 0) continue;
|
|
50
|
+
attributes.push({
|
|
51
|
+
name,
|
|
52
|
+
value,
|
|
53
|
+
domain,
|
|
54
|
+
space: identitySpace({
|
|
55
|
+
elementName: element.name ?? "",
|
|
56
|
+
attributeName: name,
|
|
57
|
+
domain
|
|
58
|
+
})
|
|
59
|
+
});
|
|
60
|
+
}
|
|
61
|
+
return attributes;
|
|
62
|
+
};
|
|
63
|
+
/**
|
|
64
|
+
* Repair out-of-range imported integers before any model or opaque XML is captured.
|
|
65
|
+
* Reserve the complete package first, then remap each old value once per space.
|
|
66
|
+
* Source splices preserve unrelated XML bytes; a valid package takes the fast path.
|
|
67
|
+
*/
|
|
68
|
+
const normalizeImportedNumericIds = (parts, options = {}) => {
|
|
69
|
+
if (![...parts.values()].some((xml) => mayContainInvalidOoxmlNumericIds(xml, "range"))) return parts;
|
|
70
|
+
const normalized = new Map(parts);
|
|
71
|
+
const candidates = [...parts].filter(([, xml]) => mayContainOoxmlNumericIds(xml)).toSorted(([left], [right]) => left.localeCompare(right));
|
|
72
|
+
const spaces = /* @__PURE__ */ new Map();
|
|
73
|
+
const reservedPools = /* @__PURE__ */ new Map();
|
|
74
|
+
const changedSpans = /* @__PURE__ */ new Map();
|
|
75
|
+
let parsedDocument;
|
|
76
|
+
const scan = ({ path, xml, visitor }) => {
|
|
77
|
+
const scanned = path.toLowerCase() === "word/document.xml" && options.onParsedDocument !== void 0 ? parseStreamingXmlWithIdentityVisitor(xml, visitor) : scanStreamingXmlNumericIdAttributes(xml, visitor);
|
|
78
|
+
if (scanned.status === "unsupported") throw new XmlResourceLimitError({
|
|
79
|
+
message: `Numeric-id normalization could not safely scan ${path}`,
|
|
80
|
+
limit: "syntax",
|
|
81
|
+
observed: 0,
|
|
82
|
+
allowed: 0
|
|
83
|
+
});
|
|
84
|
+
if (scanned.status === "parsed") parsedDocument = scanned.value;
|
|
85
|
+
};
|
|
86
|
+
for (const [path, xml] of candidates) scan({
|
|
87
|
+
path,
|
|
88
|
+
xml,
|
|
89
|
+
visitor: (element, attributeValueSpans) => {
|
|
90
|
+
for (const attribute of identifiedAttributes(element)) {
|
|
91
|
+
let space = spaces.get(attribute.space);
|
|
92
|
+
if (space === void 0) {
|
|
93
|
+
const poolName = attribute.space.startsWith("annotation:") ? "annotation" : attribute.space;
|
|
94
|
+
let reserved = reservedPools.get(poolName);
|
|
95
|
+
if (reserved === void 0) {
|
|
96
|
+
reserved = /* @__PURE__ */ new Set();
|
|
97
|
+
reservedPools.set(poolName, reserved);
|
|
98
|
+
}
|
|
99
|
+
space = {
|
|
100
|
+
reserved,
|
|
101
|
+
replacements: /* @__PURE__ */ new Map(),
|
|
102
|
+
next: 1
|
|
103
|
+
};
|
|
104
|
+
spaces.set(attribute.space, space);
|
|
105
|
+
}
|
|
106
|
+
if (isValidOoxmlNumericId(attribute.value, attribute.domain)) {
|
|
107
|
+
space.reserved.add(Number(attribute.value));
|
|
108
|
+
continue;
|
|
109
|
+
}
|
|
110
|
+
const key = identityKey(attribute.value);
|
|
111
|
+
space.replacements.set(key, "");
|
|
112
|
+
const span = attributeValueSpans.get(attribute.name);
|
|
113
|
+
if (span === void 0) panic("Missing imported numeric identity source span");
|
|
114
|
+
let spans = changedSpans.get(path);
|
|
115
|
+
if (spans === void 0) {
|
|
116
|
+
spans = [];
|
|
117
|
+
changedSpans.set(path, spans);
|
|
118
|
+
}
|
|
119
|
+
spans.push({
|
|
120
|
+
start: span.start,
|
|
121
|
+
end: span.end,
|
|
122
|
+
space: attribute.space,
|
|
123
|
+
key,
|
|
124
|
+
element,
|
|
125
|
+
attributeName: attribute.name
|
|
126
|
+
});
|
|
127
|
+
}
|
|
128
|
+
return null;
|
|
129
|
+
}
|
|
130
|
+
});
|
|
131
|
+
for (const space of spaces.values()) for (const key of space.replacements.keys()) {
|
|
132
|
+
while (space.reserved.has(space.next)) space.next += 1;
|
|
133
|
+
assertValidOoxmlNumericId({
|
|
134
|
+
value: space.next,
|
|
135
|
+
partPath: "import",
|
|
136
|
+
elementName: "allocated identity",
|
|
137
|
+
attributeName: "id"
|
|
138
|
+
});
|
|
139
|
+
const replacement = String(space.next);
|
|
140
|
+
space.replacements.set(key, replacement);
|
|
141
|
+
space.reserved.add(space.next);
|
|
142
|
+
space.next += 1;
|
|
143
|
+
}
|
|
144
|
+
for (const [path, spans] of changedSpans) {
|
|
145
|
+
const xml = parts.get(path);
|
|
146
|
+
if (xml === void 0) panic(`Missing imported numeric identity part: ${path}`);
|
|
147
|
+
const chunks = [];
|
|
148
|
+
let cursor = 0;
|
|
149
|
+
for (const span of spans.toSorted((left, right) => left.start - right.start)) {
|
|
150
|
+
const replacement = spaces.get(span.space)?.replacements.get(span.key);
|
|
151
|
+
if (replacement === void 0 || replacement === "") panic("Missing imported numeric identity replacement");
|
|
152
|
+
if (span.element.attributes === void 0) panic("Missing imported numeric identity attributes");
|
|
153
|
+
span.element.attributes[span.attributeName] = replacement;
|
|
154
|
+
chunks.push(xml.slice(cursor, span.start), replacement);
|
|
155
|
+
cursor = span.end;
|
|
156
|
+
}
|
|
157
|
+
chunks.push(xml.slice(cursor));
|
|
158
|
+
normalized.set(path, chunks.join(""));
|
|
159
|
+
}
|
|
160
|
+
if (parsedDocument !== void 0) options.onParsedDocument?.(parsedDocument);
|
|
161
|
+
return normalized;
|
|
162
|
+
};
|
|
163
|
+
/** Parser boundary: all captures and selective-save bytes see the same identities. */
|
|
164
|
+
const normalizeRawDocxNumericIds = async (raw) => {
|
|
165
|
+
let documentTree;
|
|
166
|
+
const normalized = normalizeImportedNumericIds(raw.allXml, { onParsedDocument: (tree) => {
|
|
167
|
+
documentTree = tree;
|
|
168
|
+
} });
|
|
169
|
+
if (normalized === raw.allXml) return documentTree;
|
|
170
|
+
await replaceRawDocxXmlParts(raw, normalized);
|
|
171
|
+
return documentTree;
|
|
172
|
+
};
|
|
173
|
+
//#endregion
|
|
174
|
+
export { normalizeImportedNumericIds, normalizeRawDocxNumericIds };
|
|
@@ -542,7 +542,7 @@ const tableCellParagraphPropertySourceBindingForTransport = (paragraph, inspecti
|
|
|
542
542
|
};
|
|
543
543
|
/** Prepare opaque continuation cells for ProseMirror and Yjs transport. */
|
|
544
544
|
const transportTableCellsWithParagraphPropertySources = (cells) => {
|
|
545
|
-
const cloned = structuredClone(
|
|
545
|
+
const cloned = cells.map((cell) => structuredClone(cell));
|
|
546
546
|
const sources = paragraphsInTableCells(cells);
|
|
547
547
|
const targets = paragraphsInTableCells(cloned);
|
|
548
548
|
if (sources.length !== targets.length) panic("The cloned table cells changed paragraph graph ownership.");
|
|
@@ -6,7 +6,7 @@ import { consolidateParagraphContent } from "./runConsolidator.js";
|
|
|
6
6
|
import { captureShapeAlternateContent } from "./shapeAlternateContent.js";
|
|
7
7
|
import { getTextBoxContentElement, parseTextBox, parseTextBoxContent, parseTextBoxFromShape, scanRunForTextBoxDrawings } from "./textBoxParser.js";
|
|
8
8
|
import { isVmlPictParsedByRunParser } from "./vmlImageParser.js";
|
|
9
|
-
import { findChildByLocalName, findDeep, getAttribute, getChildElements, getLocalName } from "./xmlParser.js";
|
|
9
|
+
import { findChildByLocalName, findDeep, findWordprocessingChild, getAttribute, getChildElements, getLocalName } from "./xmlParser.js";
|
|
10
10
|
//#region src/docx/paragraphTextBoxEnrichment.ts
|
|
11
11
|
const VML_HORIZONTAL_RELATIVES = /* @__PURE__ */ new Set([
|
|
12
12
|
"character",
|
|
@@ -396,6 +396,7 @@ const enrichTextBoxRuns = ({ content, xmlChildren, styles, theme, numbering, rel
|
|
|
396
396
|
lastConsumedRun = parsedRun;
|
|
397
397
|
parsedIndex += 1;
|
|
398
398
|
}
|
|
399
|
+
if (findWordprocessingChild(xmlChild, "commentReference") && content[parsedIndex]?.type === "commentReference") parsedIndex += 1;
|
|
399
400
|
}
|
|
400
401
|
};
|
|
401
402
|
const parseVmlTextBoxShape = (pictEl, { styles, theme, numbering, rels, media, parseTable, previews }) => {
|
package/dist/docx/parser.js
CHANGED
|
@@ -8,7 +8,7 @@ import { DANGLING_COMMENT_REFERENCE_WARNING, UNBALANCED_COMMENT_RANGE_WARNING, n
|
|
|
8
8
|
import { detectDocxConformanceClass } from "./conformance.js";
|
|
9
9
|
import { parseCoreProperties } from "./corePropertiesParser.js";
|
|
10
10
|
import { countDanglingRelationshipReferences } from "./danglingRelationshipReferences.js";
|
|
11
|
-
import { extractAllTemplateVariables, parseDocumentBody } from "./documentParser.js";
|
|
11
|
+
import { extractAllTemplateVariables, parseDocumentBody, parseDocumentBodyTree } from "./documentParser.js";
|
|
12
12
|
import { normalizeDrawingIds } from "./drawingIdNormalization.js";
|
|
13
13
|
import { DocxEncryptionError } from "./encryption/errors.js";
|
|
14
14
|
import { parseFontTable } from "./fontTableParser.js";
|
|
@@ -21,11 +21,13 @@ import { renderEmfSvg } from "./metafileSvg.js";
|
|
|
21
21
|
import { DocxModelValidationError, formatDocumentModelIssues, validateFolioDocumentModel } from "./modelValidation.js";
|
|
22
22
|
import { parseNumbering } from "./numberingParser.js";
|
|
23
23
|
import { UNNUMBERED_PARAGRAPH_WARNING, UNNUMBERED_STYLE_WARNING, normalizeNumberingReferences, normalizeStyleNumberingReferences } from "./numberingReferenceNormalization.js";
|
|
24
|
+
import { normalizeRawDocxNumericIds } from "./numericIdNormalization.js";
|
|
24
25
|
import { countOpaqueRevisionWrappers } from "./opaqueCarrier.js";
|
|
25
26
|
import { assignDocumentParagraphPropertySourceContract } from "./paragraphPropertySource.js";
|
|
26
27
|
import { createParseWarningCollector } from "./parseContext.js";
|
|
27
28
|
import { formatParseWarnings } from "./parseWarningMessage.js";
|
|
28
29
|
import { createPackagePreviewBudget } from "./previewBudget.js";
|
|
30
|
+
import { detectRasterMimeType } from "./rasterMime.js";
|
|
29
31
|
import { RELATIONSHIP_TYPES, parseRelationships, resolveRelativePath } from "./relsParser.js";
|
|
30
32
|
import { normalizeRenderedPageBreakHints } from "./renderedPageBreakNormalization.js";
|
|
31
33
|
import { parseSettings } from "./settingsParser.js";
|
|
@@ -33,7 +35,7 @@ import { parseStylesPackage } from "./styleParser.js";
|
|
|
33
35
|
import { applyThemeFontLang, parseTheme } from "./themeParser.js";
|
|
34
36
|
import { UNBALANCED_MOVE_RANGE_WARNING, normalizeTrackedMoveRanges } from "./trackedMoveRangeNormalization.js";
|
|
35
37
|
import { getEntryUncompressedSize, getMediaMimeType, mediaToDataUrl, unzipDocx } from "./unzip.js";
|
|
36
|
-
import { getAttribute, getChildElements, getLocalName, parseXmlDocument } from "./xmlParser.js";
|
|
38
|
+
import { getAttribute, getChildElements, getLocalName, parseXml, parseXmlDocument } from "./xmlParser.js";
|
|
37
39
|
import { FOLIO_XML_RESOURCE_LIMITS } from "./xmlResourceLimits.js";
|
|
38
40
|
import { TaggedError, panic } from "better-result";
|
|
39
41
|
import { PARSE_WARNING_CODES } from "@stll/docx-core/model";
|
|
@@ -184,13 +186,14 @@ async function parseDocxWithPreviewBudget(input, options, previewBudget) {
|
|
|
184
186
|
try {
|
|
185
187
|
const timeStage = (_name, fn) => fn();
|
|
186
188
|
const timeStageAsync = async (_name, fn) => await fn();
|
|
187
|
-
const paragraphPropertySourceDigest = sha256Hex(buffer);
|
|
188
189
|
onProgress("Extracting DOCX...", 0);
|
|
189
190
|
const raw = await timeStageAsync("unzip", () => unzipDocx(buffer, {
|
|
190
191
|
...unzipLimits,
|
|
191
192
|
password,
|
|
192
193
|
extractAllXml: false
|
|
193
194
|
}));
|
|
195
|
+
const repairedDocumentTree = await normalizeRawDocxNumericIds(raw);
|
|
196
|
+
const paragraphPropertySourceDigest = sha256Hex(raw.originalBuffer);
|
|
194
197
|
if (raw.wasEncrypted) parseContext.warn({ code: PARSE_WARNING_CODES.packageDecrypted });
|
|
195
198
|
for (const message of raw.warnings) parseContext.warn({
|
|
196
199
|
code: PARSE_WARNING_CODES.packageArchive,
|
|
@@ -227,7 +230,16 @@ async function parseDocxWithPreviewBudget(input, options, previewBudget) {
|
|
|
227
230
|
onProgress("Parsing document body...", 40);
|
|
228
231
|
let documentBody = { content: [] };
|
|
229
232
|
timeStage("documentBody", () => {
|
|
230
|
-
if (raw.documentXml) documentBody =
|
|
233
|
+
if (raw.documentXml) documentBody = parseDocumentBodyTree({
|
|
234
|
+
doc: repairedDocumentTree ?? parseXml(raw.documentXml),
|
|
235
|
+
styles,
|
|
236
|
+
theme,
|
|
237
|
+
numbering,
|
|
238
|
+
rels,
|
|
239
|
+
media,
|
|
240
|
+
context: parseContext.scoped({ part: "word/document.xml" }),
|
|
241
|
+
previews: previews.ledger
|
|
242
|
+
});
|
|
231
243
|
else parseContext.warn({ code: PARSE_WARNING_CODES.documentPartMissing });
|
|
232
244
|
});
|
|
233
245
|
onProgress("Parsed document body", 55);
|
|
@@ -507,7 +519,7 @@ async function buildMediaMap(raw, rels) {
|
|
|
507
519
|
let remainingEmfPreviewBytes = MAX_PACKAGE_EMF_PREVIEW_BYTES;
|
|
508
520
|
for (const [path, data] of raw.media.entries()) {
|
|
509
521
|
const filename = path.split("/").pop() || path;
|
|
510
|
-
const mimeType = getMediaMimeType(path);
|
|
522
|
+
const mimeType = detectRasterMimeType(data) ?? getMediaMimeType(path);
|
|
511
523
|
const isReferenced = referenced.has(path.toLowerCase());
|
|
512
524
|
if (isReferenced && isTiffMimeType(mimeType) && remainingTiffPixels > 0) {
|
|
513
525
|
const converted = await convertTiffToPngDataUrl(data, remainingTiffPixels);
|
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
//#region src/docx/rasterMime.d.ts
|
|
2
|
+
declare const RASTER_SIGNATURES: readonly [{
|
|
3
|
+
readonly mimeType: "image/png";
|
|
4
|
+
readonly chunks: readonly [{
|
|
5
|
+
readonly offset: 0;
|
|
6
|
+
readonly bytes: readonly [137, 80, 78, 71];
|
|
7
|
+
}];
|
|
8
|
+
}, {
|
|
9
|
+
readonly mimeType: "image/jpeg";
|
|
10
|
+
readonly chunks: readonly [{
|
|
11
|
+
readonly offset: 0;
|
|
12
|
+
readonly bytes: readonly [255, 216];
|
|
13
|
+
}];
|
|
14
|
+
}, {
|
|
15
|
+
readonly mimeType: "image/gif";
|
|
16
|
+
readonly chunks: readonly [{
|
|
17
|
+
readonly offset: 0;
|
|
18
|
+
readonly bytes: readonly [71, 73, 70, 56];
|
|
19
|
+
}];
|
|
20
|
+
}, {
|
|
21
|
+
readonly mimeType: "image/bmp";
|
|
22
|
+
readonly chunks: readonly [{
|
|
23
|
+
readonly offset: 0;
|
|
24
|
+
readonly bytes: readonly [66, 77];
|
|
25
|
+
}];
|
|
26
|
+
}, {
|
|
27
|
+
readonly mimeType: "image/webp";
|
|
28
|
+
readonly chunks: readonly [{
|
|
29
|
+
readonly offset: 0;
|
|
30
|
+
readonly bytes: readonly [82, 73, 70, 70];
|
|
31
|
+
}, {
|
|
32
|
+
readonly offset: 8;
|
|
33
|
+
readonly bytes: readonly [87, 69, 66, 80];
|
|
34
|
+
}];
|
|
35
|
+
}, {
|
|
36
|
+
readonly mimeType: "image/tiff";
|
|
37
|
+
readonly chunks: readonly [{
|
|
38
|
+
readonly offset: 0;
|
|
39
|
+
readonly bytes: readonly [73, 73, 42, 0];
|
|
40
|
+
}];
|
|
41
|
+
}, {
|
|
42
|
+
readonly mimeType: "image/tiff";
|
|
43
|
+
readonly chunks: readonly [{
|
|
44
|
+
readonly offset: 0;
|
|
45
|
+
readonly bytes: readonly [77, 77, 0, 42];
|
|
46
|
+
}];
|
|
47
|
+
}, {
|
|
48
|
+
readonly mimeType: "image/tiff";
|
|
49
|
+
readonly chunks: readonly [{
|
|
50
|
+
readonly offset: 0;
|
|
51
|
+
readonly bytes: readonly [73, 73, 43, 0];
|
|
52
|
+
}];
|
|
53
|
+
}, {
|
|
54
|
+
readonly mimeType: "image/tiff";
|
|
55
|
+
readonly chunks: readonly [{
|
|
56
|
+
readonly offset: 0;
|
|
57
|
+
readonly bytes: readonly [77, 77, 0, 43];
|
|
58
|
+
}];
|
|
59
|
+
}];
|
|
60
|
+
type RasterMimeType = (typeof RASTER_SIGNATURES)[number]["mimeType"];
|
|
61
|
+
declare const RASTER_MIME_TYPES: ReadonlySet<string>;
|
|
62
|
+
declare const detectRasterMimeType: (data: ArrayBuffer) => RasterMimeType | undefined;
|
|
63
|
+
//#endregion
|
|
64
|
+
export { RASTER_MIME_TYPES, RasterMimeType, detectRasterMimeType };
|