reamkit 1.32.0 → 1.32.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/esm/core/crypto/offcrypto.js +56 -12
- package/dist/esm/core/drawingml/chart-geometry.d.ts +21 -11
- package/dist/esm/core/drawingml/chart-geometry.js +205 -69
- package/dist/esm/core/drawingml/chart-parser.d.ts +19 -0
- package/dist/esm/core/drawingml/chart-parser.js +60 -23
- package/dist/esm/core/number-format.d.ts +40 -15
- package/dist/esm/core/number-format.js +280 -73
- package/dist/esm/core/opc/tag-scan.d.ts +15 -0
- package/dist/esm/core/opc/tag-scan.js +41 -0
- package/dist/esm/core/opc/xml-entities.js +39 -2
- package/dist/esm/excel/formula/functions.js +37 -14
- package/dist/esm/excel/print-model.js +14 -3
- package/dist/esm/excel/vml-drawing.js +1 -1
- package/dist/esm/excel/xlsx-reader.js +39 -5
- package/dist/esm/layout/styled-layout.js +1 -1
- package/dist/esm/markdown/markdown-writer.js +58 -7
- package/dist/esm/pdf-reader/annot-draw.js +2 -2
- package/dist/esm/pdf-reader/annots.js +3 -1
- package/dist/esm/pdf-reader/font.js +1 -1
- package/dist/esm/pdf-reader/layout.js +1 -1
- package/dist/esm/word/document-parser.js +1 -1
- package/dist/esm/word/docx-reader.js +13 -8
- package/dist/esm/word/docx-to-pdf.js +3 -3
- package/dist/esm/word/font-table.js +89 -10
- package/dist/esm/word/numbering-parser.js +1 -1
- package/package.json +3 -3
|
@@ -537,7 +537,7 @@ function parseCriteria(s) {
|
|
|
537
537
|
return {
|
|
538
538
|
kind: "text",
|
|
539
539
|
negate: op === "<>",
|
|
540
|
-
|
|
540
|
+
wildcard: wildcard(text)
|
|
541
541
|
};
|
|
542
542
|
}
|
|
543
543
|
function matchCriteria(cell, crit) {
|
|
@@ -555,22 +555,45 @@ function matchCriteria(cell, crit) {
|
|
|
555
555
|
}
|
|
556
556
|
const text = cell.t === "str" ? cell.v : cell.t === "blank" ? "" : void 0;
|
|
557
557
|
if (text === void 0) return false;
|
|
558
|
-
const hit = crit.
|
|
558
|
+
const hit = crit.wildcard.test(text);
|
|
559
559
|
return crit.negate ? !hit : hit;
|
|
560
560
|
}
|
|
561
|
-
function
|
|
562
|
-
|
|
561
|
+
function wildcard(pattern) {
|
|
562
|
+
const pieces = [[]];
|
|
563
563
|
for (let i = 0; i < pattern.length; i++) {
|
|
564
564
|
const ch = pattern[i];
|
|
565
|
+
const atoms = pieces[pieces.length - 1];
|
|
565
566
|
if (ch === "~" && (pattern[i + 1] === "*" || pattern[i + 1] === "?" || pattern[i + 1] === "~")) {
|
|
566
|
-
|
|
567
|
+
atoms.push(escapeRe(pattern[i + 1]));
|
|
567
568
|
i++;
|
|
568
|
-
} else if (ch === "*")
|
|
569
|
-
else
|
|
570
|
-
else out += escapeRe(ch);
|
|
569
|
+
} else if (ch === "*") pieces.push([]);
|
|
570
|
+
else atoms.push(ch === "?" ? "." : escapeRe(ch));
|
|
571
571
|
}
|
|
572
|
-
|
|
573
|
-
|
|
572
|
+
const all = pieces.map((atoms) => ({
|
|
573
|
+
at: new RegExp(atoms.join(""), "isy"),
|
|
574
|
+
search: new RegExp(atoms.join(""), "gis"),
|
|
575
|
+
width: atoms.length
|
|
576
|
+
}));
|
|
577
|
+
const fits = (piece, text, at) => {
|
|
578
|
+
piece.at.lastIndex = at;
|
|
579
|
+
return piece.at.test(text);
|
|
580
|
+
};
|
|
581
|
+
const first = all[0];
|
|
582
|
+
if (all.length === 1) return { test: (text) => text.length === first.width && fits(first, text, 0) };
|
|
583
|
+
const last = all[all.length - 1];
|
|
584
|
+
const between = all.slice(1, -1).filter((piece) => piece.width > 0);
|
|
585
|
+
return { test(text) {
|
|
586
|
+
const end = text.length - last.width;
|
|
587
|
+
if (end < first.width || !fits(first, text, 0) || !fits(last, text, end)) return false;
|
|
588
|
+
let at = first.width;
|
|
589
|
+
for (const piece of between) {
|
|
590
|
+
piece.search.lastIndex = at;
|
|
591
|
+
const found = piece.search.exec(text);
|
|
592
|
+
if (found === null || found.index + piece.width > end) return false;
|
|
593
|
+
at = found.index + piece.width;
|
|
594
|
+
}
|
|
595
|
+
return true;
|
|
596
|
+
} };
|
|
574
597
|
}
|
|
575
598
|
function escapeRe(ch) {
|
|
576
599
|
return /[.*+?^${}()|[\]\\]/.test(ch) ? `\\${ch}` : ch;
|
|
@@ -1090,10 +1113,10 @@ function matchFn(args, ev, ctx) {
|
|
|
1090
1113
|
if (len > MAX_SCAN) return err("#N/A");
|
|
1091
1114
|
const at = (i) => horizontal ? refGet(range.sheet, ctx, rect.r0, rect.c0 + i) : refGet(range.sheet, ctx, rect.r0 + i, rect.c0);
|
|
1092
1115
|
if (type === 0) {
|
|
1093
|
-
const
|
|
1116
|
+
const wild = target.t === "str" ? wildcard(target.v) : void 0;
|
|
1094
1117
|
for (let i = 0; i < len; i++) {
|
|
1095
1118
|
const cell = at(i);
|
|
1096
|
-
if (
|
|
1119
|
+
if (wild ? cell.t === "str" && wild.test(cell.v) : scalarEquals(cell, target)) return num(i + 1);
|
|
1097
1120
|
}
|
|
1098
1121
|
return err("#N/A");
|
|
1099
1122
|
}
|
|
@@ -1156,10 +1179,10 @@ function lookupFn(args, ev, ctx, vh) {
|
|
|
1156
1179
|
const keyAt = (i) => vertical ? refGet(sh, ctx, rect.r0 + i, rect.c0) : refGet(sh, ctx, rect.r0, rect.c0 + i);
|
|
1157
1180
|
const resultAt = (i) => vertical ? refGet(sh, ctx, rect.r0 + i, rect.c0 + line - 1) : refGet(sh, ctx, rect.r0 + line - 1, rect.c0 + i);
|
|
1158
1181
|
if (!approx) {
|
|
1159
|
-
const
|
|
1182
|
+
const wild = target.t === "str" ? wildcard(target.v) : void 0;
|
|
1160
1183
|
for (let i = 0; i < vecLen; i++) {
|
|
1161
1184
|
const k = keyAt(i);
|
|
1162
|
-
if (
|
|
1185
|
+
if (wild ? k.t === "str" && wild.test(k.v) : scalarEquals(k, target)) return resultAt(i);
|
|
1163
1186
|
}
|
|
1164
1187
|
return err("#N/A");
|
|
1165
1188
|
}
|
|
@@ -813,12 +813,23 @@ function gridBody(worksheet, sharedStrings, styles, date1904, print) {
|
|
|
813
813
|
}
|
|
814
814
|
const ws = cellMatrix[r]?.[c];
|
|
815
815
|
let text = ws ? resolveCellText(ws, sharedStrings, styles, date1904) : "";
|
|
816
|
+
let generalOverflow = false;
|
|
816
817
|
if (ws?.type === "n" && ws.rawValue.length > 0 && (styles.cellXfs[ws.styleIndex ?? 0]?.numFmtId ?? 0) === 0) {
|
|
817
818
|
const unit = charTwips(styles.cellXfs[ws.styleIndex ?? 0], styles, textTwipsUnit);
|
|
818
819
|
const digit = charWidthUnits("0");
|
|
819
820
|
if (unit > 0 && digit > 0) {
|
|
820
|
-
|
|
821
|
-
|
|
821
|
+
let widthTwips = columnWidths[c];
|
|
822
|
+
if (merge) {
|
|
823
|
+
widthTwips = 0;
|
|
824
|
+
for (let a = merge.startColumn; a <= Math.min(merge.endColumn, colWindowEnd); a++) {
|
|
825
|
+
const i = a - colStart;
|
|
826
|
+
if (i >= 0 && i < colCount && !hiddenCols.has(i)) widthTwips += columnWidths[i];
|
|
827
|
+
}
|
|
828
|
+
}
|
|
829
|
+
const room = (widthTwips - 75) / unit;
|
|
830
|
+
const fits = (candidate) => estimateChars(candidate) / digit <= room;
|
|
831
|
+
text = generalToWidth(ws.rawValue, fits);
|
|
832
|
+
generalOverflow = !fits(text);
|
|
822
833
|
}
|
|
823
834
|
}
|
|
824
835
|
const cfText = text.length > 0 ? text : void 0;
|
|
@@ -964,7 +975,7 @@ function gridBody(worksheet, sharedStrings, styles, date1904, print) {
|
|
|
964
975
|
...dropdown ? { dropdown: true } : {},
|
|
965
976
|
...noteFlag ? { noteFlag } : {},
|
|
966
977
|
...!wrapText && !rotated ? { noWrap: true } : {},
|
|
967
|
-
...!wrapText && !rotated && !shrinkToFit && !merge && ws?.type === "n" && (xf?.numFmtId ?? 0) !== 0 &&
|
|
978
|
+
...!wrapText && !rotated && !shrinkToFit && !merge && ws?.type === "n" && text.length > 0 && (generalOverflow || (xf?.numFmtId ?? 0) !== 0 && tooWideToShow(text, charTwips(xf, styles, textTwipsUnit), columnWidths[c])) ? { hashOnOverflow: true } : {},
|
|
968
979
|
verticalAlign: verticalAlignOf(xf)
|
|
969
980
|
};
|
|
970
981
|
const indentLevels = xf?.alignment?.indent ?? 0;
|
|
@@ -153,7 +153,7 @@ var CSS_UNITS = new Map([
|
|
|
153
153
|
]);
|
|
154
154
|
function cssLengthPt(value) {
|
|
155
155
|
if (value === void 0) return void 0;
|
|
156
|
-
const m =
|
|
156
|
+
const m = /^(-?[\d.]+)\s*([a-z]*)$/i.exec(value.trim());
|
|
157
157
|
if (!m) return void 0;
|
|
158
158
|
const n = Number(m[1]);
|
|
159
159
|
if (!Number.isFinite(n)) return void 0;
|
|
@@ -4,7 +4,7 @@ import { detectImageFormat } from "../core/images.js";
|
|
|
4
4
|
import { INDEXED_COLORS } from "../core/indexed-colors.js";
|
|
5
5
|
import { bytesInclude, packageHasPart } from "../core/bytes.js";
|
|
6
6
|
import { OFFICE_2023_THEME_PALETTE, makeColorResolver } from "../core/drawingml/colors.js";
|
|
7
|
-
import { parseChart, withChartColorStyle } from "../core/drawingml/chart-parser.js";
|
|
7
|
+
import { parseChart, pointsPerSeries, withChartColorStyle } from "../core/drawingml/chart-parser.js";
|
|
8
8
|
import { parseTheme, parseThemeEffectStyles, parseThemeFillStyles, parseThemeLineWidths } from "../core/drawingml/theme-parser.js";
|
|
9
9
|
import { isOoxmlRel } from "../core/opc/relationship-types.js";
|
|
10
10
|
import { OpcPackage } from "../core/opc/package.js";
|
|
@@ -409,6 +409,11 @@ function readXlsxToSheetDoc(xlsx) {
|
|
|
409
409
|
* them, and a reference naming a sheet this workbook does not have resolves to
|
|
410
410
|
* nothing rather than to zeros.
|
|
411
411
|
*
|
|
412
|
+
* A reference reads no more cells than the series' share of the chart's points
|
|
413
|
+
* ({@link pointsPerSeries}), and finds each in an index of its sheet. The range
|
|
414
|
+
* is the file's to name and unbounded — `B2:B99999999999` — and every cell of
|
|
415
|
+
* it was searched for through the whole sheet.
|
|
416
|
+
*
|
|
412
417
|
* @param chart The parsed chart.
|
|
413
418
|
* @param sheets Every sheet in the workbook, by tab order.
|
|
414
419
|
* @param styles The style table (for a referenced cell's number format).
|
|
@@ -417,13 +422,27 @@ function readXlsxToSheetDoc(xlsx) {
|
|
|
417
422
|
* @returns The chart, unchanged when nothing needed resolving.
|
|
418
423
|
*/
|
|
419
424
|
function withWorkbookData(chart, sheets, styles, sharedStrings, date1904) {
|
|
425
|
+
const most = pointsPerSeries(chart.series.length);
|
|
426
|
+
const indexes = /* @__PURE__ */ new Map();
|
|
427
|
+
const cellAt = (grid, row, column) => {
|
|
428
|
+
let index = indexes.get(grid);
|
|
429
|
+
if (!index) {
|
|
430
|
+
index = /* @__PURE__ */ new Map();
|
|
431
|
+
for (const cell of grid.cells) {
|
|
432
|
+
const key = `${String(cell.row)}:${String(cell.column)}`;
|
|
433
|
+
if (!index.has(key)) index.set(key, cell);
|
|
434
|
+
}
|
|
435
|
+
indexes.set(grid, index);
|
|
436
|
+
}
|
|
437
|
+
return index.get(`${String(row)}:${String(column)}`);
|
|
438
|
+
};
|
|
420
439
|
const cellsOf = (ref) => {
|
|
421
440
|
if (ref === void 0) return void 0;
|
|
422
441
|
const area = resolveChartRef(ref, sheets);
|
|
423
442
|
if (!area) return void 0;
|
|
424
443
|
const out = [];
|
|
425
|
-
for (let row = area.startRow; row <= area.endRow; row++) for (let col = area.startColumn; col <= area.endColumn; col++) {
|
|
426
|
-
const cell = area.grid
|
|
444
|
+
for (let row = area.startRow; row <= area.endRow && out.length < most; row++) for (let col = area.startColumn; col <= area.endColumn && out.length < most; col++) {
|
|
445
|
+
const cell = cellAt(area.grid, row, col);
|
|
427
446
|
out.push(cell ? resolveCellText(cell, sharedStrings, styles, date1904) : "");
|
|
428
447
|
}
|
|
429
448
|
return out;
|
|
@@ -810,9 +829,24 @@ function resolveTableSlicerItems(cache, tableIndex, styles, sharedStrings, date1
|
|
|
810
829
|
}
|
|
811
830
|
return out;
|
|
812
831
|
}
|
|
832
|
+
function slicerStyleNumber(name) {
|
|
833
|
+
const prefix = /SlicerStyle/gi;
|
|
834
|
+
const isLetter = (i) => /[A-Za-z]/.test(name[i] ?? "");
|
|
835
|
+
const isDigit = (i) => /\d/.test(name[i] ?? "");
|
|
836
|
+
for (let found = prefix.exec(name); found; found = prefix.exec(name)) {
|
|
837
|
+
let end = prefix.lastIndex;
|
|
838
|
+
while (isLetter(end)) end++;
|
|
839
|
+
if (isDigit(end)) {
|
|
840
|
+
let digits = end;
|
|
841
|
+
while (isDigit(digits)) digits++;
|
|
842
|
+
return name.slice(end, digits);
|
|
843
|
+
}
|
|
844
|
+
prefix.lastIndex = end;
|
|
845
|
+
}
|
|
846
|
+
}
|
|
813
847
|
function resolveSlicerStyle(styleName, palette) {
|
|
814
|
-
const
|
|
815
|
-
const column =
|
|
848
|
+
const number = styleName ? slicerStyleNumber(styleName) : void 0;
|
|
849
|
+
const column = number !== void 0 ? (Number(number) - 1) % 7 : 1;
|
|
816
850
|
const base = column === 0 ? "7F7F7F" : palette.get(`accent${column}`) ?? "4472C4";
|
|
817
851
|
return {
|
|
818
852
|
headerHex: base,
|
|
@@ -3174,7 +3174,7 @@ function collectFontResources(body, headersFooters, options, imageResources) {
|
|
|
3174
3174
|
if (chart.catAxisTitle) add(chart.catAxisTitle);
|
|
3175
3175
|
if (chart.valAxisTitle) add(chart.valAxisTitle);
|
|
3176
3176
|
for (const sr of chart.series) for (const label of sr.pointLabels ?? []) add(label.text);
|
|
3177
|
-
add("0123456789
|
|
3177
|
+
add("0123456789.,-+E%() ");
|
|
3178
3178
|
if (chart.numberFormat) add(chart.numberFormat);
|
|
3179
3179
|
}
|
|
3180
3180
|
}
|
|
@@ -82,9 +82,29 @@ function lose(ctx, severity, feature, detail) {
|
|
|
82
82
|
*/
|
|
83
83
|
function joinBlocks(blocks) {
|
|
84
84
|
const body = blocks.map(trimHardBreaks).filter((b) => b.length > 0).join("\n\n");
|
|
85
|
-
return body.length > 0 ? `${body
|
|
85
|
+
return body.length > 0 ? `${trimLineEnds(body)}\n` : "";
|
|
86
86
|
}
|
|
87
87
|
/**
|
|
88
|
+
* The text without the spaces and tabs that end its lines — what
|
|
89
|
+
* `/[ \t]+$/gm` leaves, a line ending wherever `$` does under `m`: at LF, CR,
|
|
90
|
+
* U+2028, U+2029 and the end of the text. The expression retries every blank
|
|
91
|
+
* of a long run that something else follows, and is quadratic on it; this is
|
|
92
|
+
* one pass.
|
|
93
|
+
*/
|
|
94
|
+
function trimLineEnds(text) {
|
|
95
|
+
const out = [];
|
|
96
|
+
let start = 0;
|
|
97
|
+
for (let i = 0; i <= text.length; i++) {
|
|
98
|
+
if (i < text.length && !isLineEnd(text.charCodeAt(i))) continue;
|
|
99
|
+
let end = i;
|
|
100
|
+
while (end > start && (text[end - 1] === " " || text[end - 1] === " ")) end--;
|
|
101
|
+
out.push(text.slice(start, end), text.slice(i, i + 1));
|
|
102
|
+
start = i + 1;
|
|
103
|
+
}
|
|
104
|
+
return out.join("");
|
|
105
|
+
}
|
|
106
|
+
var isLineEnd = (c) => c === 10 || c === 13 || c === 8232 || c === 8233;
|
|
107
|
+
/**
|
|
88
108
|
* Drop a hard break sitting at either end of a block: at the end there is no
|
|
89
109
|
* next line to break to, at the start no previous one, and either way the
|
|
90
110
|
* backslash is left standing in the text as itself. Only a break is taken —
|
|
@@ -218,7 +238,7 @@ function emitListItem(out, resolved, marker, inline, ctx) {
|
|
|
218
238
|
top.markerWidth = bullet.length + 1;
|
|
219
239
|
const pad = " ".repeat(top.indent);
|
|
220
240
|
const cont = " ".repeat(top.indent + top.markerWidth);
|
|
221
|
-
const line = `${pad}${bullet} ${inline.replaceAll("\\\n", `\\\n${cont}`)}`.
|
|
241
|
+
const line = `${pad}${bullet} ${inline.replaceAll("\\\n", `\\\n${cont}`)}`.trimEnd();
|
|
222
242
|
if (wasOpen && out.length > 0) out[out.length - 1] += `\n${line}`;
|
|
223
243
|
else out.push(line);
|
|
224
244
|
}
|
|
@@ -232,9 +252,24 @@ function listBullet(marker, numId, ilvl, counter, ctx) {
|
|
|
232
252
|
const level = numberingLevel(numId, ilvl, ctx.numbering);
|
|
233
253
|
if (!(level ? level.format !== "bullet" && level.format !== "none" : /\d/.test(marker))) return "-";
|
|
234
254
|
if (level && level.format !== "decimal" && level.format !== "decimalZero") lose(ctx, "degraded", FEATURES.lists, `${level.format} list markers render as decimal`);
|
|
235
|
-
const digits =
|
|
236
|
-
if (marker.replace(/\D+/g, "").length > (digits?.
|
|
237
|
-
return `${digits ? Number(digits
|
|
255
|
+
const digits = lastNumber(marker);
|
|
256
|
+
if (marker.replace(/\D+/g, "").length > (digits?.length ?? 0)) lose(ctx, "degraded", FEATURES.lists, "multi-level list marker flattened to one number");
|
|
257
|
+
return `${digits !== void 0 ? Number(digits) : counter}.`;
|
|
258
|
+
}
|
|
259
|
+
/**
|
|
260
|
+
* The last run of digits in a marker — what `/(\d+)\D*$/` matched, read back
|
|
261
|
+
* from the end: that expression tried every digit of a long run as its start.
|
|
262
|
+
* The template the marker comes from is the file's, so the run can be any
|
|
263
|
+
* length, and every item of the list reads it.
|
|
264
|
+
*/
|
|
265
|
+
function lastNumber(marker) {
|
|
266
|
+
const isDigit = (i) => /\d/.test(marker[i] ?? "");
|
|
267
|
+
let end = marker.length;
|
|
268
|
+
while (end > 0 && !isDigit(end - 1)) end--;
|
|
269
|
+
if (end === 0) return void 0;
|
|
270
|
+
let start = end - 1;
|
|
271
|
+
while (start > 0 && isDigit(start - 1)) start--;
|
|
272
|
+
return marker.slice(start, end);
|
|
238
273
|
}
|
|
239
274
|
function numberingLevel(numId, ilvl, numbering) {
|
|
240
275
|
if (!numbering) return void 0;
|
|
@@ -366,7 +401,7 @@ function destination(url) {
|
|
|
366
401
|
* paragraph it belongs to — the same shape GFM's own heading anchors take.
|
|
367
402
|
*/
|
|
368
403
|
function bookmarkSlug(name) {
|
|
369
|
-
const slug = name.toLowerCase().
|
|
404
|
+
const slug = name.toLowerCase().split(/[^\p{L}\p{N}]+/u).filter((word) => word.length > 0).join("-");
|
|
370
405
|
return slug.length > 0 ? slug : "bookmark";
|
|
371
406
|
}
|
|
372
407
|
/** The bookmark names some run actually links to — the only ones worth an anchor. */
|
|
@@ -684,7 +719,7 @@ var SPAN_EDGE = String.raw`(?:\\\n|\s)`;
|
|
|
684
719
|
function applyMarks(text, marks, ctx, before, after) {
|
|
685
720
|
const escaped = escapeInline(text);
|
|
686
721
|
const lead = new RegExp(`^${SPAN_EDGE}*`).exec(escaped)?.[0] ?? "";
|
|
687
|
-
const trail =
|
|
722
|
+
const trail = trailingEdge(escaped);
|
|
688
723
|
const core = escaped.slice(lead.length, escaped.length - trail.length);
|
|
689
724
|
if (core.length === 0) return escaped;
|
|
690
725
|
const outerBefore = lead.length > 0 ? lead[lead.length - 1] : before;
|
|
@@ -718,6 +753,22 @@ function applyMarks(text, marks, ctx, before, after) {
|
|
|
718
753
|
}
|
|
719
754
|
return `${lead}${linkify(out, marks, ctx)}${trail}`;
|
|
720
755
|
}
|
|
756
|
+
/**
|
|
757
|
+
* The whitespace and hard breaks the escaped text ends in — what
|
|
758
|
+
* `${SPAN_EDGE}*$` matches, read back from the end instead. Tried from the
|
|
759
|
+
* front, that expression rescans a long run of whitespace from each of its
|
|
760
|
+
* characters whenever something other than the end follows the run, and is
|
|
761
|
+
* quadratic on it.
|
|
762
|
+
*/
|
|
763
|
+
function trailingEdge(escaped) {
|
|
764
|
+
let from = escaped.length;
|
|
765
|
+
while (from > 0) {
|
|
766
|
+
const c = escaped[from - 1];
|
|
767
|
+
if (c === "\\" ? escaped[from] !== "\n" : !/\s/u.test(c)) break;
|
|
768
|
+
from--;
|
|
769
|
+
}
|
|
770
|
+
return escaped.slice(from);
|
|
771
|
+
}
|
|
721
772
|
var isSpace = (c) => c === "" || /\s/u.test(c);
|
|
722
773
|
var isPunct = (c) => /[\p{P}\p{S}]/u.test(c);
|
|
723
774
|
/** §6.2 — the delimiter run is left-flanking, so it can open emphasis. */
|
|
@@ -260,7 +260,7 @@ function freeTextContents(file, annot) {
|
|
|
260
260
|
const found = file.get(annot, "Contents");
|
|
261
261
|
const raw = found instanceof PdfHexString ? String.fromCharCode(...found.bytes) : typeof found === "string" ? found : void 0;
|
|
262
262
|
if (raw === void 0) return void 0;
|
|
263
|
-
const text = textString(raw).
|
|
263
|
+
const text = textString(raw).trimEnd();
|
|
264
264
|
return text.length > 0 ? text : void 0;
|
|
265
265
|
}
|
|
266
266
|
/**
|
|
@@ -331,7 +331,7 @@ function defaultAppearance(file, annot) {
|
|
|
331
331
|
const form = acro instanceof Map ? file.get(acro, "DA") : void 0;
|
|
332
332
|
const da = own ?? (typeof form === "string" ? form : "");
|
|
333
333
|
const tf = /\/([^\s/]+)\s+([\d.]+)\s+Tf/u.exec(da);
|
|
334
|
-
const rgb = /([\d.]+)\s+([\d.]+)\s+([\d.]+)\s+rg/u.exec(da);
|
|
334
|
+
const rgb = /(?<![\d.])([\d.]+)\s+([\d.]+)\s+([\d.]+)\s+rg/u.exec(da);
|
|
335
335
|
const gray = /(^|\s)([\d.]+)\s+g(\s|$)/u.exec(da);
|
|
336
336
|
const color = rgb ? `${num(Number(rgb[1]))} ${num(Number(rgb[2]))} ${num(Number(rgb[3]))}` : gray ? `${num(Number(gray[2]))} ${num(Number(gray[2]))} ${num(Number(gray[2]))}` : "0 0 0";
|
|
337
337
|
return {
|
|
@@ -137,7 +137,9 @@ function typedField(file, annot) {
|
|
|
137
137
|
* field whose value is "Hello World".
|
|
138
138
|
*/
|
|
139
139
|
function withoutVariableText(file, stream) {
|
|
140
|
-
const
|
|
140
|
+
const text = new TextDecoder("latin1").decode(file.streamData(stream));
|
|
141
|
+
const end = text.lastIndexOf("EMC") + 3;
|
|
142
|
+
const stripped = text.slice(0, end).replace(/\/Tx\s+BMC[\s\S]*?EMC/gu, "") + text.slice(end);
|
|
141
143
|
const dict = new Map(stream.dict);
|
|
142
144
|
dict.delete("Filter");
|
|
143
145
|
dict.delete("DecodeParms");
|
|
@@ -1047,7 +1047,7 @@ function familyOfFace(baseFont) {
|
|
|
1047
1047
|
const name = plainFace(baseFont);
|
|
1048
1048
|
const cut = /^(.+?)[-,]([^-,]+)$/u.exec(name);
|
|
1049
1049
|
const whole = cut && STYLE_WORDS.test(cut[2]) ? cut[1] : cut ? name.replace(/-/gu, " ") : GLUED_STYLE.exec(name)?.[1] ?? name;
|
|
1050
|
-
return (whole.replace(/(?:PSMT|PS|MT)$/u, "") || whole).replace(/([a-z])([A-Z])/gu, "$1 $2").replace(/([A-Z]+)([A-Z][a-z])/gu, "$1 $2").replace(/\bDeja Vu\b/u, "DejaVu").replace(/\s+/gu, " ").trim();
|
|
1050
|
+
return (whole.replace(/(?:PSMT|PS|MT)$/u, "") || whole).replace(/([a-z])([A-Z])/gu, "$1 $2").replace(/(?<![A-Z])([A-Z]+)([A-Z][a-z])/gu, "$1 $2").replace(/\bDeja Vu\b/u, "DejaVu").replace(/\s+/gu, " ").trim();
|
|
1051
1051
|
}
|
|
1052
1052
|
/**
|
|
1053
1053
|
* A face name with what producers write around it taken off: the subset tag
|
|
@@ -1585,7 +1585,7 @@ function inkSpan(runs) {
|
|
|
1585
1585
|
const last = marked[marked.length - 1];
|
|
1586
1586
|
const space = (r) => r.spaceWidthPt !== void 0 && r.spaceWidthPt > 0 ? r.spaceWidthPt : (r.fontSizePt || 10) * .25;
|
|
1587
1587
|
const lead = (/^\s*/u.exec(first.text)?.[0].length ?? 0) * space(first);
|
|
1588
|
-
const trail = (
|
|
1588
|
+
const trail = (last.text.length - last.text.trimEnd().length) * space(last);
|
|
1589
1589
|
const x = first.x + lead;
|
|
1590
1590
|
return {
|
|
1591
1591
|
x,
|
|
@@ -619,7 +619,7 @@ function collectPositionTabs(p) {
|
|
|
619
619
|
* @returns The display text, or `undefined` when this is not a MACROBUTTON.
|
|
620
620
|
*/
|
|
621
621
|
function macroButtonText(instr) {
|
|
622
|
-
const shown = /^\s*MACROBUTTON\s+\S+\s(.*)$/su.exec(instr)?.[1]?.
|
|
622
|
+
const shown = /^\s*MACROBUTTON\s+\S+\s(.*)$/su.exec(instr)?.[1]?.trimEnd();
|
|
623
623
|
return shown !== void 0 && shown !== "" ? shown : void 0;
|
|
624
624
|
}
|
|
625
625
|
function parseFieldInstr(instr) {
|
|
@@ -15,6 +15,7 @@ import { OpcPackage } from "../core/opc/package.js";
|
|
|
15
15
|
import { parseCoreProperties } from "../core/opc/core-properties.js";
|
|
16
16
|
import "../core/opc/index.js";
|
|
17
17
|
import { parseXml } from "../pptx/pptx-reader.js";
|
|
18
|
+
import { firstTagMatch } from "../core/opc/tag-scan.js";
|
|
18
19
|
import { EMPTY_NUMBERING, parseNumbering } from "./numbering-parser.js";
|
|
19
20
|
import { EMPTY_SECTION, applyAuthorIds, bodyIndexForBlock, newBlockCounter, parseBackgroundColor, parseBackgroundFill, parseCommentThreads, parseDocument, parseHeaderFooter, parseNotes, parsePeople, parseSections } from "./document-parser.js";
|
|
20
21
|
import { parseStyles } from "./styles-parser.js";
|
|
@@ -214,21 +215,26 @@ function makeDiagramResolver(pkg, partName, store) {
|
|
|
214
215
|
return resolved;
|
|
215
216
|
};
|
|
216
217
|
}
|
|
218
|
+
function partIndex(path) {
|
|
219
|
+
const dot = path.lastIndexOf(".");
|
|
220
|
+
if (dot < 0 || dot === path.length - 1) return "";
|
|
221
|
+
let from = dot;
|
|
222
|
+
while (from > 0 && path.charCodeAt(from - 1) >= 48 && path.charCodeAt(from - 1) <= 57) from--;
|
|
223
|
+
return path.slice(from, dot);
|
|
224
|
+
}
|
|
217
225
|
function siblingPart(pkg, ownerPart, dataPart, type) {
|
|
218
226
|
const own = pkg.getPartRelationships(dataPart).find((r) => r.type.endsWith(type));
|
|
219
227
|
if (own) return pkg.resolveRelatedPart(dataPart, own)?.data;
|
|
220
|
-
const
|
|
221
|
-
const want = index(dataPart);
|
|
228
|
+
const want = partIndex(dataPart);
|
|
222
229
|
const rels = pkg.getPartRelationships(ownerPart).filter((r) => r.type.endsWith(type));
|
|
223
|
-
const rel = rels.find((r) =>
|
|
230
|
+
const rel = rels.find((r) => partIndex(r.target) === want) ?? (rels.length === 1 ? rels[0] : void 0);
|
|
224
231
|
return rel ? pkg.resolveRelatedPart(ownerPart, rel)?.data : void 0;
|
|
225
232
|
}
|
|
226
233
|
function drawingFromOwner(pkg, ownerPart, dataPart) {
|
|
227
234
|
const drawings = pkg.getPartRelationships(ownerPart).filter((r) => r.type.endsWith("/diagramDrawing"));
|
|
228
235
|
if (drawings.length === 0) return void 0;
|
|
229
|
-
const
|
|
230
|
-
const
|
|
231
|
-
const rel = drawings.find((r) => index(r.target) === want) ?? (drawings.length === 1 ? drawings[0] : void 0);
|
|
236
|
+
const want = partIndex(dataPart);
|
|
237
|
+
const rel = drawings.find((r) => partIndex(r.target) === want) ?? (drawings.length === 1 ? drawings[0] : void 0);
|
|
232
238
|
return rel ? pkg.resolveRelatedPart(ownerPart, rel) : void 0;
|
|
233
239
|
}
|
|
234
240
|
function makeImageResolver(pkg, store, partName = MAIN_DOCUMENT_PART) {
|
|
@@ -377,11 +383,10 @@ function chartOwningParts(pkg, sections, mainPart) {
|
|
|
377
383
|
return parts;
|
|
378
384
|
}
|
|
379
385
|
function detectDocxLanguage(stylesData, documentData) {
|
|
380
|
-
const re = /<w:lang\b[^>]*\bw:val="([^"]+)"/;
|
|
381
386
|
const decoder = new TextDecoder();
|
|
382
387
|
for (const data of [stylesData, documentData]) {
|
|
383
388
|
if (!data) continue;
|
|
384
|
-
const m =
|
|
389
|
+
const m = firstTagMatch(decoder.decode(data), "<w:lang", true, /<w:lang\b[^>]*\bw:val="([^"]+)"/y);
|
|
385
390
|
if (m?.[1]) return m[1];
|
|
386
391
|
}
|
|
387
392
|
}
|
|
@@ -6,6 +6,7 @@ import "../pdf/index.js";
|
|
|
6
6
|
import { parseThemeFonts } from "../core/drawingml/theme-parser.js";
|
|
7
7
|
import { OpcPackage } from "../core/opc/package.js";
|
|
8
8
|
import "../core/opc/index.js";
|
|
9
|
+
import { tagMatches } from "../core/opc/tag-scan.js";
|
|
9
10
|
import { resolveWordThemeFont } from "./theme-fonts.js";
|
|
10
11
|
import { readDocx } from "./docx-reader.js";
|
|
11
12
|
import { flowRenderOptions } from "../core/converter/project.js";
|
|
@@ -13,6 +14,7 @@ import { flowRenderOptions } from "../core/converter/project.js";
|
|
|
13
14
|
var STYLES_PART = "word/styles.xml";
|
|
14
15
|
var THEME_PART = "word/theme/theme1.xml";
|
|
15
16
|
var MAIN_DOCUMENT_PART = "word/document.xml";
|
|
17
|
+
var ASCII_FONT = /<w:rFonts[^>]*?\bw:(ascii|asciiTheme)="([^"]+)"/y;
|
|
16
18
|
/**
|
|
17
19
|
* Synchronous one-shot conversion (requires `fonts`/`fontBytes`, no network).
|
|
18
20
|
*
|
|
@@ -72,13 +74,11 @@ function detectDocxFamilyKeys(docx) {
|
|
|
72
74
|
}
|
|
73
75
|
const decoder = new TextDecoder("utf-8");
|
|
74
76
|
const themeFonts = docxThemeFonts(pkg);
|
|
75
|
-
const re = /<w:rFonts[^>]*?\bw:(ascii|asciiTheme)="([^"]+)"/g;
|
|
76
77
|
for (const part of [STYLES_PART, MAIN_DOCUMENT_PART]) {
|
|
77
78
|
const data = pkg.getPart(part);
|
|
78
79
|
if (!data) continue;
|
|
79
80
|
const xml = decoder.decode(data);
|
|
80
|
-
|
|
81
|
-
while ((m = re.exec(xml)) !== null) {
|
|
81
|
+
for (const m of tagMatches(xml, "<w:rFonts", false, ASCII_FONT)) {
|
|
82
82
|
const name = m[1] === "ascii" ? m[2] : resolveWordThemeFont(m[2], themeFonts);
|
|
83
83
|
if (name !== void 0) keys.add(resolveFamilyKey(name));
|
|
84
84
|
}
|
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import { FontRegistry } from "../core/font/font-registry.js";
|
|
2
|
+
import { isWordChar, tagMatches } from "../core/opc/tag-scan.js";
|
|
2
3
|
//#region src/word/font-table.ts
|
|
3
4
|
var FONT_TABLE_PART = "word/fontTable.xml";
|
|
4
5
|
var REL_FONT = "http://schemas.openxmlformats.org/officeDocument/2006/relationships/font";
|
|
@@ -48,19 +49,15 @@ function obfuscateEmbeddedFont(data, fontKey) {
|
|
|
48
49
|
function parseFontTable(data) {
|
|
49
50
|
const xml = new TextDecoder("utf-8").decode(data);
|
|
50
51
|
const out = [];
|
|
51
|
-
const
|
|
52
|
-
|
|
53
|
-
while ((fm = fontRe.exec(xml)) !== null) {
|
|
52
|
+
const closed = xml.slice(0, xml.lastIndexOf("</w:font>") + 9);
|
|
53
|
+
for (const fm of tagMatches(closed, "<w:font", true, FONT)) {
|
|
54
54
|
const name = fm[1];
|
|
55
|
-
const inner = fm[2];
|
|
56
55
|
const embeds = {};
|
|
57
|
-
const
|
|
58
|
-
|
|
59
|
-
while ((em = embedRe.exec(inner)) !== null) {
|
|
60
|
-
const variant = VARIANT_BY_TAG[em[1]];
|
|
56
|
+
for (const em of embedRefs(fm[2])) {
|
|
57
|
+
const variant = VARIANT_BY_TAG[em.tag];
|
|
61
58
|
if (variant) embeds[variant] = {
|
|
62
|
-
rId: em
|
|
63
|
-
fontKey: em
|
|
59
|
+
rId: em.rId,
|
|
60
|
+
fontKey: em.fontKey
|
|
64
61
|
};
|
|
65
62
|
}
|
|
66
63
|
if (Object.keys(embeds).length > 0) out.push({
|
|
@@ -70,6 +67,88 @@ function parseFontTable(data) {
|
|
|
70
67
|
}
|
|
71
68
|
return out;
|
|
72
69
|
}
|
|
70
|
+
/** A named `w:font` and everything up to its close. Sticky, for tagMatches. */
|
|
71
|
+
var FONT = /<w:font\b[^>]*\bw:name="([^"]+)"[^>]*>([\s\S]*?)<\/w:font>/y;
|
|
72
|
+
var EMBED_HEAD = /<w:embed(Regular|Bold|Italic|BoldItalic)\b/y;
|
|
73
|
+
/**
|
|
74
|
+
* The faces a font's element embeds — what
|
|
75
|
+
* `/<w:embed(Regular|Bold|Italic|BoldItalic)\b[^>]*\br:id="([^"]+)"[^>]*\bw:fontKey="([^"]+)"/g`
|
|
76
|
+
* finds in it, in one pass. The expression gave the tag back one `r:id` at a
|
|
77
|
+
* time and read the rest of it again for a `w:fontKey` after each, quadratic
|
|
78
|
+
* in a tag with many; and searching, it read each tag again from every
|
|
79
|
+
* `<w:embed` inside it. Where a start fails, every later one before the tag's
|
|
80
|
+
* `>` does too, so the search moves past it.
|
|
81
|
+
*/
|
|
82
|
+
function embedRefs(inner) {
|
|
83
|
+
const out = [];
|
|
84
|
+
let at = inner.indexOf("<w:embed");
|
|
85
|
+
while (at >= 0) {
|
|
86
|
+
EMBED_HEAD.lastIndex = at;
|
|
87
|
+
const head = EMBED_HEAD.exec(inner);
|
|
88
|
+
if (head === null) {
|
|
89
|
+
at = inner.indexOf("<w:embed", at + 1);
|
|
90
|
+
continue;
|
|
91
|
+
}
|
|
92
|
+
const from = EMBED_HEAD.lastIndex;
|
|
93
|
+
const tagEnd = endOfStretch(inner, from);
|
|
94
|
+
const found = embedRef(inner, from, tagEnd);
|
|
95
|
+
if (found) {
|
|
96
|
+
out.push({
|
|
97
|
+
tag: head[1],
|
|
98
|
+
rId: found.rId,
|
|
99
|
+
fontKey: found.fontKey
|
|
100
|
+
});
|
|
101
|
+
at = inner.indexOf("<w:embed", found.end);
|
|
102
|
+
} else if (tagEnd < inner.length) at = inner.indexOf("<w:embed", tagEnd + 1);
|
|
103
|
+
else break;
|
|
104
|
+
}
|
|
105
|
+
return out;
|
|
106
|
+
}
|
|
107
|
+
/** The first `>` at or after `from`, or the end of the text. */
|
|
108
|
+
function endOfStretch(s, from) {
|
|
109
|
+
const gt = s.indexOf(">", from);
|
|
110
|
+
return gt < 0 ? s.length : gt;
|
|
111
|
+
}
|
|
112
|
+
/**
|
|
113
|
+
* The reference one `<w:embed…` makes, its attributes read from `from` up to
|
|
114
|
+
* the tag's `>` at `tagEnd`. The first `[^>]*` gives back the last `r:id`
|
|
115
|
+
* first; the second takes the last `w:fontKey` before the `>` that follows the
|
|
116
|
+
* `r:id`'s value — and that one is the same for every `r:id` whose value ends
|
|
117
|
+
* in one stretch, so it is looked for once per stretch.
|
|
118
|
+
*/
|
|
119
|
+
function embedRef(s, from, tagEnd) {
|
|
120
|
+
const keys = /* @__PURE__ */ new Map();
|
|
121
|
+
const lastKey = (stretchEnd) => {
|
|
122
|
+
if (keys.has(stretchEnd)) return keys.get(stretchEnd);
|
|
123
|
+
const start = s.lastIndexOf(">", stretchEnd - 1) + 1;
|
|
124
|
+
let key;
|
|
125
|
+
for (let f = s.lastIndexOf("w:fontKey=\"", stretchEnd - 11); f >= start; f = s.lastIndexOf("w:fontKey=\"", f - 1)) {
|
|
126
|
+
const close = s.indexOf("\"", f + 11);
|
|
127
|
+
if (!isWordChar(s.charCodeAt(f - 1)) && close > f + 11) {
|
|
128
|
+
key = {
|
|
129
|
+
at: f,
|
|
130
|
+
value: s.slice(f + 11, close),
|
|
131
|
+
end: close + 1
|
|
132
|
+
};
|
|
133
|
+
break;
|
|
134
|
+
}
|
|
135
|
+
if (f === 0) break;
|
|
136
|
+
}
|
|
137
|
+
keys.set(stretchEnd, key);
|
|
138
|
+
return key;
|
|
139
|
+
};
|
|
140
|
+
for (let p = s.lastIndexOf("r:id=\"", tagEnd - 6); p >= from; p = s.lastIndexOf("r:id=\"", p - 1)) {
|
|
141
|
+
if (isWordChar(s.charCodeAt(p - 1))) continue;
|
|
142
|
+
const close = s.indexOf("\"", p + 6);
|
|
143
|
+
if (close <= p + 6) continue;
|
|
144
|
+
const key = lastKey(close < tagEnd ? tagEnd : endOfStretch(s, close + 1));
|
|
145
|
+
if (key !== void 0 && key.at > close) return {
|
|
146
|
+
rId: s.slice(p + 6, close),
|
|
147
|
+
fontKey: key.value,
|
|
148
|
+
end: key.end
|
|
149
|
+
};
|
|
150
|
+
}
|
|
151
|
+
}
|
|
73
152
|
/**
|
|
74
153
|
* De-obfuscate and load every embedded font in the package, building one
|
|
75
154
|
* {@link FontRegistry} per font keyed by its normalized (trimmed, lower-cased)
|
|
@@ -171,7 +171,7 @@ var CSS_UNITS = new Map([
|
|
|
171
171
|
["px", .75]
|
|
172
172
|
]);
|
|
173
173
|
function cssLengthPt(raw) {
|
|
174
|
-
const m = raw !== void 0 ?
|
|
174
|
+
const m = raw !== void 0 ? /^(-?[\d.]+)\s*([a-z%]*)$/u.exec(raw.trim()) : null;
|
|
175
175
|
if (!m) return void 0;
|
|
176
176
|
const n = Number(m[1]);
|
|
177
177
|
if (!Number.isFinite(n)) return void 0;
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "reamkit",
|
|
3
|
-
"version": "1.32.
|
|
3
|
+
"version": "1.32.1",
|
|
4
4
|
"description": "Ream — convert DOCX, XLSX, PPTX, PDF and legacy DOC, XLS and PPT to PDF, SVG, HTML, Markdown, DOCX and XLSX, built from scratch on the ECMA-376 and ISO 32000 specifications. Parse once, convert anywhere.",
|
|
5
5
|
"license": "MIT",
|
|
6
6
|
"author": "Alex Krassavin <info@reamkit.dev>",
|
|
@@ -91,7 +91,7 @@
|
|
|
91
91
|
"typecheck:dist": "tsc -p tsconfig.json --noEmit"
|
|
92
92
|
},
|
|
93
93
|
"dependencies": {
|
|
94
|
-
"fast-xml-parser": "5.7.
|
|
94
|
+
"fast-xml-parser": "5.7.3",
|
|
95
95
|
"fflate": "^0.8.2"
|
|
96
96
|
},
|
|
97
97
|
"devDependencies": {
|
|
@@ -103,7 +103,7 @@
|
|
|
103
103
|
"tsx": "^4.19.0",
|
|
104
104
|
"typescript": "^5.7.0",
|
|
105
105
|
"vite": "8.0.5",
|
|
106
|
-
"vitest": "^
|
|
106
|
+
"vitest": "^5.0.3"
|
|
107
107
|
},
|
|
108
108
|
"engines": {
|
|
109
109
|
"node": ">=20"
|