reamkit 1.15.2 → 1.15.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/esm/core/bytes.d.ts +11 -0
- package/dist/esm/core/bytes.js +15 -1
- package/dist/esm/core/opc/package.js +3 -1
- package/dist/esm/core/spreadsheet-model/types.d.ts +21 -0
- package/dist/esm/excel/column-bands.js +1 -7
- package/dist/esm/excel/print-model.d.ts +26 -6
- package/dist/esm/excel/print-model.js +189 -28
- package/dist/esm/excel/sheet-to-flow.d.ts +10 -0
- package/dist/esm/excel/sheet-to-flow.js +14 -7
- package/dist/esm/excel/worksheet-parser.js +30 -6
- package/dist/esm/excel/xlsx-reader.d.ts +7 -3
- package/dist/esm/excel/xlsx-reader.js +15 -7
- package/dist/esm/layout/styled-layout.js +20 -3
- package/dist/esm/pptx/pptx-reader.js +2 -2
- package/dist/esm/word/docx-reader.js +2 -2
- package/package.json +5 -2
package/dist/esm/core/bytes.d.ts
CHANGED
|
@@ -8,4 +8,15 @@ export declare function toBase64(bytes: Uint8Array): string;
|
|
|
8
8
|
* Naive scan for an ASCII `needle` inside raw `haystack` bytes — used by reader
|
|
9
9
|
* sniffs to spot OPC part names (e.g. `'word/document.xml'`) without unzipping.
|
|
10
10
|
*/
|
|
11
|
+
/**
|
|
12
|
+
* Scan raw package bytes for an OPC part name (e.g. `'xl/workbook.xml'`)
|
|
13
|
+
* without unzipping — the reader sniffs' cheap format probe. Accepts both the
|
|
14
|
+
* spec's `/` separator and the `\` that Windows producers write.
|
|
15
|
+
*/
|
|
16
|
+
export declare function bytesIncludePartName(haystack: Uint8Array, partName: string): boolean;
|
|
17
|
+
/**
|
|
18
|
+
* Naive scan for an ASCII `needle` inside raw `haystack` bytes. Prefer
|
|
19
|
+
* {@link bytesIncludePartName} for OPC part names — it also accepts the
|
|
20
|
+
* backslash spelling real archives use.
|
|
21
|
+
*/
|
|
11
22
|
export declare function bytesInclude(haystack: Uint8Array, needle: string): boolean;
|
package/dist/esm/core/bytes.js
CHANGED
|
@@ -13,6 +13,20 @@ function toBase64(bytes) {
|
|
|
13
13
|
* Naive scan for an ASCII `needle` inside raw `haystack` bytes — used by reader
|
|
14
14
|
* sniffs to spot OPC part names (e.g. `'word/document.xml'`) without unzipping.
|
|
15
15
|
*/
|
|
16
|
+
/**
|
|
17
|
+
* Scan raw package bytes for an OPC part name (e.g. `'xl/workbook.xml'`)
|
|
18
|
+
* without unzipping — the reader sniffs' cheap format probe. Accepts both the
|
|
19
|
+
* spec's `/` separator and the `\` that Windows producers write.
|
|
20
|
+
*/
|
|
21
|
+
function bytesIncludePartName(haystack, partName) {
|
|
22
|
+
if (bytesInclude(haystack, partName)) return true;
|
|
23
|
+
return partName.includes("/") && bytesInclude(haystack, partName.replace(/\//g, "\\"));
|
|
24
|
+
}
|
|
25
|
+
/**
|
|
26
|
+
* Naive scan for an ASCII `needle` inside raw `haystack` bytes. Prefer
|
|
27
|
+
* {@link bytesIncludePartName} for OPC part names — it also accepts the
|
|
28
|
+
* backslash spelling real archives use.
|
|
29
|
+
*/
|
|
16
30
|
function bytesInclude(haystack, needle) {
|
|
17
31
|
const n = new TextEncoder().encode(needle);
|
|
18
32
|
outer: for (let i = 0; i + n.length <= haystack.length; i++) {
|
|
@@ -22,4 +36,4 @@ function bytesInclude(haystack, needle) {
|
|
|
22
36
|
return false;
|
|
23
37
|
}
|
|
24
38
|
//#endregion
|
|
25
|
-
export { bytesInclude, toBase64 };
|
|
39
|
+
export { bytesInclude, bytesIncludePartName, toBase64 };
|
|
@@ -42,6 +42,7 @@ var OpcPackage = class OpcPackage {
|
|
|
42
42
|
violation ??= `more than ${maxEntries} entries`;
|
|
43
43
|
return false;
|
|
44
44
|
}
|
|
45
|
+
if (info.originalSize === 4294967295) return true;
|
|
45
46
|
if (info.originalSize > maxEntry) {
|
|
46
47
|
violation ??= `entry "${info.name}" declares ${info.originalSize} bytes (limit ${maxEntry})`;
|
|
47
48
|
return false;
|
|
@@ -138,7 +139,8 @@ var OpcPackage = class OpcPackage {
|
|
|
138
139
|
}
|
|
139
140
|
};
|
|
140
141
|
function normalizePath(p) {
|
|
141
|
-
|
|
142
|
+
const slashed = p.includes("\\") ? p.replace(/\\/g, "/") : p;
|
|
143
|
+
return slashed.startsWith("/") ? slashed.slice(1) : slashed;
|
|
142
144
|
}
|
|
143
145
|
function resolveTarget(sourcePath, target) {
|
|
144
146
|
if (target.startsWith("/")) return normalizePath(target);
|
|
@@ -83,6 +83,27 @@ export interface ParsedWorksheet {
|
|
|
83
83
|
readonly columns: ReadonlyArray<ColumnWidth>;
|
|
84
84
|
readonly merges: ReadonlyArray<MergedRange>;
|
|
85
85
|
readonly rowHeights: ReadonlyArray<RowHeight>;
|
|
86
|
+
/**
|
|
87
|
+
* ECMA-376 §18.3.1.81 `<sheetFormatPr defaultRowHeight>` — the height, in
|
|
88
|
+
* points, of every row that carries no `ht` of its own. A row in a
|
|
89
|
+
* spreadsheet always has a definite height; it is not decided by whatever
|
|
90
|
+
* leading the text happens to want. Absent ⇒ Excel's 15pt (its default theme
|
|
91
|
+
* font, Calibri 11).
|
|
92
|
+
*/
|
|
93
|
+
readonly defaultRowHeightPt?: number;
|
|
94
|
+
/**
|
|
95
|
+
* §18.3.1.81 `<sheetFormatPr defaultColWidth>` — the width, in characters, of
|
|
96
|
+
* every column no `<col>` covers. Absent ⇒ derived from
|
|
97
|
+
* {@link ParsedWorksheet.baseColWidthChars}, else Excel's 8.43.
|
|
98
|
+
*/
|
|
99
|
+
readonly defaultColWidthChars?: number;
|
|
100
|
+
/**
|
|
101
|
+
* §18.3.1.81 `<sheetFormatPr baseColWidth>` — the character width the default
|
|
102
|
+
* column is computed FROM when `defaultColWidth` is absent. Excel's own
|
|
103
|
+
* default is 8; a sheet that says 10 wants every unlisted column that much
|
|
104
|
+
* wider, and ignoring it makes the whole grid narrow.
|
|
105
|
+
*/
|
|
106
|
+
readonly baseColWidthChars?: number;
|
|
86
107
|
readonly pageMargins?: XlsxPageMargins;
|
|
87
108
|
readonly pageSetup?: XlsxPageSetup;
|
|
88
109
|
/**
|
|
@@ -100,13 +100,7 @@ function withColSpan(cell, span) {
|
|
|
100
100
|
function blankCell(span) {
|
|
101
101
|
return {
|
|
102
102
|
properties: span > 1 ? { colSpan: span } : {},
|
|
103
|
-
content: [
|
|
104
|
-
kind: "paragraph",
|
|
105
|
-
paragraph: {
|
|
106
|
-
properties: {},
|
|
107
|
-
runs: []
|
|
108
|
-
}
|
|
109
|
-
}]
|
|
103
|
+
content: []
|
|
110
104
|
};
|
|
111
105
|
}
|
|
112
106
|
//#endregion
|
|
@@ -1,12 +1,25 @@
|
|
|
1
1
|
import { BodyElement, SectionProperties, Table } from '../core/document-model/index.js';
|
|
2
2
|
import { CellRange, DefinedName, ParsedWorksheet, SheetRichRun, WorksheetCell, XlsxStyles } from './index.js';
|
|
3
3
|
import { SheetHyperlink, SheetSlicer } from '../core/ir/sheet.js';
|
|
4
|
+
import { Loss } from '../core/ir/loss.js';
|
|
4
5
|
/**
|
|
5
|
-
* Excel "character width" → twips
|
|
6
|
-
* ~7 px ≈ 5.25 pt ≈ 105 twips.
|
|
7
|
-
* pass refines column widths against the actual cell text anyway.
|
|
6
|
+
* Excel "character width" → twips: the default font's Maximum Digit Width,
|
|
7
|
+
* ~7 px at 96 DPI ≈ 5.25 pt ≈ 105 twips.
|
|
8
8
|
*/
|
|
9
9
|
export declare const TWIPS_PER_EXCEL_CHAR = 105;
|
|
10
|
+
/**
|
|
11
|
+
* ECMA-376 §18.3.1.13 — a `<col width>` measures characters of text, and the
|
|
12
|
+
* rendered column is that many Maximum Digit Widths PLUS a fixed 5-pixel
|
|
13
|
+
* padding: `px = chars × MDW + 5`. 5 px at 96 DPI = 3.75 pt = 75 twips.
|
|
14
|
+
*
|
|
15
|
+
* Omitting it made every column 75 twips narrow, which compounds: on a sheet of
|
|
16
|
+
* equal 12-character columns the third one landed ~37 pt left of where
|
|
17
|
+
* LibreOffice puts it. That {@link DEFAULT_COL_TWIPS} below is 960 is the proof
|
|
18
|
+
* the padding belongs — 8.43 characters only reaches Excel's documented 64 px
|
|
19
|
+
* default with it (8.43 × 7 + 5 = 64.01 px = 48.01 pt = 960 twips), so the
|
|
20
|
+
* default was derived from the full formula while the explicit path dropped it.
|
|
21
|
+
*/
|
|
22
|
+
export declare const COL_PADDING_TWIPS = 75;
|
|
10
23
|
/**
|
|
11
24
|
* Excel's default column width is 8.43 "characters" ≈ 64px ≈ 960 twips. Used for
|
|
12
25
|
* columns without an explicit `<col width="..">`.
|
|
@@ -19,10 +32,16 @@ export declare const DEFAULT_COL_TWIPS = 960;
|
|
|
19
32
|
export declare const DEFAULT_ROW_TWIPS = 300;
|
|
20
33
|
/**
|
|
21
34
|
* Build the page section (paper size + margins) from a worksheet's `<pageSetup>`
|
|
22
|
-
* / `<pageMargins>`.
|
|
23
|
-
*
|
|
35
|
+
* / `<pageMargins>`.
|
|
36
|
+
*
|
|
37
|
+
* Margins are always set, to Excel's own defaults when the worksheet declares
|
|
38
|
+
* none (§18.3.1.62) — the renderer's fallback is a word processor's inch, which
|
|
39
|
+
* is not what a spreadsheet prints. The paper size is left unset when the
|
|
40
|
+
* worksheet names none, because there the file genuinely holds no answer: Excel
|
|
41
|
+
* picks by locale and printer, and the renderer's deterministic A4 is as good
|
|
42
|
+
* as anything we could invent.
|
|
24
43
|
*/
|
|
25
|
-
export declare function sectionFromWorksheet(worksheet: ParsedWorksheet): SectionProperties
|
|
44
|
+
export declare function sectionFromWorksheet(worksheet: ParsedWorksheet): SectionProperties;
|
|
26
45
|
/**
|
|
27
46
|
* ECMA-376 §18.2.5 — resolve the sheet-scoped `_xlnm.Print_Area` defined name
|
|
28
47
|
* (`localSheetId` = the sheet's 0-based index) into a clipping range.
|
|
@@ -46,6 +65,7 @@ interface PrintModelOptions {
|
|
|
46
65
|
readonly hyperlinks?: ReadonlyArray<SheetHyperlink>;
|
|
47
66
|
readonly sharedStringRuns?: ReadonlyArray<ReadonlyArray<SheetRichRun> | undefined>;
|
|
48
67
|
readonly now?: Date;
|
|
68
|
+
readonly losses?: Array<Loss>;
|
|
49
69
|
}
|
|
50
70
|
/**
|
|
51
71
|
* Project one worksheet's grid into Flow body elements — a single {@link Table}
|
|
@@ -1,9 +1,23 @@
|
|
|
1
1
|
import { eighthPtToPt, halfPtToPt, pt, twipsToPt } from "../core/ir/units.js";
|
|
2
|
+
import { FEATURES } from "../core/ir/features.js";
|
|
2
3
|
import { parseAreaRef, parseTitleRowRange } from "./defined-name-ref.js";
|
|
3
4
|
import { applyNumberFormat } from "./number-format.js";
|
|
4
5
|
import "./index.js";
|
|
5
6
|
import { bandedTables, computeColumnBands } from "./column-bands.js";
|
|
6
7
|
import { buildConditionalFormatter } from "./conditional-format.js";
|
|
8
|
+
/**
|
|
9
|
+
* Excel insets a cell's text by ~2 px each side (1.5 pt at 96 DPI). The layout
|
|
10
|
+
* engine's default is a word processor's 108 twips (5.4 pt), which is nearly
|
|
11
|
+
* four times as much.
|
|
12
|
+
*/
|
|
13
|
+
var EXCEL_CELL_INSET_PT = 1.5;
|
|
14
|
+
/**
|
|
15
|
+
* ECMA-376 §18.3.1.81 — the row height Excel uses when a sheet declares no
|
|
16
|
+
* `<sheetFormatPr defaultRowHeight>`: 15pt, the line height of its default
|
|
17
|
+
* theme font (Calibri 11). Independent of whatever font we end up rendering
|
|
18
|
+
* with — the height belongs to the document, not to the typesetter.
|
|
19
|
+
*/
|
|
20
|
+
var EXCEL_DEFAULT_ROW_HEIGHT_PT = 15;
|
|
7
21
|
var TWIPS_PER_POINT = 20;
|
|
8
22
|
var TWIPS_PER_INCH = 1440;
|
|
9
23
|
var PAPER_SIZES_TWIPS = new Map([
|
|
@@ -18,16 +32,20 @@ var PAPER_SIZES_TWIPS = new Map([
|
|
|
18
32
|
var DEFAULT_PAPER_TWIPS = [11906, 16838];
|
|
19
33
|
/**
|
|
20
34
|
* Build the page section (paper size + margins) from a worksheet's `<pageSetup>`
|
|
21
|
-
* / `<pageMargins>`.
|
|
22
|
-
*
|
|
35
|
+
* / `<pageMargins>`.
|
|
36
|
+
*
|
|
37
|
+
* Margins are always set, to Excel's own defaults when the worksheet declares
|
|
38
|
+
* none (§18.3.1.62) — the renderer's fallback is a word processor's inch, which
|
|
39
|
+
* is not what a spreadsheet prints. The paper size is left unset when the
|
|
40
|
+
* worksheet names none, because there the file genuinely holds no answer: Excel
|
|
41
|
+
* picks by locale and printer, and the renderer's deterministic A4 is as good
|
|
42
|
+
* as anything we could invent.
|
|
23
43
|
*/
|
|
24
44
|
function sectionFromWorksheet(worksheet) {
|
|
25
45
|
const pageSize = pageSizeFromSetup(worksheet.pageSetup);
|
|
26
|
-
const margins = marginsFromXlsx(worksheet.pageMargins);
|
|
27
|
-
if (!pageSize && !margins) return void 0;
|
|
28
46
|
return {
|
|
29
47
|
...pageSize ? { pageSize } : {},
|
|
30
|
-
|
|
48
|
+
margins: marginsFromXlsx(worksheet.pageMargins),
|
|
31
49
|
headers: [],
|
|
32
50
|
footers: []
|
|
33
51
|
};
|
|
@@ -44,8 +62,23 @@ function pageSizeFromSetup(setup) {
|
|
|
44
62
|
orientation
|
|
45
63
|
};
|
|
46
64
|
}
|
|
65
|
+
/**
|
|
66
|
+
* ECMA-376 §18.3.1.62 — the page margins Excel writes when the user has not
|
|
67
|
+
* touched them, in inches. A worksheet may omit `<pageMargins>` entirely, and
|
|
68
|
+
* falling through to the renderer's default (a word processor's 1 inch) put the
|
|
69
|
+
* grid 0.3 inch — 21.6 pt — right of where Excel and LibreOffice print it, on
|
|
70
|
+
* every such sheet.
|
|
71
|
+
*/
|
|
72
|
+
var EXCEL_DEFAULT_MARGINS = {
|
|
73
|
+
left: twipsToPt(Math.round(.7 * TWIPS_PER_INCH)),
|
|
74
|
+
right: twipsToPt(Math.round(.7 * TWIPS_PER_INCH)),
|
|
75
|
+
top: twipsToPt(Math.round(.75 * TWIPS_PER_INCH)),
|
|
76
|
+
bottom: twipsToPt(Math.round(.75 * TWIPS_PER_INCH)),
|
|
77
|
+
header: twipsToPt(Math.round(.3 * TWIPS_PER_INCH)),
|
|
78
|
+
footer: twipsToPt(Math.round(.3 * TWIPS_PER_INCH))
|
|
79
|
+
};
|
|
47
80
|
function marginsFromXlsx(margins) {
|
|
48
|
-
if (!margins) return
|
|
81
|
+
if (!margins) return EXCEL_DEFAULT_MARGINS;
|
|
49
82
|
return {
|
|
50
83
|
top: twipsToPt(Math.round(margins.topInches * TWIPS_PER_INCH)),
|
|
51
84
|
right: twipsToPt(Math.round(margins.rightInches * TWIPS_PER_INCH)),
|
|
@@ -73,16 +106,16 @@ function sheetContentWidthTwips(worksheet) {
|
|
|
73
106
|
const pageSize = pageSizeFromSetup(worksheet.pageSetup);
|
|
74
107
|
const pageWidthTwips = pageSize ? Math.round(pageSize.width * 20) : DEFAULT_PAPER_TWIPS[0];
|
|
75
108
|
const margins = marginsFromXlsx(worksheet.pageMargins);
|
|
76
|
-
const left =
|
|
77
|
-
const right =
|
|
109
|
+
const left = Math.round(margins.left * 20);
|
|
110
|
+
const right = Math.round(margins.right * 20);
|
|
78
111
|
return Math.max(TWIPS_PER_INCH / 2, pageWidthTwips - left - right);
|
|
79
112
|
}
|
|
80
113
|
function sheetContentHeightTwips(worksheet) {
|
|
81
114
|
const pageSize = pageSizeFromSetup(worksheet.pageSetup);
|
|
82
115
|
const pageHeightTwips = pageSize ? Math.round(pageSize.height * 20) : DEFAULT_PAPER_TWIPS[1];
|
|
83
116
|
const margins = marginsFromXlsx(worksheet.pageMargins);
|
|
84
|
-
const top =
|
|
85
|
-
const bottom =
|
|
117
|
+
const top = Math.round(margins.top * 20);
|
|
118
|
+
const bottom = Math.round(margins.bottom * 20);
|
|
86
119
|
return Math.max(TWIPS_PER_INCH / 2, pageHeightTwips - top - bottom);
|
|
87
120
|
}
|
|
88
121
|
function computePrintScale(worksheet, totalGridTwips, contentWidthTwips, totalGridHeightTwips, contentHeightTwips) {
|
|
@@ -154,11 +187,34 @@ function worksheetToBody(worksheet, sharedStrings, styles, date1904, print) {
|
|
|
154
187
|
}
|
|
155
188
|
if (rowEnd < rowStart || colEnd < colStart) return [];
|
|
156
189
|
const MAX_GRID_COLS = 1024;
|
|
157
|
-
const
|
|
158
|
-
const
|
|
190
|
+
const MAX_GRID_ROWS = 5e4;
|
|
191
|
+
const MAX_GRID_CELLS = 1e6;
|
|
192
|
+
const wantRows = rowEnd - rowStart + 1;
|
|
193
|
+
const wantCols = colEnd - colStart + 1;
|
|
194
|
+
let colCount = Math.min(wantCols, MAX_GRID_COLS);
|
|
195
|
+
let rowCount = Math.max(1, Math.min(wantRows, MAX_GRID_ROWS, Math.floor(MAX_GRID_CELLS / colCount)));
|
|
196
|
+
if (rowCount < wantRows) rowCount = Math.max(1, lastContentRow(worksheet, rowStart, rowCount, colStart, colCount) + 1);
|
|
197
|
+
if (colCount < wantCols) colCount = Math.max(1, lastContentColumn(worksheet, rowStart, rowCount, colStart, colCount) + 1);
|
|
198
|
+
if (rowCount < wantRows) print.losses?.push({
|
|
199
|
+
severity: "dropped",
|
|
200
|
+
feature: FEATURES.tables,
|
|
201
|
+
detail: `grid clipped to the first ${rowCount} rows of ${wantRows} in the used range (memory guard)`,
|
|
202
|
+
...print.sheetName ? { where: `sheet "${print.sheetName}"` } : {}
|
|
203
|
+
});
|
|
204
|
+
if (colCount < wantCols) print.losses?.push({
|
|
205
|
+
severity: "dropped",
|
|
206
|
+
feature: FEATURES.tables,
|
|
207
|
+
detail: `grid clipped to the first ${colCount} columns of ${wantCols} in the used range (memory guard)`,
|
|
208
|
+
...print.sheetName ? { where: `sheet "${print.sheetName}"` } : {}
|
|
209
|
+
});
|
|
159
210
|
const rowWindowEnd = rowStart + rowCount - 1;
|
|
160
211
|
const colWindowEnd = colStart + colCount - 1;
|
|
212
|
+
const defaultRowTwips = Math.round((worksheet.defaultRowHeightPt ?? EXCEL_DEFAULT_ROW_HEIGHT_PT) * TWIPS_PER_POINT);
|
|
161
213
|
const rowHeightMap = /* @__PURE__ */ new Map();
|
|
214
|
+
for (let r = 0; r < rowCount; r++) rowHeightMap.set(r, {
|
|
215
|
+
heightTwips: defaultRowTwips,
|
|
216
|
+
heightRule: "atLeast"
|
|
217
|
+
});
|
|
162
218
|
for (const h of worksheet.rowHeights) {
|
|
163
219
|
const local = h.row - rowStart;
|
|
164
220
|
if (local < 0 || local >= rowCount) continue;
|
|
@@ -184,9 +240,11 @@ function worksheetToBody(worksheet, sharedStrings, styles, date1904, print) {
|
|
|
184
240
|
const cEnd = Math.min(m.endColumn, colWindowEnd);
|
|
185
241
|
for (let r = Math.max(m.startRow, rowStart); r <= rEnd; r++) for (let c = Math.max(m.startColumn, colStart); c <= cEnd; c++) if (!(r === m.startRow && c === m.startColumn)) insideMerge.add(key(r, c));
|
|
186
242
|
}
|
|
187
|
-
const
|
|
243
|
+
const defaultColChars = worksheet.defaultColWidthChars ?? worksheet.baseColWidthChars;
|
|
244
|
+
const defaultColTwips = defaultColChars !== void 0 ? Math.round(defaultColChars * 105 + 75) : 960;
|
|
245
|
+
const columnWidths = new Array(colCount).fill(defaultColTwips);
|
|
188
246
|
for (const col of worksheet.columns) {
|
|
189
|
-
const twips = Math.round(col.widthChars * 105);
|
|
247
|
+
const twips = Math.round(col.widthChars * 105 + 75);
|
|
190
248
|
for (let abs = col.min - 1; abs <= col.max - 1; abs++) {
|
|
191
249
|
const i = abs - colStart;
|
|
192
250
|
if (i >= 0 && i < colCount) columnWidths[i] = twips;
|
|
@@ -199,6 +257,7 @@ function worksheetToBody(worksheet, sharedStrings, styles, date1904, print) {
|
|
|
199
257
|
const scaled = printScale < .999;
|
|
200
258
|
const breakRows = new Set(worksheet.rowBreaks ?? []);
|
|
201
259
|
let textBudget = MAX_SHEET_TEXT_CHARS;
|
|
260
|
+
let textBudgetReported = false;
|
|
202
261
|
const runPropsByXf = /* @__PURE__ */ new Map();
|
|
203
262
|
const cellRunProps = (xf) => {
|
|
204
263
|
let props = runPropsByXf.get(xf);
|
|
@@ -226,7 +285,9 @@ function worksheetToBody(worksheet, sharedStrings, styles, date1904, print) {
|
|
|
226
285
|
for (let r = 0; r < rowCount; r++) {
|
|
227
286
|
const absR = r + rowStart;
|
|
228
287
|
const cells = [];
|
|
288
|
+
const overflowed = /* @__PURE__ */ new Set();
|
|
229
289
|
for (let c = 0; c < colCount; c++) {
|
|
290
|
+
if (overflowed.has(c)) continue;
|
|
230
291
|
const absC = c + colStart;
|
|
231
292
|
const merge = mergeOrigins.get(key(absR, absC));
|
|
232
293
|
if (insideMerge.has(key(absR, absC)) && !merge) {
|
|
@@ -237,11 +298,22 @@ function worksheetToBody(worksheet, sharedStrings, styles, date1904, print) {
|
|
|
237
298
|
const ws = cellMatrix[r]?.[c];
|
|
238
299
|
let text = ws ? resolveCellText(ws, sharedStrings, styles, date1904) : "";
|
|
239
300
|
const cfText = text.length > 0 ? text : void 0;
|
|
240
|
-
if (text.length > textBudget)
|
|
301
|
+
if (text.length > textBudget) {
|
|
302
|
+
text = text.slice(0, Math.max(0, textBudget));
|
|
303
|
+
if (!textBudgetReported) {
|
|
304
|
+
textBudgetReported = true;
|
|
305
|
+
print.losses?.push({
|
|
306
|
+
severity: "dropped",
|
|
307
|
+
feature: FEATURES.text,
|
|
308
|
+
detail: `per-sheet text budget of ${MAX_SHEET_TEXT_CHARS} characters exhausted; the remaining cells render empty`,
|
|
309
|
+
...print.sheetName ? { where: `sheet "${print.sheetName}"` } : {}
|
|
310
|
+
});
|
|
311
|
+
}
|
|
312
|
+
}
|
|
241
313
|
textBudget -= text.length;
|
|
242
314
|
const xf = ws && ws.styleIndex !== void 0 ? styles.cellXfs[ws.styleIndex] : void 0;
|
|
243
315
|
let runProps = cellRunProps(xf);
|
|
244
|
-
const alignment =
|
|
316
|
+
const alignment = alignmentFromXf(xf, ws?.type);
|
|
245
317
|
let shading = xf ? shadingFromXf(xf, styles) : void 0;
|
|
246
318
|
const tableFmt = tableFormatByCell.get(key(absR, absC));
|
|
247
319
|
if (!shading && tableFmt?.shading) shading = tableFmt.shading;
|
|
@@ -262,14 +334,16 @@ function worksheetToBody(worksheet, sharedStrings, styles, date1904, print) {
|
|
|
262
334
|
runProps = applyCfOverride(runProps, over);
|
|
263
335
|
}
|
|
264
336
|
}
|
|
337
|
+
let overflowSpan = 1;
|
|
265
338
|
const wrapText = xf?.alignment?.wrapText === true;
|
|
266
339
|
const rotation = xf?.alignment?.textRotation;
|
|
267
340
|
const rotated = rotation !== void 0 && rotation !== 0 && text.length > 0;
|
|
268
341
|
const shrinkToFit = xf?.alignment?.shrinkToFit === true && !wrapText && !rotated;
|
|
269
|
-
if (text.length > 0 && !merge && !wrapText && !rotated && !shrinkToFit && ws && (ws.type === "s" || ws.type === "str" || ws.type === "inlineStr") &&
|
|
342
|
+
if (text.length > 0 && !merge && !wrapText && !rotated && !shrinkToFit && ws && (ws.type === "s" || ws.type === "str" || ws.type === "inlineStr") && alignment === "left") {
|
|
270
343
|
let availTwips = columnWidths[c];
|
|
271
344
|
let cc = c + 1;
|
|
272
|
-
|
|
345
|
+
const neighbourIsFree = (col) => !cellHasContent(cellMatrix[r]?.[col]) && !cellPaintsSomething(cellMatrix[r]?.[col], styles) && !sparklineByCell.has(key(absR, col + colStart)) && !(dropdownRanges.length > 0 && rangesCover(dropdownRanges, absR, col + colStart));
|
|
346
|
+
while (cc < colCount && neighbourIsFree(cc)) {
|
|
273
347
|
availTwips += columnWidths[cc];
|
|
274
348
|
cc++;
|
|
275
349
|
}
|
|
@@ -277,6 +351,10 @@ function worksheetToBody(worksheet, sharedStrings, styles, date1904, print) {
|
|
|
277
351
|
const charsFit = Math.max(1, Math.round(availTwips / 105));
|
|
278
352
|
if (text.length > charsFit) text = text.slice(0, charsFit);
|
|
279
353
|
}
|
|
354
|
+
if (cc > c + 1) {
|
|
355
|
+
overflowSpan = cc - c;
|
|
356
|
+
for (let k = c + 1; k < cc; k++) overflowed.add(k);
|
|
357
|
+
}
|
|
280
358
|
}
|
|
281
359
|
if (shrinkToFit && ws && text.length > 0) {
|
|
282
360
|
const charsFit = Math.max(1, columnWidths[c] / 105);
|
|
@@ -293,7 +371,7 @@ function worksheetToBody(worksheet, sharedStrings, styles, date1904, print) {
|
|
|
293
371
|
const href = print.hyperlinks && print.hyperlinks.length > 0 ? hyperlinkUrlAt(print.hyperlinks, absR, absC) : void 0;
|
|
294
372
|
const visibleEndCol = merge ? Math.min(merge.endColumn, colWindowEnd) : 0;
|
|
295
373
|
const properties = {
|
|
296
|
-
...merge && visibleEndCol > merge.startColumn ? { colSpan: visibleEndCol - merge.startColumn + 1 } : {},
|
|
374
|
+
...merge && visibleEndCol > merge.startColumn ? { colSpan: visibleEndCol - merge.startColumn + 1 } : overflowSpan > 1 ? { colSpan: overflowSpan } : {},
|
|
297
375
|
...merge && Math.min(merge.endRow, rowWindowEnd) > merge.startRow ? { merge: "start" } : {},
|
|
298
376
|
...shading ? { shading } : {},
|
|
299
377
|
...dataBar ? { dataBar } : {},
|
|
@@ -318,7 +396,7 @@ function worksheetToBody(worksheet, sharedStrings, styles, date1904, print) {
|
|
|
318
396
|
properties: runProps,
|
|
319
397
|
...href ? { href } : {}
|
|
320
398
|
}] : [];
|
|
321
|
-
const content = rotated ? stackedVerticalContent(text, runProps, href) : [{
|
|
399
|
+
const content = rotated ? stackedVerticalContent(text, runProps, href) : cellRuns.length === 0 ? [] : [{
|
|
322
400
|
kind: "paragraph",
|
|
323
401
|
paragraph: {
|
|
324
402
|
properties: paragraphProps,
|
|
@@ -352,6 +430,13 @@ function worksheetToBody(worksheet, sharedStrings, styles, date1904, print) {
|
|
|
352
430
|
};
|
|
353
431
|
const centered = worksheet.printOptions?.horizontalCentered === true;
|
|
354
432
|
const tableProperties = {
|
|
433
|
+
defaultCellMargins: {
|
|
434
|
+
left: pt(EXCEL_CELL_INSET_PT),
|
|
435
|
+
right: pt(EXCEL_CELL_INSET_PT),
|
|
436
|
+
top: pt(0),
|
|
437
|
+
bottom: pt(0)
|
|
438
|
+
},
|
|
439
|
+
layout: "fixed",
|
|
355
440
|
...print.gridLines ? { borders: {
|
|
356
441
|
top: thin,
|
|
357
442
|
bottom: thin,
|
|
@@ -386,7 +471,7 @@ function worksheetToBody(worksheet, sharedStrings, styles, date1904, print) {
|
|
|
386
471
|
...tableProperties,
|
|
387
472
|
frozen
|
|
388
473
|
} : tableProperties,
|
|
389
|
-
grid:
|
|
474
|
+
grid: bandWidths.map((w) => twipsToPt(w)),
|
|
390
475
|
rows
|
|
391
476
|
}
|
|
392
477
|
}];
|
|
@@ -477,10 +562,29 @@ function applyCfOverride(base, o) {
|
|
|
477
562
|
...o.italic !== void 0 ? { italic: o.italic } : {}
|
|
478
563
|
};
|
|
479
564
|
}
|
|
480
|
-
|
|
481
|
-
|
|
482
|
-
|
|
483
|
-
|
|
565
|
+
/**
|
|
566
|
+
* ECMA-376 §18.8.1 — a cell's horizontal alignment, falling back to `general`.
|
|
567
|
+
*
|
|
568
|
+
* "General" is not "left": it is decided by the VALUE. Numbers, dates and times
|
|
569
|
+
* go right, booleans and errors centre, text goes left. Treating an absent
|
|
570
|
+
* `<alignment>` as no alignment at all left every number hugging the left edge
|
|
571
|
+
* of its column, tens of points from where Excel and LibreOffice put it — on
|
|
572
|
+
* the sheets where it matters most, since a column of figures is the common
|
|
573
|
+
* case.
|
|
574
|
+
*/
|
|
575
|
+
function alignmentFromXf(xf, type) {
|
|
576
|
+
const explicit = xf?.alignment ? mapAlignment(xf.alignment.horizontal) : void 0;
|
|
577
|
+
if (explicit) return explicit;
|
|
578
|
+
return generalAlignment(type);
|
|
579
|
+
}
|
|
580
|
+
function generalAlignment(type) {
|
|
581
|
+
switch (type) {
|
|
582
|
+
case "n":
|
|
583
|
+
case "d": return "right";
|
|
584
|
+
case "b":
|
|
585
|
+
case "e": return "center";
|
|
586
|
+
default: return "left";
|
|
587
|
+
}
|
|
484
588
|
}
|
|
485
589
|
function mapAlignment(h) {
|
|
486
590
|
if (!h) return void 0;
|
|
@@ -640,7 +744,7 @@ function buildSparklineLookup(worksheet, print) {
|
|
|
640
744
|
const host = parseAreaRef(sp.sqref);
|
|
641
745
|
const area = parseAreaRef(sp.dataRange);
|
|
642
746
|
if (!host || !area) continue;
|
|
643
|
-
const values = collectSeriesValues(resolveSeriesGrid(sp.dataRange, worksheet, print.sheetGrids).cells, area);
|
|
747
|
+
const values = collectSeriesValues(resolveSeriesGrid(sp.dataRange, worksheet, print.sheetGrids).cells, area, print);
|
|
644
748
|
if (values.length === 0 || values.every((v) => v === null)) continue;
|
|
645
749
|
out.set(key(host.startRow, host.startColumn), {
|
|
646
750
|
kind: sp.kind,
|
|
@@ -660,7 +764,7 @@ function resolveSeriesGrid(dataRange, current, sheetGrids) {
|
|
|
660
764
|
return sheetGrids.get(name) ?? current;
|
|
661
765
|
}
|
|
662
766
|
var MAX_SPARKLINE_POINTS = 1e3;
|
|
663
|
-
function collectSeriesValues(cells, area) {
|
|
767
|
+
function collectSeriesValues(cells, area, print) {
|
|
664
768
|
const byKey = /* @__PURE__ */ new Map();
|
|
665
769
|
for (const c of cells) {
|
|
666
770
|
if (c.row < area.startRow || c.row > area.endRow) continue;
|
|
@@ -669,7 +773,14 @@ function collectSeriesValues(cells, area) {
|
|
|
669
773
|
const v = Number(c.rawValue);
|
|
670
774
|
if (Number.isFinite(v)) byKey.set(key(c.row, c.column), v);
|
|
671
775
|
}
|
|
672
|
-
|
|
776
|
+
const cellCount = (area.endRow - area.startRow + 1) * (area.endColumn - area.startColumn + 1);
|
|
777
|
+
if (cellCount > MAX_SPARKLINE_POINTS) {
|
|
778
|
+
print.losses?.push({
|
|
779
|
+
severity: "degraded",
|
|
780
|
+
feature: FEATURES.charts,
|
|
781
|
+
detail: `sparkline range spans ${cellCount} cells (over ${MAX_SPARKLINE_POINTS}); plotted as the compact populated series, so blank gaps no longer hold their x-positions`,
|
|
782
|
+
...print.sheetName ? { where: `sheet "${print.sheetName}"` } : {}
|
|
783
|
+
});
|
|
673
784
|
const pts = [...byKey.entries()].map(([k, v]) => {
|
|
674
785
|
const [r, col] = k.split(",").map(Number);
|
|
675
786
|
return {
|
|
@@ -732,6 +843,56 @@ function buildTableFormatLookup(worksheet) {
|
|
|
732
843
|
function cellHasContent(cell) {
|
|
733
844
|
return !!cell && (cell.rawValue !== "" || cell.inlineText !== void 0);
|
|
734
845
|
}
|
|
846
|
+
/**
|
|
847
|
+
* The window-local index of the last row inside `[rowStart, rowStart+rowCount)`
|
|
848
|
+
* that carries anything — a value, or a merge reaching into the window. `-1`
|
|
849
|
+
* when the window is entirely blank.
|
|
850
|
+
*/
|
|
851
|
+
function lastContentRow(worksheet, rowStart, rowCount, colStart, colCount) {
|
|
852
|
+
const rowEnd = rowStart + rowCount - 1;
|
|
853
|
+
const colEnd = colStart + colCount - 1;
|
|
854
|
+
let last = -1;
|
|
855
|
+
for (const c of worksheet.cells) {
|
|
856
|
+
if (!cellHasContent(c)) continue;
|
|
857
|
+
if (c.row < rowStart || c.row > rowEnd || c.column < colStart || c.column > colEnd) continue;
|
|
858
|
+
if (c.row - rowStart > last) last = c.row - rowStart;
|
|
859
|
+
}
|
|
860
|
+
for (const m of worksheet.merges) {
|
|
861
|
+
if (m.startColumn > colEnd || m.endColumn < colStart) continue;
|
|
862
|
+
const reach = Math.min(m.endRow, rowEnd) - rowStart;
|
|
863
|
+
if (reach > last) last = reach;
|
|
864
|
+
}
|
|
865
|
+
return last;
|
|
866
|
+
}
|
|
867
|
+
/** The column twin of {@link lastContentRow}. */
|
|
868
|
+
function lastContentColumn(worksheet, rowStart, rowCount, colStart, colCount) {
|
|
869
|
+
const rowEnd = rowStart + rowCount - 1;
|
|
870
|
+
const colEnd = colStart + colCount - 1;
|
|
871
|
+
let last = -1;
|
|
872
|
+
for (const c of worksheet.cells) {
|
|
873
|
+
if (!cellHasContent(c)) continue;
|
|
874
|
+
if (c.row < rowStart || c.row > rowEnd || c.column < colStart || c.column > colEnd) continue;
|
|
875
|
+
if (c.column - colStart > last) last = c.column - colStart;
|
|
876
|
+
}
|
|
877
|
+
for (const m of worksheet.merges) {
|
|
878
|
+
if (m.startRow > rowEnd || m.endRow < rowStart) continue;
|
|
879
|
+
const reach = Math.min(m.endColumn, colEnd) - colStart;
|
|
880
|
+
if (reach > last) last = reach;
|
|
881
|
+
}
|
|
882
|
+
return last;
|
|
883
|
+
}
|
|
884
|
+
/**
|
|
885
|
+
* Whether an EMPTY cell still draws something of its own — a fill or a border.
|
|
886
|
+
*
|
|
887
|
+
* Such a cell cannot be swallowed by a neighbour's overflowing text: the span
|
|
888
|
+
* that gives the text its width would take the paint with it.
|
|
889
|
+
*/
|
|
890
|
+
function cellPaintsSomething(cell, styles) {
|
|
891
|
+
if (!cell || cell.styleIndex === void 0) return false;
|
|
892
|
+
const xf = styles.cellXfs[cell.styleIndex];
|
|
893
|
+
if (!xf) return false;
|
|
894
|
+
return shadingFromXf(xf, styles) !== void 0 || bordersFromXf(xf, styles) !== void 0;
|
|
895
|
+
}
|
|
735
896
|
var SLICER_WIDTH_PT = 108;
|
|
736
897
|
var SLICER_ROW_PT = 16;
|
|
737
898
|
var SLICER_UNSELECTED_HEX = "F2F2F2";
|
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import { FlowDoc } from '../core/ir/flow.js';
|
|
2
|
+
import { Loss } from '../core/ir/loss.js';
|
|
2
3
|
import { SheetDoc } from '../core/ir/sheet.js';
|
|
3
4
|
/**
|
|
4
5
|
* Projection knobs (E-SHEET W9).
|
|
@@ -11,6 +12,15 @@ export interface ProjectSheetOptions {
|
|
|
11
12
|
* byte-identical to before.
|
|
12
13
|
*/
|
|
13
14
|
readonly now?: Date;
|
|
15
|
+
/**
|
|
16
|
+
* Sink the projection writes its {@link Loss} entries into — the print
|
|
17
|
+
* model's defence-in-depth caps (grid size, per-sheet text budget, sparkline
|
|
18
|
+
* range) fire on pathological input, and a cap that fires without saying so
|
|
19
|
+
* is a silent wrongness. {@link readXlsx} always supplies one and returns it
|
|
20
|
+
* as the read result's loss report; omitted ⇒ the caps still apply but go
|
|
21
|
+
* unreported.
|
|
22
|
+
*/
|
|
23
|
+
readonly losses?: Array<Loss>;
|
|
14
24
|
}
|
|
15
25
|
/**
|
|
16
26
|
* Project a {@link SheetDoc} into a {@link FlowDoc} (E-SHEET SA2): each grid sheet
|
|
@@ -18,12 +18,16 @@ var FOOTER_REL = "_xlsxFooterDefault";
|
|
|
18
18
|
*/
|
|
19
19
|
function projectSheetDoc(sheet, options = {}) {
|
|
20
20
|
const body = [];
|
|
21
|
+
const sheetSections = [];
|
|
22
|
+
const sheetEnds = [];
|
|
21
23
|
let firstSheetSection;
|
|
22
24
|
const headersFooters = /* @__PURE__ */ new Map();
|
|
23
25
|
const sheetGrids = new Map(sheet.sheets.map((s) => [s.name, s.grid]));
|
|
24
26
|
for (let sheetIdx = 0; sheetIdx < sheet.sheets.length; sheetIdx++) {
|
|
25
27
|
const ws = sheet.sheets[sheetIdx];
|
|
26
|
-
|
|
28
|
+
const sheetSection = sheetIdx === 0 ? withHeaderFooter(sectionFromWorksheet(ws.grid), ws, headersFooters) : sectionFromWorksheet(ws.grid);
|
|
29
|
+
if (sheetIdx === 0) firstSheetSection = sheetSection;
|
|
30
|
+
sheetSections.push(sheetSection);
|
|
27
31
|
if (sheetIdx > 0) body.push({
|
|
28
32
|
kind: "paragraph",
|
|
29
33
|
paragraph: {
|
|
@@ -43,7 +47,8 @@ function projectSheetDoc(sheet, options = {}) {
|
|
|
43
47
|
definedNames: sheet.definedNames,
|
|
44
48
|
...ws.hyperlinks ? { hyperlinks: ws.hyperlinks } : {},
|
|
45
49
|
...sheet.sharedStringRuns ? { sharedStringRuns: sheet.sharedStringRuns } : {},
|
|
46
|
-
...options.now ? { now: options.now } : {}
|
|
50
|
+
...options.now ? { now: options.now } : {},
|
|
51
|
+
...options.losses ? { losses: options.losses } : {}
|
|
47
52
|
}));
|
|
48
53
|
for (const ref of ws.charts ?? []) body.push({
|
|
49
54
|
kind: "chart",
|
|
@@ -74,11 +79,16 @@ function projectSheetDoc(sheet, options = {}) {
|
|
|
74
79
|
if (ws.comments && ws.comments.length > 0) body.push(...commentBlocks(ws.comments));
|
|
75
80
|
if (ws.formControls && ws.formControls.length > 0) body.push(...formControlBlocks(ws.formControls));
|
|
76
81
|
if (ws.activeXControls && ws.activeXControls.length > 0) body.push(...activeXBlocks(ws.activeXControls));
|
|
82
|
+
sheetEnds.push(body.length);
|
|
77
83
|
}
|
|
84
|
+
const sections = sheetSections.length > 1 ? sheetSections.map((properties, i) => ({
|
|
85
|
+
properties,
|
|
86
|
+
endIndex: sheetEnds[i]
|
|
87
|
+
})) : [];
|
|
78
88
|
return {
|
|
79
89
|
kind: "flow",
|
|
80
90
|
body: resolveBodyStyles(body, EMPTY_STYLE_SHEET),
|
|
81
|
-
sections
|
|
91
|
+
sections,
|
|
82
92
|
...firstSheetSection ? { section: firstSheetSection } : {},
|
|
83
93
|
styles: EMPTY_STYLE_SHEET,
|
|
84
94
|
resources: sheet.resources,
|
|
@@ -220,10 +230,7 @@ function withHeaderFooter(section, ws, headersFooters) {
|
|
|
220
230
|
}
|
|
221
231
|
if (headers.length === 0 && footers.length === 0) return section;
|
|
222
232
|
return {
|
|
223
|
-
...section
|
|
224
|
-
headers: [],
|
|
225
|
-
footers: []
|
|
226
|
-
},
|
|
233
|
+
...section,
|
|
227
234
|
headers,
|
|
228
235
|
footers
|
|
229
236
|
};
|
|
@@ -54,6 +54,7 @@ function parseWorksheet(data) {
|
|
|
54
54
|
const sparklines = parseSparklines(wsObj);
|
|
55
55
|
const tablePartRelIds = parseTableParts(wsObj);
|
|
56
56
|
const printModel = {
|
|
57
|
+
...parseSheetFormatPr(wsObj),
|
|
57
58
|
...pageMargins ? { pageMargins } : {},
|
|
58
59
|
...pageSetup ? { pageSetup } : {},
|
|
59
60
|
...fitToPage ? { fitToPage } : {},
|
|
@@ -245,6 +246,27 @@ function parseNumericAttr(obj, key) {
|
|
|
245
246
|
const n = Number(raw);
|
|
246
247
|
return Number.isFinite(n) ? n : void 0;
|
|
247
248
|
}
|
|
249
|
+
/**
|
|
250
|
+
* ECMA-376 §18.3.1.81 `<sheetFormatPr>` — the sheet's default row height and
|
|
251
|
+
* column width, which apply to every row/column that does not override them.
|
|
252
|
+
*
|
|
253
|
+
* Both were previously ignored, so a row without an explicit `ht` had no height
|
|
254
|
+
* at all and ended up however tall its text wanted to be. A spreadsheet row has
|
|
255
|
+
* a definite height; text metrics do not get a vote.
|
|
256
|
+
*/
|
|
257
|
+
function parseSheetFormatPr(ws) {
|
|
258
|
+
const node = ws["sheetFormatPr"];
|
|
259
|
+
if (!node || typeof node !== "object") return {};
|
|
260
|
+
const obj = node;
|
|
261
|
+
const height = parseNumericAttr(obj, "defaultRowHeight");
|
|
262
|
+
const width = parseNumericAttr(obj, "defaultColWidth");
|
|
263
|
+
const base = parseNumericAttr(obj, "baseColWidth");
|
|
264
|
+
return {
|
|
265
|
+
...height !== void 0 && height > 0 ? { defaultRowHeightPt: height } : {},
|
|
266
|
+
...width !== void 0 && width > 0 ? { defaultColWidthChars: width } : {},
|
|
267
|
+
...base !== void 0 && base > 0 ? { baseColWidthChars: base } : {}
|
|
268
|
+
};
|
|
269
|
+
}
|
|
248
270
|
function parseColumns(ws) {
|
|
249
271
|
const colsNode = ws["cols"];
|
|
250
272
|
if (!colsNode || typeof colsNode !== "object") return [];
|
|
@@ -762,16 +784,17 @@ function parseCell(c, fallbackRow, fallbackCol) {
|
|
|
762
784
|
if (!c || typeof c !== "object") return null;
|
|
763
785
|
const obj = c;
|
|
764
786
|
const ref = strAttr(obj, "r");
|
|
787
|
+
const implied = {
|
|
788
|
+
column: fallbackCol,
|
|
789
|
+
row: fallbackRow
|
|
790
|
+
};
|
|
765
791
|
let address;
|
|
766
792
|
if (ref) try {
|
|
767
793
|
address = parseCellRef(ref);
|
|
768
794
|
} catch {
|
|
769
|
-
|
|
795
|
+
address = implied;
|
|
770
796
|
}
|
|
771
|
-
else address =
|
|
772
|
-
column: fallbackCol,
|
|
773
|
-
row: fallbackRow
|
|
774
|
-
};
|
|
797
|
+
else address = implied;
|
|
775
798
|
const type = validateCellType(strAttr(obj, "t") ?? "n");
|
|
776
799
|
const styleStr = strAttr(obj, "s");
|
|
777
800
|
const styleIndex = styleStr !== void 0 ? Number(styleStr) : void 0;
|
|
@@ -784,7 +807,8 @@ function parseCell(c, fallbackRow, fallbackCol) {
|
|
|
784
807
|
};
|
|
785
808
|
if (type === "inlineStr") {
|
|
786
809
|
const is = obj["is"];
|
|
787
|
-
const
|
|
810
|
+
const fromIs = inlineStringText(is);
|
|
811
|
+
const inlineText = fromIs !== "" ? fromIs : rawValue;
|
|
788
812
|
return {
|
|
789
813
|
...base,
|
|
790
814
|
rawValue: "",
|
|
@@ -4,12 +4,16 @@ import { SheetDoc } from '../core/ir/sheet.js';
|
|
|
4
4
|
import { ProjectSheetOptions } from './sheet-to-flow.js';
|
|
5
5
|
/**
|
|
6
6
|
* Read a `.xlsx` and project it to a {@link FlowDoc} in one step:
|
|
7
|
-
* {@link readXlsxToSheetDoc} then {@link projectSheetDoc}.
|
|
8
|
-
*
|
|
7
|
+
* {@link readXlsxToSheetDoc} then {@link projectSheetDoc}.
|
|
8
|
+
*
|
|
9
|
+
* Parsing itself is lossless — SpreadsheetML maps cleanly onto the sheet IR.
|
|
10
|
+
* The losses come from the projection: the print model's defence-in-depth caps
|
|
11
|
+
* (grid size, per-sheet text budget, sparkline range) clip pathological sheets,
|
|
12
|
+
* and each clip is reported rather than applied in silence.
|
|
9
13
|
*
|
|
10
14
|
* @param xlsx The `.xlsx` (OPC ZIP) bytes.
|
|
11
15
|
* @param options Projection knobs (the W9 reference date).
|
|
12
|
-
* @returns The flow document plus
|
|
16
|
+
* @returns The flow document plus whatever the projection had to clip.
|
|
13
17
|
*/
|
|
14
18
|
export declare function readXlsx(xlsx: Uint8Array, options?: ProjectSheetOptions): ReadResult<FlowDoc>;
|
|
15
19
|
/**
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { ResourceStore } from "../core/ir/resources.js";
|
|
2
2
|
import { FEATURES } from "../core/ir/features.js";
|
|
3
|
-
import { bytesInclude } from "../core/bytes.js";
|
|
3
|
+
import { bytesInclude, bytesIncludePartName } from "../core/bytes.js";
|
|
4
4
|
import { DEFAULT_THEME_PALETTE, makeColorResolver } from "../core/drawingml/colors.js";
|
|
5
5
|
import { parseChart, withChartColorStyle } from "../core/drawingml/chart-parser.js";
|
|
6
6
|
import { parseTheme } from "../core/drawingml/theme-parser.js";
|
|
@@ -36,17 +36,25 @@ var PERSON_REL_TAIL = "/person";
|
|
|
36
36
|
var MAX_SLICER_ITEMS = 256;
|
|
37
37
|
/**
|
|
38
38
|
* Read a `.xlsx` and project it to a {@link FlowDoc} in one step:
|
|
39
|
-
* {@link readXlsxToSheetDoc} then {@link projectSheetDoc}.
|
|
40
|
-
*
|
|
39
|
+
* {@link readXlsxToSheetDoc} then {@link projectSheetDoc}.
|
|
40
|
+
*
|
|
41
|
+
* Parsing itself is lossless — SpreadsheetML maps cleanly onto the sheet IR.
|
|
42
|
+
* The losses come from the projection: the print model's defence-in-depth caps
|
|
43
|
+
* (grid size, per-sheet text budget, sparkline range) clip pathological sheets,
|
|
44
|
+
* and each clip is reported rather than applied in silence.
|
|
41
45
|
*
|
|
42
46
|
* @param xlsx The `.xlsx` (OPC ZIP) bytes.
|
|
43
47
|
* @param options Projection knobs (the W9 reference date).
|
|
44
|
-
* @returns The flow document plus
|
|
48
|
+
* @returns The flow document plus whatever the projection had to clip.
|
|
45
49
|
*/
|
|
46
50
|
function readXlsx(xlsx, options = {}) {
|
|
51
|
+
const losses = [];
|
|
47
52
|
return {
|
|
48
|
-
doc: projectSheetDoc(readXlsxToSheetDoc(xlsx),
|
|
49
|
-
|
|
53
|
+
doc: projectSheetDoc(readXlsxToSheetDoc(xlsx), {
|
|
54
|
+
...options,
|
|
55
|
+
losses
|
|
56
|
+
}),
|
|
57
|
+
losses
|
|
50
58
|
};
|
|
51
59
|
}
|
|
52
60
|
/**
|
|
@@ -414,7 +422,7 @@ var xlsxReader = {
|
|
|
414
422
|
id: "xlsx",
|
|
415
423
|
produces: "sheet",
|
|
416
424
|
supports: new Set([FEATURES.text, FEATURES.tables]),
|
|
417
|
-
sniff: (bytes) => bytes[0] === 80 && bytes[1] === 75 &&
|
|
425
|
+
sniff: (bytes) => bytes[0] === 80 && bytes[1] === 75 && bytesIncludePartName(bytes, "xl/workbook.xml"),
|
|
418
426
|
read: (bytes) => ({
|
|
419
427
|
doc: readXlsxToSheetDoc(bytes),
|
|
420
428
|
losses: []
|
|
@@ -1891,6 +1891,7 @@ var PageAssembler = class {
|
|
|
1891
1891
|
this.bookmarkPositions = bookmarkPositions;
|
|
1892
1892
|
this.ctx = sectionCtxs[0];
|
|
1893
1893
|
this.cursorY = this.ctx.pageHeight - this.ctx.marginTop;
|
|
1894
|
+
this.colStartY = this.cursorY;
|
|
1894
1895
|
}
|
|
1895
1896
|
pages = [];
|
|
1896
1897
|
ctx;
|
|
@@ -1908,18 +1909,33 @@ var PageAssembler = class {
|
|
|
1908
1909
|
*/
|
|
1909
1910
|
colIdx = 0;
|
|
1910
1911
|
colStartLen = 0;
|
|
1912
|
+
/** The cursor's y at the top of the current column — see {@link colHasContent}. */
|
|
1913
|
+
colStartY;
|
|
1911
1914
|
/** The current column's left edge (`marginLeft` plus the column x-offset). */
|
|
1912
1915
|
colLeft = () => this.ctx.marginLeft + (this.ctx.columns?.[this.colIdx]?.xOffsetPt ?? 0);
|
|
1913
1916
|
/** The current column's width (the section content width when single-column). */
|
|
1914
1917
|
colWidth = () => this.ctx.columns?.[this.colIdx]?.widthPt ?? this.ctx.contentWidth;
|
|
1915
|
-
/**
|
|
1916
|
-
|
|
1918
|
+
/**
|
|
1919
|
+
* Whether the current column is already in use.
|
|
1920
|
+
*
|
|
1921
|
+
* This gates every overflow break, so that a block too tall for an empty page
|
|
1922
|
+
* is placed rather than looping forever. It used to ask whether the column had
|
|
1923
|
+
* received any drawable ITEM, which is not the same question: a run of empty
|
|
1924
|
+
* table rows consumes vertical space and draws nothing, so the guard stayed
|
|
1925
|
+
* false, no break ever fired, and the rows marched off the bottom of the page.
|
|
1926
|
+
* A spreadsheet is full of such rows — the sheet in tdf171828.xlsx has 162 of
|
|
1927
|
+
* them — and they were silently costing whole pages of pagination.
|
|
1928
|
+
*
|
|
1929
|
+
* Consumed space counts as use, whether or not any ink went with it.
|
|
1930
|
+
*/
|
|
1931
|
+
colHasContent = () => this.current.length > this.colStartLen || this.cursorY < this.colStartY;
|
|
1917
1932
|
/** Overflow step: next column on this page, or a fresh page after the last. */
|
|
1918
1933
|
advanceColumn = () => {
|
|
1919
1934
|
if (this.ctx.columns && this.colIdx + 1 < this.ctx.columns.length) {
|
|
1920
1935
|
this.colIdx++;
|
|
1921
1936
|
this.colStartLen = this.current.length;
|
|
1922
1937
|
this.cursorY = this.ctx.pageHeight - this.ctx.marginTop;
|
|
1938
|
+
this.colStartY = this.cursorY;
|
|
1923
1939
|
} else this.flushPage();
|
|
1924
1940
|
};
|
|
1925
1941
|
/**
|
|
@@ -2098,7 +2114,7 @@ var PageAssembler = class {
|
|
|
2098
2114
|
* guarantee one page for a header/footer-only document).
|
|
2099
2115
|
*/
|
|
2100
2116
|
flushPage = (force = false) => {
|
|
2101
|
-
if (this.current.length === 0 && !force) return;
|
|
2117
|
+
if (this.current.length === 0 && this.cursorY >= this.colStartY && !force) return;
|
|
2102
2118
|
const band = bandForPage(this.pageInSection, this.globalPageIdx, this.ctx.titlePg, this.ctx.evenAndOddHeaders);
|
|
2103
2119
|
const header = pickBand(this.ctx.headerSet, band);
|
|
2104
2120
|
const footer = pickBand(this.ctx.footerSet, band);
|
|
@@ -2137,6 +2153,7 @@ var PageAssembler = class {
|
|
|
2137
2153
|
this.pageInSection++;
|
|
2138
2154
|
this.globalPageIdx++;
|
|
2139
2155
|
this.cursorY = this.ctx.pageHeight - this.ctx.marginTop;
|
|
2156
|
+
this.colStartY = this.cursorY;
|
|
2140
2157
|
};
|
|
2141
2158
|
};
|
|
2142
2159
|
function paginateSections(blocks, sectionCtxs, builder, defaultLang = "en-US", notes, bookmarkPositions, reflowParagraph) {
|
|
@@ -4,7 +4,7 @@ import { FEATURES } from "../core/ir/features.js";
|
|
|
4
4
|
import { EMPTY_STYLE_SHEET, resolveBodyStyles } from "../core/style-cascade/resolver.js";
|
|
5
5
|
import "../core/style-cascade/index.js";
|
|
6
6
|
import { poAttr, poChildren, poFindDescendant, poIntAttr, poIs } from "../core/po-helpers.js";
|
|
7
|
-
import {
|
|
7
|
+
import { bytesIncludePartName } from "../core/bytes.js";
|
|
8
8
|
import { DEFAULT_THEME_PALETTE, defaultColorResolver, makeColorResolver } from "../core/drawingml/colors.js";
|
|
9
9
|
import { parseChart, withChartColorStyle } from "../core/drawingml/chart-parser.js";
|
|
10
10
|
import { parseTheme } from "../core/drawingml/theme-parser.js";
|
|
@@ -269,7 +269,7 @@ var pptxReader = {
|
|
|
269
269
|
FEATURES.charts,
|
|
270
270
|
FEATURES.tables
|
|
271
271
|
]),
|
|
272
|
-
sniff: (bytes) => bytes[0] === 80 && bytes[1] === 75 &&
|
|
272
|
+
sniff: (bytes) => bytes[0] === 80 && bytes[1] === 75 && bytesIncludePartName(bytes, "ppt/presentation.xml"),
|
|
273
273
|
read: (bytes) => readPptx(bytes)
|
|
274
274
|
};
|
|
275
275
|
//#endregion
|
|
@@ -5,7 +5,7 @@ import { EMPTY_STYLE_SHEET, resolveBodyStyles, resolveHeadersFootersStyles } fro
|
|
|
5
5
|
import { resolveTableStyles } from "../core/style-cascade/table.js";
|
|
6
6
|
import "../core/style-cascade/index.js";
|
|
7
7
|
import { poFindDescendant } from "../core/po-helpers.js";
|
|
8
|
-
import {
|
|
8
|
+
import { bytesIncludePartName } from "../core/bytes.js";
|
|
9
9
|
import { DEFAULT_THEME_PALETTE, makeColorResolver } from "../core/drawingml/colors.js";
|
|
10
10
|
import { parseChart, withChartColorStyle } from "../core/drawingml/chart-parser.js";
|
|
11
11
|
import { parseTheme } from "../core/drawingml/theme-parser.js";
|
|
@@ -144,7 +144,7 @@ var docxReader = {
|
|
|
144
144
|
FEATURES.trackedChanges,
|
|
145
145
|
FEATURES.fontsEmbedding
|
|
146
146
|
]),
|
|
147
|
-
sniff: (bytes) => bytes[0] === 80 && bytes[1] === 75 &&
|
|
147
|
+
sniff: (bytes) => bytes[0] === 80 && bytes[1] === 75 && bytesIncludePartName(bytes, "word/document.xml"),
|
|
148
148
|
read: (bytes) => readDocx(bytes)
|
|
149
149
|
};
|
|
150
150
|
function infoFromCore(core) {
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "reamkit",
|
|
3
|
-
"version": "1.15.
|
|
3
|
+
"version": "1.15.3",
|
|
4
4
|
"description": "Ream — convert DOCX, XLSX, PPTX and PDF to PDF, SVG, HTML, DOCX and XLSX, built from scratch on the ECMA-376 and ISO 32000 specifications. Parse once, convert anywhere.",
|
|
5
5
|
"license": "MIT",
|
|
6
6
|
"author": "Alex Krassavin <info@reamkit.dev>",
|
|
@@ -70,7 +70,10 @@
|
|
|
70
70
|
"corpus:sandbox:build": "docker build -t docgen-losandbox:latest scripts/corpus/sandbox/",
|
|
71
71
|
"corpus": "tsx scripts/corpus/run.ts",
|
|
72
72
|
"corpus:roundtrip": "tsx scripts/corpus/roundtrip.ts",
|
|
73
|
-
"corpus:roundtrip:xlsx": "tsx scripts/corpus/xlsx-roundtrip.ts"
|
|
73
|
+
"corpus:roundtrip:xlsx": "tsx scripts/corpus/xlsx-roundtrip.ts",
|
|
74
|
+
"corpus:xlsx:invariants": "tsx scripts/corpus/xlsx-invariants.ts",
|
|
75
|
+
"corpus:fixtures": "tsx scripts/corpus/sync-real-fixtures.ts",
|
|
76
|
+
"corpus:golden": "tsx scripts/corpus/make-golden.ts"
|
|
74
77
|
},
|
|
75
78
|
"dependencies": {
|
|
76
79
|
"fast-xml-parser": "5.7.0",
|