reamkit 1.15.2 → 1.15.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -8,4 +8,15 @@ export declare function toBase64(bytes: Uint8Array): string;
8
8
  * Naive scan for an ASCII `needle` inside raw `haystack` bytes — used by reader
9
9
  * sniffs to spot OPC part names (e.g. `'word/document.xml'`) without unzipping.
10
10
  */
11
+ /**
12
+ * Scan raw package bytes for an OPC part name (e.g. `'xl/workbook.xml'`)
13
+ * without unzipping — the reader sniffs' cheap format probe. Accepts both the
14
+ * spec's `/` separator and the `\` that Windows producers write.
15
+ */
16
+ export declare function bytesIncludePartName(haystack: Uint8Array, partName: string): boolean;
17
+ /**
18
+ * Naive scan for an ASCII `needle` inside raw `haystack` bytes. Prefer
19
+ * {@link bytesIncludePartName} for OPC part names — it also accepts the
20
+ * backslash spelling real archives use.
21
+ */
11
22
  export declare function bytesInclude(haystack: Uint8Array, needle: string): boolean;
@@ -13,6 +13,20 @@ function toBase64(bytes) {
13
13
  * Naive scan for an ASCII `needle` inside raw `haystack` bytes — used by reader
14
14
  * sniffs to spot OPC part names (e.g. `'word/document.xml'`) without unzipping.
15
15
  */
16
+ /**
17
+ * Scan raw package bytes for an OPC part name (e.g. `'xl/workbook.xml'`)
18
+ * without unzipping — the reader sniffs' cheap format probe. Accepts both the
19
+ * spec's `/` separator and the `\` that Windows producers write.
20
+ */
21
+ function bytesIncludePartName(haystack, partName) {
22
+ if (bytesInclude(haystack, partName)) return true;
23
+ return partName.includes("/") && bytesInclude(haystack, partName.replace(/\//g, "\\"));
24
+ }
25
+ /**
26
+ * Naive scan for an ASCII `needle` inside raw `haystack` bytes. Prefer
27
+ * {@link bytesIncludePartName} for OPC part names — it also accepts the
28
+ * backslash spelling real archives use.
29
+ */
16
30
  function bytesInclude(haystack, needle) {
17
31
  const n = new TextEncoder().encode(needle);
18
32
  outer: for (let i = 0; i + n.length <= haystack.length; i++) {
@@ -22,4 +36,4 @@ function bytesInclude(haystack, needle) {
22
36
  return false;
23
37
  }
24
38
  //#endregion
25
- export { bytesInclude, toBase64 };
39
+ export { bytesInclude, bytesIncludePartName, toBase64 };
@@ -80,12 +80,15 @@ export interface SheetFormControl {
80
80
  }
81
81
  /**
82
82
  * An ActiveX control resolved against its activeX part (E-SHEET W10): the control
83
- * class (`type`, from the `<oleObject progId>`) plus the visible state persisted
84
- * in the property bag — `caption`, `value` (checked/text/number, as a string)
85
- * and OptionButton `groupName`. Render-only.
83
+ * class (`type`, from the `<oleObject progId>` or, for a §18.3.1.19 `<control>`,
84
+ * its class id) plus the visible state persisted in the property bag —
85
+ * `caption`, `value` (checked/text/number, as a string) and OptionButton
86
+ * `groupName`. Render-only.
86
87
  */
87
88
  export interface SheetActiveXControl {
88
89
  readonly type: string;
90
+ /** §18.3.1.19 `<control name>` — the control's identifier, when it came that way. */
91
+ readonly name?: string;
89
92
  readonly caption?: string;
90
93
  readonly value?: string;
91
94
  readonly groupName?: string;
@@ -42,6 +42,7 @@ var OpcPackage = class OpcPackage {
42
42
  violation ??= `more than ${maxEntries} entries`;
43
43
  return false;
44
44
  }
45
+ if (info.originalSize === 4294967295) return true;
45
46
  if (info.originalSize > maxEntry) {
46
47
  violation ??= `entry "${info.name}" declares ${info.originalSize} bytes (limit ${maxEntry})`;
47
48
  return false;
@@ -138,7 +139,8 @@ var OpcPackage = class OpcPackage {
138
139
  }
139
140
  };
140
141
  function normalizePath(p) {
141
- return p.startsWith("/") ? p.slice(1) : p;
142
+ const slashed = p.includes("\\") ? p.replace(/\\/g, "/") : p;
143
+ return slashed.startsWith("/") ? slashed.slice(1) : slashed;
142
144
  }
143
145
  function resolveTarget(sourcePath, target) {
144
146
  if (target.startsWith("/")) return normalizePath(target);
@@ -83,6 +83,27 @@ export interface ParsedWorksheet {
83
83
  readonly columns: ReadonlyArray<ColumnWidth>;
84
84
  readonly merges: ReadonlyArray<MergedRange>;
85
85
  readonly rowHeights: ReadonlyArray<RowHeight>;
86
+ /**
87
+ * ECMA-376 §18.3.1.81 `<sheetFormatPr defaultRowHeight>` — the height, in
88
+ * points, of every row that carries no `ht` of its own. A row in a
89
+ * spreadsheet always has a definite height; it is not decided by whatever
90
+ * leading the text happens to want. Absent ⇒ Excel's 15pt (its default theme
91
+ * font, Calibri 11).
92
+ */
93
+ readonly defaultRowHeightPt?: number;
94
+ /**
95
+ * §18.3.1.81 `<sheetFormatPr defaultColWidth>` — the width, in characters, of
96
+ * every column no `<col>` covers. Absent ⇒ derived from
97
+ * {@link ParsedWorksheet.baseColWidthChars}, else Excel's 8.43.
98
+ */
99
+ readonly defaultColWidthChars?: number;
100
+ /**
101
+ * §18.3.1.81 `<sheetFormatPr baseColWidth>` — the character width the default
102
+ * column is computed FROM when `defaultColWidth` is absent. Excel's own
103
+ * default is 8; a sheet that says 10 wants every unlisted column that much
104
+ * wider, and ignoring it makes the whole grid narrow.
105
+ */
106
+ readonly baseColWidthChars?: number;
86
107
  readonly pageMargins?: XlsxPageMargins;
87
108
  readonly pageSetup?: XlsxPageSetup;
88
109
  /**
@@ -22,6 +22,15 @@ export declare function parseActiveX(data: Uint8Array): ActiveXProps;
22
22
  * unknown progId falls back to a generic `'control'`.
23
23
  */
24
24
  export declare function activeXType(progId: string | undefined): string;
25
+ /**
26
+ * The affordance key for an `<ax:ocx ax:classid>`, for controls reached through
27
+ * §18.3.1.19 `<control>` rather than `<oleObject progId>` — the `<control>`
28
+ * element carries no progId, so the class id is all there is to type it by.
29
+ *
30
+ * @param xmlData The `activeX#.xml` part bytes.
31
+ * @returns The affordance key, or `'control'` when the class id is unknown.
32
+ */
33
+ export declare function activeXTypeFromPart(xmlData: Uint8Array): string;
25
34
  /**
26
35
  * The `<ax:ocx r:id>` of a control whose state is persisted to a binary stream
27
36
  * (`persistStreamInit` / `persistStream` / `persistStorage`) rather than to
@@ -62,6 +62,20 @@ function strAttr(obj, key) {
62
62
  const v = obj[`@_${key}`];
63
63
  return typeof v === "string" ? v : void 0;
64
64
  }
65
+ var ACTIVEX_CLASS_TYPES = new Map([["8BD21D40-EC42-11CE-9E0D-00AA006002F3", "checkbox"], ["8BD21D50-EC42-11CE-9E0D-00AA006002F3", "option"]]);
66
+ /**
67
+ * The affordance key for an `<ax:ocx ax:classid>`, for controls reached through
68
+ * §18.3.1.19 `<control>` rather than `<oleObject progId>` — the `<control>`
69
+ * element carries no progId, so the class id is all there is to type it by.
70
+ *
71
+ * @param xmlData The `activeX#.xml` part bytes.
72
+ * @returns The affordance key, or `'control'` when the class id is unknown.
73
+ */
74
+ function activeXTypeFromPart(xmlData) {
75
+ const ocx = asObject(parser.parse(decoder.decode(xmlData))["ocx"]);
76
+ const key = (ocx ? strAttr(ocx, "classid") : void 0)?.replace(/[{}]/g, "").toUpperCase() ?? "";
77
+ return ACTIVEX_CLASS_TYPES.get(key) ?? "control";
78
+ }
65
79
  /**
66
80
  * The `<ax:ocx r:id>` of a control whose state is persisted to a binary stream
67
81
  * (`persistStreamInit` / `persistStream` / `persistStorage`) rather than to
@@ -209,4 +223,4 @@ function parseActiveXBin(data) {
209
223
  };
210
224
  }
211
225
  //#endregion
212
- export { activeXBinRelId, activeXType, parseActiveX, parseActiveXBin };
226
+ export { activeXBinRelId, activeXType, activeXTypeFromPart, parseActiveX, parseActiveXBin };
@@ -100,13 +100,7 @@ function withColSpan(cell, span) {
100
100
  function blankCell(span) {
101
101
  return {
102
102
  properties: span > 1 ? { colSpan: span } : {},
103
- content: [{
104
- kind: "paragraph",
105
- paragraph: {
106
- properties: {},
107
- runs: []
108
- }
109
- }]
103
+ content: []
110
104
  };
111
105
  }
112
106
  //#endregion
@@ -1,12 +1,25 @@
1
1
  import { BodyElement, SectionProperties, Table } from '../core/document-model/index.js';
2
2
  import { CellRange, DefinedName, ParsedWorksheet, SheetRichRun, WorksheetCell, XlsxStyles } from './index.js';
3
3
  import { SheetHyperlink, SheetSlicer } from '../core/ir/sheet.js';
4
+ import { Loss } from '../core/ir/loss.js';
4
5
  /**
5
- * Excel "character width" → twips. Calibri 11pt default Maximum Digit Width is
6
- * ~7 px ≈ 5.25 pt ≈ 105 twips. This is a coarse approximation but the auto-fit
7
- * pass refines column widths against the actual cell text anyway.
6
+ * Excel "character width" → twips: the default font's Maximum Digit Width,
7
+ * ~7 px at 96 DPI ≈ 5.25 pt ≈ 105 twips.
8
8
  */
9
9
  export declare const TWIPS_PER_EXCEL_CHAR = 105;
10
+ /**
11
+ * ECMA-376 §18.3.1.13 — a `<col width>` measures characters of text, and the
12
+ * rendered column is that many Maximum Digit Widths PLUS a fixed 5-pixel
13
+ * padding: `px = chars × MDW + 5`. 5 px at 96 DPI = 3.75 pt = 75 twips.
14
+ *
15
+ * Omitting it made every column 75 twips narrow, which compounds: on a sheet of
16
+ * equal 12-character columns the third one landed ~37 pt left of where
17
+ * LibreOffice puts it. That {@link DEFAULT_COL_TWIPS} below is 960 is the proof
18
+ * the padding belongs — 8.43 characters only reaches Excel's documented 64 px
19
+ * default with it (8.43 × 7 + 5 = 64.01 px = 48.01 pt = 960 twips), so the
20
+ * default was derived from the full formula while the explicit path dropped it.
21
+ */
22
+ export declare const COL_PADDING_TWIPS = 75;
10
23
  /**
11
24
  * Excel's default column width is 8.43 "characters" ≈ 64px ≈ 960 twips. Used for
12
25
  * columns without an explicit `<col width="..">`.
@@ -19,10 +32,16 @@ export declare const DEFAULT_COL_TWIPS = 960;
19
32
  export declare const DEFAULT_ROW_TWIPS = 300;
20
33
  /**
21
34
  * Build the page section (paper size + margins) from a worksheet's `<pageSetup>`
22
- * / `<pageMargins>`. Returns `undefined` when neither is set, so the renderer
23
- * applies its A4 default.
35
+ * / `<pageMargins>`.
36
+ *
37
+ * Margins are always set, to Excel's own defaults when the worksheet declares
38
+ * none (§18.3.1.62) — the renderer's fallback is a word processor's inch, which
39
+ * is not what a spreadsheet prints. The paper size is left unset when the
40
+ * worksheet names none, because there the file genuinely holds no answer: Excel
41
+ * picks by locale and printer, and the renderer's deterministic A4 is as good
42
+ * as anything we could invent.
24
43
  */
25
- export declare function sectionFromWorksheet(worksheet: ParsedWorksheet): SectionProperties | undefined;
44
+ export declare function sectionFromWorksheet(worksheet: ParsedWorksheet): SectionProperties;
26
45
  /**
27
46
  * ECMA-376 §18.2.5 — resolve the sheet-scoped `_xlnm.Print_Area` defined name
28
47
  * (`localSheetId` = the sheet's 0-based index) into a clipping range.
@@ -46,6 +65,7 @@ interface PrintModelOptions {
46
65
  readonly hyperlinks?: ReadonlyArray<SheetHyperlink>;
47
66
  readonly sharedStringRuns?: ReadonlyArray<ReadonlyArray<SheetRichRun> | undefined>;
48
67
  readonly now?: Date;
68
+ readonly losses?: Array<Loss>;
49
69
  }
50
70
  /**
51
71
  * Project one worksheet's grid into Flow body elements — a single {@link Table}
@@ -1,9 +1,23 @@
1
1
  import { eighthPtToPt, halfPtToPt, pt, twipsToPt } from "../core/ir/units.js";
2
+ import { FEATURES } from "../core/ir/features.js";
2
3
  import { parseAreaRef, parseTitleRowRange } from "./defined-name-ref.js";
3
4
  import { applyNumberFormat } from "./number-format.js";
4
5
  import "./index.js";
5
6
  import { bandedTables, computeColumnBands } from "./column-bands.js";
6
7
  import { buildConditionalFormatter } from "./conditional-format.js";
8
+ /**
9
+ * Excel insets a cell's text by ~2 px each side (1.5 pt at 96 DPI). The layout
10
+ * engine's default is a word processor's 108 twips (5.4 pt), which is nearly
11
+ * four times as much.
12
+ */
13
+ var EXCEL_CELL_INSET_PT = 1.5;
14
+ /**
15
+ * ECMA-376 §18.3.1.81 — the row height Excel uses when a sheet declares no
16
+ * `<sheetFormatPr defaultRowHeight>`: 15pt, the line height of its default
17
+ * theme font (Calibri 11). Independent of whatever font we end up rendering
18
+ * with — the height belongs to the document, not to the typesetter.
19
+ */
20
+ var EXCEL_DEFAULT_ROW_HEIGHT_PT = 15;
7
21
  var TWIPS_PER_POINT = 20;
8
22
  var TWIPS_PER_INCH = 1440;
9
23
  var PAPER_SIZES_TWIPS = new Map([
@@ -18,16 +32,20 @@ var PAPER_SIZES_TWIPS = new Map([
18
32
  var DEFAULT_PAPER_TWIPS = [11906, 16838];
19
33
  /**
20
34
  * Build the page section (paper size + margins) from a worksheet's `<pageSetup>`
21
- * / `<pageMargins>`. Returns `undefined` when neither is set, so the renderer
22
- * applies its A4 default.
35
+ * / `<pageMargins>`.
36
+ *
37
+ * Margins are always set, to Excel's own defaults when the worksheet declares
38
+ * none (§18.3.1.62) — the renderer's fallback is a word processor's inch, which
39
+ * is not what a spreadsheet prints. The paper size is left unset when the
40
+ * worksheet names none, because there the file genuinely holds no answer: Excel
41
+ * picks by locale and printer, and the renderer's deterministic A4 is as good
42
+ * as anything we could invent.
23
43
  */
24
44
  function sectionFromWorksheet(worksheet) {
25
45
  const pageSize = pageSizeFromSetup(worksheet.pageSetup);
26
- const margins = marginsFromXlsx(worksheet.pageMargins);
27
- if (!pageSize && !margins) return void 0;
28
46
  return {
29
47
  ...pageSize ? { pageSize } : {},
30
- ...margins ? { margins } : {},
48
+ margins: marginsFromXlsx(worksheet.pageMargins),
31
49
  headers: [],
32
50
  footers: []
33
51
  };
@@ -44,8 +62,23 @@ function pageSizeFromSetup(setup) {
44
62
  orientation
45
63
  };
46
64
  }
65
+ /**
66
+ * ECMA-376 §18.3.1.62 — the page margins Excel writes when the user has not
67
+ * touched them, in inches. A worksheet may omit `<pageMargins>` entirely, and
68
+ * falling through to the renderer's default (a word processor's 1 inch) put the
69
+ * grid 0.3 inch — 21.6 pt — right of where Excel and LibreOffice print it, on
70
+ * every such sheet.
71
+ */
72
+ var EXCEL_DEFAULT_MARGINS = {
73
+ left: twipsToPt(Math.round(.7 * TWIPS_PER_INCH)),
74
+ right: twipsToPt(Math.round(.7 * TWIPS_PER_INCH)),
75
+ top: twipsToPt(Math.round(.75 * TWIPS_PER_INCH)),
76
+ bottom: twipsToPt(Math.round(.75 * TWIPS_PER_INCH)),
77
+ header: twipsToPt(Math.round(.3 * TWIPS_PER_INCH)),
78
+ footer: twipsToPt(Math.round(.3 * TWIPS_PER_INCH))
79
+ };
47
80
  function marginsFromXlsx(margins) {
48
- if (!margins) return void 0;
81
+ if (!margins) return EXCEL_DEFAULT_MARGINS;
49
82
  return {
50
83
  top: twipsToPt(Math.round(margins.topInches * TWIPS_PER_INCH)),
51
84
  right: twipsToPt(Math.round(margins.rightInches * TWIPS_PER_INCH)),
@@ -73,16 +106,16 @@ function sheetContentWidthTwips(worksheet) {
73
106
  const pageSize = pageSizeFromSetup(worksheet.pageSetup);
74
107
  const pageWidthTwips = pageSize ? Math.round(pageSize.width * 20) : DEFAULT_PAPER_TWIPS[0];
75
108
  const margins = marginsFromXlsx(worksheet.pageMargins);
76
- const left = margins ? Math.round(margins.left * 20) : TWIPS_PER_INCH;
77
- const right = margins ? Math.round(margins.right * 20) : TWIPS_PER_INCH;
109
+ const left = Math.round(margins.left * 20);
110
+ const right = Math.round(margins.right * 20);
78
111
  return Math.max(TWIPS_PER_INCH / 2, pageWidthTwips - left - right);
79
112
  }
80
113
  function sheetContentHeightTwips(worksheet) {
81
114
  const pageSize = pageSizeFromSetup(worksheet.pageSetup);
82
115
  const pageHeightTwips = pageSize ? Math.round(pageSize.height * 20) : DEFAULT_PAPER_TWIPS[1];
83
116
  const margins = marginsFromXlsx(worksheet.pageMargins);
84
- const top = margins ? Math.round(margins.top * 20) : TWIPS_PER_INCH;
85
- const bottom = margins ? Math.round(margins.bottom * 20) : TWIPS_PER_INCH;
117
+ const top = Math.round(margins.top * 20);
118
+ const bottom = Math.round(margins.bottom * 20);
86
119
  return Math.max(TWIPS_PER_INCH / 2, pageHeightTwips - top - bottom);
87
120
  }
88
121
  function computePrintScale(worksheet, totalGridTwips, contentWidthTwips, totalGridHeightTwips, contentHeightTwips) {
@@ -127,11 +160,14 @@ function worksheetToBody(worksheet, sharedStrings, styles, date1904, print) {
127
160
  if (worksheet.maxRow < 0 || worksheet.maxColumn < 0) return [];
128
161
  let usedRow = -1;
129
162
  let usedCol = -1;
163
+ const contentAt = /* @__PURE__ */ new Set();
130
164
  for (const c of worksheet.cells) if (c.rawValue !== "" || c.inlineText !== void 0) {
165
+ contentAt.add(key(c.row, c.column));
131
166
  if (c.row > usedRow) usedRow = c.row;
132
167
  if (c.column > usedCol) usedCol = c.column;
133
168
  }
134
169
  for (const m of worksheet.merges) {
170
+ if (!contentAt.has(key(m.startRow, m.startColumn))) continue;
135
171
  if (m.endRow > usedRow) usedRow = m.endRow;
136
172
  if (m.endColumn > usedCol) usedCol = m.endColumn;
137
173
  }
@@ -153,12 +189,35 @@ function worksheetToBody(worksheet, sharedStrings, styles, date1904, print) {
153
189
  colEnd = Math.min(usedCol, print.printArea.endColumn);
154
190
  }
155
191
  if (rowEnd < rowStart || colEnd < colStart) return [];
156
- const MAX_GRID_COLS = 1024;
157
- const rowCount = Math.min(rowEnd - rowStart + 1, 5e4);
158
- const colCount = Math.min(colEnd - colStart + 1, MAX_GRID_COLS);
192
+ const MAX_GRID_COLS = 16384;
193
+ const MAX_GRID_ROWS = 5e4;
194
+ const MAX_GRID_CELLS = 1e6;
195
+ const wantRows = rowEnd - rowStart + 1;
196
+ const wantCols = colEnd - colStart + 1;
197
+ let colCount = Math.min(wantCols, MAX_GRID_COLS);
198
+ let rowCount = Math.max(1, Math.min(wantRows, MAX_GRID_ROWS, Math.floor(MAX_GRID_CELLS / colCount)));
199
+ if (rowCount < wantRows) rowCount = Math.max(1, lastContentRow(worksheet, rowStart, rowCount, colStart, colCount) + 1);
200
+ if (colCount < wantCols) colCount = Math.max(1, lastContentColumn(worksheet, rowStart, rowCount, colStart, colCount) + 1);
201
+ if (rowCount < wantRows) print.losses?.push({
202
+ severity: "dropped",
203
+ feature: FEATURES.tables,
204
+ detail: `grid clipped to the first ${rowCount} rows of ${wantRows} in the used range (memory guard)`,
205
+ ...print.sheetName ? { where: `sheet "${print.sheetName}"` } : {}
206
+ });
207
+ if (colCount < wantCols) print.losses?.push({
208
+ severity: "dropped",
209
+ feature: FEATURES.tables,
210
+ detail: `grid clipped to the first ${colCount} columns of ${wantCols} in the used range (memory guard)`,
211
+ ...print.sheetName ? { where: `sheet "${print.sheetName}"` } : {}
212
+ });
159
213
  const rowWindowEnd = rowStart + rowCount - 1;
160
214
  const colWindowEnd = colStart + colCount - 1;
215
+ const defaultRowTwips = Math.round((worksheet.defaultRowHeightPt ?? EXCEL_DEFAULT_ROW_HEIGHT_PT) * TWIPS_PER_POINT);
161
216
  const rowHeightMap = /* @__PURE__ */ new Map();
217
+ for (let r = 0; r < rowCount; r++) rowHeightMap.set(r, {
218
+ heightTwips: defaultRowTwips,
219
+ heightRule: "atLeast"
220
+ });
162
221
  for (const h of worksheet.rowHeights) {
163
222
  const local = h.row - rowStart;
164
223
  if (local < 0 || local >= rowCount) continue;
@@ -184,9 +243,11 @@ function worksheetToBody(worksheet, sharedStrings, styles, date1904, print) {
184
243
  const cEnd = Math.min(m.endColumn, colWindowEnd);
185
244
  for (let r = Math.max(m.startRow, rowStart); r <= rEnd; r++) for (let c = Math.max(m.startColumn, colStart); c <= cEnd; c++) if (!(r === m.startRow && c === m.startColumn)) insideMerge.add(key(r, c));
186
245
  }
187
- const columnWidths = new Array(colCount).fill(960);
246
+ const defaultColChars = worksheet.defaultColWidthChars ?? worksheet.baseColWidthChars;
247
+ const defaultColTwips = defaultColChars !== void 0 ? Math.round(defaultColChars * 105 + 75) : 960;
248
+ const columnWidths = new Array(colCount).fill(defaultColTwips);
188
249
  for (const col of worksheet.columns) {
189
- const twips = Math.round(col.widthChars * 105);
250
+ const twips = Math.round(col.widthChars * 105 + 75);
190
251
  for (let abs = col.min - 1; abs <= col.max - 1; abs++) {
191
252
  const i = abs - colStart;
192
253
  if (i >= 0 && i < colCount) columnWidths[i] = twips;
@@ -199,6 +260,7 @@ function worksheetToBody(worksheet, sharedStrings, styles, date1904, print) {
199
260
  const scaled = printScale < .999;
200
261
  const breakRows = new Set(worksheet.rowBreaks ?? []);
201
262
  let textBudget = MAX_SHEET_TEXT_CHARS;
263
+ let textBudgetReported = false;
202
264
  const runPropsByXf = /* @__PURE__ */ new Map();
203
265
  const cellRunProps = (xf) => {
204
266
  let props = runPropsByXf.get(xf);
@@ -226,7 +288,9 @@ function worksheetToBody(worksheet, sharedStrings, styles, date1904, print) {
226
288
  for (let r = 0; r < rowCount; r++) {
227
289
  const absR = r + rowStart;
228
290
  const cells = [];
291
+ const overflowed = /* @__PURE__ */ new Set();
229
292
  for (let c = 0; c < colCount; c++) {
293
+ if (overflowed.has(c)) continue;
230
294
  const absC = c + colStart;
231
295
  const merge = mergeOrigins.get(key(absR, absC));
232
296
  if (insideMerge.has(key(absR, absC)) && !merge) {
@@ -237,11 +301,22 @@ function worksheetToBody(worksheet, sharedStrings, styles, date1904, print) {
237
301
  const ws = cellMatrix[r]?.[c];
238
302
  let text = ws ? resolveCellText(ws, sharedStrings, styles, date1904) : "";
239
303
  const cfText = text.length > 0 ? text : void 0;
240
- if (text.length > textBudget) text = text.slice(0, Math.max(0, textBudget));
304
+ if (text.length > textBudget) {
305
+ text = text.slice(0, Math.max(0, textBudget));
306
+ if (!textBudgetReported) {
307
+ textBudgetReported = true;
308
+ print.losses?.push({
309
+ severity: "dropped",
310
+ feature: FEATURES.text,
311
+ detail: `per-sheet text budget of ${MAX_SHEET_TEXT_CHARS} characters exhausted; the remaining cells render empty`,
312
+ ...print.sheetName ? { where: `sheet "${print.sheetName}"` } : {}
313
+ });
314
+ }
315
+ }
241
316
  textBudget -= text.length;
242
317
  const xf = ws && ws.styleIndex !== void 0 ? styles.cellXfs[ws.styleIndex] : void 0;
243
318
  let runProps = cellRunProps(xf);
244
- const alignment = xf ? alignmentFromXf(xf) : void 0;
319
+ const alignment = alignmentFromXf(xf, ws?.type);
245
320
  let shading = xf ? shadingFromXf(xf, styles) : void 0;
246
321
  const tableFmt = tableFormatByCell.get(key(absR, absC));
247
322
  if (!shading && tableFmt?.shading) shading = tableFmt.shading;
@@ -262,14 +337,16 @@ function worksheetToBody(worksheet, sharedStrings, styles, date1904, print) {
262
337
  runProps = applyCfOverride(runProps, over);
263
338
  }
264
339
  }
340
+ let overflowSpan = 1;
265
341
  const wrapText = xf?.alignment?.wrapText === true;
266
342
  const rotation = xf?.alignment?.textRotation;
267
343
  const rotated = rotation !== void 0 && rotation !== 0 && text.length > 0;
268
344
  const shrinkToFit = xf?.alignment?.shrinkToFit === true && !wrapText && !rotated;
269
- if (text.length > 0 && !merge && !wrapText && !rotated && !shrinkToFit && ws && (ws.type === "s" || ws.type === "str" || ws.type === "inlineStr") && (alignment === void 0 || alignment === "left")) {
345
+ if (text.length > 0 && !merge && !wrapText && !rotated && !shrinkToFit && ws && (ws.type === "s" || ws.type === "str" || ws.type === "inlineStr") && alignment === "left") {
270
346
  let availTwips = columnWidths[c];
271
347
  let cc = c + 1;
272
- while (cc < colCount && !cellHasContent(cellMatrix[r]?.[cc])) {
348
+ const neighbourIsFree = (col) => !cellHasContent(cellMatrix[r]?.[col]) && !cellPaintsSomething(cellMatrix[r]?.[col], styles) && !sparklineByCell.has(key(absR, col + colStart)) && !(dropdownRanges.length > 0 && rangesCover(dropdownRanges, absR, col + colStart));
349
+ while (cc < colCount && neighbourIsFree(cc)) {
273
350
  availTwips += columnWidths[cc];
274
351
  cc++;
275
352
  }
@@ -277,6 +354,10 @@ function worksheetToBody(worksheet, sharedStrings, styles, date1904, print) {
277
354
  const charsFit = Math.max(1, Math.round(availTwips / 105));
278
355
  if (text.length > charsFit) text = text.slice(0, charsFit);
279
356
  }
357
+ if (cc > c + 1) {
358
+ overflowSpan = cc - c;
359
+ for (let k = c + 1; k < cc; k++) overflowed.add(k);
360
+ }
280
361
  }
281
362
  if (shrinkToFit && ws && text.length > 0) {
282
363
  const charsFit = Math.max(1, columnWidths[c] / 105);
@@ -293,7 +374,7 @@ function worksheetToBody(worksheet, sharedStrings, styles, date1904, print) {
293
374
  const href = print.hyperlinks && print.hyperlinks.length > 0 ? hyperlinkUrlAt(print.hyperlinks, absR, absC) : void 0;
294
375
  const visibleEndCol = merge ? Math.min(merge.endColumn, colWindowEnd) : 0;
295
376
  const properties = {
296
- ...merge && visibleEndCol > merge.startColumn ? { colSpan: visibleEndCol - merge.startColumn + 1 } : {},
377
+ ...merge && visibleEndCol > merge.startColumn ? { colSpan: visibleEndCol - merge.startColumn + 1 } : overflowSpan > 1 ? { colSpan: overflowSpan } : {},
297
378
  ...merge && Math.min(merge.endRow, rowWindowEnd) > merge.startRow ? { merge: "start" } : {},
298
379
  ...shading ? { shading } : {},
299
380
  ...dataBar ? { dataBar } : {},
@@ -318,7 +399,7 @@ function worksheetToBody(worksheet, sharedStrings, styles, date1904, print) {
318
399
  properties: runProps,
319
400
  ...href ? { href } : {}
320
401
  }] : [];
321
- const content = rotated ? stackedVerticalContent(text, runProps, href) : [{
402
+ const content = rotated ? stackedVerticalContent(text, runProps, href) : cellRuns.length === 0 ? [] : [{
322
403
  kind: "paragraph",
323
404
  paragraph: {
324
405
  properties: paragraphProps,
@@ -352,6 +433,13 @@ function worksheetToBody(worksheet, sharedStrings, styles, date1904, print) {
352
433
  };
353
434
  const centered = worksheet.printOptions?.horizontalCentered === true;
354
435
  const tableProperties = {
436
+ defaultCellMargins: {
437
+ left: pt(EXCEL_CELL_INSET_PT),
438
+ right: pt(EXCEL_CELL_INSET_PT),
439
+ top: pt(0),
440
+ bottom: pt(0)
441
+ },
442
+ layout: "fixed",
355
443
  ...print.gridLines ? { borders: {
356
444
  top: thin,
357
445
  bottom: thin,
@@ -386,7 +474,7 @@ function worksheetToBody(worksheet, sharedStrings, styles, date1904, print) {
386
474
  ...tableProperties,
387
475
  frozen
388
476
  } : tableProperties,
389
- grid: columnWidths.map((w) => twipsToPt(w)),
477
+ grid: bandWidths.map((w) => twipsToPt(w)),
390
478
  rows
391
479
  }
392
480
  }];
@@ -477,10 +565,29 @@ function applyCfOverride(base, o) {
477
565
  ...o.italic !== void 0 ? { italic: o.italic } : {}
478
566
  };
479
567
  }
480
- function alignmentFromXf(xf) {
481
- const align = xf.alignment;
482
- if (!align) return void 0;
483
- return mapAlignment(align.horizontal);
568
+ /**
569
+ * ECMA-376 §18.8.1 — a cell's horizontal alignment, falling back to `general`.
570
+ *
571
+ * "General" is not "left": it is decided by the VALUE. Numbers, dates and times
572
+ * go right, booleans and errors centre, text goes left. Treating an absent
573
+ * `<alignment>` as no alignment at all left every number hugging the left edge
574
+ * of its column, tens of points from where Excel and LibreOffice put it — on
575
+ * the sheets where it matters most, since a column of figures is the common
576
+ * case.
577
+ */
578
+ function alignmentFromXf(xf, type) {
579
+ const explicit = xf?.alignment ? mapAlignment(xf.alignment.horizontal) : void 0;
580
+ if (explicit) return explicit;
581
+ return generalAlignment(type);
582
+ }
583
+ function generalAlignment(type) {
584
+ switch (type) {
585
+ case "n":
586
+ case "d": return "right";
587
+ case "b":
588
+ case "e": return "center";
589
+ default: return "left";
590
+ }
484
591
  }
485
592
  function mapAlignment(h) {
486
593
  if (!h) return void 0;
@@ -640,7 +747,7 @@ function buildSparklineLookup(worksheet, print) {
640
747
  const host = parseAreaRef(sp.sqref);
641
748
  const area = parseAreaRef(sp.dataRange);
642
749
  if (!host || !area) continue;
643
- const values = collectSeriesValues(resolveSeriesGrid(sp.dataRange, worksheet, print.sheetGrids).cells, area);
750
+ const values = collectSeriesValues(resolveSeriesGrid(sp.dataRange, worksheet, print.sheetGrids).cells, area, print);
644
751
  if (values.length === 0 || values.every((v) => v === null)) continue;
645
752
  out.set(key(host.startRow, host.startColumn), {
646
753
  kind: sp.kind,
@@ -660,7 +767,7 @@ function resolveSeriesGrid(dataRange, current, sheetGrids) {
660
767
  return sheetGrids.get(name) ?? current;
661
768
  }
662
769
  var MAX_SPARKLINE_POINTS = 1e3;
663
- function collectSeriesValues(cells, area) {
770
+ function collectSeriesValues(cells, area, print) {
664
771
  const byKey = /* @__PURE__ */ new Map();
665
772
  for (const c of cells) {
666
773
  if (c.row < area.startRow || c.row > area.endRow) continue;
@@ -669,7 +776,14 @@ function collectSeriesValues(cells, area) {
669
776
  const v = Number(c.rawValue);
670
777
  if (Number.isFinite(v)) byKey.set(key(c.row, c.column), v);
671
778
  }
672
- if ((area.endRow - area.startRow + 1) * (area.endColumn - area.startColumn + 1) > MAX_SPARKLINE_POINTS) {
779
+ const cellCount = (area.endRow - area.startRow + 1) * (area.endColumn - area.startColumn + 1);
780
+ if (cellCount > MAX_SPARKLINE_POINTS) {
781
+ print.losses?.push({
782
+ severity: "degraded",
783
+ feature: FEATURES.charts,
784
+ detail: `sparkline range spans ${cellCount} cells (over ${MAX_SPARKLINE_POINTS}); plotted as the compact populated series, so blank gaps no longer hold their x-positions`,
785
+ ...print.sheetName ? { where: `sheet "${print.sheetName}"` } : {}
786
+ });
673
787
  const pts = [...byKey.entries()].map(([k, v]) => {
674
788
  const [r, col] = k.split(",").map(Number);
675
789
  return {
@@ -732,6 +846,56 @@ function buildTableFormatLookup(worksheet) {
732
846
  function cellHasContent(cell) {
733
847
  return !!cell && (cell.rawValue !== "" || cell.inlineText !== void 0);
734
848
  }
849
+ /**
850
+ * The window-local index of the last row inside `[rowStart, rowStart+rowCount)`
851
+ * that carries anything — a value, or a merge reaching into the window. `-1`
852
+ * when the window is entirely blank.
853
+ */
854
+ function lastContentRow(worksheet, rowStart, rowCount, colStart, colCount) {
855
+ const rowEnd = rowStart + rowCount - 1;
856
+ const colEnd = colStart + colCount - 1;
857
+ let last = -1;
858
+ for (const c of worksheet.cells) {
859
+ if (!cellHasContent(c)) continue;
860
+ if (c.row < rowStart || c.row > rowEnd || c.column < colStart || c.column > colEnd) continue;
861
+ if (c.row - rowStart > last) last = c.row - rowStart;
862
+ }
863
+ for (const m of worksheet.merges) {
864
+ if (m.startColumn > colEnd || m.endColumn < colStart) continue;
865
+ const reach = Math.min(m.endRow, rowEnd) - rowStart;
866
+ if (reach > last) last = reach;
867
+ }
868
+ return last;
869
+ }
870
+ /** The column twin of {@link lastContentRow}. */
871
+ function lastContentColumn(worksheet, rowStart, rowCount, colStart, colCount) {
872
+ const rowEnd = rowStart + rowCount - 1;
873
+ const colEnd = colStart + colCount - 1;
874
+ let last = -1;
875
+ for (const c of worksheet.cells) {
876
+ if (!cellHasContent(c)) continue;
877
+ if (c.row < rowStart || c.row > rowEnd || c.column < colStart || c.column > colEnd) continue;
878
+ if (c.column - colStart > last) last = c.column - colStart;
879
+ }
880
+ for (const m of worksheet.merges) {
881
+ if (m.startRow > rowEnd || m.endRow < rowStart) continue;
882
+ const reach = Math.min(m.endColumn, colEnd) - colStart;
883
+ if (reach > last) last = reach;
884
+ }
885
+ return last;
886
+ }
887
+ /**
888
+ * Whether an EMPTY cell still draws something of its own — a fill or a border.
889
+ *
890
+ * Such a cell cannot be swallowed by a neighbour's overflowing text: the span
891
+ * that gives the text its width would take the paint with it.
892
+ */
893
+ function cellPaintsSomething(cell, styles) {
894
+ if (!cell || cell.styleIndex === void 0) return false;
895
+ const xf = styles.cellXfs[cell.styleIndex];
896
+ if (!xf) return false;
897
+ return shadingFromXf(xf, styles) !== void 0 || bordersFromXf(xf, styles) !== void 0;
898
+ }
735
899
  var SLICER_WIDTH_PT = 108;
736
900
  var SLICER_ROW_PT = 16;
737
901
  var SLICER_UNSELECTED_HEX = "F2F2F2";
@@ -1,4 +1,5 @@
1
1
  import { FlowDoc } from '../core/ir/flow.js';
2
+ import { Loss } from '../core/ir/loss.js';
2
3
  import { SheetDoc } from '../core/ir/sheet.js';
3
4
  /**
4
5
  * Projection knobs (E-SHEET W9).
@@ -11,6 +12,15 @@ export interface ProjectSheetOptions {
11
12
  * byte-identical to before.
12
13
  */
13
14
  readonly now?: Date;
15
+ /**
16
+ * Sink the projection writes its {@link Loss} entries into — the print
17
+ * model's defence-in-depth caps (grid size, per-sheet text budget, sparkline
18
+ * range) fire on pathological input, and a cap that fires without saying so
19
+ * is a silent wrongness. {@link readXlsx} always supplies one and returns it
20
+ * as the read result's loss report; omitted ⇒ the caps still apply but go
21
+ * unreported.
22
+ */
23
+ readonly losses?: Array<Loss>;
14
24
  }
15
25
  /**
16
26
  * Project a {@link SheetDoc} into a {@link FlowDoc} (E-SHEET SA2): each grid sheet
@@ -18,12 +18,16 @@ var FOOTER_REL = "_xlsxFooterDefault";
18
18
  */
19
19
  function projectSheetDoc(sheet, options = {}) {
20
20
  const body = [];
21
+ const sheetSections = [];
22
+ const sheetEnds = [];
21
23
  let firstSheetSection;
22
24
  const headersFooters = /* @__PURE__ */ new Map();
23
25
  const sheetGrids = new Map(sheet.sheets.map((s) => [s.name, s.grid]));
24
26
  for (let sheetIdx = 0; sheetIdx < sheet.sheets.length; sheetIdx++) {
25
27
  const ws = sheet.sheets[sheetIdx];
26
- if (sheetIdx === 0) firstSheetSection = withHeaderFooter(sectionFromWorksheet(ws.grid), ws, headersFooters);
28
+ const sheetSection = sheetIdx === 0 ? withHeaderFooter(sectionFromWorksheet(ws.grid), ws, headersFooters) : sectionFromWorksheet(ws.grid);
29
+ if (sheetIdx === 0) firstSheetSection = sheetSection;
30
+ sheetSections.push(sheetSection);
27
31
  if (sheetIdx > 0) body.push({
28
32
  kind: "paragraph",
29
33
  paragraph: {
@@ -43,7 +47,8 @@ function projectSheetDoc(sheet, options = {}) {
43
47
  definedNames: sheet.definedNames,
44
48
  ...ws.hyperlinks ? { hyperlinks: ws.hyperlinks } : {},
45
49
  ...sheet.sharedStringRuns ? { sharedStringRuns: sheet.sharedStringRuns } : {},
46
- ...options.now ? { now: options.now } : {}
50
+ ...options.now ? { now: options.now } : {},
51
+ ...options.losses ? { losses: options.losses } : {}
47
52
  }));
48
53
  for (const ref of ws.charts ?? []) body.push({
49
54
  kind: "chart",
@@ -74,11 +79,16 @@ function projectSheetDoc(sheet, options = {}) {
74
79
  if (ws.comments && ws.comments.length > 0) body.push(...commentBlocks(ws.comments));
75
80
  if (ws.formControls && ws.formControls.length > 0) body.push(...formControlBlocks(ws.formControls));
76
81
  if (ws.activeXControls && ws.activeXControls.length > 0) body.push(...activeXBlocks(ws.activeXControls));
82
+ sheetEnds.push(body.length);
77
83
  }
84
+ const sections = sheetSections.length > 1 ? sheetSections.map((properties, i) => ({
85
+ properties,
86
+ endIndex: sheetEnds[i]
87
+ })) : [];
78
88
  return {
79
89
  kind: "flow",
80
90
  body: resolveBodyStyles(body, EMPTY_STYLE_SHEET),
81
- sections: [],
91
+ sections,
82
92
  ...firstSheetSection ? { section: firstSheetSection } : {},
83
93
  styles: EMPTY_STYLE_SHEET,
84
94
  resources: sheet.resources,
@@ -220,10 +230,7 @@ function withHeaderFooter(section, ws, headersFooters) {
220
230
  }
221
231
  if (headers.length === 0 && footers.length === 0) return section;
222
232
  return {
223
- ...section ?? {
224
- headers: [],
225
- footers: []
226
- },
233
+ ...section,
227
234
  headers,
228
235
  footers
229
236
  };
@@ -54,6 +54,7 @@ function parseWorksheet(data) {
54
54
  const sparklines = parseSparklines(wsObj);
55
55
  const tablePartRelIds = parseTableParts(wsObj);
56
56
  const printModel = {
57
+ ...parseSheetFormatPr(wsObj),
57
58
  ...pageMargins ? { pageMargins } : {},
58
59
  ...pageSetup ? { pageSetup } : {},
59
60
  ...fitToPage ? { fitToPage } : {},
@@ -245,6 +246,27 @@ function parseNumericAttr(obj, key) {
245
246
  const n = Number(raw);
246
247
  return Number.isFinite(n) ? n : void 0;
247
248
  }
249
+ /**
250
+ * ECMA-376 §18.3.1.81 `<sheetFormatPr>` — the sheet's default row height and
251
+ * column width, which apply to every row/column that does not override them.
252
+ *
253
+ * Both were previously ignored, so a row without an explicit `ht` had no height
254
+ * at all and ended up however tall its text wanted to be. A spreadsheet row has
255
+ * a definite height; text metrics do not get a vote.
256
+ */
257
+ function parseSheetFormatPr(ws) {
258
+ const node = ws["sheetFormatPr"];
259
+ if (!node || typeof node !== "object") return {};
260
+ const obj = node;
261
+ const height = parseNumericAttr(obj, "defaultRowHeight");
262
+ const width = parseNumericAttr(obj, "defaultColWidth");
263
+ const base = parseNumericAttr(obj, "baseColWidth");
264
+ return {
265
+ ...height !== void 0 && height > 0 ? { defaultRowHeightPt: height } : {},
266
+ ...width !== void 0 && width > 0 ? { defaultColWidthChars: width } : {},
267
+ ...base !== void 0 && base > 0 ? { baseColWidthChars: base } : {}
268
+ };
269
+ }
248
270
  function parseColumns(ws) {
249
271
  const colsNode = ws["cols"];
250
272
  if (!colsNode || typeof colsNode !== "object") return [];
@@ -762,16 +784,17 @@ function parseCell(c, fallbackRow, fallbackCol) {
762
784
  if (!c || typeof c !== "object") return null;
763
785
  const obj = c;
764
786
  const ref = strAttr(obj, "r");
787
+ const implied = {
788
+ column: fallbackCol,
789
+ row: fallbackRow
790
+ };
765
791
  let address;
766
792
  if (ref) try {
767
793
  address = parseCellRef(ref);
768
794
  } catch {
769
- return null;
795
+ address = numericRef(ref, fallbackRow) ?? implied;
770
796
  }
771
- else address = {
772
- column: fallbackCol,
773
- row: fallbackRow
774
- };
797
+ else address = implied;
775
798
  const type = validateCellType(strAttr(obj, "t") ?? "n");
776
799
  const styleStr = strAttr(obj, "s");
777
800
  const styleIndex = styleStr !== void 0 ? Number(styleStr) : void 0;
@@ -784,7 +807,8 @@ function parseCell(c, fallbackRow, fallbackCol) {
784
807
  };
785
808
  if (type === "inlineStr") {
786
809
  const is = obj["is"];
787
- const inlineText = inlineStringText(is);
810
+ const fromIs = inlineStringText(is);
811
+ const inlineText = fromIs !== "" ? fromIs : rawValue;
788
812
  return {
789
813
  ...base,
790
814
  rawValue: "",
@@ -798,6 +822,38 @@ function parseCell(c, fallbackRow, fallbackCol) {
798
822
  ...Number.isFinite(styleIndex) ? { styleIndex } : {}
799
823
  };
800
824
  }
825
+ var NUMERIC_REF = /^(\d+)_(\d+)$/;
826
+ /**
827
+ * Decode a numeric `"4_2"`-style reference, but only when the file corroborates
828
+ * the reading.
829
+ *
830
+ * This is not a spec spelling and there is no obligation to understand it. What
831
+ * makes it safe to act on is that the file states its own convention and can be
832
+ * checked against itself: every ref inside `<row r="2">` ends in `_2`. So the
833
+ * row half must agree with the row the cell actually sits in — and when it does
834
+ * not, the reference has told us nothing and document order (§18.3.1.4) remains
835
+ * the answer. A guess that cannot be checked is worse than the fallback; this
836
+ * one can be.
837
+ *
838
+ * The alternative is not harmless: these cells are sparse (columns 1, 4, 5, 7,
839
+ * 11, 13, 22 …), so packing them consecutively files every value under the
840
+ * wrong heading.
841
+ *
842
+ * @param ref The unparseable `r` attribute.
843
+ * @param rowIndex The 0-based row the cell sits in.
844
+ * @returns The address, or undefined when the reading is not corroborated.
845
+ */
846
+ function numericRef(ref, rowIndex) {
847
+ const m = NUMERIC_REF.exec(ref);
848
+ if (!m) return void 0;
849
+ const column = Number(m[1]);
850
+ const row = Number(m[2]);
851
+ if (column < 1 || row !== rowIndex + 1) return void 0;
852
+ return {
853
+ column: column - 1,
854
+ row: rowIndex
855
+ };
856
+ }
801
857
  function validateCellType(t) {
802
858
  if (t === "n" || t === "s" || t === "str" || t === "b" || t === "d" || t === "e" || t === "inlineStr") return t;
803
859
  return "n";
@@ -4,12 +4,16 @@ import { SheetDoc } from '../core/ir/sheet.js';
4
4
  import { ProjectSheetOptions } from './sheet-to-flow.js';
5
5
  /**
6
6
  * Read a `.xlsx` and project it to a {@link FlowDoc} in one step:
7
- * {@link readXlsxToSheetDoc} then {@link projectSheetDoc}. The xlsx reader
8
- * records no read-time losses.
7
+ * {@link readXlsxToSheetDoc} then {@link projectSheetDoc}.
8
+ *
9
+ * Parsing itself is lossless — SpreadsheetML maps cleanly onto the sheet IR.
10
+ * The losses come from the projection: the print model's defence-in-depth caps
11
+ * (grid size, per-sheet text budget, sparkline range) clip pathological sheets,
12
+ * and each clip is reported rather than applied in silence.
9
13
  *
10
14
  * @param xlsx The `.xlsx` (OPC ZIP) bytes.
11
15
  * @param options Projection knobs (the W9 reference date).
12
- * @returns The flow document plus an (empty) loss list.
16
+ * @returns The flow document plus whatever the projection had to clip.
13
17
  */
14
18
  export declare function readXlsx(xlsx: Uint8Array, options?: ProjectSheetOptions): ReadResult<FlowDoc>;
15
19
  /**
@@ -1,6 +1,6 @@
1
1
  import { ResourceStore } from "../core/ir/resources.js";
2
2
  import { FEATURES } from "../core/ir/features.js";
3
- import { bytesInclude } from "../core/bytes.js";
3
+ import { bytesInclude, bytesIncludePartName } from "../core/bytes.js";
4
4
  import { DEFAULT_THEME_PALETTE, makeColorResolver } from "../core/drawingml/colors.js";
5
5
  import { parseChart, withChartColorStyle } from "../core/drawingml/chart-parser.js";
6
6
  import { parseTheme } from "../core/drawingml/theme-parser.js";
@@ -21,7 +21,7 @@ import { parsePivotTablePart } from "./pivot-table-parser.js";
21
21
  import { parseSlicerCachePart, parseSlicerPart } from "./slicer-parser.js";
22
22
  import { parseLegacyComments, parsePersons, parseThreadedComments } from "./comments-parser.js";
23
23
  import { parseFormControlProps } from "./form-control-parser.js";
24
- import { activeXBinRelId, activeXType, parseActiveX, parseActiveXBin } from "./activex-parser.js";
24
+ import { activeXBinRelId, activeXType, activeXTypeFromPart, parseActiveX, parseActiveXBin } from "./activex-parser.js";
25
25
  import { parseSheetShapes } from "./sheet-shape-parser.js";
26
26
  import { projectSheetDoc } from "./sheet-to-flow.js";
27
27
  //#region src/excel/xlsx-reader.ts
@@ -34,19 +34,28 @@ var SLICER_CACHE_REL_TAIL = "/slicerCache";
34
34
  var THREADED_COMMENTS_REL_TAIL = "/threadedComment";
35
35
  var PERSON_REL_TAIL = "/person";
36
36
  var MAX_SLICER_ITEMS = 256;
37
+ var ACTIVEX_PART = /^xl\/activeX\//i;
37
38
  /**
38
39
  * Read a `.xlsx` and project it to a {@link FlowDoc} in one step:
39
- * {@link readXlsxToSheetDoc} then {@link projectSheetDoc}. The xlsx reader
40
- * records no read-time losses.
40
+ * {@link readXlsxToSheetDoc} then {@link projectSheetDoc}.
41
+ *
42
+ * Parsing itself is lossless — SpreadsheetML maps cleanly onto the sheet IR.
43
+ * The losses come from the projection: the print model's defence-in-depth caps
44
+ * (grid size, per-sheet text budget, sparkline range) clip pathological sheets,
45
+ * and each clip is reported rather than applied in silence.
41
46
  *
42
47
  * @param xlsx The `.xlsx` (OPC ZIP) bytes.
43
48
  * @param options Projection knobs (the W9 reference date).
44
- * @returns The flow document plus an (empty) loss list.
49
+ * @returns The flow document plus whatever the projection had to clip.
45
50
  */
46
51
  function readXlsx(xlsx, options = {}) {
52
+ const losses = [];
47
53
  return {
48
- doc: projectSheetDoc(readXlsxToSheetDoc(xlsx), options),
49
- losses: []
54
+ doc: projectSheetDoc(readXlsxToSheetDoc(xlsx), {
55
+ ...options,
56
+ losses
57
+ }),
58
+ losses
50
59
  };
51
60
  }
52
61
  /**
@@ -211,13 +220,33 @@ function readXlsxToSheetDoc(xlsx) {
211
220
  }
212
221
  if (resolvedComments.length > 0) comments = resolvedComments;
213
222
  }
223
+ const wsRels = pkg.getPartRelationships(resolved.path);
224
+ const resolvedAx = [];
225
+ const activeXState = (part) => {
226
+ const props = parseActiveX(part.data);
227
+ const binRelId = activeXBinRelId(part.data);
228
+ const binRel = binRelId ? pkg.getPartRelationships(part.path).find((r) => r.id === binRelId) : void 0;
229
+ const binPart = binRel ? pkg.resolveRelatedPart(part.path, binRel) : void 0;
230
+ return {
231
+ type: "control",
232
+ ...binPart ? parseActiveXBin(binPart.data) : {},
233
+ ...props
234
+ };
235
+ };
214
236
  let formControls;
215
237
  if (worksheet.formControls && worksheet.formControls.length > 0) {
216
- const wsRels = pkg.getPartRelationships(resolved.path);
217
238
  const resolvedControls = [];
218
239
  for (const fc of worksheet.formControls) {
219
240
  const rel = wsRels.find((r) => r.id === fc.relId);
220
241
  const part = rel ? pkg.resolveRelatedPart(resolved.path, rel) : void 0;
242
+ if (part && ACTIVEX_PART.test(part.path)) {
243
+ resolvedAx.push({
244
+ ...activeXState(part),
245
+ type: activeXTypeFromPart(part.data),
246
+ ...fc.name ? { name: fc.name } : {}
247
+ });
248
+ continue;
249
+ }
221
250
  const props = part ? parseFormControlProps(part.data) : {};
222
251
  resolvedControls.push({
223
252
  ...fc.name ? { name: fc.name } : {},
@@ -226,29 +255,15 @@ function readXlsxToSheetDoc(xlsx) {
226
255
  }
227
256
  if (resolvedControls.length > 0) formControls = resolvedControls;
228
257
  }
229
- let activeXControls;
230
- if (worksheet.oleObjects && worksheet.oleObjects.length > 0) {
231
- const wsRels = pkg.getPartRelationships(resolved.path);
232
- const resolvedAx = [];
233
- for (const ole of worksheet.oleObjects) {
234
- const rel = wsRels.find((r) => r.id === ole.relId);
235
- const part = rel ? pkg.resolveRelatedPart(resolved.path, rel) : void 0;
236
- const props = part ? parseActiveX(part.data) : {};
237
- let binProps = {};
238
- if (part) {
239
- const binRelId = activeXBinRelId(part.data);
240
- const binRel = binRelId ? pkg.getPartRelationships(part.path).find((r) => r.id === binRelId) : void 0;
241
- const binPart = binRel ? pkg.resolveRelatedPart(part.path, binRel) : void 0;
242
- if (binPart) binProps = parseActiveXBin(binPart.data);
243
- }
244
- resolvedAx.push({
245
- type: activeXType(ole.progId),
246
- ...binProps,
247
- ...props
248
- });
249
- }
250
- if (resolvedAx.length > 0) activeXControls = resolvedAx;
258
+ if (worksheet.oleObjects && worksheet.oleObjects.length > 0) for (const ole of worksheet.oleObjects) {
259
+ const rel = wsRels.find((r) => r.id === ole.relId);
260
+ const part = rel ? pkg.resolveRelatedPart(resolved.path, rel) : void 0;
261
+ resolvedAx.push({
262
+ ...part ? activeXState(part) : {},
263
+ type: activeXType(ole.progId)
264
+ });
251
265
  }
266
+ const activeXControls = resolvedAx.length > 0 ? resolvedAx : void 0;
252
267
  const grid = tables || pivotTables ? {
253
268
  ...worksheet,
254
269
  ...tables ? { tables } : {},
@@ -414,7 +429,7 @@ var xlsxReader = {
414
429
  id: "xlsx",
415
430
  produces: "sheet",
416
431
  supports: new Set([FEATURES.text, FEATURES.tables]),
417
- sniff: (bytes) => bytes[0] === 80 && bytes[1] === 75 && bytesInclude(bytes, "xl/workbook.xml"),
432
+ sniff: (bytes) => bytes[0] === 80 && bytes[1] === 75 && bytesIncludePartName(bytes, "xl/workbook.xml"),
418
433
  read: (bytes) => ({
419
434
  doc: readXlsxToSheetDoc(bytes),
420
435
  losses: []
@@ -1891,6 +1891,7 @@ var PageAssembler = class {
1891
1891
  this.bookmarkPositions = bookmarkPositions;
1892
1892
  this.ctx = sectionCtxs[0];
1893
1893
  this.cursorY = this.ctx.pageHeight - this.ctx.marginTop;
1894
+ this.colStartY = this.cursorY;
1894
1895
  }
1895
1896
  pages = [];
1896
1897
  ctx;
@@ -1908,18 +1909,33 @@ var PageAssembler = class {
1908
1909
  */
1909
1910
  colIdx = 0;
1910
1911
  colStartLen = 0;
1912
+ /** The cursor's y at the top of the current column — see {@link colHasContent}. */
1913
+ colStartY;
1911
1914
  /** The current column's left edge (`marginLeft` plus the column x-offset). */
1912
1915
  colLeft = () => this.ctx.marginLeft + (this.ctx.columns?.[this.colIdx]?.xOffsetPt ?? 0);
1913
1916
  /** The current column's width (the section content width when single-column). */
1914
1917
  colWidth = () => this.ctx.columns?.[this.colIdx]?.widthPt ?? this.ctx.contentWidth;
1915
- /** Whether the current column has received any items yet. */
1916
- colHasContent = () => this.current.length > this.colStartLen;
1918
+ /**
1919
+ * Whether the current column is already in use.
1920
+ *
1921
+ * This gates every overflow break, so that a block too tall for an empty page
1922
+ * is placed rather than looping forever. It used to ask whether the column had
1923
+ * received any drawable ITEM, which is not the same question: a run of empty
1924
+ * table rows consumes vertical space and draws nothing, so the guard stayed
1925
+ * false, no break ever fired, and the rows marched off the bottom of the page.
1926
+ * A spreadsheet is full of such rows — the sheet in tdf171828.xlsx has 162 of
1927
+ * them — and they were silently costing whole pages of pagination.
1928
+ *
1929
+ * Consumed space counts as use, whether or not any ink went with it.
1930
+ */
1931
+ colHasContent = () => this.current.length > this.colStartLen || this.cursorY < this.colStartY;
1917
1932
  /** Overflow step: next column on this page, or a fresh page after the last. */
1918
1933
  advanceColumn = () => {
1919
1934
  if (this.ctx.columns && this.colIdx + 1 < this.ctx.columns.length) {
1920
1935
  this.colIdx++;
1921
1936
  this.colStartLen = this.current.length;
1922
1937
  this.cursorY = this.ctx.pageHeight - this.ctx.marginTop;
1938
+ this.colStartY = this.cursorY;
1923
1939
  } else this.flushPage();
1924
1940
  };
1925
1941
  /**
@@ -2098,7 +2114,7 @@ var PageAssembler = class {
2098
2114
  * guarantee one page for a header/footer-only document).
2099
2115
  */
2100
2116
  flushPage = (force = false) => {
2101
- if (this.current.length === 0 && !force) return;
2117
+ if (this.current.length === 0 && this.cursorY >= this.colStartY && !force) return;
2102
2118
  const band = bandForPage(this.pageInSection, this.globalPageIdx, this.ctx.titlePg, this.ctx.evenAndOddHeaders);
2103
2119
  const header = pickBand(this.ctx.headerSet, band);
2104
2120
  const footer = pickBand(this.ctx.footerSet, band);
@@ -2137,6 +2153,7 @@ var PageAssembler = class {
2137
2153
  this.pageInSection++;
2138
2154
  this.globalPageIdx++;
2139
2155
  this.cursorY = this.ctx.pageHeight - this.ctx.marginTop;
2156
+ this.colStartY = this.cursorY;
2140
2157
  };
2141
2158
  };
2142
2159
  function paginateSections(blocks, sectionCtxs, builder, defaultLang = "en-US", notes, bookmarkPositions, reflowParagraph) {
@@ -4,7 +4,7 @@ import { FEATURES } from "../core/ir/features.js";
4
4
  import { EMPTY_STYLE_SHEET, resolveBodyStyles } from "../core/style-cascade/resolver.js";
5
5
  import "../core/style-cascade/index.js";
6
6
  import { poAttr, poChildren, poFindDescendant, poIntAttr, poIs } from "../core/po-helpers.js";
7
- import { bytesInclude } from "../core/bytes.js";
7
+ import { bytesIncludePartName } from "../core/bytes.js";
8
8
  import { DEFAULT_THEME_PALETTE, defaultColorResolver, makeColorResolver } from "../core/drawingml/colors.js";
9
9
  import { parseChart, withChartColorStyle } from "../core/drawingml/chart-parser.js";
10
10
  import { parseTheme } from "../core/drawingml/theme-parser.js";
@@ -269,7 +269,7 @@ var pptxReader = {
269
269
  FEATURES.charts,
270
270
  FEATURES.tables
271
271
  ]),
272
- sniff: (bytes) => bytes[0] === 80 && bytes[1] === 75 && bytesInclude(bytes, "ppt/presentation.xml"),
272
+ sniff: (bytes) => bytes[0] === 80 && bytes[1] === 75 && bytesIncludePartName(bytes, "ppt/presentation.xml"),
273
273
  read: (bytes) => readPptx(bytes)
274
274
  };
275
275
  //#endregion
@@ -5,7 +5,7 @@ import { EMPTY_STYLE_SHEET, resolveBodyStyles, resolveHeadersFootersStyles } fro
5
5
  import { resolveTableStyles } from "../core/style-cascade/table.js";
6
6
  import "../core/style-cascade/index.js";
7
7
  import { poFindDescendant } from "../core/po-helpers.js";
8
- import { bytesInclude } from "../core/bytes.js";
8
+ import { bytesIncludePartName } from "../core/bytes.js";
9
9
  import { DEFAULT_THEME_PALETTE, makeColorResolver } from "../core/drawingml/colors.js";
10
10
  import { parseChart, withChartColorStyle } from "../core/drawingml/chart-parser.js";
11
11
  import { parseTheme } from "../core/drawingml/theme-parser.js";
@@ -144,7 +144,7 @@ var docxReader = {
144
144
  FEATURES.trackedChanges,
145
145
  FEATURES.fontsEmbedding
146
146
  ]),
147
- sniff: (bytes) => bytes[0] === 80 && bytes[1] === 75 && bytesInclude(bytes, "word/document.xml"),
147
+ sniff: (bytes) => bytes[0] === 80 && bytes[1] === 75 && bytesIncludePartName(bytes, "word/document.xml"),
148
148
  read: (bytes) => readDocx(bytes)
149
149
  };
150
150
  function infoFromCore(core) {
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "reamkit",
3
- "version": "1.15.2",
3
+ "version": "1.15.4",
4
4
  "description": "Ream — convert DOCX, XLSX, PPTX and PDF to PDF, SVG, HTML, DOCX and XLSX, built from scratch on the ECMA-376 and ISO 32000 specifications. Parse once, convert anywhere.",
5
5
  "license": "MIT",
6
6
  "author": "Alex Krassavin <info@reamkit.dev>",
@@ -70,7 +70,10 @@
70
70
  "corpus:sandbox:build": "docker build -t docgen-losandbox:latest scripts/corpus/sandbox/",
71
71
  "corpus": "tsx scripts/corpus/run.ts",
72
72
  "corpus:roundtrip": "tsx scripts/corpus/roundtrip.ts",
73
- "corpus:roundtrip:xlsx": "tsx scripts/corpus/xlsx-roundtrip.ts"
73
+ "corpus:roundtrip:xlsx": "tsx scripts/corpus/xlsx-roundtrip.ts",
74
+ "corpus:xlsx:invariants": "tsx scripts/corpus/xlsx-invariants.ts",
75
+ "corpus:fixtures": "tsx scripts/corpus/sync-real-fixtures.ts",
76
+ "corpus:golden": "tsx scripts/corpus/make-golden.ts"
74
77
  },
75
78
  "dependencies": {
76
79
  "fast-xml-parser": "5.7.0",