pdf-parser.js 5.1.0 → 5.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +19 -1
- package/dist/codec.d.cts +1 -0
- package/dist/codec.d.ts +1 -0
- package/dist/font-read.cjs +94 -12
- package/dist/font-read.d.cts +7 -1
- package/dist/font-read.d.ts +7 -1
- package/dist/font-read.js +94 -12
- package/dist/interpret.cjs +23 -7
- package/dist/interpret.d.cts +9 -1
- package/dist/interpret.d.ts +9 -1
- package/dist/interpret.js +23 -7
- package/dist/layout.cjs +1 -0
- package/dist/layout.d.cts +4 -0
- package/dist/layout.d.ts +4 -0
- package/dist/layout.js +1 -0
- package/dist/raster.cjs +14 -8
- package/dist/raster.js +14 -8
- package/dist/read.cjs +1 -0
- package/dist/read.js +1 -0
- package/dist/text-group.cjs +26 -16
- package/dist/text-group.d.cts +3 -1
- package/dist/text-group.d.ts +3 -1
- package/dist/text-group.js +26 -16
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -333,7 +333,25 @@ A run whose `widthPt` is absent states no advance, so where it ends is unknown a
|
|
|
333
333
|
|
|
334
334
|
Gaps that are derivable become a word space at an eighth of an em (the narrowest genuine one: the standard fourteen faces set their space glyph at 250/1000 em, and justified setting compresses to no less than half of that) and a column boundary past a full em (wider than the em space, the widest single space character there is), so a table's row reads as `"North\t4.2m\t11%"` rather than as one sentence. `runsShareBaseline` and `runGapPt` are exported on their own for a consumer with its own pipeline around them, and every threshold is overridable.
|
|
335
335
|
|
|
336
|
-
Limits, each inherent to what a PDF states rather than to the implementation: grouping is by baseline alone with no page segmentation, so on a multi-column page whose columns sit at different vertical offsets a line of one can fall within tolerance of a line of the next and the two are reported as one line with a column boundary between them, which is the cut a caller needs (`document-outline.js`'s `segmentPdfRegions` finds the gutter first if you want them genuinely apart); text comes out in visual order, so a right-to-left script needs the Unicode Bidi Algorithm applied to the result;
|
|
336
|
+
Limits, each inherent to what a PDF states rather than to the implementation: grouping is by baseline alone with no page segmentation, so on a multi-column page whose columns sit at different vertical offsets a line of one can fall within tolerance of a line of the next and the two are reported as one line with a column boundary between them, which is the cut a caller needs (`document-outline.js`'s `segmentPdfRegions` finds the gutter first if you want them genuinely apart); text comes out in visual order, so a right-to-left script needs the Unicode Bidi Algorithm applied to the result; a vertically set run groups into a column instead of a line, reported with a `rotationDeg` of 270 and its own `writingMode`, with columns ordered right to left; and a word hyphenated across a line end stays split, hyphen intact, because rejoining it is a paragraph-level judgement this package leaves to its consumers.
|
|
337
|
+
|
|
338
|
+
## Vertical writing mode
|
|
339
|
+
|
|
340
|
+
A composite font's `/Encoding` CMap chooses a writing mode (ISO 32000-1 9.7.5.1). Mode 1 is vertical: the glyphs stay upright but the text position advances **down** the page, by the descendant CIDFont's own `/DW2` and `/W2` vertical metrics rather than by its horizontal widths, and each glyph paints offset from that position by its own position vector. `readPdf` reads all of it ([#1358](https://github.com/ExaDev/documents.js/issues/1358)), from a predefined CMap's `-V` name, from the bare `V` CMap, or from an embedded CMap stream's own `/WMode`.
|
|
341
|
+
|
|
342
|
+
A vertically set run carries `writingMode: "vertical"`, and that is a separate fact from `rotationDeg`, which reports the text rendering matrix turning the glyphs themselves on their side. An ordinary Japanese column is `writingMode: "vertical"` with a `rotationDeg` of 0: upright glyphs, running downward. On such a run, `xPt`/`yPt` is the top of the column and `widthPt` measures downward.
|
|
343
|
+
|
|
344
|
+
```ts
|
|
345
|
+
const runs = page.items.filter((item) => item.kind === "text");
|
|
346
|
+
for (const line of groupPdfTextRuns(runs)) {
|
|
347
|
+
// A vertical column reports rotationDeg 270 (the direction it advances) and writingMode "vertical".
|
|
348
|
+
console.log(line.writingMode ?? "horizontal", line.text);
|
|
349
|
+
}
|
|
350
|
+
```
|
|
351
|
+
|
|
352
|
+
`groupPdfTextRuns` groups such runs into columns, ordered right to left, and never lets a column and a line merge even when they advance along the same axis: a column of upright glyphs and a line of glyphs turned on their side are not the same text, whatever direction they share.
|
|
353
|
+
|
|
354
|
+
What is not supported: a **predefined non-Identity CMap's own byte decoding**. `90ms-RKSJ-V` and its siblings remap multi-byte character codes to CIDs through tables this package does not carry, so a font using one still has its codes read as two-byte CIDs. That is this package's long-standing composite-font limit, unchanged by and unrelated to writing mode; `Identity-V` (what modern producers emit) is read exactly. `writePdf` produces horizontal text only, and has no way to request a vertical one.
|
|
337
355
|
|
|
338
356
|
## JBIG2 scope
|
|
339
357
|
|
package/dist/codec.d.cts
CHANGED
|
@@ -54,6 +54,7 @@ declare const pdfCodec: z.ZodCodec<z.ZodCustom<Uint8Array<ArrayBuffer>, Uint8Arr
|
|
|
54
54
|
}, z.core.$strip>;
|
|
55
55
|
widthPt: z.ZodOptional<z.ZodNumber>;
|
|
56
56
|
rotationDeg: z.ZodOptional<z.ZodNumber>;
|
|
57
|
+
writingMode: z.ZodOptional<z.ZodLiteral<"vertical">>;
|
|
57
58
|
underline: z.ZodOptional<z.ZodBoolean>;
|
|
58
59
|
layer: z.ZodOptional<z.ZodString>;
|
|
59
60
|
actualText: z.ZodOptional<z.ZodString>;
|
package/dist/codec.d.ts
CHANGED
|
@@ -54,6 +54,7 @@ declare const pdfCodec: z.ZodCodec<z.ZodCustom<Uint8Array<ArrayBuffer>, Uint8Arr
|
|
|
54
54
|
}, z.core.$strip>;
|
|
55
55
|
widthPt: z.ZodOptional<z.ZodNumber>;
|
|
56
56
|
rotationDeg: z.ZodOptional<z.ZodNumber>;
|
|
57
|
+
writingMode: z.ZodOptional<z.ZodLiteral<"vertical">>;
|
|
57
58
|
underline: z.ZodOptional<z.ZodBoolean>;
|
|
58
59
|
layer: z.ZodOptional<z.ZodString>;
|
|
59
60
|
actualText: z.ZodOptional<z.ZodString>;
|
package/dist/font-read.cjs
CHANGED
|
@@ -177,6 +177,63 @@ function readCidWidths(w) {
|
|
|
177
177
|
return map;
|
|
178
178
|
}
|
|
179
179
|
const DEFAULT_CID_WIDTH = 1e3;
|
|
180
|
+
const DEFAULT_VERTICAL_POSITION_Y = 880;
|
|
181
|
+
const DEFAULT_VERTICAL_DISPLACEMENT_Y = -1e3;
|
|
182
|
+
const DW2_ENTRY_COUNT = 2;
|
|
183
|
+
const W2_TRIPLET_LENGTH = 3;
|
|
184
|
+
const VERTICAL_WRITING_MODE = 1;
|
|
185
|
+
function readCidVerticalMetrics(w2) {
|
|
186
|
+
const map = /* @__PURE__ */ new Map();
|
|
187
|
+
if (w2 === void 0) return map;
|
|
188
|
+
let i = 0;
|
|
189
|
+
while (i < w2.length) {
|
|
190
|
+
const first = require_objects.asNumber(w2[i]);
|
|
191
|
+
if (first === void 0) {
|
|
192
|
+
i++;
|
|
193
|
+
continue;
|
|
194
|
+
}
|
|
195
|
+
const next = w2[i + 1];
|
|
196
|
+
if (next?.kind === "array") {
|
|
197
|
+
tripletsOf(next.items).forEach((triplet, index) => {
|
|
198
|
+
const entry = verticalEntryFrom(require_objects.asNumber(triplet[0]), require_objects.asNumber(triplet[1]), require_objects.asNumber(triplet[2]));
|
|
199
|
+
if (entry !== void 0) map.set(first + index, entry);
|
|
200
|
+
});
|
|
201
|
+
i += 2;
|
|
202
|
+
continue;
|
|
203
|
+
}
|
|
204
|
+
const last = require_objects.asNumber(next);
|
|
205
|
+
const entry = verticalEntryFrom(require_objects.asNumber(w2[i + 2]), require_objects.asNumber(w2[i + 3]), require_objects.asNumber(w2[i + 4]));
|
|
206
|
+
if (last !== void 0 && entry !== void 0) for (let cid = first; cid <= last; cid++) map.set(cid, entry);
|
|
207
|
+
i += 5;
|
|
208
|
+
}
|
|
209
|
+
return map;
|
|
210
|
+
}
|
|
211
|
+
function tripletsOf(items) {
|
|
212
|
+
const groups = [];
|
|
213
|
+
for (const item of items) {
|
|
214
|
+
const last = groups[groups.length - 1];
|
|
215
|
+
if (last !== void 0 && last.length < W2_TRIPLET_LENGTH) last.push(item);
|
|
216
|
+
else groups.push([item]);
|
|
217
|
+
}
|
|
218
|
+
return groups;
|
|
219
|
+
}
|
|
220
|
+
function verticalEntryFrom(displacementY, positionX, positionY) {
|
|
221
|
+
if (displacementY === void 0 || positionX === void 0 || positionY === void 0) return;
|
|
222
|
+
return {
|
|
223
|
+
displacementY,
|
|
224
|
+
positionX,
|
|
225
|
+
positionY
|
|
226
|
+
};
|
|
227
|
+
}
|
|
228
|
+
function predefinedCMapIsVertical(name) {
|
|
229
|
+
return name === "V" || name.endsWith("-V");
|
|
230
|
+
}
|
|
231
|
+
function readsVertically(fontDict, context) {
|
|
232
|
+
const encoding = context.resolver.resolve(require_objects.dictGet(fontDict, "Encoding"));
|
|
233
|
+
if (encoding?.kind === "name") return predefinedCMapIsVertical(encoding.name);
|
|
234
|
+
if (encoding?.kind === "stream") return require_objects.asNumber(require_objects.dictGet(encoding.dict, "WMode")) === VERTICAL_WRITING_MODE;
|
|
235
|
+
return false;
|
|
236
|
+
}
|
|
180
237
|
function buildCompositeFont(fontDict, context) {
|
|
181
238
|
const baseFont = require_objects.asName(require_objects.dictGet(fontDict, "BaseFont")) ?? "Helvetica";
|
|
182
239
|
const descendants = require_objects.asArray(require_objects.dictGet(fontDict, "DescendantFonts"));
|
|
@@ -186,9 +243,22 @@ function buildCompositeFont(fontDict, context) {
|
|
|
186
243
|
const dw = descendantDict !== void 0 ? require_objects.asNumber(require_objects.dictGet(descendantDict, "DW")) ?? DEFAULT_CID_WIDTH : DEFAULT_CID_WIDTH;
|
|
187
244
|
const widthMap = readCidWidths(descendantDict !== void 0 ? require_objects.asArray(require_objects.dictGet(descendantDict, "W")) : void 0);
|
|
188
245
|
const widthOf = (cid) => widthMap.get(cid) ?? dw;
|
|
246
|
+
const dw2 = descendantDict !== void 0 ? require_objects.asArray(require_objects.dictGet(descendantDict, "DW2")) : void 0;
|
|
247
|
+
const defaultPositionY = (dw2?.length === DW2_ENTRY_COUNT ? require_objects.asNumber(dw2[0]) : void 0) ?? DEFAULT_VERTICAL_POSITION_Y;
|
|
248
|
+
const defaultDisplacementY = (dw2?.length === DW2_ENTRY_COUNT ? require_objects.asNumber(dw2[1]) : void 0) ?? DEFAULT_VERTICAL_DISPLACEMENT_Y;
|
|
249
|
+
const verticalMap = readCidVerticalMetrics(descendantDict !== void 0 ? require_objects.asArray(require_objects.dictGet(descendantDict, "W2")) : void 0);
|
|
250
|
+
const verticalMetricsOf = (cid) => {
|
|
251
|
+
const entry = verticalMap.get(cid);
|
|
252
|
+
return {
|
|
253
|
+
displacementY: entry?.displacementY ?? defaultDisplacementY,
|
|
254
|
+
positionX: entry?.positionX ?? widthOf(cid) / 2,
|
|
255
|
+
positionY: entry?.positionY ?? defaultPositionY
|
|
256
|
+
};
|
|
257
|
+
};
|
|
189
258
|
const toUnicode = readToUnicodeCMap(fontDict, context);
|
|
190
259
|
const cidToGidMap = descendantDict !== void 0 ? require_objects.dictGet(descendantDict, "CIDToGIDMap") : void 0;
|
|
191
|
-
const
|
|
260
|
+
const encodingName = require_objects.dictGet(fontDict, "Encoding");
|
|
261
|
+
const programEncoding = (require_objects.isName(encodingName, "Identity-H") || require_objects.isName(encodingName, "Identity-V")) && (cidToGidMap === void 0 || require_objects.isName(cidToGidMap, "Identity")) ? lazyFontProgram(descriptor, context) : void 0;
|
|
192
262
|
const decodeToUnicode = (codes) => {
|
|
193
263
|
let out = "";
|
|
194
264
|
let glyphCount = 0;
|
|
@@ -222,7 +292,8 @@ function buildCompositeFont(fontDict, context) {
|
|
|
222
292
|
bold,
|
|
223
293
|
italic,
|
|
224
294
|
widthOf,
|
|
225
|
-
decodeToUnicode
|
|
295
|
+
decodeToUnicode,
|
|
296
|
+
...readsVertically(fontDict, context) ? { verticalMetricsOf } : {}
|
|
226
297
|
};
|
|
227
298
|
}
|
|
228
299
|
function createFontResolver(context) {
|
|
@@ -237,16 +308,27 @@ function createFontResolver(context) {
|
|
|
237
308
|
return resolved;
|
|
238
309
|
};
|
|
239
310
|
return {
|
|
240
|
-
metrics: {
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
|
|
244
|
-
|
|
245
|
-
|
|
246
|
-
|
|
247
|
-
|
|
248
|
-
|
|
249
|
-
|
|
311
|
+
metrics: {
|
|
312
|
+
glyphAdvance(fontResourceName, resources, codes, byteOffset) {
|
|
313
|
+
const font = resolve(fontResourceName, resources);
|
|
314
|
+
if (font === void 0) return;
|
|
315
|
+
const byteLengthConsumed = font.composite ? 2 : 1;
|
|
316
|
+
const code = font.composite ? (codes[byteOffset] ?? 0) << 8 | (codes[byteOffset + 1] ?? 0) : codes[byteOffset] ?? 0;
|
|
317
|
+
const vertical = font.verticalMetricsOf?.(code);
|
|
318
|
+
return {
|
|
319
|
+
widthPer1000: font.widthOf(code),
|
|
320
|
+
byteLengthConsumed,
|
|
321
|
+
...vertical === void 0 ? {} : { vertical: {
|
|
322
|
+
displacementPer1000: vertical.displacementY,
|
|
323
|
+
positionXPer1000: vertical.positionX,
|
|
324
|
+
positionYPer1000: vertical.positionY
|
|
325
|
+
} }
|
|
326
|
+
};
|
|
327
|
+
},
|
|
328
|
+
isVertical(fontResourceName, resources) {
|
|
329
|
+
return resolve(fontResourceName, resources)?.verticalMetricsOf !== void 0;
|
|
330
|
+
}
|
|
331
|
+
},
|
|
250
332
|
resolve
|
|
251
333
|
};
|
|
252
334
|
}
|
package/dist/font-read.d.cts
CHANGED
|
@@ -2,6 +2,11 @@ import { PdfDiagnosticSink } from "./diagnostics.cjs";
|
|
|
2
2
|
import { n as PdfDict } from "./objects-B7HdMeWS.cjs";
|
|
3
3
|
import { FontMetricsPort, PdfObjectResolver } from "./interpret.cjs";
|
|
4
4
|
//#region src/font-read.d.ts
|
|
5
|
+
interface VerticalGlyphMetrics {
|
|
6
|
+
readonly displacementY: number;
|
|
7
|
+
readonly positionX: number;
|
|
8
|
+
readonly positionY: number;
|
|
9
|
+
}
|
|
5
10
|
interface PdfFont {
|
|
6
11
|
readonly composite: boolean;
|
|
7
12
|
readonly family: string;
|
|
@@ -9,6 +14,7 @@ interface PdfFont {
|
|
|
9
14
|
readonly italic: boolean;
|
|
10
15
|
widthOf(code: number): number;
|
|
11
16
|
decodeToUnicode(codes: Uint8Array<ArrayBuffer>): string;
|
|
17
|
+
readonly verticalMetricsOf?: (code: number) => VerticalGlyphMetrics;
|
|
12
18
|
}
|
|
13
19
|
interface FontReadContext {
|
|
14
20
|
readonly resolver: PdfObjectResolver;
|
|
@@ -20,4 +26,4 @@ interface FontResolverService {
|
|
|
20
26
|
}
|
|
21
27
|
declare function createFontResolver(context: FontReadContext): FontResolverService;
|
|
22
28
|
//#endregion
|
|
23
|
-
export { FontReadContext, FontResolverService, PdfFont, createFontResolver };
|
|
29
|
+
export { FontReadContext, FontResolverService, PdfFont, VerticalGlyphMetrics, createFontResolver };
|
package/dist/font-read.d.ts
CHANGED
|
@@ -2,6 +2,11 @@ import { PdfDiagnosticSink } from "./diagnostics.js";
|
|
|
2
2
|
import { n as PdfDict } from "./objects-B7HdMeWS.js";
|
|
3
3
|
import { FontMetricsPort, PdfObjectResolver } from "./interpret.js";
|
|
4
4
|
//#region src/font-read.d.ts
|
|
5
|
+
interface VerticalGlyphMetrics {
|
|
6
|
+
readonly displacementY: number;
|
|
7
|
+
readonly positionX: number;
|
|
8
|
+
readonly positionY: number;
|
|
9
|
+
}
|
|
5
10
|
interface PdfFont {
|
|
6
11
|
readonly composite: boolean;
|
|
7
12
|
readonly family: string;
|
|
@@ -9,6 +14,7 @@ interface PdfFont {
|
|
|
9
14
|
readonly italic: boolean;
|
|
10
15
|
widthOf(code: number): number;
|
|
11
16
|
decodeToUnicode(codes: Uint8Array<ArrayBuffer>): string;
|
|
17
|
+
readonly verticalMetricsOf?: (code: number) => VerticalGlyphMetrics;
|
|
12
18
|
}
|
|
13
19
|
interface FontReadContext {
|
|
14
20
|
readonly resolver: PdfObjectResolver;
|
|
@@ -20,4 +26,4 @@ interface FontResolverService {
|
|
|
20
26
|
}
|
|
21
27
|
declare function createFontResolver(context: FontReadContext): FontResolverService;
|
|
22
28
|
//#endregion
|
|
23
|
-
export { FontReadContext, FontResolverService, PdfFont, createFontResolver };
|
|
29
|
+
export { FontReadContext, FontResolverService, PdfFont, VerticalGlyphMetrics, createFontResolver };
|
package/dist/font-read.js
CHANGED
|
@@ -176,6 +176,63 @@ function readCidWidths(w) {
|
|
|
176
176
|
return map;
|
|
177
177
|
}
|
|
178
178
|
const DEFAULT_CID_WIDTH = 1e3;
|
|
179
|
+
const DEFAULT_VERTICAL_POSITION_Y = 880;
|
|
180
|
+
const DEFAULT_VERTICAL_DISPLACEMENT_Y = -1e3;
|
|
181
|
+
const DW2_ENTRY_COUNT = 2;
|
|
182
|
+
const W2_TRIPLET_LENGTH = 3;
|
|
183
|
+
const VERTICAL_WRITING_MODE = 1;
|
|
184
|
+
function readCidVerticalMetrics(w2) {
|
|
185
|
+
const map = /* @__PURE__ */ new Map();
|
|
186
|
+
if (w2 === void 0) return map;
|
|
187
|
+
let i = 0;
|
|
188
|
+
while (i < w2.length) {
|
|
189
|
+
const first = asNumber(w2[i]);
|
|
190
|
+
if (first === void 0) {
|
|
191
|
+
i++;
|
|
192
|
+
continue;
|
|
193
|
+
}
|
|
194
|
+
const next = w2[i + 1];
|
|
195
|
+
if (next?.kind === "array") {
|
|
196
|
+
tripletsOf(next.items).forEach((triplet, index) => {
|
|
197
|
+
const entry = verticalEntryFrom(asNumber(triplet[0]), asNumber(triplet[1]), asNumber(triplet[2]));
|
|
198
|
+
if (entry !== void 0) map.set(first + index, entry);
|
|
199
|
+
});
|
|
200
|
+
i += 2;
|
|
201
|
+
continue;
|
|
202
|
+
}
|
|
203
|
+
const last = asNumber(next);
|
|
204
|
+
const entry = verticalEntryFrom(asNumber(w2[i + 2]), asNumber(w2[i + 3]), asNumber(w2[i + 4]));
|
|
205
|
+
if (last !== void 0 && entry !== void 0) for (let cid = first; cid <= last; cid++) map.set(cid, entry);
|
|
206
|
+
i += 5;
|
|
207
|
+
}
|
|
208
|
+
return map;
|
|
209
|
+
}
|
|
210
|
+
function tripletsOf(items) {
|
|
211
|
+
const groups = [];
|
|
212
|
+
for (const item of items) {
|
|
213
|
+
const last = groups[groups.length - 1];
|
|
214
|
+
if (last !== void 0 && last.length < W2_TRIPLET_LENGTH) last.push(item);
|
|
215
|
+
else groups.push([item]);
|
|
216
|
+
}
|
|
217
|
+
return groups;
|
|
218
|
+
}
|
|
219
|
+
function verticalEntryFrom(displacementY, positionX, positionY) {
|
|
220
|
+
if (displacementY === void 0 || positionX === void 0 || positionY === void 0) return;
|
|
221
|
+
return {
|
|
222
|
+
displacementY,
|
|
223
|
+
positionX,
|
|
224
|
+
positionY
|
|
225
|
+
};
|
|
226
|
+
}
|
|
227
|
+
function predefinedCMapIsVertical(name) {
|
|
228
|
+
return name === "V" || name.endsWith("-V");
|
|
229
|
+
}
|
|
230
|
+
function readsVertically(fontDict, context) {
|
|
231
|
+
const encoding = context.resolver.resolve(dictGet(fontDict, "Encoding"));
|
|
232
|
+
if (encoding?.kind === "name") return predefinedCMapIsVertical(encoding.name);
|
|
233
|
+
if (encoding?.kind === "stream") return asNumber(dictGet(encoding.dict, "WMode")) === VERTICAL_WRITING_MODE;
|
|
234
|
+
return false;
|
|
235
|
+
}
|
|
179
236
|
function buildCompositeFont(fontDict, context) {
|
|
180
237
|
const baseFont = asName(dictGet(fontDict, "BaseFont")) ?? "Helvetica";
|
|
181
238
|
const descendants = asArray(dictGet(fontDict, "DescendantFonts"));
|
|
@@ -185,9 +242,22 @@ function buildCompositeFont(fontDict, context) {
|
|
|
185
242
|
const dw = descendantDict !== void 0 ? asNumber(dictGet(descendantDict, "DW")) ?? DEFAULT_CID_WIDTH : DEFAULT_CID_WIDTH;
|
|
186
243
|
const widthMap = readCidWidths(descendantDict !== void 0 ? asArray(dictGet(descendantDict, "W")) : void 0);
|
|
187
244
|
const widthOf = (cid) => widthMap.get(cid) ?? dw;
|
|
245
|
+
const dw2 = descendantDict !== void 0 ? asArray(dictGet(descendantDict, "DW2")) : void 0;
|
|
246
|
+
const defaultPositionY = (dw2?.length === DW2_ENTRY_COUNT ? asNumber(dw2[0]) : void 0) ?? DEFAULT_VERTICAL_POSITION_Y;
|
|
247
|
+
const defaultDisplacementY = (dw2?.length === DW2_ENTRY_COUNT ? asNumber(dw2[1]) : void 0) ?? DEFAULT_VERTICAL_DISPLACEMENT_Y;
|
|
248
|
+
const verticalMap = readCidVerticalMetrics(descendantDict !== void 0 ? asArray(dictGet(descendantDict, "W2")) : void 0);
|
|
249
|
+
const verticalMetricsOf = (cid) => {
|
|
250
|
+
const entry = verticalMap.get(cid);
|
|
251
|
+
return {
|
|
252
|
+
displacementY: entry?.displacementY ?? defaultDisplacementY,
|
|
253
|
+
positionX: entry?.positionX ?? widthOf(cid) / 2,
|
|
254
|
+
positionY: entry?.positionY ?? defaultPositionY
|
|
255
|
+
};
|
|
256
|
+
};
|
|
188
257
|
const toUnicode = readToUnicodeCMap(fontDict, context);
|
|
189
258
|
const cidToGidMap = descendantDict !== void 0 ? dictGet(descendantDict, "CIDToGIDMap") : void 0;
|
|
190
|
-
const
|
|
259
|
+
const encodingName = dictGet(fontDict, "Encoding");
|
|
260
|
+
const programEncoding = (isName(encodingName, "Identity-H") || isName(encodingName, "Identity-V")) && (cidToGidMap === void 0 || isName(cidToGidMap, "Identity")) ? lazyFontProgram(descriptor, context) : void 0;
|
|
191
261
|
const decodeToUnicode = (codes) => {
|
|
192
262
|
let out = "";
|
|
193
263
|
let glyphCount = 0;
|
|
@@ -221,7 +291,8 @@ function buildCompositeFont(fontDict, context) {
|
|
|
221
291
|
bold,
|
|
222
292
|
italic,
|
|
223
293
|
widthOf,
|
|
224
|
-
decodeToUnicode
|
|
294
|
+
decodeToUnicode,
|
|
295
|
+
...readsVertically(fontDict, context) ? { verticalMetricsOf } : {}
|
|
225
296
|
};
|
|
226
297
|
}
|
|
227
298
|
function createFontResolver(context) {
|
|
@@ -236,16 +307,27 @@ function createFontResolver(context) {
|
|
|
236
307
|
return resolved;
|
|
237
308
|
};
|
|
238
309
|
return {
|
|
239
|
-
metrics: {
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
|
|
244
|
-
|
|
245
|
-
|
|
246
|
-
|
|
247
|
-
|
|
248
|
-
|
|
310
|
+
metrics: {
|
|
311
|
+
glyphAdvance(fontResourceName, resources, codes, byteOffset) {
|
|
312
|
+
const font = resolve(fontResourceName, resources);
|
|
313
|
+
if (font === void 0) return;
|
|
314
|
+
const byteLengthConsumed = font.composite ? 2 : 1;
|
|
315
|
+
const code = font.composite ? (codes[byteOffset] ?? 0) << 8 | (codes[byteOffset + 1] ?? 0) : codes[byteOffset] ?? 0;
|
|
316
|
+
const vertical = font.verticalMetricsOf?.(code);
|
|
317
|
+
return {
|
|
318
|
+
widthPer1000: font.widthOf(code),
|
|
319
|
+
byteLengthConsumed,
|
|
320
|
+
...vertical === void 0 ? {} : { vertical: {
|
|
321
|
+
displacementPer1000: vertical.displacementY,
|
|
322
|
+
positionXPer1000: vertical.positionX,
|
|
323
|
+
positionYPer1000: vertical.positionY
|
|
324
|
+
} }
|
|
325
|
+
};
|
|
326
|
+
},
|
|
327
|
+
isVertical(fontResourceName, resources) {
|
|
328
|
+
return resolve(fontResourceName, resources)?.verticalMetricsOf !== void 0;
|
|
329
|
+
}
|
|
330
|
+
},
|
|
249
331
|
resolve
|
|
250
332
|
};
|
|
251
333
|
}
|
package/dist/interpret.cjs
CHANGED
|
@@ -433,24 +433,38 @@ function runContentStream(bytes, resources, initialState, context, items, depth,
|
|
|
433
433
|
const widthPer1000 = glyph?.widthPer1000 ?? FALLBACK_GLYPH_WIDTH_PER_1000;
|
|
434
434
|
const byteLength = glyph?.byteLengthConsumed ?? 1;
|
|
435
435
|
const isSingleByteSpace = byteLength === 1 && codes[offset] === 32;
|
|
436
|
-
const
|
|
437
|
-
|
|
436
|
+
const spacing = gs.charSpace + (isSingleByteSpace ? gs.wordSpace : 0);
|
|
437
|
+
const vertical = glyph?.vertical;
|
|
438
|
+
text.tm = vertical === void 0 ? require_matrix.multiplyMatrices(require_matrix.translationMatrix((widthPer1000 / 1e3 * gs.fontSizePt + spacing) * gs.horizScale, 0), text.tm) : require_matrix.multiplyMatrices(require_matrix.translationMatrix(0, vertical.displacementPer1000 / 1e3 * gs.fontSizePt + spacing), text.tm);
|
|
438
439
|
offset += byteLength;
|
|
439
440
|
}
|
|
440
441
|
};
|
|
442
|
+
const positionVectorOf = (codes) => {
|
|
443
|
+
const fontResourceName = gs.fontResourceName;
|
|
444
|
+
if (fontResourceName === void 0) return;
|
|
445
|
+
const vertical = context.fontMetrics.glyphAdvance(fontResourceName, resources, codes, 0)?.vertical;
|
|
446
|
+
if (vertical === void 0) return;
|
|
447
|
+
return {
|
|
448
|
+
x: vertical.positionXPer1000 / 1e3,
|
|
449
|
+
y: vertical.positionYPer1000 / 1e3
|
|
450
|
+
};
|
|
451
|
+
};
|
|
441
452
|
const showTextArray = (elements) => {
|
|
442
453
|
const fontResourceName = gs.fontResourceName;
|
|
443
454
|
if (fontResourceName === void 0) return;
|
|
455
|
+
const vertical = context.fontMetrics.isVertical(fontResourceName, resources);
|
|
444
456
|
const startMatrix = computeTrm(gs, text);
|
|
445
457
|
const chunks = [];
|
|
446
458
|
let totalLength = 0;
|
|
459
|
+
let firstGlyphPosition;
|
|
447
460
|
for (const el of elements) if (el.kind === "string") {
|
|
461
|
+
firstGlyphPosition ??= vertical ? positionVectorOf(el.bytes) : void 0;
|
|
448
462
|
chunks.push(el.bytes);
|
|
449
463
|
totalLength += el.bytes.length;
|
|
450
464
|
advanceThroughString(el.bytes);
|
|
451
465
|
} else if (el.kind === "number") {
|
|
452
|
-
const adjustment = -(el.value / 1e3) * gs.fontSizePt
|
|
453
|
-
text.tm = require_matrix.multiplyMatrices(require_matrix.translationMatrix(adjustment, 0), text.tm);
|
|
466
|
+
const adjustment = -(el.value / 1e3) * gs.fontSizePt;
|
|
467
|
+
text.tm = require_matrix.multiplyMatrices(vertical ? require_matrix.translationMatrix(0, adjustment) : require_matrix.translationMatrix(adjustment * gs.horizScale, 0), text.tm);
|
|
454
468
|
}
|
|
455
469
|
if (totalLength === 0) return;
|
|
456
470
|
const combined = new Uint8Array(totalLength);
|
|
@@ -460,15 +474,17 @@ function runContentStream(bytes, resources, initialState, context, items, depth,
|
|
|
460
474
|
at += chunk.length;
|
|
461
475
|
}
|
|
462
476
|
const endMatrix = computeTrm(gs, text);
|
|
477
|
+
const shift = firstGlyphPosition === void 0 ? void 0 : require_matrix.translationMatrix(-firstGlyphPosition.x, -firstGlyphPosition.y);
|
|
463
478
|
pushItem({
|
|
464
479
|
kind: "text",
|
|
465
480
|
codes: combined,
|
|
466
481
|
fontResourceName,
|
|
467
482
|
resources,
|
|
468
|
-
startMatrix,
|
|
469
|
-
endMatrix,
|
|
483
|
+
startMatrix: shift === void 0 ? startMatrix : require_matrix.multiplyMatrices(shift, startMatrix),
|
|
484
|
+
endMatrix: shift === void 0 ? endMatrix : require_matrix.multiplyMatrices(shift, endMatrix),
|
|
470
485
|
sizePt: gs.fontSizePt,
|
|
471
|
-
color: gs.fillColor
|
|
486
|
+
color: gs.fillColor,
|
|
487
|
+
...vertical ? { vertical: true } : {}
|
|
472
488
|
});
|
|
473
489
|
};
|
|
474
490
|
const showText = (bytes) => {
|
package/dist/interpret.d.cts
CHANGED
|
@@ -16,6 +16,7 @@ interface ExtractedTextRun {
|
|
|
16
16
|
readonly actualText?: string;
|
|
17
17
|
readonly alt?: string;
|
|
18
18
|
readonly mcid?: number;
|
|
19
|
+
readonly vertical?: boolean;
|
|
19
20
|
}
|
|
20
21
|
interface ExtractedPaint {
|
|
21
22
|
readonly fill: Color | undefined;
|
|
@@ -94,12 +95,19 @@ interface ExtractedInlineImage {
|
|
|
94
95
|
readonly mcid?: number;
|
|
95
96
|
}
|
|
96
97
|
type ExtractedItem = ExtractedTextRun | ExtractedRect | ExtractedEllipse | ExtractedLine | ExtractedPath | ExtractedImage | ExtractedInlineImage;
|
|
98
|
+
interface VerticalGlyphAdvance {
|
|
99
|
+
readonly displacementPer1000: number;
|
|
100
|
+
readonly positionXPer1000: number;
|
|
101
|
+
readonly positionYPer1000: number;
|
|
102
|
+
}
|
|
97
103
|
interface GlyphAdvance {
|
|
98
104
|
readonly widthPer1000: number;
|
|
99
105
|
readonly byteLengthConsumed: number;
|
|
106
|
+
readonly vertical?: VerticalGlyphAdvance;
|
|
100
107
|
}
|
|
101
108
|
interface FontMetricsPort {
|
|
102
109
|
glyphAdvance(fontResourceName: string, resources: PdfDict, codes: Uint8Array<ArrayBuffer>, byteOffset: number): GlyphAdvance | undefined;
|
|
110
|
+
isVertical(fontResourceName: string, resources: PdfDict): boolean;
|
|
103
111
|
}
|
|
104
112
|
interface PdfObjectResolver {
|
|
105
113
|
resolve(obj: PdfObject | undefined): PdfObject | undefined;
|
|
@@ -113,4 +121,4 @@ interface InterpretContext {
|
|
|
113
121
|
}
|
|
114
122
|
declare function interpretContentStream(bytes: Uint8Array<ArrayBuffer>, resources: PdfDict, context: InterpretContext): ExtractedItem[];
|
|
115
123
|
//#endregion
|
|
116
|
-
export { ExtractedEllipse, ExtractedImage, ExtractedInlineImage, ExtractedItem, ExtractedLine, ExtractedPaint, ExtractedPath, ExtractedPathSegment, ExtractedRect, ExtractedSubpath, ExtractedTextRun, FontMetricsPort, GlyphAdvance, InterpretContext, PdfObjectResolver, interpretContentStream };
|
|
124
|
+
export { ExtractedEllipse, ExtractedImage, ExtractedInlineImage, ExtractedItem, ExtractedLine, ExtractedPaint, ExtractedPath, ExtractedPathSegment, ExtractedRect, ExtractedSubpath, ExtractedTextRun, FontMetricsPort, GlyphAdvance, InterpretContext, PdfObjectResolver, VerticalGlyphAdvance, interpretContentStream };
|
package/dist/interpret.d.ts
CHANGED
|
@@ -16,6 +16,7 @@ interface ExtractedTextRun {
|
|
|
16
16
|
readonly actualText?: string;
|
|
17
17
|
readonly alt?: string;
|
|
18
18
|
readonly mcid?: number;
|
|
19
|
+
readonly vertical?: boolean;
|
|
19
20
|
}
|
|
20
21
|
interface ExtractedPaint {
|
|
21
22
|
readonly fill: Color | undefined;
|
|
@@ -94,12 +95,19 @@ interface ExtractedInlineImage {
|
|
|
94
95
|
readonly mcid?: number;
|
|
95
96
|
}
|
|
96
97
|
type ExtractedItem = ExtractedTextRun | ExtractedRect | ExtractedEllipse | ExtractedLine | ExtractedPath | ExtractedImage | ExtractedInlineImage;
|
|
98
|
+
interface VerticalGlyphAdvance {
|
|
99
|
+
readonly displacementPer1000: number;
|
|
100
|
+
readonly positionXPer1000: number;
|
|
101
|
+
readonly positionYPer1000: number;
|
|
102
|
+
}
|
|
97
103
|
interface GlyphAdvance {
|
|
98
104
|
readonly widthPer1000: number;
|
|
99
105
|
readonly byteLengthConsumed: number;
|
|
106
|
+
readonly vertical?: VerticalGlyphAdvance;
|
|
100
107
|
}
|
|
101
108
|
interface FontMetricsPort {
|
|
102
109
|
glyphAdvance(fontResourceName: string, resources: PdfDict, codes: Uint8Array<ArrayBuffer>, byteOffset: number): GlyphAdvance | undefined;
|
|
110
|
+
isVertical(fontResourceName: string, resources: PdfDict): boolean;
|
|
103
111
|
}
|
|
104
112
|
interface PdfObjectResolver {
|
|
105
113
|
resolve(obj: PdfObject | undefined): PdfObject | undefined;
|
|
@@ -113,4 +121,4 @@ interface InterpretContext {
|
|
|
113
121
|
}
|
|
114
122
|
declare function interpretContentStream(bytes: Uint8Array<ArrayBuffer>, resources: PdfDict, context: InterpretContext): ExtractedItem[];
|
|
115
123
|
//#endregion
|
|
116
|
-
export { ExtractedEllipse, ExtractedImage, ExtractedInlineImage, ExtractedItem, ExtractedLine, ExtractedPaint, ExtractedPath, ExtractedPathSegment, ExtractedRect, ExtractedSubpath, ExtractedTextRun, FontMetricsPort, GlyphAdvance, InterpretContext, PdfObjectResolver, interpretContentStream };
|
|
124
|
+
export { ExtractedEllipse, ExtractedImage, ExtractedInlineImage, ExtractedItem, ExtractedLine, ExtractedPaint, ExtractedPath, ExtractedPathSegment, ExtractedRect, ExtractedSubpath, ExtractedTextRun, FontMetricsPort, GlyphAdvance, InterpretContext, PdfObjectResolver, VerticalGlyphAdvance, interpretContentStream };
|
package/dist/interpret.js
CHANGED
|
@@ -432,24 +432,38 @@ function runContentStream(bytes, resources, initialState, context, items, depth,
|
|
|
432
432
|
const widthPer1000 = glyph?.widthPer1000 ?? FALLBACK_GLYPH_WIDTH_PER_1000;
|
|
433
433
|
const byteLength = glyph?.byteLengthConsumed ?? 1;
|
|
434
434
|
const isSingleByteSpace = byteLength === 1 && codes[offset] === 32;
|
|
435
|
-
const
|
|
436
|
-
|
|
435
|
+
const spacing = gs.charSpace + (isSingleByteSpace ? gs.wordSpace : 0);
|
|
436
|
+
const vertical = glyph?.vertical;
|
|
437
|
+
text.tm = vertical === void 0 ? multiplyMatrices(translationMatrix((widthPer1000 / 1e3 * gs.fontSizePt + spacing) * gs.horizScale, 0), text.tm) : multiplyMatrices(translationMatrix(0, vertical.displacementPer1000 / 1e3 * gs.fontSizePt + spacing), text.tm);
|
|
437
438
|
offset += byteLength;
|
|
438
439
|
}
|
|
439
440
|
};
|
|
441
|
+
const positionVectorOf = (codes) => {
|
|
442
|
+
const fontResourceName = gs.fontResourceName;
|
|
443
|
+
if (fontResourceName === void 0) return;
|
|
444
|
+
const vertical = context.fontMetrics.glyphAdvance(fontResourceName, resources, codes, 0)?.vertical;
|
|
445
|
+
if (vertical === void 0) return;
|
|
446
|
+
return {
|
|
447
|
+
x: vertical.positionXPer1000 / 1e3,
|
|
448
|
+
y: vertical.positionYPer1000 / 1e3
|
|
449
|
+
};
|
|
450
|
+
};
|
|
440
451
|
const showTextArray = (elements) => {
|
|
441
452
|
const fontResourceName = gs.fontResourceName;
|
|
442
453
|
if (fontResourceName === void 0) return;
|
|
454
|
+
const vertical = context.fontMetrics.isVertical(fontResourceName, resources);
|
|
443
455
|
const startMatrix = computeTrm(gs, text);
|
|
444
456
|
const chunks = [];
|
|
445
457
|
let totalLength = 0;
|
|
458
|
+
let firstGlyphPosition;
|
|
446
459
|
for (const el of elements) if (el.kind === "string") {
|
|
460
|
+
firstGlyphPosition ??= vertical ? positionVectorOf(el.bytes) : void 0;
|
|
447
461
|
chunks.push(el.bytes);
|
|
448
462
|
totalLength += el.bytes.length;
|
|
449
463
|
advanceThroughString(el.bytes);
|
|
450
464
|
} else if (el.kind === "number") {
|
|
451
|
-
const adjustment = -(el.value / 1e3) * gs.fontSizePt
|
|
452
|
-
text.tm = multiplyMatrices(translationMatrix(adjustment, 0), text.tm);
|
|
465
|
+
const adjustment = -(el.value / 1e3) * gs.fontSizePt;
|
|
466
|
+
text.tm = multiplyMatrices(vertical ? translationMatrix(0, adjustment) : translationMatrix(adjustment * gs.horizScale, 0), text.tm);
|
|
453
467
|
}
|
|
454
468
|
if (totalLength === 0) return;
|
|
455
469
|
const combined = new Uint8Array(totalLength);
|
|
@@ -459,15 +473,17 @@ function runContentStream(bytes, resources, initialState, context, items, depth,
|
|
|
459
473
|
at += chunk.length;
|
|
460
474
|
}
|
|
461
475
|
const endMatrix = computeTrm(gs, text);
|
|
476
|
+
const shift = firstGlyphPosition === void 0 ? void 0 : translationMatrix(-firstGlyphPosition.x, -firstGlyphPosition.y);
|
|
462
477
|
pushItem({
|
|
463
478
|
kind: "text",
|
|
464
479
|
codes: combined,
|
|
465
480
|
fontResourceName,
|
|
466
481
|
resources,
|
|
467
|
-
startMatrix,
|
|
468
|
-
endMatrix,
|
|
482
|
+
startMatrix: shift === void 0 ? startMatrix : multiplyMatrices(shift, startMatrix),
|
|
483
|
+
endMatrix: shift === void 0 ? endMatrix : multiplyMatrices(shift, endMatrix),
|
|
469
484
|
sizePt: gs.fontSizePt,
|
|
470
|
-
color: gs.fillColor
|
|
485
|
+
color: gs.fillColor,
|
|
486
|
+
...vertical ? { vertical: true } : {}
|
|
471
487
|
});
|
|
472
488
|
};
|
|
473
489
|
const showText = (bytes) => {
|
package/dist/layout.cjs
CHANGED
|
@@ -13,6 +13,7 @@ const LayoutTextSchema = zod.z.object({
|
|
|
13
13
|
color: document_schema_js.ColorSchema,
|
|
14
14
|
widthPt: zod.z.number().nonnegative().optional(),
|
|
15
15
|
rotationDeg: zod.z.number().optional(),
|
|
16
|
+
writingMode: zod.z.literal("vertical").optional(),
|
|
16
17
|
underline: zod.z.boolean().optional(),
|
|
17
18
|
layer: zod.z.string().optional(),
|
|
18
19
|
actualText: zod.z.string().optional(),
|
package/dist/layout.d.cts
CHANGED
|
@@ -25,6 +25,7 @@ declare const LayoutTextSchema: z.ZodObject<{
|
|
|
25
25
|
}, z.core.$strip>;
|
|
26
26
|
widthPt: z.ZodOptional<z.ZodNumber>;
|
|
27
27
|
rotationDeg: z.ZodOptional<z.ZodNumber>;
|
|
28
|
+
writingMode: z.ZodOptional<z.ZodLiteral<"vertical">>;
|
|
28
29
|
underline: z.ZodOptional<z.ZodBoolean>;
|
|
29
30
|
layer: z.ZodOptional<z.ZodString>;
|
|
30
31
|
actualText: z.ZodOptional<z.ZodString>;
|
|
@@ -243,6 +244,7 @@ declare const LayoutItemSchema: z.ZodDiscriminatedUnion<[z.ZodObject<{
|
|
|
243
244
|
}, z.core.$strip>;
|
|
244
245
|
widthPt: z.ZodOptional<z.ZodNumber>;
|
|
245
246
|
rotationDeg: z.ZodOptional<z.ZodNumber>;
|
|
247
|
+
writingMode: z.ZodOptional<z.ZodLiteral<"vertical">>;
|
|
246
248
|
underline: z.ZodOptional<z.ZodBoolean>;
|
|
247
249
|
layer: z.ZodOptional<z.ZodString>;
|
|
248
250
|
actualText: z.ZodOptional<z.ZodString>;
|
|
@@ -475,6 +477,7 @@ declare const LayoutPageSchema: z.ZodObject<{
|
|
|
475
477
|
}, z.core.$strip>;
|
|
476
478
|
widthPt: z.ZodOptional<z.ZodNumber>;
|
|
477
479
|
rotationDeg: z.ZodOptional<z.ZodNumber>;
|
|
480
|
+
writingMode: z.ZodOptional<z.ZodLiteral<"vertical">>;
|
|
478
481
|
underline: z.ZodOptional<z.ZodBoolean>;
|
|
479
482
|
layer: z.ZodOptional<z.ZodString>;
|
|
480
483
|
actualText: z.ZodOptional<z.ZodString>;
|
|
@@ -825,6 +828,7 @@ declare const LayoutDocumentSchema: z.ZodObject<{
|
|
|
825
828
|
}, z.core.$strip>;
|
|
826
829
|
widthPt: z.ZodOptional<z.ZodNumber>;
|
|
827
830
|
rotationDeg: z.ZodOptional<z.ZodNumber>;
|
|
831
|
+
writingMode: z.ZodOptional<z.ZodLiteral<"vertical">>;
|
|
828
832
|
underline: z.ZodOptional<z.ZodBoolean>;
|
|
829
833
|
layer: z.ZodOptional<z.ZodString>;
|
|
830
834
|
actualText: z.ZodOptional<z.ZodString>;
|
package/dist/layout.d.ts
CHANGED
|
@@ -25,6 +25,7 @@ declare const LayoutTextSchema: z.ZodObject<{
|
|
|
25
25
|
}, z.core.$strip>;
|
|
26
26
|
widthPt: z.ZodOptional<z.ZodNumber>;
|
|
27
27
|
rotationDeg: z.ZodOptional<z.ZodNumber>;
|
|
28
|
+
writingMode: z.ZodOptional<z.ZodLiteral<"vertical">>;
|
|
28
29
|
underline: z.ZodOptional<z.ZodBoolean>;
|
|
29
30
|
layer: z.ZodOptional<z.ZodString>;
|
|
30
31
|
actualText: z.ZodOptional<z.ZodString>;
|
|
@@ -243,6 +244,7 @@ declare const LayoutItemSchema: z.ZodDiscriminatedUnion<[z.ZodObject<{
|
|
|
243
244
|
}, z.core.$strip>;
|
|
244
245
|
widthPt: z.ZodOptional<z.ZodNumber>;
|
|
245
246
|
rotationDeg: z.ZodOptional<z.ZodNumber>;
|
|
247
|
+
writingMode: z.ZodOptional<z.ZodLiteral<"vertical">>;
|
|
246
248
|
underline: z.ZodOptional<z.ZodBoolean>;
|
|
247
249
|
layer: z.ZodOptional<z.ZodString>;
|
|
248
250
|
actualText: z.ZodOptional<z.ZodString>;
|
|
@@ -475,6 +477,7 @@ declare const LayoutPageSchema: z.ZodObject<{
|
|
|
475
477
|
}, z.core.$strip>;
|
|
476
478
|
widthPt: z.ZodOptional<z.ZodNumber>;
|
|
477
479
|
rotationDeg: z.ZodOptional<z.ZodNumber>;
|
|
480
|
+
writingMode: z.ZodOptional<z.ZodLiteral<"vertical">>;
|
|
478
481
|
underline: z.ZodOptional<z.ZodBoolean>;
|
|
479
482
|
layer: z.ZodOptional<z.ZodString>;
|
|
480
483
|
actualText: z.ZodOptional<z.ZodString>;
|
|
@@ -825,6 +828,7 @@ declare const LayoutDocumentSchema: z.ZodObject<{
|
|
|
825
828
|
}, z.core.$strip>;
|
|
826
829
|
widthPt: z.ZodOptional<z.ZodNumber>;
|
|
827
830
|
rotationDeg: z.ZodOptional<z.ZodNumber>;
|
|
831
|
+
writingMode: z.ZodOptional<z.ZodLiteral<"vertical">>;
|
|
828
832
|
underline: z.ZodOptional<z.ZodBoolean>;
|
|
829
833
|
layer: z.ZodOptional<z.ZodString>;
|
|
830
834
|
actualText: z.ZodOptional<z.ZodString>;
|
package/dist/layout.js
CHANGED
|
@@ -12,6 +12,7 @@ const LayoutTextSchema = z.object({
|
|
|
12
12
|
color: ColorSchema,
|
|
13
13
|
widthPt: z.number().nonnegative().optional(),
|
|
14
14
|
rotationDeg: z.number().optional(),
|
|
15
|
+
writingMode: z.literal("vertical").optional(),
|
|
15
16
|
underline: z.boolean().optional(),
|
|
16
17
|
layer: z.string().optional(),
|
|
17
18
|
actualText: z.string().optional(),
|
package/dist/raster.cjs
CHANGED
|
@@ -533,8 +533,8 @@ function buildTextOutlineFace(fontDict, resolver, fontResolver, fontResourceName
|
|
|
533
533
|
const subtype = require_objects.asName(require_objects.dictGet(fontDict, "Subtype"));
|
|
534
534
|
if (subtype === "Type0") {
|
|
535
535
|
const encoding = resolver.resolve(require_objects.dictGet(fontDict, "Encoding"));
|
|
536
|
-
if (encoding?.kind !== "name" || encoding.name !== "Identity-H") {
|
|
537
|
-
unavailable("a Type0 font whose /Encoding is
|
|
536
|
+
if (encoding?.kind !== "name" || encoding.name !== "Identity-H" && encoding.name !== "Identity-V") {
|
|
537
|
+
unavailable("a Type0 font whose /Encoding is neither Identity-H nor Identity-V (a predefined or embedded CMap this raster walk does not decode)");
|
|
538
538
|
return;
|
|
539
539
|
}
|
|
540
540
|
const descendants = require_objects.asArray(require_objects.dictGet(fontDict, "DescendantFonts"));
|
|
@@ -623,17 +623,19 @@ function drawTextRun(item, fontResolver, resolver, outlineFaces, interpretToDevi
|
|
|
623
623
|
let cumulative = 0;
|
|
624
624
|
while (offset < item.codes.length) {
|
|
625
625
|
const advance = fontResolver.metrics.glyphAdvance(item.fontResourceName, item.resources, item.codes, offset);
|
|
626
|
-
const widthPer1000 = advance.widthPer1000;
|
|
627
626
|
const byteLength = advance.byteLengthConsumed;
|
|
628
627
|
placements.push({
|
|
629
628
|
glyphId: face.glyphIdOf(item.codes, offset),
|
|
630
|
-
advance: cumulative
|
|
629
|
+
advance: cumulative,
|
|
630
|
+
positionX: (advance.vertical?.positionXPer1000 ?? 0) / 1e3,
|
|
631
|
+
positionY: (advance.vertical?.positionYPer1000 ?? 0) / 1e3
|
|
631
632
|
});
|
|
632
|
-
cumulative += widthPer1000 / 1e3 * item.sizePt;
|
|
633
|
+
cumulative += (advance.vertical?.displacementPer1000 ?? advance.widthPer1000) / 1e3 * item.sizePt;
|
|
633
634
|
offset += byteLength;
|
|
634
635
|
}
|
|
635
636
|
const startDeviceMatrix = require_matrix.multiplyMatrices(item.startMatrix, interpretToDeviceMatrix);
|
|
636
637
|
const endDeviceMatrix = require_matrix.multiplyMatrices(item.endMatrix, interpretToDeviceMatrix);
|
|
638
|
+
const vertical = item.vertical === true;
|
|
637
639
|
const startPoint = require_matrix.applyMatrix(startDeviceMatrix, {
|
|
638
640
|
x: 0,
|
|
639
641
|
y: 0
|
|
@@ -642,19 +644,23 @@ function drawTextRun(item, fontResolver, resolver, outlineFaces, interpretToDevi
|
|
|
642
644
|
x: 0,
|
|
643
645
|
y: 0
|
|
644
646
|
});
|
|
645
|
-
const rawEndPoint = require_matrix.applyMatrix(startDeviceMatrix, {
|
|
647
|
+
const rawEndPoint = require_matrix.applyMatrix(startDeviceMatrix, vertical ? {
|
|
648
|
+
x: 0,
|
|
649
|
+
y: cumulative
|
|
650
|
+
} : {
|
|
646
651
|
x: cumulative,
|
|
647
652
|
y: 0
|
|
648
653
|
});
|
|
649
654
|
const rawExtent = Math.hypot(rawEndPoint.x - startPoint.x, rawEndPoint.y - startPoint.y);
|
|
650
655
|
const actualExtent = Math.hypot(endPoint.x - startPoint.x, endPoint.y - startPoint.y);
|
|
651
|
-
const correction = rawExtent > 1e-9
|
|
656
|
+
const correction = rawExtent > 1e-9 ? actualExtent / rawExtent : 1;
|
|
657
|
+
const firstPlacement = placements[0];
|
|
652
658
|
const glyphScale = require_matrix.scaleMatrix(1 / face.unitsPerEm, 1 / face.unitsPerEm);
|
|
653
659
|
for (const placement of placements) {
|
|
654
660
|
if (placement.glyphId === void 0) continue;
|
|
655
661
|
const outline = require_glyf_contours.decodeGlyphOutline(face.glyf, placement.glyphId);
|
|
656
662
|
if (outline === void 0) continue;
|
|
657
|
-
const trm = require_matrix.multiplyMatrices(require_matrix.translationMatrix(placement.advance * correction, 0), item.startMatrix);
|
|
663
|
+
const trm = require_matrix.multiplyMatrices(require_matrix.translationMatrix((firstPlacement?.positionX ?? 0) - placement.positionX, (firstPlacement?.positionY ?? 0) - placement.positionY), require_matrix.multiplyMatrices(vertical ? require_matrix.translationMatrix(0, placement.advance * correction) : require_matrix.translationMatrix(placement.advance * correction, 0), item.startMatrix));
|
|
658
664
|
drawGlyphOutline(outline, require_matrix.multiplyMatrices(glyphScale, require_matrix.multiplyMatrices(trm, interpretToDeviceMatrix)), item.color, rasteriser);
|
|
659
665
|
}
|
|
660
666
|
}
|
package/dist/raster.js
CHANGED
|
@@ -532,8 +532,8 @@ function buildTextOutlineFace(fontDict, resolver, fontResolver, fontResourceName
|
|
|
532
532
|
const subtype = asName(dictGet(fontDict, "Subtype"));
|
|
533
533
|
if (subtype === "Type0") {
|
|
534
534
|
const encoding = resolver.resolve(dictGet(fontDict, "Encoding"));
|
|
535
|
-
if (encoding?.kind !== "name" || encoding.name !== "Identity-H") {
|
|
536
|
-
unavailable("a Type0 font whose /Encoding is
|
|
535
|
+
if (encoding?.kind !== "name" || encoding.name !== "Identity-H" && encoding.name !== "Identity-V") {
|
|
536
|
+
unavailable("a Type0 font whose /Encoding is neither Identity-H nor Identity-V (a predefined or embedded CMap this raster walk does not decode)");
|
|
537
537
|
return;
|
|
538
538
|
}
|
|
539
539
|
const descendants = asArray(dictGet(fontDict, "DescendantFonts"));
|
|
@@ -622,17 +622,19 @@ function drawTextRun(item, fontResolver, resolver, outlineFaces, interpretToDevi
|
|
|
622
622
|
let cumulative = 0;
|
|
623
623
|
while (offset < item.codes.length) {
|
|
624
624
|
const advance = fontResolver.metrics.glyphAdvance(item.fontResourceName, item.resources, item.codes, offset);
|
|
625
|
-
const widthPer1000 = advance.widthPer1000;
|
|
626
625
|
const byteLength = advance.byteLengthConsumed;
|
|
627
626
|
placements.push({
|
|
628
627
|
glyphId: face.glyphIdOf(item.codes, offset),
|
|
629
|
-
advance: cumulative
|
|
628
|
+
advance: cumulative,
|
|
629
|
+
positionX: (advance.vertical?.positionXPer1000 ?? 0) / 1e3,
|
|
630
|
+
positionY: (advance.vertical?.positionYPer1000 ?? 0) / 1e3
|
|
630
631
|
});
|
|
631
|
-
cumulative += widthPer1000 / 1e3 * item.sizePt;
|
|
632
|
+
cumulative += (advance.vertical?.displacementPer1000 ?? advance.widthPer1000) / 1e3 * item.sizePt;
|
|
632
633
|
offset += byteLength;
|
|
633
634
|
}
|
|
634
635
|
const startDeviceMatrix = multiplyMatrices(item.startMatrix, interpretToDeviceMatrix);
|
|
635
636
|
const endDeviceMatrix = multiplyMatrices(item.endMatrix, interpretToDeviceMatrix);
|
|
637
|
+
const vertical = item.vertical === true;
|
|
636
638
|
const startPoint = applyMatrix(startDeviceMatrix, {
|
|
637
639
|
x: 0,
|
|
638
640
|
y: 0
|
|
@@ -641,19 +643,23 @@ function drawTextRun(item, fontResolver, resolver, outlineFaces, interpretToDevi
|
|
|
641
643
|
x: 0,
|
|
642
644
|
y: 0
|
|
643
645
|
});
|
|
644
|
-
const rawEndPoint = applyMatrix(startDeviceMatrix, {
|
|
646
|
+
const rawEndPoint = applyMatrix(startDeviceMatrix, vertical ? {
|
|
647
|
+
x: 0,
|
|
648
|
+
y: cumulative
|
|
649
|
+
} : {
|
|
645
650
|
x: cumulative,
|
|
646
651
|
y: 0
|
|
647
652
|
});
|
|
648
653
|
const rawExtent = Math.hypot(rawEndPoint.x - startPoint.x, rawEndPoint.y - startPoint.y);
|
|
649
654
|
const actualExtent = Math.hypot(endPoint.x - startPoint.x, endPoint.y - startPoint.y);
|
|
650
|
-
const correction = rawExtent > 1e-9
|
|
655
|
+
const correction = rawExtent > 1e-9 ? actualExtent / rawExtent : 1;
|
|
656
|
+
const firstPlacement = placements[0];
|
|
651
657
|
const glyphScale = scaleMatrix(1 / face.unitsPerEm, 1 / face.unitsPerEm);
|
|
652
658
|
for (const placement of placements) {
|
|
653
659
|
if (placement.glyphId === void 0) continue;
|
|
654
660
|
const outline = decodeGlyphOutline(face.glyf, placement.glyphId);
|
|
655
661
|
if (outline === void 0) continue;
|
|
656
|
-
const trm = multiplyMatrices(translationMatrix(placement.advance * correction, 0), item.startMatrix);
|
|
662
|
+
const trm = multiplyMatrices(translationMatrix((firstPlacement?.positionX ?? 0) - placement.positionX, (firstPlacement?.positionY ?? 0) - placement.positionY), multiplyMatrices(vertical ? translationMatrix(0, placement.advance * correction) : translationMatrix(placement.advance * correction, 0), item.startMatrix));
|
|
657
663
|
drawGlyphOutline(outline, multiplyMatrices(glyphScale, multiplyMatrices(trm, interpretToDeviceMatrix)), item.color, rasteriser);
|
|
658
664
|
}
|
|
659
665
|
}
|
package/dist/read.cjs
CHANGED
|
@@ -366,6 +366,7 @@ function convertText(item, pageMatrix, fontResolver) {
|
|
|
366
366
|
color: item.color,
|
|
367
367
|
widthPt,
|
|
368
368
|
rotationDeg: rotationDeg !== 0 ? rotationDeg : void 0,
|
|
369
|
+
...item.vertical === true ? { writingMode: "vertical" } : {},
|
|
369
370
|
...item.layerName !== void 0 ? { layer: item.layerName } : {},
|
|
370
371
|
...item.actualText !== void 0 ? { actualText: item.actualText } : {},
|
|
371
372
|
...item.alt !== void 0 ? { alt: item.alt } : {}
|
package/dist/read.js
CHANGED
|
@@ -365,6 +365,7 @@ function convertText(item, pageMatrix, fontResolver) {
|
|
|
365
365
|
color: item.color,
|
|
366
366
|
widthPt,
|
|
367
367
|
rotationDeg: rotationDeg !== 0 ? rotationDeg : void 0,
|
|
368
|
+
...item.vertical === true ? { writingMode: "vertical" } : {},
|
|
368
369
|
...item.layerName !== void 0 ? { layer: item.layerName } : {},
|
|
369
370
|
...item.actualText !== void 0 ? { actualText: item.actualText } : {},
|
|
370
371
|
...item.alt !== void 0 ? { alt: item.alt } : {}
|
package/dist/text-group.cjs
CHANGED
|
@@ -5,8 +5,10 @@ const DEFAULT_WORD_GAP_EM = 1 / 8;
|
|
|
5
5
|
const DEFAULT_COLUMN_GAP_EM = 1;
|
|
6
6
|
const ROTATION_NOISE_DEG = 1e-6;
|
|
7
7
|
const DEGREES_PER_TURN = 360;
|
|
8
|
-
|
|
9
|
-
|
|
8
|
+
const VERTICAL_ADVANCE_OFFSET_DEG = -90;
|
|
9
|
+
function advanceAngleDeg(run) {
|
|
10
|
+
const raw = (run.rotationDeg ?? 0) + (run.writingMode === "vertical" ? VERTICAL_ADVANCE_OFFSET_DEG : 0);
|
|
11
|
+
return (Math.round(raw / ROTATION_NOISE_DEG) * ROTATION_NOISE_DEG % DEGREES_PER_TURN + DEGREES_PER_TURN) % DEGREES_PER_TURN;
|
|
10
12
|
}
|
|
11
13
|
const RADIANS_PER_TURN = 2 * Math.PI;
|
|
12
14
|
function baselineAxis(rotationDeg) {
|
|
@@ -42,8 +44,8 @@ function unproject(alongPt, acrossPt, rotationDeg) {
|
|
|
42
44
|
* @returns true when the two baselines are within tolerance of each other
|
|
43
45
|
*/
|
|
44
46
|
function runsShareBaseline(a, b, options = {}) {
|
|
45
|
-
const rotationDeg =
|
|
46
|
-
if (rotationDeg !==
|
|
47
|
+
const rotationDeg = advanceAngleDeg(a);
|
|
48
|
+
if (rotationDeg !== advanceAngleDeg(b) || a.writingMode !== b.writingMode) return false;
|
|
47
49
|
const toleranceEm = options.baselineToleranceEm ?? .6666666666666666;
|
|
48
50
|
return Math.abs(project(a, rotationDeg).acrossPt - project(b, rotationDeg).acrossPt) <= toleranceEm * Math.min(a.sizePt, b.sizePt);
|
|
49
51
|
}
|
|
@@ -56,8 +58,8 @@ function runsShareBaseline(a, b, options = {}) {
|
|
|
56
58
|
* @returns the gap in points, or undefined when it cannot be derived
|
|
57
59
|
*/
|
|
58
60
|
function runGapPt(previous, next) {
|
|
59
|
-
const rotationDeg =
|
|
60
|
-
if (rotationDeg !==
|
|
61
|
+
const rotationDeg = advanceAngleDeg(previous);
|
|
62
|
+
if (rotationDeg !== advanceAngleDeg(next) || previous.writingMode !== next.writingMode) return;
|
|
61
63
|
const previousWidthPt = previous.widthPt;
|
|
62
64
|
if (previousWidthPt === void 0) return;
|
|
63
65
|
return project(next, rotationDeg).alongPt - (project(previous, rotationDeg).alongPt + previousWidthPt);
|
|
@@ -150,7 +152,7 @@ function buildWords(line, rotationDeg, wordGapEm, columnGapEm) {
|
|
|
150
152
|
...word.measurable ? { bounds: boundsOf(word.minAlongPt, word.maxAlongEndPt, line.anchor.acrossPt, word.maxSizePt, rotationDeg) } : {}
|
|
151
153
|
}));
|
|
152
154
|
}
|
|
153
|
-
function buildLine(line, rotationDeg, wordGapEm, columnGapEm) {
|
|
155
|
+
function buildLine(line, rotationDeg, vertical, wordGapEm, columnGapEm) {
|
|
154
156
|
const words = buildWords(line, rotationDeg, wordGapEm, columnGapEm);
|
|
155
157
|
const text = words.map((word) => SEPARATOR_TEXT[word.separatorBefore] + word.text).join("");
|
|
156
158
|
let minAlongPt = Number.POSITIVE_INFINITY;
|
|
@@ -169,6 +171,7 @@ function buildLine(line, rotationDeg, wordGapEm, columnGapEm) {
|
|
|
169
171
|
words,
|
|
170
172
|
baselineYPt: line.anchor.acrossPt,
|
|
171
173
|
rotationDeg,
|
|
174
|
+
...vertical ? { writingMode: "vertical" } : {},
|
|
172
175
|
...measurable ? { bounds: boundsOf(minAlongPt, maxAlongEndPt, line.anchor.acrossPt, maxSizePt, rotationDeg) } : {}
|
|
173
176
|
};
|
|
174
177
|
}
|
|
@@ -181,7 +184,7 @@ function buildLine(line, rotationDeg, wordGapEm, columnGapEm) {
|
|
|
181
184
|
*
|
|
182
185
|
* Grouping is by baseline alone, with no page segmentation: on a multi-column page whose columns are set at different vertical offsets, a line of one column can fall within tolerance of a line of the next and the two are reported as one line, because from geometry alone at this level that is what they are. The gutter between them still reads as a column boundary, so the separator says where to cut. A caller that needs the columns apart segments the page first, which is what document-outline.js's `segmentPdfRegions` is for, and groups each region's own runs.
|
|
183
186
|
*
|
|
184
|
-
* Known limitations, all of them inherent to what a PDF states rather than to this implementation. Text is reported in visual order, left to right along the baseline: a right-to-left script arrives from the content stream already laid out visually and carries no direction of its own, so a consumer needing logical order applies the Unicode Bidi Algorithm to the result.
|
|
187
|
+
* Known limitations, all of them inherent to what a PDF states rather than to this implementation. Text is reported in visual order, left to right along the baseline: a right-to-left script arrives from the content stream already laid out visually and carries no direction of its own, so a consumer needing logical order applies the Unicode Bidi Algorithm to the result. A vertically set run is grouped into columns rather than lines, from the writingMode its own reader reported, and its column is reported with a rotationDeg of 270 and its own writingMode; columns order right to left, the direction vertical setting is read in. What is NOT recognised is a predefined non-Identity CMap's own byte decoding, so a vertically set font using one (90ms-RKSJ-V and its siblings) still has its codes read as two-byte CIDs, which is this package's own long-standing composite-font limit rather than anything about writing mode. A word hyphenated across a line end is left split, with its hyphen intact, because rejoining it needs to know the two lines belong to one paragraph, which is a semantic judgement this package deliberately leaves to its consumers. Runs with empty text are dropped, and so are the empty words a whitespace-only run would otherwise produce, but no other normalisation of the decoded text is attempted: a ligature and a zero-width character each reach the output exactly as the font's ToUnicode mapping spelled them.
|
|
185
188
|
* @param runs - positioned text runs, in any order
|
|
186
189
|
* @param options - tolerance overrides; each defaults to the exported constant of the same name
|
|
187
190
|
* @returns the grouped lines, in reading order
|
|
@@ -190,20 +193,27 @@ function groupPdfTextRuns(runs, options = {}) {
|
|
|
190
193
|
const toleranceEm = options.baselineToleranceEm ?? .6666666666666666;
|
|
191
194
|
const wordGapEm = options.wordGapEm ?? .125;
|
|
192
195
|
const columnGapEm = options.columnGapEm ?? 1;
|
|
193
|
-
const
|
|
196
|
+
const buckets = /* @__PURE__ */ new Map();
|
|
194
197
|
for (const run of runs) {
|
|
195
198
|
if (run.text.length === 0) continue;
|
|
196
|
-
const rotationDeg =
|
|
197
|
-
const
|
|
198
|
-
|
|
199
|
-
|
|
199
|
+
const rotationDeg = advanceAngleDeg(run);
|
|
200
|
+
const vertical = run.writingMode === "vertical";
|
|
201
|
+
const key = `${String(rotationDeg)}|${String(vertical)}`;
|
|
202
|
+
const bucket = buckets.get(key);
|
|
203
|
+
if (bucket === void 0) buckets.set(key, {
|
|
204
|
+
rotationDeg,
|
|
205
|
+
vertical,
|
|
206
|
+
runs: [run]
|
|
207
|
+
});
|
|
208
|
+
else bucket.runs.push(run);
|
|
200
209
|
}
|
|
201
210
|
const result = [];
|
|
202
|
-
|
|
203
|
-
|
|
211
|
+
const ordered = [...buckets.values()].sort((a, b) => a.rotationDeg - b.rotationDeg || Number(a.vertical) - Number(b.vertical));
|
|
212
|
+
for (const bucket of ordered) {
|
|
213
|
+
const entries = bucket.runs.map((run) => project(run, bucket.rotationDeg)).sort((a, b) => b.acrossPt - a.acrossPt || a.alongPt - b.alongPt);
|
|
204
214
|
for (const line of clusterIntoLines(entries, toleranceEm)) {
|
|
205
215
|
line.entries.sort((a, b) => a.alongPt - b.alongPt);
|
|
206
|
-
result.push(buildLine(line, rotationDeg, wordGapEm, columnGapEm));
|
|
216
|
+
result.push(buildLine(line, bucket.rotationDeg, bucket.vertical, wordGapEm, columnGapEm));
|
|
207
217
|
}
|
|
208
218
|
}
|
|
209
219
|
return result;
|
package/dist/text-group.d.cts
CHANGED
|
@@ -6,6 +6,7 @@ interface PdfTextRunGeometry {
|
|
|
6
6
|
readonly sizePt: number;
|
|
7
7
|
readonly widthPt?: number;
|
|
8
8
|
readonly rotationDeg?: number;
|
|
9
|
+
readonly writingMode?: "vertical";
|
|
9
10
|
}
|
|
10
11
|
interface PdfTextBox {
|
|
11
12
|
readonly xPt: number;
|
|
@@ -25,6 +26,7 @@ interface PdfTextLine<TRun extends PdfTextRunGeometry> {
|
|
|
25
26
|
readonly words: readonly PdfTextWord<TRun>[];
|
|
26
27
|
readonly baselineYPt: number;
|
|
27
28
|
readonly rotationDeg: number;
|
|
29
|
+
readonly writingMode?: "vertical";
|
|
28
30
|
readonly bounds?: PdfTextBox;
|
|
29
31
|
}
|
|
30
32
|
interface PdfTextGroupingOptions {
|
|
@@ -63,7 +65,7 @@ declare function runGapPt(previous: PdfTextRunGeometry, next: PdfTextRunGeometry
|
|
|
63
65
|
*
|
|
64
66
|
* Grouping is by baseline alone, with no page segmentation: on a multi-column page whose columns are set at different vertical offsets, a line of one column can fall within tolerance of a line of the next and the two are reported as one line, because from geometry alone at this level that is what they are. The gutter between them still reads as a column boundary, so the separator says where to cut. A caller that needs the columns apart segments the page first, which is what document-outline.js's `segmentPdfRegions` is for, and groups each region's own runs.
|
|
65
67
|
*
|
|
66
|
-
* Known limitations, all of them inherent to what a PDF states rather than to this implementation. Text is reported in visual order, left to right along the baseline: a right-to-left script arrives from the content stream already laid out visually and carries no direction of its own, so a consumer needing logical order applies the Unicode Bidi Algorithm to the result.
|
|
68
|
+
* Known limitations, all of them inherent to what a PDF states rather than to this implementation. Text is reported in visual order, left to right along the baseline: a right-to-left script arrives from the content stream already laid out visually and carries no direction of its own, so a consumer needing logical order applies the Unicode Bidi Algorithm to the result. A vertically set run is grouped into columns rather than lines, from the writingMode its own reader reported, and its column is reported with a rotationDeg of 270 and its own writingMode; columns order right to left, the direction vertical setting is read in. What is NOT recognised is a predefined non-Identity CMap's own byte decoding, so a vertically set font using one (90ms-RKSJ-V and its siblings) still has its codes read as two-byte CIDs, which is this package's own long-standing composite-font limit rather than anything about writing mode. A word hyphenated across a line end is left split, with its hyphen intact, because rejoining it needs to know the two lines belong to one paragraph, which is a semantic judgement this package deliberately leaves to its consumers. Runs with empty text are dropped, and so are the empty words a whitespace-only run would otherwise produce, but no other normalisation of the decoded text is attempted: a ligature and a zero-width character each reach the output exactly as the font's ToUnicode mapping spelled them.
|
|
67
69
|
* @param runs - positioned text runs, in any order
|
|
68
70
|
* @param options - tolerance overrides; each defaults to the exported constant of the same name
|
|
69
71
|
* @returns the grouped lines, in reading order
|
package/dist/text-group.d.ts
CHANGED
|
@@ -6,6 +6,7 @@ interface PdfTextRunGeometry {
|
|
|
6
6
|
readonly sizePt: number;
|
|
7
7
|
readonly widthPt?: number;
|
|
8
8
|
readonly rotationDeg?: number;
|
|
9
|
+
readonly writingMode?: "vertical";
|
|
9
10
|
}
|
|
10
11
|
interface PdfTextBox {
|
|
11
12
|
readonly xPt: number;
|
|
@@ -25,6 +26,7 @@ interface PdfTextLine<TRun extends PdfTextRunGeometry> {
|
|
|
25
26
|
readonly words: readonly PdfTextWord<TRun>[];
|
|
26
27
|
readonly baselineYPt: number;
|
|
27
28
|
readonly rotationDeg: number;
|
|
29
|
+
readonly writingMode?: "vertical";
|
|
28
30
|
readonly bounds?: PdfTextBox;
|
|
29
31
|
}
|
|
30
32
|
interface PdfTextGroupingOptions {
|
|
@@ -63,7 +65,7 @@ declare function runGapPt(previous: PdfTextRunGeometry, next: PdfTextRunGeometry
|
|
|
63
65
|
*
|
|
64
66
|
* Grouping is by baseline alone, with no page segmentation: on a multi-column page whose columns are set at different vertical offsets, a line of one column can fall within tolerance of a line of the next and the two are reported as one line, because from geometry alone at this level that is what they are. The gutter between them still reads as a column boundary, so the separator says where to cut. A caller that needs the columns apart segments the page first, which is what document-outline.js's `segmentPdfRegions` is for, and groups each region's own runs.
|
|
65
67
|
*
|
|
66
|
-
* Known limitations, all of them inherent to what a PDF states rather than to this implementation. Text is reported in visual order, left to right along the baseline: a right-to-left script arrives from the content stream already laid out visually and carries no direction of its own, so a consumer needing logical order applies the Unicode Bidi Algorithm to the result.
|
|
68
|
+
* Known limitations, all of them inherent to what a PDF states rather than to this implementation. Text is reported in visual order, left to right along the baseline: a right-to-left script arrives from the content stream already laid out visually and carries no direction of its own, so a consumer needing logical order applies the Unicode Bidi Algorithm to the result. A vertically set run is grouped into columns rather than lines, from the writingMode its own reader reported, and its column is reported with a rotationDeg of 270 and its own writingMode; columns order right to left, the direction vertical setting is read in. What is NOT recognised is a predefined non-Identity CMap's own byte decoding, so a vertically set font using one (90ms-RKSJ-V and its siblings) still has its codes read as two-byte CIDs, which is this package's own long-standing composite-font limit rather than anything about writing mode. A word hyphenated across a line end is left split, with its hyphen intact, because rejoining it needs to know the two lines belong to one paragraph, which is a semantic judgement this package deliberately leaves to its consumers. Runs with empty text are dropped, and so are the empty words a whitespace-only run would otherwise produce, but no other normalisation of the decoded text is attempted: a ligature and a zero-width character each reach the output exactly as the font's ToUnicode mapping spelled them.
|
|
67
69
|
* @param runs - positioned text runs, in any order
|
|
68
70
|
* @param options - tolerance overrides; each defaults to the exported constant of the same name
|
|
69
71
|
* @returns the grouped lines, in reading order
|
package/dist/text-group.js
CHANGED
|
@@ -4,8 +4,10 @@ const DEFAULT_WORD_GAP_EM = 1 / 8;
|
|
|
4
4
|
const DEFAULT_COLUMN_GAP_EM = 1;
|
|
5
5
|
const ROTATION_NOISE_DEG = 1e-6;
|
|
6
6
|
const DEGREES_PER_TURN = 360;
|
|
7
|
-
|
|
8
|
-
|
|
7
|
+
const VERTICAL_ADVANCE_OFFSET_DEG = -90;
|
|
8
|
+
function advanceAngleDeg(run) {
|
|
9
|
+
const raw = (run.rotationDeg ?? 0) + (run.writingMode === "vertical" ? VERTICAL_ADVANCE_OFFSET_DEG : 0);
|
|
10
|
+
return (Math.round(raw / ROTATION_NOISE_DEG) * ROTATION_NOISE_DEG % DEGREES_PER_TURN + DEGREES_PER_TURN) % DEGREES_PER_TURN;
|
|
9
11
|
}
|
|
10
12
|
const RADIANS_PER_TURN = 2 * Math.PI;
|
|
11
13
|
function baselineAxis(rotationDeg) {
|
|
@@ -41,8 +43,8 @@ function unproject(alongPt, acrossPt, rotationDeg) {
|
|
|
41
43
|
* @returns true when the two baselines are within tolerance of each other
|
|
42
44
|
*/
|
|
43
45
|
function runsShareBaseline(a, b, options = {}) {
|
|
44
|
-
const rotationDeg =
|
|
45
|
-
if (rotationDeg !==
|
|
46
|
+
const rotationDeg = advanceAngleDeg(a);
|
|
47
|
+
if (rotationDeg !== advanceAngleDeg(b) || a.writingMode !== b.writingMode) return false;
|
|
46
48
|
const toleranceEm = options.baselineToleranceEm ?? .6666666666666666;
|
|
47
49
|
return Math.abs(project(a, rotationDeg).acrossPt - project(b, rotationDeg).acrossPt) <= toleranceEm * Math.min(a.sizePt, b.sizePt);
|
|
48
50
|
}
|
|
@@ -55,8 +57,8 @@ function runsShareBaseline(a, b, options = {}) {
|
|
|
55
57
|
* @returns the gap in points, or undefined when it cannot be derived
|
|
56
58
|
*/
|
|
57
59
|
function runGapPt(previous, next) {
|
|
58
|
-
const rotationDeg =
|
|
59
|
-
if (rotationDeg !==
|
|
60
|
+
const rotationDeg = advanceAngleDeg(previous);
|
|
61
|
+
if (rotationDeg !== advanceAngleDeg(next) || previous.writingMode !== next.writingMode) return;
|
|
60
62
|
const previousWidthPt = previous.widthPt;
|
|
61
63
|
if (previousWidthPt === void 0) return;
|
|
62
64
|
return project(next, rotationDeg).alongPt - (project(previous, rotationDeg).alongPt + previousWidthPt);
|
|
@@ -149,7 +151,7 @@ function buildWords(line, rotationDeg, wordGapEm, columnGapEm) {
|
|
|
149
151
|
...word.measurable ? { bounds: boundsOf(word.minAlongPt, word.maxAlongEndPt, line.anchor.acrossPt, word.maxSizePt, rotationDeg) } : {}
|
|
150
152
|
}));
|
|
151
153
|
}
|
|
152
|
-
function buildLine(line, rotationDeg, wordGapEm, columnGapEm) {
|
|
154
|
+
function buildLine(line, rotationDeg, vertical, wordGapEm, columnGapEm) {
|
|
153
155
|
const words = buildWords(line, rotationDeg, wordGapEm, columnGapEm);
|
|
154
156
|
const text = words.map((word) => SEPARATOR_TEXT[word.separatorBefore] + word.text).join("");
|
|
155
157
|
let minAlongPt = Number.POSITIVE_INFINITY;
|
|
@@ -168,6 +170,7 @@ function buildLine(line, rotationDeg, wordGapEm, columnGapEm) {
|
|
|
168
170
|
words,
|
|
169
171
|
baselineYPt: line.anchor.acrossPt,
|
|
170
172
|
rotationDeg,
|
|
173
|
+
...vertical ? { writingMode: "vertical" } : {},
|
|
171
174
|
...measurable ? { bounds: boundsOf(minAlongPt, maxAlongEndPt, line.anchor.acrossPt, maxSizePt, rotationDeg) } : {}
|
|
172
175
|
};
|
|
173
176
|
}
|
|
@@ -180,7 +183,7 @@ function buildLine(line, rotationDeg, wordGapEm, columnGapEm) {
|
|
|
180
183
|
*
|
|
181
184
|
* Grouping is by baseline alone, with no page segmentation: on a multi-column page whose columns are set at different vertical offsets, a line of one column can fall within tolerance of a line of the next and the two are reported as one line, because from geometry alone at this level that is what they are. The gutter between them still reads as a column boundary, so the separator says where to cut. A caller that needs the columns apart segments the page first, which is what document-outline.js's `segmentPdfRegions` is for, and groups each region's own runs.
|
|
182
185
|
*
|
|
183
|
-
* Known limitations, all of them inherent to what a PDF states rather than to this implementation. Text is reported in visual order, left to right along the baseline: a right-to-left script arrives from the content stream already laid out visually and carries no direction of its own, so a consumer needing logical order applies the Unicode Bidi Algorithm to the result.
|
|
186
|
+
* Known limitations, all of them inherent to what a PDF states rather than to this implementation. Text is reported in visual order, left to right along the baseline: a right-to-left script arrives from the content stream already laid out visually and carries no direction of its own, so a consumer needing logical order applies the Unicode Bidi Algorithm to the result. A vertically set run is grouped into columns rather than lines, from the writingMode its own reader reported, and its column is reported with a rotationDeg of 270 and its own writingMode; columns order right to left, the direction vertical setting is read in. What is NOT recognised is a predefined non-Identity CMap's own byte decoding, so a vertically set font using one (90ms-RKSJ-V and its siblings) still has its codes read as two-byte CIDs, which is this package's own long-standing composite-font limit rather than anything about writing mode. A word hyphenated across a line end is left split, with its hyphen intact, because rejoining it needs to know the two lines belong to one paragraph, which is a semantic judgement this package deliberately leaves to its consumers. Runs with empty text are dropped, and so are the empty words a whitespace-only run would otherwise produce, but no other normalisation of the decoded text is attempted: a ligature and a zero-width character each reach the output exactly as the font's ToUnicode mapping spelled them.
|
|
184
187
|
* @param runs - positioned text runs, in any order
|
|
185
188
|
* @param options - tolerance overrides; each defaults to the exported constant of the same name
|
|
186
189
|
* @returns the grouped lines, in reading order
|
|
@@ -189,20 +192,27 @@ function groupPdfTextRuns(runs, options = {}) {
|
|
|
189
192
|
const toleranceEm = options.baselineToleranceEm ?? .6666666666666666;
|
|
190
193
|
const wordGapEm = options.wordGapEm ?? .125;
|
|
191
194
|
const columnGapEm = options.columnGapEm ?? 1;
|
|
192
|
-
const
|
|
195
|
+
const buckets = /* @__PURE__ */ new Map();
|
|
193
196
|
for (const run of runs) {
|
|
194
197
|
if (run.text.length === 0) continue;
|
|
195
|
-
const rotationDeg =
|
|
196
|
-
const
|
|
197
|
-
|
|
198
|
-
|
|
198
|
+
const rotationDeg = advanceAngleDeg(run);
|
|
199
|
+
const vertical = run.writingMode === "vertical";
|
|
200
|
+
const key = `${String(rotationDeg)}|${String(vertical)}`;
|
|
201
|
+
const bucket = buckets.get(key);
|
|
202
|
+
if (bucket === void 0) buckets.set(key, {
|
|
203
|
+
rotationDeg,
|
|
204
|
+
vertical,
|
|
205
|
+
runs: [run]
|
|
206
|
+
});
|
|
207
|
+
else bucket.runs.push(run);
|
|
199
208
|
}
|
|
200
209
|
const result = [];
|
|
201
|
-
|
|
202
|
-
|
|
210
|
+
const ordered = [...buckets.values()].sort((a, b) => a.rotationDeg - b.rotationDeg || Number(a.vertical) - Number(b.vertical));
|
|
211
|
+
for (const bucket of ordered) {
|
|
212
|
+
const entries = bucket.runs.map((run) => project(run, bucket.rotationDeg)).sort((a, b) => b.acrossPt - a.acrossPt || a.alongPt - b.alongPt);
|
|
203
213
|
for (const line of clusterIntoLines(entries, toleranceEm)) {
|
|
204
214
|
line.entries.sort((a, b) => a.alongPt - b.alongPt);
|
|
205
|
-
result.push(buildLine(line, rotationDeg, wordGapEm, columnGapEm));
|
|
215
|
+
result.push(buildLine(line, bucket.rotationDeg, bucket.vertical, wordGapEm, columnGapEm));
|
|
206
216
|
}
|
|
207
217
|
}
|
|
208
218
|
return result;
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pdf-parser.js",
|
|
3
|
-
"version": "5.
|
|
3
|
+
"version": "5.2.0",
|
|
4
4
|
"description": "Hand-written, dependency-minimal PDF codec: parses arbitrary real-world PDFs and generates new ones, built on its own codec-owned LayoutDocument item model and Zod 4 codecs.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"repository": {
|