reamkit 1.25.0 → 1.26.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +5 -3
- package/dist/esm/pdf-reader/content.d.ts +2 -16
- package/dist/esm/pdf-reader/content.js +69 -3
- package/dist/esm/pdf-reader/font.js +21 -1
- package/dist/esm/pdf-reader/glyph-names.d.ts +6 -0
- package/dist/esm/pdf-reader/glyph-names.js +408 -0
- package/dist/esm/pdf-reader/image-decode.js +32 -1
- package/dist/esm/pdf-reader/jbig2.d.ts +86 -0
- package/dist/esm/pdf-reader/jbig2.js +2597 -0
- package/dist/esm/pdf-reader/reader.js +1 -1
- package/dist/esm/pdf-reader/shading.d.ts +13 -0
- package/dist/esm/pdf-reader/shading.js +68 -2
- package/dist/esm/pdf-reader/vector.js +5 -4
- package/dist/esm/word/docx-writer.js +208 -28
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -191,10 +191,12 @@ preview, flowed HTML export, and **docx + xlsx output** (write WordprocessingML
|
|
|
191
191
|
tagged PDF from its structure tree (headings, tables, lists, reading order), an
|
|
192
192
|
untagged one heuristically from glyph positions (lines, paragraphs, headings,
|
|
193
193
|
and a clean two-column split). It lifts back the text (via each font's
|
|
194
|
-
`/ToUnicode`, or the embedded program's own `cmap` where there is none
|
|
194
|
+
`/ToUnicode`, or the embedded program's own `cmap` where there is none, or the
|
|
195
|
+
glyph names its `/Encoding` states — which is all a PDF from TeX gives), the
|
|
195
196
|
font programs themselves, raster images (JPEG verbatim; PNG/Flate/LZW/CCITT-fax
|
|
196
|
-
decoded and re-encoded), `/Link` hyperlinks, form-XObject content,
|
|
197
|
-
appearances,
|
|
197
|
+
and **JBIG2** decoded and re-encoded), `/Link` hyperlinks, form-XObject content,
|
|
198
|
+
annotation appearances, colour set through a named space, and the page's
|
|
199
|
+
artwork: filled / stroked / gradient shapes,
|
|
198
200
|
clipping paths, tiling patterns, constant alpha, and the Type 3 glyphs that are
|
|
199
201
|
drawings rather than letters. It reads modern compressed files (cross-reference
|
|
200
202
|
+ object streams) and encrypted ones (RC4 / AES — the user password is passed to
|
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { ColorSpaceInfo } from './shading.js';
|
|
1
2
|
import { ShapeGradient } from '../core/vector.js';
|
|
2
3
|
import { PdfDict, PdfStream } from '../pdf/objects.js';
|
|
3
4
|
/**
|
|
@@ -221,22 +222,7 @@ export type Matrix = readonly [number, number, number, number, number, number];
|
|
|
221
222
|
export declare const IDENTITY: Matrix;
|
|
222
223
|
/** Compose two {@link Matrix matrices}: `a` applied first, then `b`. */
|
|
223
224
|
export declare function multiply(a: Matrix, b: Matrix): Matrix;
|
|
224
|
-
|
|
225
|
-
* Walk a page's decoded content stream (§9.4) and extract its positioned text,
|
|
226
|
-
* painted XObjects and painted paths. Tracks the graphics state (CTM via
|
|
227
|
-
* `q`/`Q`/`cm`) and text state (text/line matrices, font, size, spacing), and
|
|
228
|
-
* emits one {@link TextRun} per show operator (`Tj` / `TJ` / `'` / `"`) in page
|
|
229
|
-
* space (points).
|
|
230
|
-
*
|
|
231
|
-
* @param bytes The page's decoded content-stream bytes.
|
|
232
|
-
* @param fonts Page fonts by resource key (`Tf` name), for decoding + widths;
|
|
233
|
-
* an unmapped key falls back to Latin-1 with a half-em advance.
|
|
234
|
-
* @param initialCtm The starting CTM mapping user space to page space.
|
|
235
|
-
* @param shadings Shading patterns by name, selected by `scn`/`sc` (EP16c).
|
|
236
|
-
* @param alphas Constant fill alphas by `/ExtGState` name, selected by `gs`.
|
|
237
|
-
* @returns The extracted text runs, image placements and vector paths.
|
|
238
|
-
*/
|
|
239
|
-
export declare function interpretContent(bytes: Uint8Array, fonts: ReadonlyMap<string, ContentFont>, initialCtm?: Matrix, shadings?: ReadonlyMap<string, ShapeGradient>, alphas?: ReadonlyMap<string, number>): InterpretResult;
|
|
225
|
+
export declare function interpretContent(bytes: Uint8Array, fonts: ReadonlyMap<string, ContentFont>, initialCtm?: Matrix, shadings?: ReadonlyMap<string, ShapeGradient>, alphas?: ReadonlyMap<string, number>, spaces?: ReadonlyMap<string, ColorSpaceInfo>): InterpretResult;
|
|
240
226
|
/**
|
|
241
227
|
* Whether a string is wholly right-to-left: at least one letter of an RTL
|
|
242
228
|
* script and nothing of any other, spaces and joiners aside.
|
|
@@ -73,6 +73,8 @@ var FALLBACK_FONT = {
|
|
|
73
73
|
var UPRIGHT_TOLERANCE_DEG = .5;
|
|
74
74
|
function initialState() {
|
|
75
75
|
return {
|
|
76
|
+
fillSpace: void 0,
|
|
77
|
+
strokeSpace: void 0,
|
|
76
78
|
ctm: IDENTITY,
|
|
77
79
|
fontKey: "",
|
|
78
80
|
font: FALLBACK_FONT,
|
|
@@ -107,7 +109,49 @@ function initialState() {
|
|
|
107
109
|
* @param alphas Constant fill alphas by `/ExtGState` name, selected by `gs`.
|
|
108
110
|
* @returns The extracted text runs, image placements and vector paths.
|
|
109
111
|
*/
|
|
110
|
-
|
|
112
|
+
/** §8.6.8 — the space a `cs` / `CS` operand names, if it is one we read. */
|
|
113
|
+
function spaceOf(operands, spaces) {
|
|
114
|
+
const name = operands[operands.length - 1];
|
|
115
|
+
if (!(name instanceof PdfName)) return void 0;
|
|
116
|
+
const direct = spaces.get(name.value);
|
|
117
|
+
if (direct) return direct;
|
|
118
|
+
switch (name.value) {
|
|
119
|
+
case "DeviceGray": return {
|
|
120
|
+
kind: "gray",
|
|
121
|
+
components: 1
|
|
122
|
+
};
|
|
123
|
+
case "DeviceRGB": return {
|
|
124
|
+
kind: "rgb",
|
|
125
|
+
components: 3
|
|
126
|
+
};
|
|
127
|
+
case "DeviceCMYK": return {
|
|
128
|
+
kind: "cmyk",
|
|
129
|
+
components: 4
|
|
130
|
+
};
|
|
131
|
+
default: return;
|
|
132
|
+
}
|
|
133
|
+
}
|
|
134
|
+
/**
|
|
135
|
+
* §8.6.8 — the colour a run of `sc` / `scn` components comes to.
|
|
136
|
+
*
|
|
137
|
+
* The space in force decides, and where it was not read the COUNT is the next
|
|
138
|
+
* best witness: three numbers are RGB and four are CMYK on every device space
|
|
139
|
+
* there is. One number is the ambiguous case — grey in a device space, but the
|
|
140
|
+
* strength of a colorant in a Separation, where 1 is the ink at full and reads
|
|
141
|
+
* dark — so a lone component is only taken where the space said what it means.
|
|
142
|
+
*/
|
|
143
|
+
function componentColor(operands, space) {
|
|
144
|
+
const nums = operands.filter((o) => typeof o === "number");
|
|
145
|
+
if (nums.length === 0) return void 0;
|
|
146
|
+
switch (space?.kind ?? (nums.length === 3 ? "rgb" : nums.length === 4 ? "cmyk" : void 0)) {
|
|
147
|
+
case "rgb": return nums.length >= 3 ? rgbHex(nums[0], nums[1], nums[2]) : void 0;
|
|
148
|
+
case "cmyk": return nums.length >= 4 ? cmykHex(nums[0], nums[1], nums[2], nums[3]) : void 0;
|
|
149
|
+
case "gray": return grayHex(nums[0]);
|
|
150
|
+
case "tint": return grayHex(1 - Math.max(...nums));
|
|
151
|
+
default: return;
|
|
152
|
+
}
|
|
153
|
+
}
|
|
154
|
+
function interpretContent(bytes, fonts, initialCtm = IDENTITY, shadings = /* @__PURE__ */ new Map(), alphas = /* @__PURE__ */ new Map(), spaces = /* @__PURE__ */ new Map()) {
|
|
111
155
|
const runs = [];
|
|
112
156
|
const images = [];
|
|
113
157
|
const vectors = [];
|
|
@@ -384,12 +428,34 @@ function interpretContent(bytes, fonts, initialCtm = IDENTITY, shadings = /* @__
|
|
|
384
428
|
state.fillGradient = void 0;
|
|
385
429
|
state.fillPattern = void 0;
|
|
386
430
|
break;
|
|
431
|
+
case "cs":
|
|
432
|
+
state.fillSpace = spaceOf(operands, spaces);
|
|
433
|
+
break;
|
|
434
|
+
case "CS":
|
|
435
|
+
state.strokeSpace = spaceOf(operands, spaces);
|
|
436
|
+
break;
|
|
387
437
|
case "scn":
|
|
388
438
|
case "sc": {
|
|
389
439
|
const last = operands[operands.length - 1];
|
|
390
440
|
const named = last instanceof PdfName ? last.value : void 0;
|
|
391
|
-
|
|
392
|
-
|
|
441
|
+
if (named !== void 0) {
|
|
442
|
+
state.fillGradient = shadings.get(named);
|
|
443
|
+
state.fillPattern = state.fillGradient ? void 0 : named;
|
|
444
|
+
break;
|
|
445
|
+
}
|
|
446
|
+
const hex = componentColor(operands, state.fillSpace);
|
|
447
|
+
if (hex !== void 0) {
|
|
448
|
+
state.fillColor = hex;
|
|
449
|
+
state.fillGradient = void 0;
|
|
450
|
+
state.fillPattern = void 0;
|
|
451
|
+
}
|
|
452
|
+
break;
|
|
453
|
+
}
|
|
454
|
+
case "SCN":
|
|
455
|
+
case "SC": {
|
|
456
|
+
if (operands[operands.length - 1] instanceof PdfName) break;
|
|
457
|
+
const hex = componentColor(operands, state.strokeSpace);
|
|
458
|
+
if (hex !== void 0) state.strokeColor = hex;
|
|
393
459
|
break;
|
|
394
460
|
}
|
|
395
461
|
case "RG":
|
|
@@ -2,6 +2,7 @@ import { parseTtf } from "../core/font/ttf-parser.js";
|
|
|
2
2
|
import { PDF_NULL, PdfName, PdfStream } from "../pdf/objects.js";
|
|
3
3
|
import { embeddedFontName } from "./embedded-fonts.js";
|
|
4
4
|
import { parseToUnicodeCMap } from "./cmap.js";
|
|
5
|
+
import { textForGlyphName } from "./glyph-names.js";
|
|
5
6
|
//#region src/pdf-reader/font.ts
|
|
6
7
|
/**
|
|
7
8
|
* Build a {@link ContentFont} (the interpreter's decode + advance hooks) from a
|
|
@@ -25,7 +26,9 @@ function buildContentFont(file, fontDict) {
|
|
|
25
26
|
toUnicode = parsed.map;
|
|
26
27
|
if (isType0) codeBytes = parsed.codeBytes;
|
|
27
28
|
}
|
|
28
|
-
const
|
|
29
|
+
const fromProgram = isType0 && toUnicode.size === 0 ? embeddedCmap(file, fontDict) : void 0;
|
|
30
|
+
const fromNames = !isType0 ? namedGlyphs(file, fontDict) : void 0;
|
|
31
|
+
const unicode = fromProgram ?? (toUnicode.size > 0 ? toUnicode : fromNames ?? toUnicode);
|
|
29
32
|
const bytesPerCode = codeBytes;
|
|
30
33
|
const style = faceStyle(file, fontDict, isType0);
|
|
31
34
|
const name = embeddedFontName(file, fontDict);
|
|
@@ -148,6 +151,23 @@ function fontMatrix(file, fontDict) {
|
|
|
148
151
|
n[5]
|
|
149
152
|
];
|
|
150
153
|
}
|
|
154
|
+
/**
|
|
155
|
+
* §9.6.6.1 — code → text for a simple font, out of the glyph names its
|
|
156
|
+
* `/Encoding /Differences` gives.
|
|
157
|
+
*
|
|
158
|
+
* Returns `undefined` where the font names nothing, so the caller keeps its
|
|
159
|
+
* Latin-1 reading rather than replacing it with an empty map.
|
|
160
|
+
*/
|
|
161
|
+
function namedGlyphs(file, fontDict) {
|
|
162
|
+
const names = differences(file, fontDict);
|
|
163
|
+
if (names.size === 0) return void 0;
|
|
164
|
+
const out = /* @__PURE__ */ new Map();
|
|
165
|
+
for (const [code, name] of names) {
|
|
166
|
+
const text = textForGlyphName(name);
|
|
167
|
+
if (text !== void 0) out.set(code, text);
|
|
168
|
+
}
|
|
169
|
+
return out.size > 0 ? out : void 0;
|
|
170
|
+
}
|
|
151
171
|
/** §9.6.6.1 `/Encoding` `/Differences` — code → glyph name, as the array runs. */
|
|
152
172
|
function differences(file, fontDict) {
|
|
153
173
|
const out = /* @__PURE__ */ new Map();
|
|
@@ -0,0 +1,408 @@
|
|
|
1
|
+
//#region src/pdf-reader/glyph-names.ts
|
|
2
|
+
/** Glyph name → the text it stands for. */
|
|
3
|
+
var GLYPHS = {
|
|
4
|
+
AE: "Æ",
|
|
5
|
+
Aacute: "Á",
|
|
6
|
+
Abreve: "Ă",
|
|
7
|
+
Acaron: "Ǎ",
|
|
8
|
+
Acircumflex: "Â",
|
|
9
|
+
Adieresis: "Ä",
|
|
10
|
+
Adotaccent: "Ȧ",
|
|
11
|
+
Agrave: "À",
|
|
12
|
+
Amacron: "Ā",
|
|
13
|
+
Aogonek: "Ą",
|
|
14
|
+
Aring: "Å",
|
|
15
|
+
Atilde: "Ã",
|
|
16
|
+
Bdotaccent: "Ḃ",
|
|
17
|
+
Cacute: "Ć",
|
|
18
|
+
Ccaron: "Č",
|
|
19
|
+
Ccedilla: "Ç",
|
|
20
|
+
Ccircumflex: "Ĉ",
|
|
21
|
+
Cdotaccent: "Ċ",
|
|
22
|
+
Dcaron: "Ď",
|
|
23
|
+
Dcedilla: "Ḑ",
|
|
24
|
+
Ddotaccent: "Ḋ",
|
|
25
|
+
Eacute: "É",
|
|
26
|
+
Ebreve: "Ĕ",
|
|
27
|
+
Ecaron: "Ě",
|
|
28
|
+
Ecedilla: "Ȩ",
|
|
29
|
+
Ecircumflex: "Ê",
|
|
30
|
+
Edieresis: "Ë",
|
|
31
|
+
Edotaccent: "Ė",
|
|
32
|
+
Egrave: "È",
|
|
33
|
+
Emacron: "Ē",
|
|
34
|
+
Eogonek: "Ę",
|
|
35
|
+
Eth: "Ð",
|
|
36
|
+
Etilde: "Ẽ",
|
|
37
|
+
Euro: "€",
|
|
38
|
+
Fdotaccent: "Ḟ",
|
|
39
|
+
Gacute: "Ǵ",
|
|
40
|
+
Gbreve: "Ğ",
|
|
41
|
+
Gcaron: "Ǧ",
|
|
42
|
+
Gcedilla: "Ģ",
|
|
43
|
+
Gcircumflex: "Ĝ",
|
|
44
|
+
Gdotaccent: "Ġ",
|
|
45
|
+
Gmacron: "Ḡ",
|
|
46
|
+
Hcaron: "Ȟ",
|
|
47
|
+
Hcedilla: "Ḩ",
|
|
48
|
+
Hcircumflex: "Ĥ",
|
|
49
|
+
Hdieresis: "Ḧ",
|
|
50
|
+
Hdotaccent: "Ḣ",
|
|
51
|
+
Iacute: "Í",
|
|
52
|
+
Ibreve: "Ĭ",
|
|
53
|
+
Icaron: "Ǐ",
|
|
54
|
+
Icircumflex: "Î",
|
|
55
|
+
Idieresis: "Ï",
|
|
56
|
+
Idotaccent: "İ",
|
|
57
|
+
Igrave: "Ì",
|
|
58
|
+
Imacron: "Ī",
|
|
59
|
+
Iogonek: "Į",
|
|
60
|
+
Itilde: "Ĩ",
|
|
61
|
+
Jcircumflex: "Ĵ",
|
|
62
|
+
Kacute: "Ḱ",
|
|
63
|
+
Kcaron: "Ǩ",
|
|
64
|
+
Kcedilla: "Ķ",
|
|
65
|
+
Lacute: "Ĺ",
|
|
66
|
+
Lcaron: "Ľ",
|
|
67
|
+
Lcedilla: "Ļ",
|
|
68
|
+
Macute: "Ḿ",
|
|
69
|
+
Mdotaccent: "Ṁ",
|
|
70
|
+
Nacute: "Ń",
|
|
71
|
+
Ncaron: "Ň",
|
|
72
|
+
Ncedilla: "Ņ",
|
|
73
|
+
Ndotaccent: "Ṅ",
|
|
74
|
+
Ngrave: "Ǹ",
|
|
75
|
+
Ntilde: "Ñ",
|
|
76
|
+
OE: "Œ",
|
|
77
|
+
Oacute: "Ó",
|
|
78
|
+
Obreve: "Ŏ",
|
|
79
|
+
Ocaron: "Ǒ",
|
|
80
|
+
Ocircumflex: "Ô",
|
|
81
|
+
Odieresis: "Ö",
|
|
82
|
+
Odotaccent: "Ȯ",
|
|
83
|
+
Ograve: "Ò",
|
|
84
|
+
Ohungarumlaut: "Ő",
|
|
85
|
+
Omacron: "Ō",
|
|
86
|
+
Oogonek: "Ǫ",
|
|
87
|
+
Oslash: "Ø",
|
|
88
|
+
Otilde: "Õ",
|
|
89
|
+
Pacute: "Ṕ",
|
|
90
|
+
Pdotaccent: "Ṗ",
|
|
91
|
+
Racute: "Ŕ",
|
|
92
|
+
Rcaron: "Ř",
|
|
93
|
+
Rcedilla: "Ŗ",
|
|
94
|
+
Rdotaccent: "Ṙ",
|
|
95
|
+
Sacute: "Ś",
|
|
96
|
+
Scaron: "Š",
|
|
97
|
+
Scedilla: "Ş",
|
|
98
|
+
Scircumflex: "Ŝ",
|
|
99
|
+
Sdotaccent: "Ṡ",
|
|
100
|
+
Tcaron: "Ť",
|
|
101
|
+
Tcedilla: "Ţ",
|
|
102
|
+
Tdotaccent: "Ṫ",
|
|
103
|
+
Thorn: "Þ",
|
|
104
|
+
Uacute: "Ú",
|
|
105
|
+
Ubreve: "Ŭ",
|
|
106
|
+
Ucaron: "Ǔ",
|
|
107
|
+
Ucircumflex: "Û",
|
|
108
|
+
Udieresis: "Ü",
|
|
109
|
+
Ugrave: "Ù",
|
|
110
|
+
Uhungarumlaut: "Ű",
|
|
111
|
+
Umacron: "Ū",
|
|
112
|
+
Uogonek: "Ų",
|
|
113
|
+
Uring: "Ů",
|
|
114
|
+
Utilde: "Ũ",
|
|
115
|
+
Vtilde: "Ṽ",
|
|
116
|
+
Wacute: "Ẃ",
|
|
117
|
+
Wcircumflex: "Ŵ",
|
|
118
|
+
Wdieresis: "Ẅ",
|
|
119
|
+
Wdotaccent: "Ẇ",
|
|
120
|
+
Wgrave: "Ẁ",
|
|
121
|
+
Xdieresis: "Ẍ",
|
|
122
|
+
Xdotaccent: "Ẋ",
|
|
123
|
+
Yacute: "Ý",
|
|
124
|
+
Ycircumflex: "Ŷ",
|
|
125
|
+
Ydieresis: "Ÿ",
|
|
126
|
+
Ydotaccent: "Ẏ",
|
|
127
|
+
Ygrave: "Ỳ",
|
|
128
|
+
Ymacron: "Ȳ",
|
|
129
|
+
Ytilde: "Ỹ",
|
|
130
|
+
Zacute: "Ź",
|
|
131
|
+
Zcaron: "Ž",
|
|
132
|
+
Zcircumflex: "Ẑ",
|
|
133
|
+
Zdotaccent: "Ż",
|
|
134
|
+
aacute: "á",
|
|
135
|
+
abreve: "ă",
|
|
136
|
+
acaron: "ǎ",
|
|
137
|
+
acircumflex: "â",
|
|
138
|
+
acute: "´",
|
|
139
|
+
adieresis: "ä",
|
|
140
|
+
adotaccent: "ȧ",
|
|
141
|
+
ae: "æ",
|
|
142
|
+
agrave: "à",
|
|
143
|
+
amacron: "ā",
|
|
144
|
+
ampersand: "&",
|
|
145
|
+
aogonek: "ą",
|
|
146
|
+
aring: "å",
|
|
147
|
+
asciicircum: "^",
|
|
148
|
+
asciitilde: "~",
|
|
149
|
+
asterisk: "*",
|
|
150
|
+
at: "@",
|
|
151
|
+
atilde: "ã",
|
|
152
|
+
backslash: "\\",
|
|
153
|
+
bar: "|",
|
|
154
|
+
bdotaccent: "ḃ",
|
|
155
|
+
braceleft: "{",
|
|
156
|
+
braceright: "}",
|
|
157
|
+
bracketleft: "[",
|
|
158
|
+
bracketright: "]",
|
|
159
|
+
breve: "˘",
|
|
160
|
+
brokenbar: "¦",
|
|
161
|
+
bullet: "•",
|
|
162
|
+
cacute: "ć",
|
|
163
|
+
caron: "ˇ",
|
|
164
|
+
ccaron: "č",
|
|
165
|
+
ccedilla: "ç",
|
|
166
|
+
ccircumflex: "ĉ",
|
|
167
|
+
cdotaccent: "ċ",
|
|
168
|
+
cedilla: "¸",
|
|
169
|
+
cent: "¢",
|
|
170
|
+
circumflex: "ˆ",
|
|
171
|
+
colon: ":",
|
|
172
|
+
comma: ",",
|
|
173
|
+
copyright: "©",
|
|
174
|
+
currency: "¤",
|
|
175
|
+
dagger: "†",
|
|
176
|
+
daggerdbl: "‡",
|
|
177
|
+
dcaron: "ď",
|
|
178
|
+
dcedilla: "ḑ",
|
|
179
|
+
ddotaccent: "ḋ",
|
|
180
|
+
degree: "°",
|
|
181
|
+
dieresis: "¨",
|
|
182
|
+
divide: "÷",
|
|
183
|
+
dollar: "$",
|
|
184
|
+
dotaccent: "˙",
|
|
185
|
+
dotlessi: "ı",
|
|
186
|
+
eacute: "é",
|
|
187
|
+
ebreve: "ĕ",
|
|
188
|
+
ecaron: "ě",
|
|
189
|
+
ecedilla: "ȩ",
|
|
190
|
+
ecircumflex: "ê",
|
|
191
|
+
edieresis: "ë",
|
|
192
|
+
edotaccent: "ė",
|
|
193
|
+
egrave: "è",
|
|
194
|
+
eight: "8",
|
|
195
|
+
ellipsis: "…",
|
|
196
|
+
emacron: "ē",
|
|
197
|
+
emdash: "—",
|
|
198
|
+
endash: "–",
|
|
199
|
+
eogonek: "ę",
|
|
200
|
+
equal: "=",
|
|
201
|
+
eth: "ð",
|
|
202
|
+
etilde: "ẽ",
|
|
203
|
+
euro: "€",
|
|
204
|
+
exclam: "!",
|
|
205
|
+
exclamdown: "¡",
|
|
206
|
+
fdotaccent: "ḟ",
|
|
207
|
+
ff: "ff",
|
|
208
|
+
ffi: "ffi",
|
|
209
|
+
ffl: "ffl",
|
|
210
|
+
fi: "fi",
|
|
211
|
+
five: "5",
|
|
212
|
+
fl: "fl",
|
|
213
|
+
florin: "ƒ",
|
|
214
|
+
four: "4",
|
|
215
|
+
fraction: "⁄",
|
|
216
|
+
gacute: "ǵ",
|
|
217
|
+
gbreve: "ğ",
|
|
218
|
+
gcaron: "ǧ",
|
|
219
|
+
gcedilla: "ģ",
|
|
220
|
+
gcircumflex: "ĝ",
|
|
221
|
+
gdotaccent: "ġ",
|
|
222
|
+
germandbls: "ß",
|
|
223
|
+
gmacron: "ḡ",
|
|
224
|
+
grave: "`",
|
|
225
|
+
greater: ">",
|
|
226
|
+
guillemotleft: "«",
|
|
227
|
+
guillemotright: "»",
|
|
228
|
+
guilsinglleft: "‹",
|
|
229
|
+
guilsinglright: "›",
|
|
230
|
+
hcaron: "ȟ",
|
|
231
|
+
hcedilla: "ḩ",
|
|
232
|
+
hcircumflex: "ĥ",
|
|
233
|
+
hdieresis: "ḧ",
|
|
234
|
+
hdotaccent: "ḣ",
|
|
235
|
+
hungarumlaut: "˝",
|
|
236
|
+
hyphen: "-",
|
|
237
|
+
iacute: "í",
|
|
238
|
+
ibreve: "ĭ",
|
|
239
|
+
icaron: "ǐ",
|
|
240
|
+
icircumflex: "î",
|
|
241
|
+
idieresis: "ï",
|
|
242
|
+
igrave: "ì",
|
|
243
|
+
imacron: "ī",
|
|
244
|
+
iogonek: "į",
|
|
245
|
+
itilde: "ĩ",
|
|
246
|
+
jcaron: "ǰ",
|
|
247
|
+
jcircumflex: "ĵ",
|
|
248
|
+
kacute: "ḱ",
|
|
249
|
+
kcaron: "ǩ",
|
|
250
|
+
kcedilla: "ķ",
|
|
251
|
+
lacute: "ĺ",
|
|
252
|
+
lcaron: "ľ",
|
|
253
|
+
lcedilla: "ļ",
|
|
254
|
+
less: "<",
|
|
255
|
+
logicalnot: "¬",
|
|
256
|
+
macron: "¯",
|
|
257
|
+
macute: "ḿ",
|
|
258
|
+
mdotaccent: "ṁ",
|
|
259
|
+
minus: "−",
|
|
260
|
+
mu: "µ",
|
|
261
|
+
multiply: "×",
|
|
262
|
+
nacute: "ń",
|
|
263
|
+
nbspace: "\xA0",
|
|
264
|
+
ncaron: "ň",
|
|
265
|
+
ncedilla: "ņ",
|
|
266
|
+
ndotaccent: "ṅ",
|
|
267
|
+
ngrave: "ǹ",
|
|
268
|
+
nine: "9",
|
|
269
|
+
ntilde: "ñ",
|
|
270
|
+
numbersign: "#",
|
|
271
|
+
oacute: "ó",
|
|
272
|
+
obreve: "ŏ",
|
|
273
|
+
ocaron: "ǒ",
|
|
274
|
+
ocircumflex: "ô",
|
|
275
|
+
odieresis: "ö",
|
|
276
|
+
odotaccent: "ȯ",
|
|
277
|
+
oe: "œ",
|
|
278
|
+
ogonek: "˛",
|
|
279
|
+
ograve: "ò",
|
|
280
|
+
ohungarumlaut: "ő",
|
|
281
|
+
omacron: "ō",
|
|
282
|
+
one: "1",
|
|
283
|
+
onehalf: "½",
|
|
284
|
+
onequarter: "¼",
|
|
285
|
+
onesuperior: "¹",
|
|
286
|
+
oogonek: "ǫ",
|
|
287
|
+
ordfeminine: "ª",
|
|
288
|
+
ordmasculine: "º",
|
|
289
|
+
oslash: "ø",
|
|
290
|
+
otilde: "õ",
|
|
291
|
+
pacute: "ṕ",
|
|
292
|
+
paragraph: "¶",
|
|
293
|
+
parenleft: "(",
|
|
294
|
+
parenright: ")",
|
|
295
|
+
pdotaccent: "ṗ",
|
|
296
|
+
percent: "%",
|
|
297
|
+
period: ".",
|
|
298
|
+
perthousand: "‰",
|
|
299
|
+
plus: "+",
|
|
300
|
+
plusminus: "±",
|
|
301
|
+
question: "?",
|
|
302
|
+
questiondown: "¿",
|
|
303
|
+
quotedbl: "\"",
|
|
304
|
+
quotedblbase: "„",
|
|
305
|
+
quotedblleft: "“",
|
|
306
|
+
quotedblright: "”",
|
|
307
|
+
quoteleft: "‘",
|
|
308
|
+
quoteright: "’",
|
|
309
|
+
quotesinglbase: "‚",
|
|
310
|
+
quotesingle: "'",
|
|
311
|
+
racute: "ŕ",
|
|
312
|
+
rcaron: "ř",
|
|
313
|
+
rcedilla: "ŗ",
|
|
314
|
+
rdotaccent: "ṙ",
|
|
315
|
+
registered: "®",
|
|
316
|
+
ring: "˚",
|
|
317
|
+
sacute: "ś",
|
|
318
|
+
scaron: "š",
|
|
319
|
+
scedilla: "ş",
|
|
320
|
+
scircumflex: "ŝ",
|
|
321
|
+
sdotaccent: "ṡ",
|
|
322
|
+
section: "§",
|
|
323
|
+
semicolon: ";",
|
|
324
|
+
seven: "7",
|
|
325
|
+
sfthyphen: "",
|
|
326
|
+
six: "6",
|
|
327
|
+
slash: "/",
|
|
328
|
+
softhyphen: "",
|
|
329
|
+
space: " ",
|
|
330
|
+
sterling: "£",
|
|
331
|
+
tcaron: "ť",
|
|
332
|
+
tcedilla: "ţ",
|
|
333
|
+
tdieresis: "ẗ",
|
|
334
|
+
tdotaccent: "ṫ",
|
|
335
|
+
thorn: "þ",
|
|
336
|
+
three: "3",
|
|
337
|
+
threequarters: "¾",
|
|
338
|
+
threesuperior: "³",
|
|
339
|
+
tilde: "˜",
|
|
340
|
+
trademark: "™",
|
|
341
|
+
two: "2",
|
|
342
|
+
twosuperior: "²",
|
|
343
|
+
uacute: "ú",
|
|
344
|
+
ubreve: "ŭ",
|
|
345
|
+
ucaron: "ǔ",
|
|
346
|
+
ucircumflex: "û",
|
|
347
|
+
udieresis: "ü",
|
|
348
|
+
ugrave: "ù",
|
|
349
|
+
uhungarumlaut: "ű",
|
|
350
|
+
umacron: "ū",
|
|
351
|
+
underscore: "_",
|
|
352
|
+
uogonek: "ų",
|
|
353
|
+
uring: "ů",
|
|
354
|
+
utilde: "ũ",
|
|
355
|
+
vtilde: "ṽ",
|
|
356
|
+
wacute: "ẃ",
|
|
357
|
+
wcircumflex: "ŵ",
|
|
358
|
+
wdieresis: "ẅ",
|
|
359
|
+
wdotaccent: "ẇ",
|
|
360
|
+
wgrave: "ẁ",
|
|
361
|
+
wring: "ẘ",
|
|
362
|
+
xdieresis: "ẍ",
|
|
363
|
+
xdotaccent: "ẋ",
|
|
364
|
+
yacute: "ý",
|
|
365
|
+
ycircumflex: "ŷ",
|
|
366
|
+
ydieresis: "ÿ",
|
|
367
|
+
ydotaccent: "ẏ",
|
|
368
|
+
yen: "¥",
|
|
369
|
+
ygrave: "ỳ",
|
|
370
|
+
ymacron: "ȳ",
|
|
371
|
+
yring: "ẙ",
|
|
372
|
+
ytilde: "ỹ",
|
|
373
|
+
zacute: "ź",
|
|
374
|
+
zcaron: "ž",
|
|
375
|
+
zcircumflex: "ẑ",
|
|
376
|
+
zdotaccent: "ż",
|
|
377
|
+
zero: "0"
|
|
378
|
+
};
|
|
379
|
+
/**
|
|
380
|
+
* The text a glyph name stands for, or `undefined` when the name says nothing.
|
|
381
|
+
*
|
|
382
|
+
* @param name The name as `/Differences` (or a font program's charset) gives it.
|
|
383
|
+
*/
|
|
384
|
+
function textForGlyphName(name) {
|
|
385
|
+
const base = name.includes(".") ? name.split(".")[0] ?? name : name;
|
|
386
|
+
if (base === "") return void 0;
|
|
387
|
+
if (base.includes("_")) {
|
|
388
|
+
const parts = base.split("_").map((p) => textForGlyphName(p));
|
|
389
|
+
return parts.every((p) => p !== void 0) ? parts.join("") : void 0;
|
|
390
|
+
}
|
|
391
|
+
const known = GLYPHS[base];
|
|
392
|
+
if (known !== void 0) return known;
|
|
393
|
+
if (base.length === 1) return base;
|
|
394
|
+
const uni = /^uni((?:[0-9A-Fa-f]{4})+)$/u.exec(base);
|
|
395
|
+
if (uni) {
|
|
396
|
+
const hex = uni[1];
|
|
397
|
+
let out = "";
|
|
398
|
+
for (let i = 0; i < hex.length; i += 4) out += String.fromCodePoint(parseInt(hex.slice(i, i + 4), 16));
|
|
399
|
+
return out;
|
|
400
|
+
}
|
|
401
|
+
const u = /^u([0-9A-Fa-f]{4,6})$/u.exec(base);
|
|
402
|
+
if (u) {
|
|
403
|
+
const cp = parseInt(u[1], 16);
|
|
404
|
+
return cp <= 1114111 ? String.fromCodePoint(cp) : void 0;
|
|
405
|
+
}
|
|
406
|
+
}
|
|
407
|
+
//#endregion
|
|
408
|
+
export { textForGlyphName };
|
|
@@ -3,6 +3,7 @@ import { encodePng } from "../core/png-encode.js";
|
|
|
3
3
|
import { lzwDecodeMsb } from "../core/lzw.js";
|
|
4
4
|
import { reversePredictor } from "./predictor.js";
|
|
5
5
|
import { decodeCcitt } from "./ccitt.js";
|
|
6
|
+
import { decodeJbig2 } from "./jbig2.js";
|
|
6
7
|
import { decodeJpeg } from "./jpeg.js";
|
|
7
8
|
import { unzlibSync } from "fflate";
|
|
8
9
|
//#region src/pdf-reader/image-decode.ts
|
|
@@ -52,7 +53,17 @@ function decodePdfImage(file, stream) {
|
|
|
52
53
|
heightPx: height,
|
|
53
54
|
degraded: "JPEG 2000 image — limited viewer support"
|
|
54
55
|
};
|
|
55
|
-
if (last === "JBIG2Decode")
|
|
56
|
+
if (last === "JBIG2Decode") {
|
|
57
|
+
const raster = decodeJbig2Image(file, stream, filters, width, height);
|
|
58
|
+
if (typeof raster === "string") return fail("dropped", raster);
|
|
59
|
+
return {
|
|
60
|
+
ok: true,
|
|
61
|
+
bytes: encodePng(width, height, raster.color, raster.samples),
|
|
62
|
+
format: "png",
|
|
63
|
+
widthPx: width,
|
|
64
|
+
heightPx: height
|
|
65
|
+
};
|
|
66
|
+
}
|
|
56
67
|
const decoded = last === "CCITTFaxDecode" || last === "CCF" ? decodeCcittImage(file, stream, filters, width, height) : decodeToSamples(file, stream, filters, width, height);
|
|
57
68
|
if (typeof decoded === "string") return fail("dropped", decoded);
|
|
58
69
|
const alpha = decodeSMask(file, d, width, height);
|
|
@@ -75,6 +86,26 @@ function decodeToSamples(file, stream, filters, width, height) {
|
|
|
75
86
|
const decodeArr = decodeArrayOf(file, d);
|
|
76
87
|
return toColor(cs, unpackSamples(raw, width, height, cs.components, bpc), width * height, bpc, decodeArr);
|
|
77
88
|
}
|
|
89
|
+
/**
|
|
90
|
+
* §7.4.7 — a `/JBIG2Decode` image: the stream's own segments, plus the shared
|
|
91
|
+
* ones a `/JBIG2Globals` holds (the symbol dictionary a scanned book stores
|
|
92
|
+
* once for all its pages).
|
|
93
|
+
*/
|
|
94
|
+
function decodeJbig2Image(file, stream, filters, width, height) {
|
|
95
|
+
const parms = decodeParmsOf(file, stream.dict);
|
|
96
|
+
const globalsVal = parms ? file.get(parms, "JBIG2Globals") : void 0;
|
|
97
|
+
const globalsStream = globalsVal instanceof PdfStream ? globalsVal : void 0;
|
|
98
|
+
const globals = globalsStream ? file.streamData(globalsStream) : void 0;
|
|
99
|
+
const packed = decodeJbig2(applyChainExceptLast(filters, stream.data), globals, width, height);
|
|
100
|
+
if (!packed) return "JBIG2 image not decoded";
|
|
101
|
+
const rowBytes = width + 7 >> 3;
|
|
102
|
+
const samples = new Uint8Array(width * height);
|
|
103
|
+
for (let y = 0; y < height; y++) for (let x = 0; x < width; x++) samples[y * width + x] = packed[y * rowBytes + (x >> 3)] >> 7 - (x & 7) & 1 ? 0 : 255;
|
|
104
|
+
return {
|
|
105
|
+
color: "gray",
|
|
106
|
+
samples
|
|
107
|
+
};
|
|
108
|
+
}
|
|
78
109
|
function decodeCcittImage(file, stream, filters, width, height) {
|
|
79
110
|
const parms = decodeParmsOf(file, stream.dict);
|
|
80
111
|
const columns = (parms ? intOf(file.get(parms, "Columns")) : 0) || 1728;
|