@readium/shared 1.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (150) hide show
  1. package/LICENSE +28 -0
  2. package/README.MD +27 -0
  3. package/dist/index.js +2536 -0
  4. package/dist/index.umd.cjs +2 -0
  5. package/package.json +73 -0
  6. package/src/fetcher/Fetcher.ts +37 -0
  7. package/src/fetcher/HttpFetcher.ts +122 -0
  8. package/src/fetcher/Resource.ts +37 -0
  9. package/src/fetcher/index.ts +2 -0
  10. package/src/index.ts +4 -0
  11. package/src/opds/Acquisition.ts +54 -0
  12. package/src/opds/Availability.ts +61 -0
  13. package/src/opds/Copies.ts +43 -0
  14. package/src/opds/Holds.ts +43 -0
  15. package/src/opds/Price.ts +51 -0
  16. package/src/opds/index.ts +5 -0
  17. package/src/publication/BelongsTo.ts +51 -0
  18. package/src/publication/Contributor.ts +129 -0
  19. package/src/publication/GuidedNavigation.ts +176 -0
  20. package/src/publication/Link.ts +332 -0
  21. package/src/publication/LocalizedString.ts +74 -0
  22. package/src/publication/Locator.ts +225 -0
  23. package/src/publication/LocatorCollection.ts +133 -0
  24. package/src/publication/Manifest.ts +256 -0
  25. package/src/publication/Metadata.ts +329 -0
  26. package/src/publication/Properties.ts +44 -0
  27. package/src/publication/Publication.ts +148 -0
  28. package/src/publication/PublicationCollection.ts +144 -0
  29. package/src/publication/ReadingProgression.ts +28 -0
  30. package/src/publication/Subject.ts +115 -0
  31. package/src/publication/encryption/Encryption.ts +68 -0
  32. package/src/publication/encryption/Properties.ts +20 -0
  33. package/src/publication/encryption/index.ts +2 -0
  34. package/src/publication/epub/EPUBLayout.ts +10 -0
  35. package/src/publication/epub/Presentation.ts +16 -0
  36. package/src/publication/epub/Properties.ts +28 -0
  37. package/src/publication/epub/Publication.ts +51 -0
  38. package/src/publication/epub/index.ts +4 -0
  39. package/src/publication/html/DomRange.ts +63 -0
  40. package/src/publication/html/DomRangePoint.ts +64 -0
  41. package/src/publication/html/Locations.ts +129 -0
  42. package/src/publication/html/index.ts +3 -0
  43. package/src/publication/index.ts +19 -0
  44. package/src/publication/opds/Properties.ts +84 -0
  45. package/src/publication/opds/Publication.ts +14 -0
  46. package/src/publication/opds/index.ts +2 -0
  47. package/src/publication/presentation/Metadata.ts +19 -0
  48. package/src/publication/presentation/Presentation.ts +155 -0
  49. package/src/publication/presentation/Properties.ts +70 -0
  50. package/src/publication/presentation/index.ts +3 -0
  51. package/src/publication/services/content/Content.ts +36 -0
  52. package/src/publication/services/content/ContentTokenizer.ts +62 -0
  53. package/src/publication/services/content/Iterator.ts +46 -0
  54. package/src/publication/services/content/element/attributes.ts +44 -0
  55. package/src/publication/services/content/element/element.ts +118 -0
  56. package/src/publication/services/content/element/index.ts +3 -0
  57. package/src/publication/services/content/element/text_role.ts +60 -0
  58. package/src/publication/services/content/index.ts +5 -0
  59. package/src/publication/services/content/iterators/HTMLResourceContentIterator.ts +571 -0
  60. package/src/publication/services/content/iterators/PDFTextContentIterator.ts +184 -0
  61. package/src/publication/services/content/iterators/PublicationContentIterator.ts +144 -0
  62. package/src/publication/services/content/iterators/helpers.ts +146 -0
  63. package/src/publication/services/content/iterators/index.ts +3 -0
  64. package/src/publication/services/index.ts +1 -0
  65. package/src/util/JSONParse.ts +31 -0
  66. package/src/util/Language.ts +5 -0
  67. package/src/util/URITemplate.ts +83 -0
  68. package/src/util/index.ts +5 -0
  69. package/src/util/mediatype/MediaType.ts +563 -0
  70. package/src/util/mediatype/index.ts +1 -0
  71. package/src/util/tokenizer/TextTokenizer.ts +126 -0
  72. package/src/util/tokenizer/Tokenizer.ts +8 -0
  73. package/src/util/tokenizer/index.ts +2 -0
  74. package/src/util/tokenizer/tokenize-english/README.md +7 -0
  75. package/src/util/tokenizer/tokenize-english/abbreviations.js +52 -0
  76. package/src/util/tokenizer/tokenize-english/index.d.ts +3 -0
  77. package/src/util/tokenizer/tokenize-english/index.js +243 -0
  78. package/src/util/tokenizer/tokenize-english/utils.js +16 -0
  79. package/src/util/tokenizer/tokenize-text/README.MD +6 -0
  80. package/src/util/tokenizer/tokenize-text/index.d.ts +52 -0
  81. package/src/util/tokenizer/tokenize-text/index.js +256 -0
  82. package/src/util/tokenizer/tokenize-text/tokens.js +100 -0
  83. package/types/src/fetcher/Fetcher.d.ts +27 -0
  84. package/types/src/fetcher/HttpFetcher.d.ts +28 -0
  85. package/types/src/fetcher/Resource.d.ts +15 -0
  86. package/types/src/fetcher/index.d.ts +2 -0
  87. package/types/src/index.d.ts +4 -0
  88. package/types/src/opds/Acquisition.d.ts +26 -0
  89. package/types/src/opds/Availability.d.ts +32 -0
  90. package/types/src/opds/Copies.d.ts +22 -0
  91. package/types/src/opds/Holds.d.ts +22 -0
  92. package/types/src/opds/Price.d.ts +29 -0
  93. package/types/src/opds/index.d.ts +5 -0
  94. package/types/src/publication/BelongsTo.d.ts +23 -0
  95. package/types/src/publication/Contributor.d.ts +60 -0
  96. package/types/src/publication/GuidedNavigation.d.ts +70 -0
  97. package/types/src/publication/Link.d.ts +128 -0
  98. package/types/src/publication/LocalizedString.d.ts +48 -0
  99. package/types/src/publication/Locator.d.ts +101 -0
  100. package/types/src/publication/LocatorCollection.d.ts +54 -0
  101. package/types/src/publication/Manifest.d.ts +59 -0
  102. package/types/src/publication/Metadata.d.ts +101 -0
  103. package/types/src/publication/Properties.d.ts +28 -0
  104. package/types/src/publication/Publication.d.ts +53 -0
  105. package/types/src/publication/PublicationCollection.d.ts +34 -0
  106. package/types/src/publication/ReadingProgression.d.ts +9 -0
  107. package/types/src/publication/Subject.d.ts +56 -0
  108. package/types/src/publication/encryption/Encryption.d.ts +33 -0
  109. package/types/src/publication/encryption/Properties.d.ts +10 -0
  110. package/types/src/publication/encryption/index.d.ts +2 -0
  111. package/types/src/publication/epub/EPUBLayout.d.ts +5 -0
  112. package/types/src/publication/epub/Presentation.d.ts +7 -0
  113. package/types/src/publication/epub/Properties.d.ts +14 -0
  114. package/types/src/publication/epub/Publication.d.ts +19 -0
  115. package/types/src/publication/epub/index.d.ts +4 -0
  116. package/types/src/publication/html/DomRange.d.ts +40 -0
  117. package/types/src/publication/html/DomRangePoint.d.ts +36 -0
  118. package/types/src/publication/html/Locations.d.ts +45 -0
  119. package/types/src/publication/html/index.d.ts +3 -0
  120. package/types/src/publication/index.d.ts +19 -0
  121. package/types/src/publication/opds/Properties.d.ts +38 -0
  122. package/types/src/publication/opds/Publication.d.ts +6 -0
  123. package/types/src/publication/opds/index.d.ts +2 -0
  124. package/types/src/publication/presentation/Metadata.d.ts +6 -0
  125. package/types/src/publication/presentation/Presentation.d.ts +95 -0
  126. package/types/src/publication/presentation/Properties.d.ts +32 -0
  127. package/types/src/publication/presentation/index.d.ts +3 -0
  128. package/types/src/publication/services/content/Content.d.ts +20 -0
  129. package/types/src/publication/services/content/ContentTokenizer.d.ts +19 -0
  130. package/types/src/publication/services/content/Iterator.d.ts +38 -0
  131. package/types/src/publication/services/content/element/attributes.d.ts +31 -0
  132. package/types/src/publication/services/content/element/element.d.ts +107 -0
  133. package/types/src/publication/services/content/element/index.d.ts +3 -0
  134. package/types/src/publication/services/content/element/text_role.d.ts +53 -0
  135. package/types/src/publication/services/content/index.d.ts +5 -0
  136. package/types/src/publication/services/content/iterators/HTMLResourceContentIterator.d.ts +30 -0
  137. package/types/src/publication/services/content/iterators/PDFTextContentIterator.d.ts +60 -0
  138. package/types/src/publication/services/content/iterators/PublicationContentIterator.d.ts +59 -0
  139. package/types/src/publication/services/content/iterators/helpers.d.ts +10 -0
  140. package/types/src/publication/services/content/iterators/index.d.ts +3 -0
  141. package/types/src/publication/services/index.d.ts +1 -0
  142. package/types/src/util/JSONParse.d.ts +11 -0
  143. package/types/src/util/Language.d.ts +1 -0
  144. package/types/src/util/URITemplate.d.ts +24 -0
  145. package/types/src/util/index.d.ts +5 -0
  146. package/types/src/util/mediatype/MediaType.d.ts +148 -0
  147. package/types/src/util/mediatype/index.d.ts +1 -0
  148. package/types/src/util/tokenizer/TextTokenizer.d.ts +39 -0
  149. package/types/src/util/tokenizer/Tokenizer.d.ts +6 -0
  150. package/types/src/util/tokenizer/index.d.ts +2 -0
@@ -0,0 +1,563 @@
1
+ /* Copyright 2021 Readium Foundation. All rights reserved.
2
+ * Use of this source code is governed by a BSD-style license,
3
+ * available in the LICENSE file present in the Github repository of the project.
4
+ */
5
+
6
+ type ParametersMap = {
7
+ [param: string]: string;
8
+ };
9
+
10
+ /** Represents a string media type.
11
+ * `MediaType` handles:
12
+ * - components parsing – eg. type, subtype and parameters,
13
+ * - media types comparison.
14
+ */
15
+ export class MediaType {
16
+ /** The type component, e.g. `application` in `application/epub+zip`. */
17
+ public type: string;
18
+
19
+ /** The subtype component, e.g. `epub+zip` in `application/epub+zip`. */
20
+ public subtype: string;
21
+
22
+ /** The parameters in the media type, such as `charset=utf-8`. */
23
+ public parameters: ParametersMap;
24
+
25
+ /** The string representation of this media type. */
26
+ public string: string;
27
+
28
+ /** Encoding as declared in the `charset` parameter, if there's any. */
29
+ public encoding?: string;
30
+
31
+ /** A human readable name identifying the media type, which may be presented to the user. */
32
+ public name?: string;
33
+
34
+ /** The default file extension to use for this media type. */
35
+ public fileExtension?: string;
36
+
37
+ /** Creates a MediaType object. */
38
+ constructor(values: {
39
+ mediaType: string;
40
+ name?: string;
41
+ fileExtension?: string;
42
+ }) {
43
+ let type: string;
44
+ let subtype: string;
45
+ let components = values.mediaType.replace(/\s/g, '').split(';');
46
+ const types = components[0].split('/');
47
+ if (types.length === 2) {
48
+ type = types[0].toLowerCase().trim();
49
+ subtype = types[1].toLowerCase().trim();
50
+
51
+ if (type.length === 0 || subtype.length === 0) {
52
+ throw new Error('Invalid media type');
53
+ }
54
+ } else {
55
+ throw new Error('Invalid media type');
56
+ }
57
+
58
+ const _parameters: ParametersMap = {};
59
+ for (let i = 1; i < components.length; i++) {
60
+ const component = components[i].split('=');
61
+ if (component.length === 2) {
62
+ const key = component[0].toLocaleLowerCase();
63
+ const value =
64
+ key === 'charset' ? component[1].toUpperCase() : component[1];
65
+ _parameters[key] = value;
66
+ }
67
+ }
68
+
69
+ const parameters: ParametersMap = {};
70
+ const keys = Object.keys(_parameters);
71
+ keys.sort((a, b) => a.localeCompare(b));
72
+
73
+ keys.forEach(x => (parameters[x] = _parameters[x]));
74
+
75
+ let parametersString: string = '';
76
+ for (const p in parameters) {
77
+ const value = parameters[p];
78
+ parametersString += `;${p}=${value}`;
79
+ }
80
+ const string = `${type}/${subtype}${parametersString}`;
81
+
82
+ const encoding = parameters['encoding'];
83
+
84
+ this.string = string;
85
+ this.type = type;
86
+ this.subtype = subtype;
87
+ this.parameters = parameters;
88
+ this.encoding = encoding;
89
+ this.name = values.name;
90
+ this.fileExtension = values.fileExtension;
91
+ }
92
+
93
+ public static parse(values: {
94
+ mediaType: string;
95
+ name?: string;
96
+ fileExtension?: string;
97
+ }): MediaType {
98
+ return new MediaType(values);
99
+ }
100
+
101
+ /** Structured syntax suffix, e.g. `+zip` in `application/epub+zip`.
102
+ * Gives a hint on the underlying structure of this media type.
103
+ * See. https://tools.ietf.org/html/rfc6838#section-4.2.8
104
+ */
105
+ public get structuredSyntaxSuffix(): string | undefined {
106
+ const parts = this.subtype.split('+');
107
+ return parts.length > 1 ? `+${parts[parts.length - 1]}` : undefined;
108
+ }
109
+
110
+ /** Parameter values might or might not be case-sensitive, depending on the semantics of
111
+ * the parameter name.
112
+ * https://tools.ietf.org/html/rfc2616#section-3.7
113
+ *
114
+ * The character set names may be up to 40 characters taken from the printable characters
115
+ * of US-ASCII. However, no distinction is made between use of upper and lower case
116
+ * letters.
117
+ * https://www.iana.org/assignments/character-sets/character-sets.xhtml
118
+ */
119
+ public get charset(): string | undefined {
120
+ return this.parameters['charset'];
121
+ }
122
+
123
+ /** Returns whether the given `other` media type is included in this media type.
124
+ * For example, `text/html` contains `text/html;charset=utf-8`.
125
+ * - `other` must match the parameters in the `parameters` property, but extra parameters
126
+ * are ignored.
127
+ * - Order of parameters is ignored.
128
+ * - Wildcards are supported, meaning that `image/*` contains `image/png`
129
+ */
130
+ public contains(other: MediaType | string): boolean {
131
+ const _other =
132
+ typeof other === 'string' ? MediaType.parse({ mediaType: other }) : other;
133
+
134
+ if (
135
+ !(
136
+ (this.type === '*' || this.type === _other.type) &&
137
+ (this.subtype === '*' || this.subtype === _other.subtype)
138
+ )
139
+ ) {
140
+ return false;
141
+ }
142
+
143
+ const paramSet = new Set(
144
+ Object.entries(this.parameters).map(([key, value]) => `${key}=${value}`)
145
+ );
146
+ const otherParamSet = new Set(
147
+ Object.entries(_other.parameters).map(([key, value]) => `${key}=${value}`)
148
+ );
149
+
150
+ // check weather otherParamSet contains all parameters
151
+ for (const key of Array.from(paramSet.values())) {
152
+ if (!otherParamSet.has(key)) {
153
+ return false;
154
+ }
155
+ }
156
+
157
+ return true;
158
+ }
159
+
160
+ /** Returns whether this media type and `other` are the same, ignoring parameters that
161
+ * are not in both media types.
162
+ * For example, `text/html` matches `text/html;charset=utf-8`, but `text/html;charset=ascii`
163
+ * doesn't. This is basically like `contains`, but working in both direction.
164
+ */
165
+ public matches(other: MediaType | string): boolean {
166
+ const _other =
167
+ typeof other === 'string' ? MediaType.parse({ mediaType: other }) : other;
168
+ return this.contains(_other) || _other.contains(this);
169
+ }
170
+
171
+ /**
172
+ * Returns whether this media type matches any of the [others] media types.
173
+ */
174
+ public matchesAny(...others: MediaType[] | string[]): boolean {
175
+ for (const other of others) {
176
+ if (this.matches(other)) {
177
+ return true;
178
+ }
179
+ }
180
+ return false;
181
+ }
182
+
183
+ /** Checks the MediaType equals another one (comparing their string) */
184
+ public equals(other: MediaType): boolean {
185
+ return this.string === other.string;
186
+ }
187
+
188
+ /** Returns whether this media type is structured as a ZIP archive. */
189
+ public get isZIP(): boolean {
190
+ return (
191
+ this.matchesAny(
192
+ MediaType.ZIP,
193
+ MediaType.LCP_PROTECTED_AUDIOBOOK,
194
+ MediaType.LCP_PROTECTED_PDF
195
+ ) || this.structuredSyntaxSuffix === '+zip'
196
+ );
197
+ }
198
+
199
+ /** Returns whether this media type is structured as a JSON file. */
200
+ public get isJSON(): boolean {
201
+ return (
202
+ this.matchesAny(MediaType.JSON) || this.structuredSyntaxSuffix === '+json'
203
+ );
204
+ }
205
+
206
+ /** Returns whether this media type is of an OPDS feed. */
207
+ public get isOPDS(): boolean {
208
+ return (
209
+ this.matchesAny(
210
+ MediaType.OPDS1,
211
+ MediaType.OPDS1_ENTRY,
212
+ MediaType.OPDS2,
213
+ MediaType.OPDS2_PUBLICATION,
214
+ MediaType.OPDS_AUTHENTICATION
215
+ ) || this.structuredSyntaxSuffix === '+json'
216
+ );
217
+ }
218
+
219
+ /** Returns whether this media type is of an HTML document. */
220
+ public get isHTML(): boolean {
221
+ return this.matchesAny(MediaType.HTML, MediaType.XHTML);
222
+ }
223
+
224
+ /** Returns whether this media type is of a bitmap image, so excluding vectorial formats. */
225
+ public get isBitmap(): boolean {
226
+ return this.matchesAny(
227
+ MediaType.BMP,
228
+ MediaType.GIF,
229
+ MediaType.JPEG,
230
+ MediaType.PNG,
231
+ MediaType.TIFF,
232
+ MediaType.WEBP
233
+ );
234
+ }
235
+
236
+ /** Returns whether this media type is of an audio clip. */
237
+ public get isAudio(): boolean {
238
+ return this.type === 'audio';
239
+ }
240
+
241
+ /** Returns whether this media type is of a video clip. */
242
+ public get isVideo(): boolean {
243
+ return this.type === 'video';
244
+ }
245
+
246
+ /** Returns whether this media type is of a Readium Web Publication Manifest. */
247
+ public get isRWPM(): boolean {
248
+ return this.matchesAny(
249
+ MediaType.READIUM_AUDIOBOOK_MANIFEST,
250
+ MediaType.DIVINA_MANIFEST,
251
+ MediaType.READIUM_WEBPUB_MANIFEST
252
+ );
253
+ }
254
+
255
+ /** Returns whether this media type is of a publication file. */
256
+ public get isPublication(): boolean {
257
+ return this.matchesAny(
258
+ MediaType.READIUM_AUDIOBOOK,
259
+ MediaType.READIUM_AUDIOBOOK_MANIFEST,
260
+ MediaType.CBZ,
261
+ MediaType.DIVINA,
262
+ MediaType.DIVINA_MANIFEST,
263
+ MediaType.EPUB,
264
+ MediaType.LCP_PROTECTED_AUDIOBOOK,
265
+ MediaType.LCP_PROTECTED_PDF,
266
+ MediaType.LPF,
267
+ MediaType.PDF,
268
+ MediaType.W3C_WPUB_MANIFEST,
269
+ MediaType.READIUM_WEBPUB,
270
+ MediaType.READIUM_WEBPUB_MANIFEST,
271
+ MediaType.ZAB
272
+ );
273
+ }
274
+
275
+ // Known Media Types
276
+ public static get AAC(): MediaType {
277
+ return MediaType.parse({ mediaType: 'audio/aac', fileExtension: 'aac' });
278
+ }
279
+ public static get ACSM(): MediaType {
280
+ return MediaType.parse({
281
+ mediaType: 'application/vnd.adobe.adept+xml',
282
+ name: 'Adobe Content Server Message',
283
+ fileExtension: 'acsm',
284
+ });
285
+ }
286
+ public static get AIFF(): MediaType {
287
+ return MediaType.parse({ mediaType: 'audio/aiff', fileExtension: 'aiff' });
288
+ }
289
+ public static get AVI(): MediaType {
290
+ return MediaType.parse({
291
+ mediaType: 'video/x-msvideo',
292
+ fileExtension: 'avi',
293
+ });
294
+ }
295
+ public static get BINARY(): MediaType {
296
+ return MediaType.parse({ mediaType: 'application/octet-stream' });
297
+ }
298
+ public static get BMP(): MediaType {
299
+ return MediaType.parse({ mediaType: 'image/bmp', fileExtension: 'bmp' });
300
+ }
301
+ public static get CBZ(): MediaType {
302
+ return MediaType.parse({
303
+ mediaType: 'application/vnd.comicbook+zip',
304
+ name: 'Comic Book Archive',
305
+ fileExtension: 'cbz',
306
+ });
307
+ }
308
+ public static get CSS(): MediaType {
309
+ return MediaType.parse({ mediaType: 'text/css', fileExtension: 'css' });
310
+ }
311
+ public static get DIVINA(): MediaType {
312
+ return MediaType.parse({
313
+ mediaType: 'application/divina+zip',
314
+ name: 'Digital Visual Narratives',
315
+ fileExtension: 'divina',
316
+ });
317
+ }
318
+ public static get DIVINA_MANIFEST(): MediaType {
319
+ return MediaType.parse({
320
+ mediaType: 'application/divina+json',
321
+ name: 'Digital Visual Narratives',
322
+ fileExtension: 'json',
323
+ });
324
+ }
325
+ public static get EPUB(): MediaType {
326
+ return MediaType.parse({
327
+ mediaType: 'application/epub+zip',
328
+ name: 'EPUB',
329
+ fileExtension: 'epub',
330
+ });
331
+ }
332
+ public static get GIF(): MediaType {
333
+ return MediaType.parse({ mediaType: 'image/gif', fileExtension: 'gif' });
334
+ }
335
+ public static get GZ(): MediaType {
336
+ return MediaType.parse({
337
+ mediaType: 'application/gzip',
338
+ fileExtension: 'gz',
339
+ });
340
+ }
341
+ public static get HTML(): MediaType {
342
+ return MediaType.parse({ mediaType: 'text/html', fileExtension: 'html' });
343
+ }
344
+ public static get JAVASCRIPT(): MediaType {
345
+ return MediaType.parse({
346
+ mediaType: 'text/javascript',
347
+ fileExtension: 'js',
348
+ });
349
+ }
350
+ public static get JPEG(): MediaType {
351
+ return MediaType.parse({ mediaType: 'image/jpeg', fileExtension: 'jpeg' });
352
+ }
353
+ public static get JSON(): MediaType {
354
+ return MediaType.parse({ mediaType: 'application/json' });
355
+ }
356
+ public static get LCP_LICENSE_DOCUMENT(): MediaType {
357
+ return MediaType.parse({
358
+ mediaType: 'application/vnd.readium.lcp.license.v1.0+json',
359
+ name: 'LCP License',
360
+ fileExtension: 'lcpl',
361
+ });
362
+ }
363
+ public static get LCP_PROTECTED_AUDIOBOOK(): MediaType {
364
+ return MediaType.parse({
365
+ mediaType: 'application/audiobook+lcp',
366
+ name: 'LCP Protected Audiobook',
367
+ fileExtension: 'lcpa',
368
+ });
369
+ }
370
+ public static get LCP_PROTECTED_PDF(): MediaType {
371
+ return MediaType.parse({
372
+ mediaType: 'application/pdf+lcp',
373
+ name: 'LCP Protected PDF',
374
+ fileExtension: 'lcpdf',
375
+ });
376
+ }
377
+ public static get LCP_STATUS_DOCUMENT(): MediaType {
378
+ return MediaType.parse({
379
+ mediaType: 'application/vnd.readium.license.status.v1.0+json',
380
+ });
381
+ }
382
+ public static get LPF(): MediaType {
383
+ return MediaType.parse({
384
+ mediaType: 'application/lpf+zip',
385
+ fileExtension: 'lpf',
386
+ });
387
+ }
388
+ public static get MP3(): MediaType {
389
+ return MediaType.parse({ mediaType: 'audio/mpeg', fileExtension: 'mp3' });
390
+ }
391
+ public static get MPEG(): MediaType {
392
+ return MediaType.parse({ mediaType: 'video/mpeg', fileExtension: 'mpeg' });
393
+ }
394
+ public static get NCX(): MediaType {
395
+ return MediaType.parse({
396
+ mediaType: 'application/x-dtbncx+xml',
397
+ fileExtension: 'ncx',
398
+ });
399
+ }
400
+ public static get OGG(): MediaType {
401
+ return MediaType.parse({ mediaType: 'audio/ogg', fileExtension: 'oga' });
402
+ }
403
+ public static get OGV(): MediaType {
404
+ return MediaType.parse({ mediaType: 'video/ogg', fileExtension: 'ogv' });
405
+ }
406
+ public static get OPDS1(): MediaType {
407
+ return MediaType.parse({
408
+ mediaType: 'application/atom+xml;profile=opds-catalog',
409
+ });
410
+ }
411
+ public static get OPDS1_ENTRY(): MediaType {
412
+ return MediaType.parse({
413
+ mediaType: 'application/atom+xml;type=entry;profile=opds-catalog',
414
+ });
415
+ }
416
+ public static get OPDS2(): MediaType {
417
+ return MediaType.parse({ mediaType: 'application/opds+json' });
418
+ }
419
+ public static get OPDS2_PUBLICATION(): MediaType {
420
+ return MediaType.parse({ mediaType: 'application/opds-publication+json' });
421
+ }
422
+ public static get OPDS_AUTHENTICATION(): MediaType {
423
+ return MediaType.parse({
424
+ mediaType: 'application/opds-authentication+json',
425
+ });
426
+ }
427
+ public static get OPUS(): MediaType {
428
+ return MediaType.parse({ mediaType: 'audio/opus', fileExtension: 'opus' });
429
+ }
430
+ public static get OTF(): MediaType {
431
+ return MediaType.parse({ mediaType: 'font/otf', fileExtension: 'otf' });
432
+ }
433
+ public static get PDF(): MediaType {
434
+ return MediaType.parse({
435
+ mediaType: 'application/pdf',
436
+ name: 'PDF',
437
+ fileExtension: 'pdf',
438
+ });
439
+ }
440
+ public static get PNG(): MediaType {
441
+ return MediaType.parse({ mediaType: 'image/png', fileExtension: 'png' });
442
+ }
443
+ public static get READIUM_AUDIOBOOK(): MediaType {
444
+ return MediaType.parse({
445
+ mediaType: 'application/audiobook+zip',
446
+ name: 'Readium Audiobook',
447
+ fileExtension: 'audiobook',
448
+ });
449
+ }
450
+ public static get READIUM_AUDIOBOOK_MANIFEST(): MediaType {
451
+ return MediaType.parse({
452
+ mediaType: 'application/audiobook+json',
453
+ name: 'Readium Audiobook',
454
+ fileExtension: 'json',
455
+ });
456
+ }
457
+ public static get READIUM_CONTENT_DOCUMENT(): MediaType {
458
+ return MediaType.parse({
459
+ mediaType: 'application/vnd.readium.content+json',
460
+ name: 'Readium Content Document',
461
+ fileExtension: 'json',
462
+ });
463
+ }
464
+ public static get READIUM_GUIDED_NAVIGATION_DOCUMENT(): MediaType {
465
+ return MediaType.parse({
466
+ mediaType: 'application/guided-navigation+json',
467
+ name: 'Readium Guided Navigation Document',
468
+ fileExtension: 'json',
469
+ });
470
+ }
471
+ public static get READIUM_POSITION_LIST(): MediaType {
472
+ return MediaType.parse({
473
+ mediaType: 'application/vnd.readium.position-list+json',
474
+ name: 'Readium Position List',
475
+ fileExtension: 'json',
476
+ });
477
+ }
478
+ public static get READIUM_WEBPUB(): MediaType {
479
+ return MediaType.parse({
480
+ mediaType: 'application/webpub+zip',
481
+ name: 'Readium Web Publication',
482
+ fileExtension: 'webpub',
483
+ });
484
+ }
485
+ public static get READIUM_WEBPUB_MANIFEST(): MediaType {
486
+ return MediaType.parse({
487
+ mediaType: 'application/webpub+json',
488
+ name: 'Readium Web Publication',
489
+ fileExtension: 'json',
490
+ });
491
+ }
492
+ public static get SMIL(): MediaType {
493
+ return MediaType.parse({
494
+ mediaType: 'application/smil+xml',
495
+ fileExtension: 'smil',
496
+ });
497
+ }
498
+ public static get SVG(): MediaType {
499
+ return MediaType.parse({
500
+ mediaType: 'image/svg+xml',
501
+ fileExtension: 'svg',
502
+ });
503
+ }
504
+ public static get TEXT(): MediaType {
505
+ return MediaType.parse({ mediaType: 'text/plain', fileExtension: 'txt' });
506
+ }
507
+ public static get TIFF(): MediaType {
508
+ return MediaType.parse({ mediaType: 'image/tiff', fileExtension: 'tiff' });
509
+ }
510
+ public static get TTF(): MediaType {
511
+ return MediaType.parse({ mediaType: 'font/ttf', fileExtension: 'ttf' });
512
+ }
513
+ public static get W3C_WPUB_MANIFEST(): MediaType {
514
+ return MediaType.parse({
515
+ mediaType: 'application/x.readium.w3c.wpub+json',
516
+ name: 'Web Publication',
517
+ fileExtension: 'json',
518
+ });
519
+ }
520
+ public static get WAV(): MediaType {
521
+ return MediaType.parse({ mediaType: 'audio/wav', fileExtension: 'wav' });
522
+ }
523
+ public static get WEBM_AUDIO(): MediaType {
524
+ return MediaType.parse({ mediaType: 'audio/webm', fileExtension: 'webm' });
525
+ }
526
+ public static get WEBM_VIDEO(): MediaType {
527
+ return MediaType.parse({ mediaType: 'video/webm', fileExtension: 'webm' });
528
+ }
529
+ public static get WEBP(): MediaType {
530
+ return MediaType.parse({ mediaType: 'image/webp', fileExtension: 'webp' });
531
+ }
532
+ public static get WOFF(): MediaType {
533
+ return MediaType.parse({ mediaType: 'font/woff', fileExtension: 'woff' });
534
+ }
535
+ public static get WOFF2(): MediaType {
536
+ return MediaType.parse({ mediaType: 'font/woff2', fileExtension: 'woff2' });
537
+ }
538
+ public static get XHTML(): MediaType {
539
+ return MediaType.parse({
540
+ mediaType: 'application/xhtml+xml',
541
+ fileExtension: 'xhtml',
542
+ });
543
+ }
544
+ public static get XML(): MediaType {
545
+ return MediaType.parse({
546
+ mediaType: 'application/xml',
547
+ fileExtension: 'xml',
548
+ });
549
+ }
550
+ public static get ZAB(): MediaType {
551
+ return MediaType.parse({
552
+ mediaType: 'application/x.readium.zab+zip',
553
+ name: 'Zipped Audio Book',
554
+ fileExtension: 'zab',
555
+ });
556
+ }
557
+ public static get ZIP(): MediaType {
558
+ return MediaType.parse({
559
+ mediaType: 'application/zip',
560
+ fileExtension: 'zip',
561
+ });
562
+ }
563
+ }
@@ -0,0 +1 @@
1
+ export * from './MediaType';
@@ -0,0 +1,126 @@
1
+ import { Language } from "../Language";
2
+ import { Tokenizer } from "./Tokenizer";
3
+ import BasicEnglishTokenizer from "./tokenize-english";
4
+ import BasicTokenizer, { TextlintSegment } from "./tokenize-text";
5
+
6
+ // Start / End
7
+ export type Range = [number, number];
8
+
9
+ /**
10
+ * A tokenizer splitting a String into range tokens (e.g. words, sentences, etc.).
11
+ */
12
+ export type TextTokenizer = Tokenizer<string, Range>;
13
+
14
+ /**
15
+ * A text token unit which can be used with a [TextTokenizer].
16
+ */
17
+ export enum TextUnit {
18
+ Word = "word",
19
+ Sentence = "sentence",
20
+ Paragraph = "paragraph",
21
+ }
22
+
23
+ // A default cluster [TextTokenizer] taking advantage of the best capabilities of the navigator
24
+ export const DefaultTextContentTokenizer = (language: Language | null, unit: TextUnit): TextTokenizer => {
25
+ if("Segmenter" in Intl) {
26
+ // Available in any evergreen browser EXCEPT for Firefox.
27
+ // See: https://caniuse.com/mdn-javascript_builtins_intl_segmenter
28
+ return new IntlTextTokenizer(language, unit);
29
+ } else {
30
+ // Fallback that works mainly for English
31
+ return new NaiveTextTokenizer(language, unit);
32
+ }
33
+ };
34
+
35
+ /**
36
+ * A [TextTokenizer] using the Intl.Segmenter API.
37
+ * Very aware of language-specific rules since it uses ICU behind the scenes.
38
+ */
39
+ export class IntlTextTokenizer implements TextTokenizer {
40
+ private segmenter: Intl.Segmenter;
41
+
42
+ constructor(
43
+ language: Language | null,
44
+ private unit: TextUnit
45
+ ) {
46
+ language = language ?? navigator?.language; // Fallback to browser language
47
+ if("Segmenter" in Intl === false) throw new Error("Intl.Segmenter is not supported in this environment");
48
+ if(unit === TextUnit.Paragraph) throw new Error("IntlTextTokenizer does not handle TextUnit.Paragraph");
49
+ this.segmenter = new Intl.Segmenter(language, {
50
+ granularity: unit
51
+ });
52
+ }
53
+
54
+ tokenize(data: string): Range[] {
55
+ const segments = this.segmenter.segment(data);
56
+ const ranges: Range[] = [];
57
+ for (let segment of segments) {
58
+ if(this.unit === TextUnit.Word && segment.isWordLike === false) continue;
59
+ const s = speakableToken(segment.segment);
60
+ if(s === null) continue;
61
+ ranges.push([segment.index, segment.index + s.length]);
62
+ }
63
+ return ranges;
64
+ }
65
+ }
66
+
67
+ /**
68
+ * A [TextTokenizer] using a naive approach to splitting text into tokens.
69
+ * This is a fallback for browsers that don't support Intl.Segmenter.
70
+ * It works mainly on English and similar languages. Don't use unless necessary.
71
+ */
72
+ export class NaiveTextTokenizer {
73
+ private tokenizer: BasicTokenizer;
74
+ private isEnglish: boolean;
75
+
76
+ constructor(
77
+ language: Language | null,
78
+ private unit: TextUnit
79
+ ) {
80
+ language = language ?? navigator?.language;
81
+ this.isEnglish = language.toLowerCase().split("-")[0] === "en";
82
+ if(unit === TextUnit.Paragraph) throw new Error("NaiveTextTokenizer does not handle TextUnit.Paragraph");
83
+ this.tokenizer = new BasicTokenizer();
84
+ }
85
+
86
+ tokenize(data: string): Range[] {
87
+ let segments: TextlintSegment[] = [];
88
+ switch (this.unit) {
89
+ case TextUnit.Word:
90
+ segments = this.tokenizer.words()(data);
91
+ break;
92
+ case TextUnit.Sentence:
93
+ if(this.isEnglish) {
94
+ const englishTokenizer = BasicEnglishTokenizer(this.tokenizer);
95
+ segments = englishTokenizer.sentences()(data);
96
+ } else {
97
+ segments = this.tokenizer.sections()(data);
98
+ }
99
+ break;
100
+ default:
101
+ segments = [];
102
+ }
103
+ if(!segments) return [];
104
+
105
+ const ranges: Range[] = [];
106
+ segments.forEach(segment => {
107
+ if(segment.value.length === 0) return;
108
+ const s = speakableToken(segment.value);
109
+ if(s === null) return;
110
+ ranges.push([segment.index, segment.index + s.length]);
111
+ });
112
+
113
+ return ranges;
114
+ }
115
+ }
116
+
117
+ const trimmedMatcher = new RegExp("[\\p{L}\\p{N}]+", "u");
118
+
119
+ // Unicode-aware of checking if there's anything that can be spoken in a string
120
+ // "Spoken" in this case means at least one unicode letter or unicode number character
121
+ export const speakableToken = (token: string): string | null => {
122
+ const trimmedToken = token.trimEnd();
123
+ if(trimmedToken.length === 0) return null;
124
+ if(trimmedToken.match(trimmedMatcher) === null) return null;
125
+ return trimmedToken;
126
+ }
@@ -0,0 +1,8 @@
1
+
2
+
3
+ /**
4
+ * A tokenizer splits a piece of data [D] into a list of [T] tokens.
5
+ */
6
+ export interface Tokenizer<D, T> {
7
+ tokenize(data: D): T[];
8
+ }
@@ -0,0 +1,2 @@
1
+ export * from './Tokenizer';
2
+ export * from './TextTokenizer';