@readium/shared 1.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (150) hide show
  1. package/LICENSE +28 -0
  2. package/README.MD +27 -0
  3. package/dist/index.js +2536 -0
  4. package/dist/index.umd.cjs +2 -0
  5. package/package.json +73 -0
  6. package/src/fetcher/Fetcher.ts +37 -0
  7. package/src/fetcher/HttpFetcher.ts +122 -0
  8. package/src/fetcher/Resource.ts +37 -0
  9. package/src/fetcher/index.ts +2 -0
  10. package/src/index.ts +4 -0
  11. package/src/opds/Acquisition.ts +54 -0
  12. package/src/opds/Availability.ts +61 -0
  13. package/src/opds/Copies.ts +43 -0
  14. package/src/opds/Holds.ts +43 -0
  15. package/src/opds/Price.ts +51 -0
  16. package/src/opds/index.ts +5 -0
  17. package/src/publication/BelongsTo.ts +51 -0
  18. package/src/publication/Contributor.ts +129 -0
  19. package/src/publication/GuidedNavigation.ts +176 -0
  20. package/src/publication/Link.ts +332 -0
  21. package/src/publication/LocalizedString.ts +74 -0
  22. package/src/publication/Locator.ts +225 -0
  23. package/src/publication/LocatorCollection.ts +133 -0
  24. package/src/publication/Manifest.ts +256 -0
  25. package/src/publication/Metadata.ts +329 -0
  26. package/src/publication/Properties.ts +44 -0
  27. package/src/publication/Publication.ts +148 -0
  28. package/src/publication/PublicationCollection.ts +144 -0
  29. package/src/publication/ReadingProgression.ts +28 -0
  30. package/src/publication/Subject.ts +115 -0
  31. package/src/publication/encryption/Encryption.ts +68 -0
  32. package/src/publication/encryption/Properties.ts +20 -0
  33. package/src/publication/encryption/index.ts +2 -0
  34. package/src/publication/epub/EPUBLayout.ts +10 -0
  35. package/src/publication/epub/Presentation.ts +16 -0
  36. package/src/publication/epub/Properties.ts +28 -0
  37. package/src/publication/epub/Publication.ts +51 -0
  38. package/src/publication/epub/index.ts +4 -0
  39. package/src/publication/html/DomRange.ts +63 -0
  40. package/src/publication/html/DomRangePoint.ts +64 -0
  41. package/src/publication/html/Locations.ts +129 -0
  42. package/src/publication/html/index.ts +3 -0
  43. package/src/publication/index.ts +19 -0
  44. package/src/publication/opds/Properties.ts +84 -0
  45. package/src/publication/opds/Publication.ts +14 -0
  46. package/src/publication/opds/index.ts +2 -0
  47. package/src/publication/presentation/Metadata.ts +19 -0
  48. package/src/publication/presentation/Presentation.ts +155 -0
  49. package/src/publication/presentation/Properties.ts +70 -0
  50. package/src/publication/presentation/index.ts +3 -0
  51. package/src/publication/services/content/Content.ts +36 -0
  52. package/src/publication/services/content/ContentTokenizer.ts +62 -0
  53. package/src/publication/services/content/Iterator.ts +46 -0
  54. package/src/publication/services/content/element/attributes.ts +44 -0
  55. package/src/publication/services/content/element/element.ts +118 -0
  56. package/src/publication/services/content/element/index.ts +3 -0
  57. package/src/publication/services/content/element/text_role.ts +60 -0
  58. package/src/publication/services/content/index.ts +5 -0
  59. package/src/publication/services/content/iterators/HTMLResourceContentIterator.ts +571 -0
  60. package/src/publication/services/content/iterators/PDFTextContentIterator.ts +184 -0
  61. package/src/publication/services/content/iterators/PublicationContentIterator.ts +144 -0
  62. package/src/publication/services/content/iterators/helpers.ts +146 -0
  63. package/src/publication/services/content/iterators/index.ts +3 -0
  64. package/src/publication/services/index.ts +1 -0
  65. package/src/util/JSONParse.ts +31 -0
  66. package/src/util/Language.ts +5 -0
  67. package/src/util/URITemplate.ts +83 -0
  68. package/src/util/index.ts +5 -0
  69. package/src/util/mediatype/MediaType.ts +563 -0
  70. package/src/util/mediatype/index.ts +1 -0
  71. package/src/util/tokenizer/TextTokenizer.ts +126 -0
  72. package/src/util/tokenizer/Tokenizer.ts +8 -0
  73. package/src/util/tokenizer/index.ts +2 -0
  74. package/src/util/tokenizer/tokenize-english/README.md +7 -0
  75. package/src/util/tokenizer/tokenize-english/abbreviations.js +52 -0
  76. package/src/util/tokenizer/tokenize-english/index.d.ts +3 -0
  77. package/src/util/tokenizer/tokenize-english/index.js +243 -0
  78. package/src/util/tokenizer/tokenize-english/utils.js +16 -0
  79. package/src/util/tokenizer/tokenize-text/README.MD +6 -0
  80. package/src/util/tokenizer/tokenize-text/index.d.ts +52 -0
  81. package/src/util/tokenizer/tokenize-text/index.js +256 -0
  82. package/src/util/tokenizer/tokenize-text/tokens.js +100 -0
  83. package/types/src/fetcher/Fetcher.d.ts +27 -0
  84. package/types/src/fetcher/HttpFetcher.d.ts +28 -0
  85. package/types/src/fetcher/Resource.d.ts +15 -0
  86. package/types/src/fetcher/index.d.ts +2 -0
  87. package/types/src/index.d.ts +4 -0
  88. package/types/src/opds/Acquisition.d.ts +26 -0
  89. package/types/src/opds/Availability.d.ts +32 -0
  90. package/types/src/opds/Copies.d.ts +22 -0
  91. package/types/src/opds/Holds.d.ts +22 -0
  92. package/types/src/opds/Price.d.ts +29 -0
  93. package/types/src/opds/index.d.ts +5 -0
  94. package/types/src/publication/BelongsTo.d.ts +23 -0
  95. package/types/src/publication/Contributor.d.ts +60 -0
  96. package/types/src/publication/GuidedNavigation.d.ts +70 -0
  97. package/types/src/publication/Link.d.ts +128 -0
  98. package/types/src/publication/LocalizedString.d.ts +48 -0
  99. package/types/src/publication/Locator.d.ts +101 -0
  100. package/types/src/publication/LocatorCollection.d.ts +54 -0
  101. package/types/src/publication/Manifest.d.ts +59 -0
  102. package/types/src/publication/Metadata.d.ts +101 -0
  103. package/types/src/publication/Properties.d.ts +28 -0
  104. package/types/src/publication/Publication.d.ts +53 -0
  105. package/types/src/publication/PublicationCollection.d.ts +34 -0
  106. package/types/src/publication/ReadingProgression.d.ts +9 -0
  107. package/types/src/publication/Subject.d.ts +56 -0
  108. package/types/src/publication/encryption/Encryption.d.ts +33 -0
  109. package/types/src/publication/encryption/Properties.d.ts +10 -0
  110. package/types/src/publication/encryption/index.d.ts +2 -0
  111. package/types/src/publication/epub/EPUBLayout.d.ts +5 -0
  112. package/types/src/publication/epub/Presentation.d.ts +7 -0
  113. package/types/src/publication/epub/Properties.d.ts +14 -0
  114. package/types/src/publication/epub/Publication.d.ts +19 -0
  115. package/types/src/publication/epub/index.d.ts +4 -0
  116. package/types/src/publication/html/DomRange.d.ts +40 -0
  117. package/types/src/publication/html/DomRangePoint.d.ts +36 -0
  118. package/types/src/publication/html/Locations.d.ts +45 -0
  119. package/types/src/publication/html/index.d.ts +3 -0
  120. package/types/src/publication/index.d.ts +19 -0
  121. package/types/src/publication/opds/Properties.d.ts +38 -0
  122. package/types/src/publication/opds/Publication.d.ts +6 -0
  123. package/types/src/publication/opds/index.d.ts +2 -0
  124. package/types/src/publication/presentation/Metadata.d.ts +6 -0
  125. package/types/src/publication/presentation/Presentation.d.ts +95 -0
  126. package/types/src/publication/presentation/Properties.d.ts +32 -0
  127. package/types/src/publication/presentation/index.d.ts +3 -0
  128. package/types/src/publication/services/content/Content.d.ts +20 -0
  129. package/types/src/publication/services/content/ContentTokenizer.d.ts +19 -0
  130. package/types/src/publication/services/content/Iterator.d.ts +38 -0
  131. package/types/src/publication/services/content/element/attributes.d.ts +31 -0
  132. package/types/src/publication/services/content/element/element.d.ts +107 -0
  133. package/types/src/publication/services/content/element/index.d.ts +3 -0
  134. package/types/src/publication/services/content/element/text_role.d.ts +53 -0
  135. package/types/src/publication/services/content/index.d.ts +5 -0
  136. package/types/src/publication/services/content/iterators/HTMLResourceContentIterator.d.ts +30 -0
  137. package/types/src/publication/services/content/iterators/PDFTextContentIterator.d.ts +60 -0
  138. package/types/src/publication/services/content/iterators/PublicationContentIterator.d.ts +59 -0
  139. package/types/src/publication/services/content/iterators/helpers.d.ts +10 -0
  140. package/types/src/publication/services/content/iterators/index.d.ts +3 -0
  141. package/types/src/publication/services/index.d.ts +1 -0
  142. package/types/src/util/JSONParse.d.ts +11 -0
  143. package/types/src/util/Language.d.ts +1 -0
  144. package/types/src/util/URITemplate.d.ts +24 -0
  145. package/types/src/util/index.d.ts +5 -0
  146. package/types/src/util/mediatype/MediaType.d.ts +148 -0
  147. package/types/src/util/mediatype/index.d.ts +1 -0
  148. package/types/src/util/tokenizer/TextTokenizer.d.ts +39 -0
  149. package/types/src/util/tokenizer/Tokenizer.d.ts +6 -0
  150. package/types/src/util/tokenizer/index.d.ts +2 -0
@@ -0,0 +1,7 @@
1
+ This is a copy of https://github.com/textlint-rule/rousseau/tree/master/packages/tokenize-english
2
+ which is also available as the npm package `@textlint-rule/tokenize-english`.
3
+
4
+ It has been included here because a slight modification needs to be made since
5
+ the code does not run in JavaScript's strict mode due to a poor global variable declaration.
6
+ It has also been modified to remove the lodash dependency so it doesn't have to be
7
+ included directly in our dependencies.
@@ -0,0 +1,52 @@
1
+ export default [
2
+ "ie",
3
+ "eg",
4
+ "ext", // + number?
5
+ "Fig",
6
+ "fig",
7
+ "Figs",
8
+ "figs",
9
+ "et al",
10
+ "Co",
11
+ "Corp",
12
+ "Ave",
13
+ "Inc",
14
+ "Ex",
15
+ "Viz",
16
+ "vs",
17
+ "Vs",
18
+ "repr",
19
+ "Rep",
20
+ "Dem",
21
+ "trans",
22
+ "Vol",
23
+ "pp",
24
+ "rev",
25
+ "est",
26
+ "Ref",
27
+ "Refs",
28
+ "Eq",
29
+ "Eqs",
30
+ "Ch",
31
+ "Sec",
32
+ "Secs",
33
+ "mi",
34
+ "Dept",
35
+
36
+ "Univ",
37
+ "Nos",
38
+ "No",
39
+ "Mol",
40
+ "Cell",
41
+
42
+ "Miss", "Mrs", "Mr", "Ms",
43
+ "Prof", "Dr",
44
+ "Sgt", "Col", "Gen", "Rep", "Sen",'Gov', "Lt", "Maj", "Capt","St",
45
+
46
+ "Sr", "Jr", "jr", "Rev",
47
+ "PhD", "MD", "BA", "MA", "MM",
48
+ "BSc", "MSc",
49
+
50
+ "Jan","Feb","Mar","Apr","Jun","Jul","Aug","Sep","Sept","Oct","Nov","Dec",
51
+ "Sun","Mon","Tu","Tue","Tues","Wed","Th","Thu","Thur","Thurs","Fri","Sat"
52
+ ];
@@ -0,0 +1,3 @@
1
+ export default function (tokenize: Tokenizer): {
2
+ sentences: () => (text: string) => Segment[];
3
+ };
@@ -0,0 +1,243 @@
1
+ /*
2
+ Sentence Boundary Detection (SBD)
3
+ Split text into sentences with a vanilla rule based approach (i.e working ~95% of the time).
4
+
5
+ Split a text based on period, question and exclamation marks.
6
+ Skips (most) abbreviations (Mr., Mrs., PhD.)
7
+ Skips numbers/currency
8
+ Skips urls, websites, email addresses, phone nr.
9
+ Counts ellipsis and ?! as single punctuation
10
+ */
11
+
12
+ import utils from "./utils";
13
+ import abbreviations from "./abbreviations";
14
+
15
+ function isCapitalized(str) {
16
+ return /^[A-Z][a-z].*/.test(str) || isNumber(str);
17
+ }
18
+
19
+ // Start with opening quotes or capitalized letter
20
+ function isSentenceStarter(str) {
21
+ return isCapitalized(str) || /``|"|'/.test(str.substring(0,2));
22
+ }
23
+
24
+ function isCommonAbbreviation(str) {
25
+ return ~abbreviations.indexOf(str.replace(/\W+/g, ''));
26
+ }
27
+
28
+ // This is going towards too much rule based
29
+ function isTimeAbbreviation(word, next) {
30
+ if (word === "a.m." || word === "p.m.") {
31
+ var tmp = next.replace(/\W+/g, '').slice(-3).toLowerCase();
32
+
33
+ if (tmp === "day") {
34
+ return true;
35
+ }
36
+ }
37
+
38
+ return false;
39
+ }
40
+
41
+ function isDottedAbbreviation(word) {
42
+ var matches = word.replace(/[\(\)\[\]\{\}]/g, '').match(/(.\.)*/);
43
+ return matches && matches[0].length > 0;
44
+ }
45
+
46
+ // TODO look for next words, if multiple capitalized -> not sentence ending
47
+ function isCustomAbbreviation(str) {
48
+ if (str.length <= 3)
49
+ return true;
50
+
51
+ return isCapitalized(str);
52
+ }
53
+
54
+ // Uses current word count in sentence and next few words to check if it is
55
+ // more likely an abbreviation + name or new sentence.
56
+ function isNameAbbreviation(wordCount, words) {
57
+ if (words.length > 0) {
58
+ if (wordCount < 5 && words[0].length < 6 && isCapitalized(words[0])) {
59
+ return true;
60
+ }
61
+
62
+ var capitalized = words.filter(function(str) {
63
+ return /[A-Z]/.test(str.charAt(0));
64
+ });
65
+
66
+ return capitalized.length >= 3;
67
+ }
68
+
69
+ return false;
70
+ }
71
+
72
+ function isNumber(str, dotPos) {
73
+ if (dotPos) {
74
+ str = str.slice(dotPos-1, dotPos+2);
75
+ }
76
+
77
+ return !isNaN(Number(str));
78
+ }
79
+
80
+ // Phone number matching
81
+ // http://stackoverflow.com/a/123666/951517
82
+ function isPhoneNr(str) {
83
+ return str.match(/^(?:(?:\+?1\s*(?:[.-]\s*)?)?(?:\(\s*([2-9]1[02-9]|[2-9][02-8]1|[2-9][02-8][02-9])\s*\)|([2-9]1[02-9]|[2-9][02-8]1|[2-9][02-8][02-9]))\s*(?:[.-]\s*)?)?([2-9]1[02-9]|[2-9][02-9]1|[2-9][02-9]{2})\s*(?:[.-]\s*)?([0-9]{4})(?:\s*(?:#|x\.?|ext\.?|extension)\s*(\d+))?$/);
84
+ }
85
+
86
+ // Match urls / emails
87
+ // http://stackoverflow.com/a/3809435/951517
88
+ function isURL(str) {
89
+ return str.match(/[-a-zA-Z0-9@:%._\+~#=]{2,256}\.[a-z]{2,6}\b([-a-zA-Z0-9@:%_\+.~#?&//=]*)/);
90
+ }
91
+
92
+ // Starting a new sentence if beginning with capital letter
93
+ // Exception: The word is enclosed in brackets
94
+ function isConcatenated(word) {
95
+ var i = 0;
96
+
97
+ if ((i = word.indexOf(".")) > -1 ||
98
+ (i = word.indexOf("!")) > -1 ||
99
+ (i = word.indexOf("?")) > -1)
100
+ {
101
+ var c = word.charAt(i + 1);
102
+
103
+ // Check if the next word starts with a letter
104
+ if (c.match(/[a-zA-Z].*/)) {
105
+ return [word.slice(0, i), word.charAt(i), word.slice(i+1)];
106
+ }
107
+ }
108
+
109
+ return false;
110
+ }
111
+
112
+ function isBoundaryChar(word) {
113
+ return word === "." ||
114
+ word === "!" ||
115
+ word === "?";
116
+ }
117
+
118
+ // http://tech.grammarly.com/blog/posts/How-to-Split-Sentences.html
119
+ function tokenizeSentences(tokenize, opts) {
120
+ opts = Object.assign(opts || {}, {
121
+ newlineBoundary: opts && opts.newlineBoundary || true
122
+ })
123
+
124
+ var splitRe = /\s/;
125
+
126
+ return tokenize.serie(
127
+ // Split into words
128
+ tokenize.re(splitRe, { split: true }),
129
+
130
+ // Merge words as sentences
131
+ tokenize.splitAndMerge(function(word, token, prev, next, i, tokens) {
132
+ var tmp;
133
+ var endOfSentence = [word, null];
134
+ var sentenceNotOver = word;
135
+
136
+ // Find the next word
137
+ var nextWord = tokens.slice(i+1).find(function(tok) {
138
+ return !splitRe.test(tok.value)
139
+ });
140
+
141
+ // Newline boundaries
142
+ if (word === '\n' && opts.newlineBoundary) return endOfSentence;
143
+
144
+ if (isBoundaryChar(word) ||
145
+ utils.endsWithChar(word, "?!"))
146
+ {
147
+ return endOfSentence;
148
+ }
149
+
150
+ // A dot might indicate the end sentences
151
+ // Exception: The next sentence starts with a word (non abbreviation)
152
+ // that has a capital letter.
153
+ if (utils.endsWithChar(word, '.')) {
154
+ // Check if there is a next word
155
+ if (nextWord) {
156
+ // This should be improved with machine learning
157
+
158
+ // Single character abbr.
159
+ if (word.length === 2 && isNaN(word.charAt(0)) && word.match(/[a-zA-Z]/)) {
160
+ return sentenceNotOver;
161
+ }
162
+
163
+ // Common abbr. that often do not end sentences
164
+ if (isCommonAbbreviation(word)) {
165
+ return sentenceNotOver;
166
+ }
167
+
168
+ // Next word starts with capital word, but current sentence is
169
+ // quite short
170
+ if (isSentenceStarter(nextWord.value)) {
171
+ if (isTimeAbbreviation(word, nextWord.value)) {
172
+ return sentenceNotOver;
173
+ }
174
+
175
+ // Dealing with names at the start of sentences
176
+ /*if (isNameAbbreviation(wordCount, words.slice(i, 6))) {
177
+ return word;
178
+ }*/
179
+
180
+ if (isNumber(nextWord.value) && isCustomAbbreviation(word)) {
181
+ return sentenceNotOver;
182
+ }
183
+ }
184
+ else {
185
+ // Skip ellipsis
186
+ if (utils.endsWith(word, "..")) {
187
+ return sentenceNotOver;
188
+ }
189
+
190
+ //// Skip abbreviations
191
+ // Short words + dot or a dot after each letter
192
+ if (isDottedAbbreviation(word) || isCustomAbbreviation(word)) {
193
+ return sentenceNotOver;
194
+ }
195
+ }
196
+ }
197
+
198
+ return endOfSentence;
199
+ }
200
+
201
+ // Check if the word has a dot in it
202
+ let index;
203
+ if ((index = word.indexOf(".")) > -1) {
204
+ if (isNumber(word, index)) {
205
+ return sentenceNotOver;
206
+ }
207
+
208
+ // Custom dotted abbreviations (like K.L.M or I.C.T)
209
+ if (isDottedAbbreviation(word)) {
210
+ return sentenceNotOver;
211
+ }
212
+
213
+ // Skip urls / emails and the like
214
+ if (isURL(word) || isPhoneNr(word)) {
215
+ return sentenceNotOver;
216
+ }
217
+ }
218
+
219
+ if (tmp = isConcatenated(word)) {
220
+ return [
221
+ tmp[0]+tmp[1], null, tmp[2]
222
+ ];
223
+ }
224
+
225
+ return sentenceNotOver;
226
+ }),
227
+
228
+ // Filter empty sentences
229
+ tokenize.filter(function(sentence) {
230
+ if (sentence.trim() == '') return false;
231
+ return true;
232
+ })
233
+ );
234
+ };
235
+
236
+ // From lodash in original
237
+ const partial = (fn, ...partialArgs) => (...args) => fn(...partialArgs, ...args);
238
+
239
+ export default function(tokenize) {
240
+ return {
241
+ sentences: partial(tokenizeSentences, tokenize)
242
+ };
243
+ };
@@ -0,0 +1,16 @@
1
+ function endsWithChar(word, c) {
2
+ if (c.length > 1) {
3
+ return c.indexOf(word.slice(-1)) > -1;
4
+ }
5
+
6
+ return word.slice(-1) === c;
7
+ }
8
+
9
+ function endsWith(word, end) {
10
+ return word.slice(word.length - end.length) === end;
11
+ }
12
+
13
+ export default {
14
+ endsWith: endsWith,
15
+ endsWithChar: endsWithChar
16
+ };
@@ -0,0 +1,6 @@
1
+ This is a copy of https://github.com/textlint-rule/rousseau/tree/master/packages/tokenize-text
2
+ which is also available as the npm package `@textlint-rule/tokenize-text`.
3
+
4
+ It has been modified to remove the lodash dependency so it doesn't have to be
5
+ included directly in our dependencies, since it becomes incredibly large with
6
+ lodash, especially since the entire lodash lib was imported.
@@ -0,0 +1,52 @@
1
+ // Only covers what's necessary to use these packages from TS
2
+
3
+ export declare interface TextlintSegment {
4
+ value: string;
5
+ index: number;
6
+ offset: number;
7
+ }
8
+
9
+ declare class Tokenizer {
10
+ constructor(opts?: {
11
+ cacheGet?: (key: any) => any;
12
+ cacheSet?: (key: any, value: any) => void;
13
+ });
14
+
15
+ split(
16
+ fn: Function,
17
+ opts?: {
18
+ preserveProperties?: boolean;
19
+ cache?: Function;
20
+ }
21
+ ): (text: string | Object[], tok?: any) => any[];
22
+
23
+ debug(prefix?: string): (text: string, tok: any) => boolean;
24
+
25
+ re(re: RegExp, opts?: { split?: boolean }): Function;
26
+
27
+ splitAndMerge(
28
+ fn: Function,
29
+ opts?: {
30
+ mergeWith?: string;
31
+ }
32
+ ): (tokens: any[]) => any[];
33
+
34
+ filter(fn: Function): Function;
35
+
36
+ extend(fn: Function | Object): Function;
37
+
38
+ ifthen(condition: Function, then: Function): Function;
39
+
40
+ test(re: RegExp): Function;
41
+
42
+ flow(...fns: Function[]): Function;
43
+
44
+ serie(...fns: Function[]): Function;
45
+
46
+ merge: Function;
47
+ sections: Function;
48
+ words: Function;
49
+ characters: Function;
50
+ }
51
+
52
+ export default Tokenizer;
@@ -0,0 +1,256 @@
1
+ import tokenUtils from './tokens';
2
+
3
+ var WORD_BOUNDARY_CHARS = '\t\r\n\u00A0 !\"#$%&()*+,\-.\\/:;<=>?@\[\\\]^_`{|}~';
4
+ var WORD_BOUNDARY_REGEX = new RegExp('[' + WORD_BOUNDARY_CHARS + ']');
5
+ var SPLIT_REGEX = new RegExp(
6
+ '([^' + WORD_BOUNDARY_CHARS + ']+)');
7
+
8
+ function Tokenizer(opts) {
9
+ if (!(this instanceof Tokenizer)) return new Tokenizer(opts);
10
+
11
+ this.opts = Object.assign({
12
+ cacheGet: function(key) { return null; },
13
+ cacheSet: function(key, value) { }
14
+ }, opts);
15
+ }
16
+
17
+ Tokenizer.prototype.split = function tokenizeSplit(fn, opts = {}) {
18
+ var that = this;
19
+ opts = Object.assign({ preserveProperties: true, cache: () => null }, opts);
20
+
21
+ return function(text, tok) {
22
+ if (arguments.length === 6) return fn.apply(null, arguments);
23
+
24
+ var prev;
25
+ var cacheId, cacheValue;
26
+
27
+ if(text === undefined) return [];
28
+
29
+ if (typeof text === "string") {
30
+ text = [{
31
+ value: text,
32
+ index: 0,
33
+ offset: text.length
34
+ }];
35
+ } else if (!Array.isArray(text)) {
36
+ text = [text];
37
+ }
38
+
39
+ cacheId = tokenUtils.tokensId(text, opts.cache());
40
+ if (cacheId) {
41
+ cacheValue = that.opts.cacheGet(cacheId);
42
+ if (cacheValue) {
43
+ return cacheValue;
44
+ }
45
+ }
46
+
47
+ var result = text.map(function(token, i) {
48
+ var next = text[i + 1];
49
+ var tokens = fn(
50
+ token.value,
51
+ Object.assign({}, token),
52
+ prev ? Object.assign({}, prev) : null,
53
+ next ? Object.assign({}, next) : null,
54
+ i,
55
+ text
56
+ ) || [];
57
+
58
+ tokens = tokenUtils.normalize(token, tokens);
59
+
60
+ if (opts.preserveProperties) {
61
+ var props = tokenUtils.properties(token);
62
+ tokens = tokens.map(function(_tok) {
63
+ return Object.assign({}, _tok, props);
64
+ });
65
+ }
66
+
67
+ prev = token;
68
+
69
+ return tokens;
70
+ }).filter(Boolean).flat();
71
+
72
+ if (cacheId) {
73
+ that.opts.cacheSet(cacheId, result);
74
+ }
75
+
76
+ return result;
77
+ };
78
+ };
79
+
80
+ // Tokenize a text using a RegExp
81
+ Tokenizer.prototype.re = function tokenizeRe(re, opts = {}) {
82
+ opts = Object.assign({ split: false }, opts);
83
+
84
+ return this.split(function(text, tok) {
85
+ var originalText = text;
86
+ var tokens = [];
87
+ var match;
88
+ var start = 0;
89
+ var lastIndex = 0;
90
+
91
+ while (match = re.exec(text)) {
92
+ // Index in the current text section
93
+ var index = match.index;
94
+
95
+ // Index in the original text
96
+ var absoluteIndex = start + index;
97
+
98
+ var value = match[0] || "";
99
+ var offset = value.length;
100
+
101
+ // If splitting, push missed text
102
+ if (opts.split && start < absoluteIndex) {
103
+ var beforeText = originalText.slice(start, absoluteIndex);
104
+ tokens.push({
105
+ value: beforeText,
106
+ index: start,
107
+ offset: beforeText.length
108
+ });
109
+ }
110
+
111
+ tokens.push({
112
+ value: value,
113
+ index: absoluteIndex,
114
+ offset: offset,
115
+ match: match
116
+ });
117
+
118
+ text = text.slice(index + offset);
119
+ start = absoluteIndex + offset;
120
+ }
121
+
122
+ // If splitting, push left text
123
+ if (opts.split && text) {
124
+ tokens.push({
125
+ value: text,
126
+ index: start,
127
+ offset: text.length
128
+ });
129
+ }
130
+
131
+ return tokens;
132
+ }, {
133
+ cache: function() {
134
+ return re.toString();
135
+ }
136
+ });
137
+ };
138
+
139
+ // Split and merge tokens
140
+ Tokenizer.prototype.splitAndMerge = function tokenizeSplitAndMerge(fn, opts = {}) {
141
+ var that = this;
142
+ opts = Object.assign({ mergeWith: '' }, opts);
143
+
144
+ return function(tokens) {
145
+ var result = [];
146
+ var accu = [];
147
+
148
+ function pushAccu() {
149
+ if (accu.length == 0) return;
150
+
151
+ // Merge accumulator into one token
152
+ var tok = tokenUtils.merge(accu, opts.mergeWith);
153
+
154
+ result.push(tok);
155
+ accu = [];
156
+ }
157
+
158
+ that.split(function(word, token) {
159
+ var toks = fn.apply(null, arguments);
160
+
161
+ // Normalize tokens
162
+ toks = tokenUtils.normalize(token, toks);
163
+
164
+ // Accumulate tokens and push to final results
165
+ toks.forEach(function(tok) {
166
+ if (tok === null) {
167
+ pushAccu();
168
+ } else {
169
+ accu.push(tok);
170
+ }
171
+ });
172
+ })(tokens);
173
+
174
+ // Push tokens left in accumulator
175
+ pushAccu();
176
+
177
+ return result;
178
+ };
179
+ };
180
+
181
+ // Filter when tokenising
182
+ Tokenizer.prototype.filter = function tokenizeFilter(fn) {
183
+ return this.split(function(text, tok) {
184
+ if (fn.apply(null, arguments)) {
185
+ return {
186
+ value: tok.value,
187
+ index: 0,
188
+ offset: tok.offset
189
+ };
190
+ }
191
+ return undefined;
192
+ });
193
+ };
194
+
195
+ // Extend a token properties
196
+ Tokenizer.prototype.extend = function tokenizeExtend(fn) {
197
+ return this.split(function(text, tok) {
198
+ var o = typeof fn === 'function' ? fn.apply(null, arguments) : fn;
199
+ return Object.assign({
200
+ value: tok.value,
201
+ index: 0,
202
+ offset: tok.offset
203
+ }, o);
204
+ });
205
+ };
206
+
207
+ // Condition for tokenizing flow
208
+ Tokenizer.prototype.ifthen = function(condition, then) {
209
+ return this.split(function(text, tok) {
210
+ if (condition.apply(null, arguments)) {
211
+ return then.apply(null, arguments);
212
+ }
213
+ const {index, ...rest} = tok; // Omit 'index'
214
+ return rest;
215
+ });
216
+ };
217
+
218
+ // Filter by testing a regex
219
+ Tokenizer.prototype.test = function tokenizeTest(re) {
220
+ return this.filter(function(text, tok) {
221
+ return re.test(text);
222
+ }, {
223
+ cache: re.toString()
224
+ });
225
+ };
226
+
227
+ // Process token by all arguments
228
+ Tokenizer.prototype.flow = function tokenizeFlow(...args) {
229
+ const fn = args.reduce((acc, cur) => (...args) => cur(acc(...args)));
230
+ return this.split(fn);
231
+ };
232
+
233
+ // Group and process a token as a group
234
+ Tokenizer.prototype.serie = function tokenizeFlow(...args) {
235
+ return args.reduce((acc, cur) => (...args) => cur(acc(...args)));
236
+ };
237
+
238
+ // Merge all tokens into one
239
+ Tokenizer.prototype.merge = function() {
240
+ return this.splitAndMerge(token => [token]);
241
+ };
242
+
243
+ Tokenizer.prototype.sections = function() {
244
+ return this.re(/([^\n\.,;!?]+)/i, { split: false });
245
+ };
246
+
247
+ Tokenizer.prototype.words = function() {
248
+ return this.re(SPLIT_REGEX);
249
+ };
250
+
251
+ Tokenizer.prototype.characters = function() {
252
+ return this.re(/[^\s]/);
253
+ };
254
+
255
+ export default Tokenizer;
256
+