@readium/shared 1.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +28 -0
- package/README.MD +27 -0
- package/dist/index.js +2536 -0
- package/dist/index.umd.cjs +2 -0
- package/package.json +73 -0
- package/src/fetcher/Fetcher.ts +37 -0
- package/src/fetcher/HttpFetcher.ts +122 -0
- package/src/fetcher/Resource.ts +37 -0
- package/src/fetcher/index.ts +2 -0
- package/src/index.ts +4 -0
- package/src/opds/Acquisition.ts +54 -0
- package/src/opds/Availability.ts +61 -0
- package/src/opds/Copies.ts +43 -0
- package/src/opds/Holds.ts +43 -0
- package/src/opds/Price.ts +51 -0
- package/src/opds/index.ts +5 -0
- package/src/publication/BelongsTo.ts +51 -0
- package/src/publication/Contributor.ts +129 -0
- package/src/publication/GuidedNavigation.ts +176 -0
- package/src/publication/Link.ts +332 -0
- package/src/publication/LocalizedString.ts +74 -0
- package/src/publication/Locator.ts +225 -0
- package/src/publication/LocatorCollection.ts +133 -0
- package/src/publication/Manifest.ts +256 -0
- package/src/publication/Metadata.ts +329 -0
- package/src/publication/Properties.ts +44 -0
- package/src/publication/Publication.ts +148 -0
- package/src/publication/PublicationCollection.ts +144 -0
- package/src/publication/ReadingProgression.ts +28 -0
- package/src/publication/Subject.ts +115 -0
- package/src/publication/encryption/Encryption.ts +68 -0
- package/src/publication/encryption/Properties.ts +20 -0
- package/src/publication/encryption/index.ts +2 -0
- package/src/publication/epub/EPUBLayout.ts +10 -0
- package/src/publication/epub/Presentation.ts +16 -0
- package/src/publication/epub/Properties.ts +28 -0
- package/src/publication/epub/Publication.ts +51 -0
- package/src/publication/epub/index.ts +4 -0
- package/src/publication/html/DomRange.ts +63 -0
- package/src/publication/html/DomRangePoint.ts +64 -0
- package/src/publication/html/Locations.ts +129 -0
- package/src/publication/html/index.ts +3 -0
- package/src/publication/index.ts +19 -0
- package/src/publication/opds/Properties.ts +84 -0
- package/src/publication/opds/Publication.ts +14 -0
- package/src/publication/opds/index.ts +2 -0
- package/src/publication/presentation/Metadata.ts +19 -0
- package/src/publication/presentation/Presentation.ts +155 -0
- package/src/publication/presentation/Properties.ts +70 -0
- package/src/publication/presentation/index.ts +3 -0
- package/src/publication/services/content/Content.ts +36 -0
- package/src/publication/services/content/ContentTokenizer.ts +62 -0
- package/src/publication/services/content/Iterator.ts +46 -0
- package/src/publication/services/content/element/attributes.ts +44 -0
- package/src/publication/services/content/element/element.ts +118 -0
- package/src/publication/services/content/element/index.ts +3 -0
- package/src/publication/services/content/element/text_role.ts +60 -0
- package/src/publication/services/content/index.ts +5 -0
- package/src/publication/services/content/iterators/HTMLResourceContentIterator.ts +571 -0
- package/src/publication/services/content/iterators/PDFTextContentIterator.ts +184 -0
- package/src/publication/services/content/iterators/PublicationContentIterator.ts +144 -0
- package/src/publication/services/content/iterators/helpers.ts +146 -0
- package/src/publication/services/content/iterators/index.ts +3 -0
- package/src/publication/services/index.ts +1 -0
- package/src/util/JSONParse.ts +31 -0
- package/src/util/Language.ts +5 -0
- package/src/util/URITemplate.ts +83 -0
- package/src/util/index.ts +5 -0
- package/src/util/mediatype/MediaType.ts +563 -0
- package/src/util/mediatype/index.ts +1 -0
- package/src/util/tokenizer/TextTokenizer.ts +126 -0
- package/src/util/tokenizer/Tokenizer.ts +8 -0
- package/src/util/tokenizer/index.ts +2 -0
- package/src/util/tokenizer/tokenize-english/README.md +7 -0
- package/src/util/tokenizer/tokenize-english/abbreviations.js +52 -0
- package/src/util/tokenizer/tokenize-english/index.d.ts +3 -0
- package/src/util/tokenizer/tokenize-english/index.js +243 -0
- package/src/util/tokenizer/tokenize-english/utils.js +16 -0
- package/src/util/tokenizer/tokenize-text/README.MD +6 -0
- package/src/util/tokenizer/tokenize-text/index.d.ts +52 -0
- package/src/util/tokenizer/tokenize-text/index.js +256 -0
- package/src/util/tokenizer/tokenize-text/tokens.js +100 -0
- package/types/src/fetcher/Fetcher.d.ts +27 -0
- package/types/src/fetcher/HttpFetcher.d.ts +28 -0
- package/types/src/fetcher/Resource.d.ts +15 -0
- package/types/src/fetcher/index.d.ts +2 -0
- package/types/src/index.d.ts +4 -0
- package/types/src/opds/Acquisition.d.ts +26 -0
- package/types/src/opds/Availability.d.ts +32 -0
- package/types/src/opds/Copies.d.ts +22 -0
- package/types/src/opds/Holds.d.ts +22 -0
- package/types/src/opds/Price.d.ts +29 -0
- package/types/src/opds/index.d.ts +5 -0
- package/types/src/publication/BelongsTo.d.ts +23 -0
- package/types/src/publication/Contributor.d.ts +60 -0
- package/types/src/publication/GuidedNavigation.d.ts +70 -0
- package/types/src/publication/Link.d.ts +128 -0
- package/types/src/publication/LocalizedString.d.ts +48 -0
- package/types/src/publication/Locator.d.ts +101 -0
- package/types/src/publication/LocatorCollection.d.ts +54 -0
- package/types/src/publication/Manifest.d.ts +59 -0
- package/types/src/publication/Metadata.d.ts +101 -0
- package/types/src/publication/Properties.d.ts +28 -0
- package/types/src/publication/Publication.d.ts +53 -0
- package/types/src/publication/PublicationCollection.d.ts +34 -0
- package/types/src/publication/ReadingProgression.d.ts +9 -0
- package/types/src/publication/Subject.d.ts +56 -0
- package/types/src/publication/encryption/Encryption.d.ts +33 -0
- package/types/src/publication/encryption/Properties.d.ts +10 -0
- package/types/src/publication/encryption/index.d.ts +2 -0
- package/types/src/publication/epub/EPUBLayout.d.ts +5 -0
- package/types/src/publication/epub/Presentation.d.ts +7 -0
- package/types/src/publication/epub/Properties.d.ts +14 -0
- package/types/src/publication/epub/Publication.d.ts +19 -0
- package/types/src/publication/epub/index.d.ts +4 -0
- package/types/src/publication/html/DomRange.d.ts +40 -0
- package/types/src/publication/html/DomRangePoint.d.ts +36 -0
- package/types/src/publication/html/Locations.d.ts +45 -0
- package/types/src/publication/html/index.d.ts +3 -0
- package/types/src/publication/index.d.ts +19 -0
- package/types/src/publication/opds/Properties.d.ts +38 -0
- package/types/src/publication/opds/Publication.d.ts +6 -0
- package/types/src/publication/opds/index.d.ts +2 -0
- package/types/src/publication/presentation/Metadata.d.ts +6 -0
- package/types/src/publication/presentation/Presentation.d.ts +95 -0
- package/types/src/publication/presentation/Properties.d.ts +32 -0
- package/types/src/publication/presentation/index.d.ts +3 -0
- package/types/src/publication/services/content/Content.d.ts +20 -0
- package/types/src/publication/services/content/ContentTokenizer.d.ts +19 -0
- package/types/src/publication/services/content/Iterator.d.ts +38 -0
- package/types/src/publication/services/content/element/attributes.d.ts +31 -0
- package/types/src/publication/services/content/element/element.d.ts +107 -0
- package/types/src/publication/services/content/element/index.d.ts +3 -0
- package/types/src/publication/services/content/element/text_role.d.ts +53 -0
- package/types/src/publication/services/content/index.d.ts +5 -0
- package/types/src/publication/services/content/iterators/HTMLResourceContentIterator.d.ts +30 -0
- package/types/src/publication/services/content/iterators/PDFTextContentIterator.d.ts +60 -0
- package/types/src/publication/services/content/iterators/PublicationContentIterator.d.ts +59 -0
- package/types/src/publication/services/content/iterators/helpers.d.ts +10 -0
- package/types/src/publication/services/content/iterators/index.d.ts +3 -0
- package/types/src/publication/services/index.d.ts +1 -0
- package/types/src/util/JSONParse.d.ts +11 -0
- package/types/src/util/Language.d.ts +1 -0
- package/types/src/util/URITemplate.d.ts +24 -0
- package/types/src/util/index.d.ts +5 -0
- package/types/src/util/mediatype/MediaType.d.ts +148 -0
- package/types/src/util/mediatype/index.d.ts +1 -0
- package/types/src/util/tokenizer/TextTokenizer.d.ts +39 -0
- package/types/src/util/tokenizer/Tokenizer.d.ts +6 -0
- package/types/src/util/tokenizer/index.d.ts +2 -0
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
This is a copy of https://github.com/textlint-rule/rousseau/tree/master/packages/tokenize-english
|
|
2
|
+
which is also available as the npm package `@textlint-rule/tokenize-english`.
|
|
3
|
+
|
|
4
|
+
It has been included here because a slight modification needs to be made since
|
|
5
|
+
the code does not run in JavaScript's strict mode due to a poor global variable declaration.
|
|
6
|
+
It has also been modified to remove the lodash dependency so it doesn't have to be
|
|
7
|
+
included directly in our dependencies.
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
export default [
|
|
2
|
+
"ie",
|
|
3
|
+
"eg",
|
|
4
|
+
"ext", // + number?
|
|
5
|
+
"Fig",
|
|
6
|
+
"fig",
|
|
7
|
+
"Figs",
|
|
8
|
+
"figs",
|
|
9
|
+
"et al",
|
|
10
|
+
"Co",
|
|
11
|
+
"Corp",
|
|
12
|
+
"Ave",
|
|
13
|
+
"Inc",
|
|
14
|
+
"Ex",
|
|
15
|
+
"Viz",
|
|
16
|
+
"vs",
|
|
17
|
+
"Vs",
|
|
18
|
+
"repr",
|
|
19
|
+
"Rep",
|
|
20
|
+
"Dem",
|
|
21
|
+
"trans",
|
|
22
|
+
"Vol",
|
|
23
|
+
"pp",
|
|
24
|
+
"rev",
|
|
25
|
+
"est",
|
|
26
|
+
"Ref",
|
|
27
|
+
"Refs",
|
|
28
|
+
"Eq",
|
|
29
|
+
"Eqs",
|
|
30
|
+
"Ch",
|
|
31
|
+
"Sec",
|
|
32
|
+
"Secs",
|
|
33
|
+
"mi",
|
|
34
|
+
"Dept",
|
|
35
|
+
|
|
36
|
+
"Univ",
|
|
37
|
+
"Nos",
|
|
38
|
+
"No",
|
|
39
|
+
"Mol",
|
|
40
|
+
"Cell",
|
|
41
|
+
|
|
42
|
+
"Miss", "Mrs", "Mr", "Ms",
|
|
43
|
+
"Prof", "Dr",
|
|
44
|
+
"Sgt", "Col", "Gen", "Rep", "Sen",'Gov', "Lt", "Maj", "Capt","St",
|
|
45
|
+
|
|
46
|
+
"Sr", "Jr", "jr", "Rev",
|
|
47
|
+
"PhD", "MD", "BA", "MA", "MM",
|
|
48
|
+
"BSc", "MSc",
|
|
49
|
+
|
|
50
|
+
"Jan","Feb","Mar","Apr","Jun","Jul","Aug","Sep","Sept","Oct","Nov","Dec",
|
|
51
|
+
"Sun","Mon","Tu","Tue","Tues","Wed","Th","Thu","Thur","Thurs","Fri","Sat"
|
|
52
|
+
];
|
|
@@ -0,0 +1,243 @@
|
|
|
1
|
+
/*
|
|
2
|
+
Sentence Boundary Detection (SBD)
|
|
3
|
+
Split text into sentences with a vanilla rule based approach (i.e working ~95% of the time).
|
|
4
|
+
|
|
5
|
+
Split a text based on period, question and exclamation marks.
|
|
6
|
+
Skips (most) abbreviations (Mr., Mrs., PhD.)
|
|
7
|
+
Skips numbers/currency
|
|
8
|
+
Skips urls, websites, email addresses, phone nr.
|
|
9
|
+
Counts ellipsis and ?! as single punctuation
|
|
10
|
+
*/
|
|
11
|
+
|
|
12
|
+
import utils from "./utils";
|
|
13
|
+
import abbreviations from "./abbreviations";
|
|
14
|
+
|
|
15
|
+
function isCapitalized(str) {
|
|
16
|
+
return /^[A-Z][a-z].*/.test(str) || isNumber(str);
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
// Start with opening quotes or capitalized letter
|
|
20
|
+
function isSentenceStarter(str) {
|
|
21
|
+
return isCapitalized(str) || /``|"|'/.test(str.substring(0,2));
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
function isCommonAbbreviation(str) {
|
|
25
|
+
return ~abbreviations.indexOf(str.replace(/\W+/g, ''));
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
// This is going towards too much rule based
|
|
29
|
+
function isTimeAbbreviation(word, next) {
|
|
30
|
+
if (word === "a.m." || word === "p.m.") {
|
|
31
|
+
var tmp = next.replace(/\W+/g, '').slice(-3).toLowerCase();
|
|
32
|
+
|
|
33
|
+
if (tmp === "day") {
|
|
34
|
+
return true;
|
|
35
|
+
}
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
return false;
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
function isDottedAbbreviation(word) {
|
|
42
|
+
var matches = word.replace(/[\(\)\[\]\{\}]/g, '').match(/(.\.)*/);
|
|
43
|
+
return matches && matches[0].length > 0;
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
// TODO look for next words, if multiple capitalized -> not sentence ending
|
|
47
|
+
function isCustomAbbreviation(str) {
|
|
48
|
+
if (str.length <= 3)
|
|
49
|
+
return true;
|
|
50
|
+
|
|
51
|
+
return isCapitalized(str);
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
// Uses current word count in sentence and next few words to check if it is
|
|
55
|
+
// more likely an abbreviation + name or new sentence.
|
|
56
|
+
function isNameAbbreviation(wordCount, words) {
|
|
57
|
+
if (words.length > 0) {
|
|
58
|
+
if (wordCount < 5 && words[0].length < 6 && isCapitalized(words[0])) {
|
|
59
|
+
return true;
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
var capitalized = words.filter(function(str) {
|
|
63
|
+
return /[A-Z]/.test(str.charAt(0));
|
|
64
|
+
});
|
|
65
|
+
|
|
66
|
+
return capitalized.length >= 3;
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
return false;
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
function isNumber(str, dotPos) {
|
|
73
|
+
if (dotPos) {
|
|
74
|
+
str = str.slice(dotPos-1, dotPos+2);
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
return !isNaN(Number(str));
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
// Phone number matching
|
|
81
|
+
// http://stackoverflow.com/a/123666/951517
|
|
82
|
+
function isPhoneNr(str) {
|
|
83
|
+
return str.match(/^(?:(?:\+?1\s*(?:[.-]\s*)?)?(?:\(\s*([2-9]1[02-9]|[2-9][02-8]1|[2-9][02-8][02-9])\s*\)|([2-9]1[02-9]|[2-9][02-8]1|[2-9][02-8][02-9]))\s*(?:[.-]\s*)?)?([2-9]1[02-9]|[2-9][02-9]1|[2-9][02-9]{2})\s*(?:[.-]\s*)?([0-9]{4})(?:\s*(?:#|x\.?|ext\.?|extension)\s*(\d+))?$/);
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
// Match urls / emails
|
|
87
|
+
// http://stackoverflow.com/a/3809435/951517
|
|
88
|
+
function isURL(str) {
|
|
89
|
+
return str.match(/[-a-zA-Z0-9@:%._\+~#=]{2,256}\.[a-z]{2,6}\b([-a-zA-Z0-9@:%_\+.~#?&//=]*)/);
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
// Starting a new sentence if beginning with capital letter
|
|
93
|
+
// Exception: The word is enclosed in brackets
|
|
94
|
+
function isConcatenated(word) {
|
|
95
|
+
var i = 0;
|
|
96
|
+
|
|
97
|
+
if ((i = word.indexOf(".")) > -1 ||
|
|
98
|
+
(i = word.indexOf("!")) > -1 ||
|
|
99
|
+
(i = word.indexOf("?")) > -1)
|
|
100
|
+
{
|
|
101
|
+
var c = word.charAt(i + 1);
|
|
102
|
+
|
|
103
|
+
// Check if the next word starts with a letter
|
|
104
|
+
if (c.match(/[a-zA-Z].*/)) {
|
|
105
|
+
return [word.slice(0, i), word.charAt(i), word.slice(i+1)];
|
|
106
|
+
}
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
return false;
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
function isBoundaryChar(word) {
|
|
113
|
+
return word === "." ||
|
|
114
|
+
word === "!" ||
|
|
115
|
+
word === "?";
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
// http://tech.grammarly.com/blog/posts/How-to-Split-Sentences.html
|
|
119
|
+
function tokenizeSentences(tokenize, opts) {
|
|
120
|
+
opts = Object.assign(opts || {}, {
|
|
121
|
+
newlineBoundary: opts && opts.newlineBoundary || true
|
|
122
|
+
})
|
|
123
|
+
|
|
124
|
+
var splitRe = /\s/;
|
|
125
|
+
|
|
126
|
+
return tokenize.serie(
|
|
127
|
+
// Split into words
|
|
128
|
+
tokenize.re(splitRe, { split: true }),
|
|
129
|
+
|
|
130
|
+
// Merge words as sentences
|
|
131
|
+
tokenize.splitAndMerge(function(word, token, prev, next, i, tokens) {
|
|
132
|
+
var tmp;
|
|
133
|
+
var endOfSentence = [word, null];
|
|
134
|
+
var sentenceNotOver = word;
|
|
135
|
+
|
|
136
|
+
// Find the next word
|
|
137
|
+
var nextWord = tokens.slice(i+1).find(function(tok) {
|
|
138
|
+
return !splitRe.test(tok.value)
|
|
139
|
+
});
|
|
140
|
+
|
|
141
|
+
// Newline boundaries
|
|
142
|
+
if (word === '\n' && opts.newlineBoundary) return endOfSentence;
|
|
143
|
+
|
|
144
|
+
if (isBoundaryChar(word) ||
|
|
145
|
+
utils.endsWithChar(word, "?!"))
|
|
146
|
+
{
|
|
147
|
+
return endOfSentence;
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
// A dot might indicate the end sentences
|
|
151
|
+
// Exception: The next sentence starts with a word (non abbreviation)
|
|
152
|
+
// that has a capital letter.
|
|
153
|
+
if (utils.endsWithChar(word, '.')) {
|
|
154
|
+
// Check if there is a next word
|
|
155
|
+
if (nextWord) {
|
|
156
|
+
// This should be improved with machine learning
|
|
157
|
+
|
|
158
|
+
// Single character abbr.
|
|
159
|
+
if (word.length === 2 && isNaN(word.charAt(0)) && word.match(/[a-zA-Z]/)) {
|
|
160
|
+
return sentenceNotOver;
|
|
161
|
+
}
|
|
162
|
+
|
|
163
|
+
// Common abbr. that often do not end sentences
|
|
164
|
+
if (isCommonAbbreviation(word)) {
|
|
165
|
+
return sentenceNotOver;
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
// Next word starts with capital word, but current sentence is
|
|
169
|
+
// quite short
|
|
170
|
+
if (isSentenceStarter(nextWord.value)) {
|
|
171
|
+
if (isTimeAbbreviation(word, nextWord.value)) {
|
|
172
|
+
return sentenceNotOver;
|
|
173
|
+
}
|
|
174
|
+
|
|
175
|
+
// Dealing with names at the start of sentences
|
|
176
|
+
/*if (isNameAbbreviation(wordCount, words.slice(i, 6))) {
|
|
177
|
+
return word;
|
|
178
|
+
}*/
|
|
179
|
+
|
|
180
|
+
if (isNumber(nextWord.value) && isCustomAbbreviation(word)) {
|
|
181
|
+
return sentenceNotOver;
|
|
182
|
+
}
|
|
183
|
+
}
|
|
184
|
+
else {
|
|
185
|
+
// Skip ellipsis
|
|
186
|
+
if (utils.endsWith(word, "..")) {
|
|
187
|
+
return sentenceNotOver;
|
|
188
|
+
}
|
|
189
|
+
|
|
190
|
+
//// Skip abbreviations
|
|
191
|
+
// Short words + dot or a dot after each letter
|
|
192
|
+
if (isDottedAbbreviation(word) || isCustomAbbreviation(word)) {
|
|
193
|
+
return sentenceNotOver;
|
|
194
|
+
}
|
|
195
|
+
}
|
|
196
|
+
}
|
|
197
|
+
|
|
198
|
+
return endOfSentence;
|
|
199
|
+
}
|
|
200
|
+
|
|
201
|
+
// Check if the word has a dot in it
|
|
202
|
+
let index;
|
|
203
|
+
if ((index = word.indexOf(".")) > -1) {
|
|
204
|
+
if (isNumber(word, index)) {
|
|
205
|
+
return sentenceNotOver;
|
|
206
|
+
}
|
|
207
|
+
|
|
208
|
+
// Custom dotted abbreviations (like K.L.M or I.C.T)
|
|
209
|
+
if (isDottedAbbreviation(word)) {
|
|
210
|
+
return sentenceNotOver;
|
|
211
|
+
}
|
|
212
|
+
|
|
213
|
+
// Skip urls / emails and the like
|
|
214
|
+
if (isURL(word) || isPhoneNr(word)) {
|
|
215
|
+
return sentenceNotOver;
|
|
216
|
+
}
|
|
217
|
+
}
|
|
218
|
+
|
|
219
|
+
if (tmp = isConcatenated(word)) {
|
|
220
|
+
return [
|
|
221
|
+
tmp[0]+tmp[1], null, tmp[2]
|
|
222
|
+
];
|
|
223
|
+
}
|
|
224
|
+
|
|
225
|
+
return sentenceNotOver;
|
|
226
|
+
}),
|
|
227
|
+
|
|
228
|
+
// Filter empty sentences
|
|
229
|
+
tokenize.filter(function(sentence) {
|
|
230
|
+
if (sentence.trim() == '') return false;
|
|
231
|
+
return true;
|
|
232
|
+
})
|
|
233
|
+
);
|
|
234
|
+
};
|
|
235
|
+
|
|
236
|
+
// From lodash in original
|
|
237
|
+
const partial = (fn, ...partialArgs) => (...args) => fn(...partialArgs, ...args);
|
|
238
|
+
|
|
239
|
+
export default function(tokenize) {
|
|
240
|
+
return {
|
|
241
|
+
sentences: partial(tokenizeSentences, tokenize)
|
|
242
|
+
};
|
|
243
|
+
};
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
function endsWithChar(word, c) {
|
|
2
|
+
if (c.length > 1) {
|
|
3
|
+
return c.indexOf(word.slice(-1)) > -1;
|
|
4
|
+
}
|
|
5
|
+
|
|
6
|
+
return word.slice(-1) === c;
|
|
7
|
+
}
|
|
8
|
+
|
|
9
|
+
function endsWith(word, end) {
|
|
10
|
+
return word.slice(word.length - end.length) === end;
|
|
11
|
+
}
|
|
12
|
+
|
|
13
|
+
export default {
|
|
14
|
+
endsWith: endsWith,
|
|
15
|
+
endsWithChar: endsWithChar
|
|
16
|
+
};
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
This is a copy of https://github.com/textlint-rule/rousseau/tree/master/packages/tokenize-text
|
|
2
|
+
which is also available as the npm package `@textlint-rule/tokenize-text`.
|
|
3
|
+
|
|
4
|
+
It has been modified to remove the lodash dependency so it doesn't have to be
|
|
5
|
+
included directly in our dependencies, since it becomes incredibly large with
|
|
6
|
+
lodash, especially since the entire lodash lib was imported.
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
// Only covers what's necessary to use these packages from TS
|
|
2
|
+
|
|
3
|
+
export declare interface TextlintSegment {
|
|
4
|
+
value: string;
|
|
5
|
+
index: number;
|
|
6
|
+
offset: number;
|
|
7
|
+
}
|
|
8
|
+
|
|
9
|
+
declare class Tokenizer {
|
|
10
|
+
constructor(opts?: {
|
|
11
|
+
cacheGet?: (key: any) => any;
|
|
12
|
+
cacheSet?: (key: any, value: any) => void;
|
|
13
|
+
});
|
|
14
|
+
|
|
15
|
+
split(
|
|
16
|
+
fn: Function,
|
|
17
|
+
opts?: {
|
|
18
|
+
preserveProperties?: boolean;
|
|
19
|
+
cache?: Function;
|
|
20
|
+
}
|
|
21
|
+
): (text: string | Object[], tok?: any) => any[];
|
|
22
|
+
|
|
23
|
+
debug(prefix?: string): (text: string, tok: any) => boolean;
|
|
24
|
+
|
|
25
|
+
re(re: RegExp, opts?: { split?: boolean }): Function;
|
|
26
|
+
|
|
27
|
+
splitAndMerge(
|
|
28
|
+
fn: Function,
|
|
29
|
+
opts?: {
|
|
30
|
+
mergeWith?: string;
|
|
31
|
+
}
|
|
32
|
+
): (tokens: any[]) => any[];
|
|
33
|
+
|
|
34
|
+
filter(fn: Function): Function;
|
|
35
|
+
|
|
36
|
+
extend(fn: Function | Object): Function;
|
|
37
|
+
|
|
38
|
+
ifthen(condition: Function, then: Function): Function;
|
|
39
|
+
|
|
40
|
+
test(re: RegExp): Function;
|
|
41
|
+
|
|
42
|
+
flow(...fns: Function[]): Function;
|
|
43
|
+
|
|
44
|
+
serie(...fns: Function[]): Function;
|
|
45
|
+
|
|
46
|
+
merge: Function;
|
|
47
|
+
sections: Function;
|
|
48
|
+
words: Function;
|
|
49
|
+
characters: Function;
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
export default Tokenizer;
|
|
@@ -0,0 +1,256 @@
|
|
|
1
|
+
import tokenUtils from './tokens';
|
|
2
|
+
|
|
3
|
+
var WORD_BOUNDARY_CHARS = '\t\r\n\u00A0 !\"#$%&()*+,\-.\\/:;<=>?@\[\\\]^_`{|}~';
|
|
4
|
+
var WORD_BOUNDARY_REGEX = new RegExp('[' + WORD_BOUNDARY_CHARS + ']');
|
|
5
|
+
var SPLIT_REGEX = new RegExp(
|
|
6
|
+
'([^' + WORD_BOUNDARY_CHARS + ']+)');
|
|
7
|
+
|
|
8
|
+
function Tokenizer(opts) {
|
|
9
|
+
if (!(this instanceof Tokenizer)) return new Tokenizer(opts);
|
|
10
|
+
|
|
11
|
+
this.opts = Object.assign({
|
|
12
|
+
cacheGet: function(key) { return null; },
|
|
13
|
+
cacheSet: function(key, value) { }
|
|
14
|
+
}, opts);
|
|
15
|
+
}
|
|
16
|
+
|
|
17
|
+
Tokenizer.prototype.split = function tokenizeSplit(fn, opts = {}) {
|
|
18
|
+
var that = this;
|
|
19
|
+
opts = Object.assign({ preserveProperties: true, cache: () => null }, opts);
|
|
20
|
+
|
|
21
|
+
return function(text, tok) {
|
|
22
|
+
if (arguments.length === 6) return fn.apply(null, arguments);
|
|
23
|
+
|
|
24
|
+
var prev;
|
|
25
|
+
var cacheId, cacheValue;
|
|
26
|
+
|
|
27
|
+
if(text === undefined) return [];
|
|
28
|
+
|
|
29
|
+
if (typeof text === "string") {
|
|
30
|
+
text = [{
|
|
31
|
+
value: text,
|
|
32
|
+
index: 0,
|
|
33
|
+
offset: text.length
|
|
34
|
+
}];
|
|
35
|
+
} else if (!Array.isArray(text)) {
|
|
36
|
+
text = [text];
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
cacheId = tokenUtils.tokensId(text, opts.cache());
|
|
40
|
+
if (cacheId) {
|
|
41
|
+
cacheValue = that.opts.cacheGet(cacheId);
|
|
42
|
+
if (cacheValue) {
|
|
43
|
+
return cacheValue;
|
|
44
|
+
}
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
var result = text.map(function(token, i) {
|
|
48
|
+
var next = text[i + 1];
|
|
49
|
+
var tokens = fn(
|
|
50
|
+
token.value,
|
|
51
|
+
Object.assign({}, token),
|
|
52
|
+
prev ? Object.assign({}, prev) : null,
|
|
53
|
+
next ? Object.assign({}, next) : null,
|
|
54
|
+
i,
|
|
55
|
+
text
|
|
56
|
+
) || [];
|
|
57
|
+
|
|
58
|
+
tokens = tokenUtils.normalize(token, tokens);
|
|
59
|
+
|
|
60
|
+
if (opts.preserveProperties) {
|
|
61
|
+
var props = tokenUtils.properties(token);
|
|
62
|
+
tokens = tokens.map(function(_tok) {
|
|
63
|
+
return Object.assign({}, _tok, props);
|
|
64
|
+
});
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
prev = token;
|
|
68
|
+
|
|
69
|
+
return tokens;
|
|
70
|
+
}).filter(Boolean).flat();
|
|
71
|
+
|
|
72
|
+
if (cacheId) {
|
|
73
|
+
that.opts.cacheSet(cacheId, result);
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
return result;
|
|
77
|
+
};
|
|
78
|
+
};
|
|
79
|
+
|
|
80
|
+
// Tokenize a text using a RegExp
|
|
81
|
+
Tokenizer.prototype.re = function tokenizeRe(re, opts = {}) {
|
|
82
|
+
opts = Object.assign({ split: false }, opts);
|
|
83
|
+
|
|
84
|
+
return this.split(function(text, tok) {
|
|
85
|
+
var originalText = text;
|
|
86
|
+
var tokens = [];
|
|
87
|
+
var match;
|
|
88
|
+
var start = 0;
|
|
89
|
+
var lastIndex = 0;
|
|
90
|
+
|
|
91
|
+
while (match = re.exec(text)) {
|
|
92
|
+
// Index in the current text section
|
|
93
|
+
var index = match.index;
|
|
94
|
+
|
|
95
|
+
// Index in the original text
|
|
96
|
+
var absoluteIndex = start + index;
|
|
97
|
+
|
|
98
|
+
var value = match[0] || "";
|
|
99
|
+
var offset = value.length;
|
|
100
|
+
|
|
101
|
+
// If splitting, push missed text
|
|
102
|
+
if (opts.split && start < absoluteIndex) {
|
|
103
|
+
var beforeText = originalText.slice(start, absoluteIndex);
|
|
104
|
+
tokens.push({
|
|
105
|
+
value: beforeText,
|
|
106
|
+
index: start,
|
|
107
|
+
offset: beforeText.length
|
|
108
|
+
});
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
tokens.push({
|
|
112
|
+
value: value,
|
|
113
|
+
index: absoluteIndex,
|
|
114
|
+
offset: offset,
|
|
115
|
+
match: match
|
|
116
|
+
});
|
|
117
|
+
|
|
118
|
+
text = text.slice(index + offset);
|
|
119
|
+
start = absoluteIndex + offset;
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
// If splitting, push left text
|
|
123
|
+
if (opts.split && text) {
|
|
124
|
+
tokens.push({
|
|
125
|
+
value: text,
|
|
126
|
+
index: start,
|
|
127
|
+
offset: text.length
|
|
128
|
+
});
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
return tokens;
|
|
132
|
+
}, {
|
|
133
|
+
cache: function() {
|
|
134
|
+
return re.toString();
|
|
135
|
+
}
|
|
136
|
+
});
|
|
137
|
+
};
|
|
138
|
+
|
|
139
|
+
// Split and merge tokens
|
|
140
|
+
Tokenizer.prototype.splitAndMerge = function tokenizeSplitAndMerge(fn, opts = {}) {
|
|
141
|
+
var that = this;
|
|
142
|
+
opts = Object.assign({ mergeWith: '' }, opts);
|
|
143
|
+
|
|
144
|
+
return function(tokens) {
|
|
145
|
+
var result = [];
|
|
146
|
+
var accu = [];
|
|
147
|
+
|
|
148
|
+
function pushAccu() {
|
|
149
|
+
if (accu.length == 0) return;
|
|
150
|
+
|
|
151
|
+
// Merge accumulator into one token
|
|
152
|
+
var tok = tokenUtils.merge(accu, opts.mergeWith);
|
|
153
|
+
|
|
154
|
+
result.push(tok);
|
|
155
|
+
accu = [];
|
|
156
|
+
}
|
|
157
|
+
|
|
158
|
+
that.split(function(word, token) {
|
|
159
|
+
var toks = fn.apply(null, arguments);
|
|
160
|
+
|
|
161
|
+
// Normalize tokens
|
|
162
|
+
toks = tokenUtils.normalize(token, toks);
|
|
163
|
+
|
|
164
|
+
// Accumulate tokens and push to final results
|
|
165
|
+
toks.forEach(function(tok) {
|
|
166
|
+
if (tok === null) {
|
|
167
|
+
pushAccu();
|
|
168
|
+
} else {
|
|
169
|
+
accu.push(tok);
|
|
170
|
+
}
|
|
171
|
+
});
|
|
172
|
+
})(tokens);
|
|
173
|
+
|
|
174
|
+
// Push tokens left in accumulator
|
|
175
|
+
pushAccu();
|
|
176
|
+
|
|
177
|
+
return result;
|
|
178
|
+
};
|
|
179
|
+
};
|
|
180
|
+
|
|
181
|
+
// Filter when tokenising
|
|
182
|
+
Tokenizer.prototype.filter = function tokenizeFilter(fn) {
|
|
183
|
+
return this.split(function(text, tok) {
|
|
184
|
+
if (fn.apply(null, arguments)) {
|
|
185
|
+
return {
|
|
186
|
+
value: tok.value,
|
|
187
|
+
index: 0,
|
|
188
|
+
offset: tok.offset
|
|
189
|
+
};
|
|
190
|
+
}
|
|
191
|
+
return undefined;
|
|
192
|
+
});
|
|
193
|
+
};
|
|
194
|
+
|
|
195
|
+
// Extend a token properties
|
|
196
|
+
Tokenizer.prototype.extend = function tokenizeExtend(fn) {
|
|
197
|
+
return this.split(function(text, tok) {
|
|
198
|
+
var o = typeof fn === 'function' ? fn.apply(null, arguments) : fn;
|
|
199
|
+
return Object.assign({
|
|
200
|
+
value: tok.value,
|
|
201
|
+
index: 0,
|
|
202
|
+
offset: tok.offset
|
|
203
|
+
}, o);
|
|
204
|
+
});
|
|
205
|
+
};
|
|
206
|
+
|
|
207
|
+
// Condition for tokenizing flow
|
|
208
|
+
Tokenizer.prototype.ifthen = function(condition, then) {
|
|
209
|
+
return this.split(function(text, tok) {
|
|
210
|
+
if (condition.apply(null, arguments)) {
|
|
211
|
+
return then.apply(null, arguments);
|
|
212
|
+
}
|
|
213
|
+
const {index, ...rest} = tok; // Omit 'index'
|
|
214
|
+
return rest;
|
|
215
|
+
});
|
|
216
|
+
};
|
|
217
|
+
|
|
218
|
+
// Filter by testing a regex
|
|
219
|
+
Tokenizer.prototype.test = function tokenizeTest(re) {
|
|
220
|
+
return this.filter(function(text, tok) {
|
|
221
|
+
return re.test(text);
|
|
222
|
+
}, {
|
|
223
|
+
cache: re.toString()
|
|
224
|
+
});
|
|
225
|
+
};
|
|
226
|
+
|
|
227
|
+
// Process token by all arguments
|
|
228
|
+
Tokenizer.prototype.flow = function tokenizeFlow(...args) {
|
|
229
|
+
const fn = args.reduce((acc, cur) => (...args) => cur(acc(...args)));
|
|
230
|
+
return this.split(fn);
|
|
231
|
+
};
|
|
232
|
+
|
|
233
|
+
// Group and process a token as a group
|
|
234
|
+
Tokenizer.prototype.serie = function tokenizeFlow(...args) {
|
|
235
|
+
return args.reduce((acc, cur) => (...args) => cur(acc(...args)));
|
|
236
|
+
};
|
|
237
|
+
|
|
238
|
+
// Merge all tokens into one
|
|
239
|
+
Tokenizer.prototype.merge = function() {
|
|
240
|
+
return this.splitAndMerge(token => [token]);
|
|
241
|
+
};
|
|
242
|
+
|
|
243
|
+
Tokenizer.prototype.sections = function() {
|
|
244
|
+
return this.re(/([^\n\.,;!?]+)/i, { split: false });
|
|
245
|
+
};
|
|
246
|
+
|
|
247
|
+
Tokenizer.prototype.words = function() {
|
|
248
|
+
return this.re(SPLIT_REGEX);
|
|
249
|
+
};
|
|
250
|
+
|
|
251
|
+
Tokenizer.prototype.characters = function() {
|
|
252
|
+
return this.re(/[^\s]/);
|
|
253
|
+
};
|
|
254
|
+
|
|
255
|
+
export default Tokenizer;
|
|
256
|
+
|