@echogarden/text-segmentation 0.2.0 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +30 -25
- package/dist/EastAsianCharacterPatterns.d.ts +0 -6
- package/dist/EastAsianCharacterPatterns.js +6 -6
- package/dist/EastAsianCharacterPatterns.js.map +1 -1
- package/dist/Patterns.d.ts +4 -29
- package/dist/Patterns.js +136 -61
- package/dist/Patterns.js.map +1 -1
- package/dist/Suppressions.d.ts +1 -0
- package/dist/Suppressions.js +19 -9
- package/dist/Suppressions.js.map +1 -1
- package/dist/Test.js +13 -5
- package/dist/Test.js.map +1 -1
- package/dist/TextSegmentation.d.ts +4 -5
- package/dist/TextSegmentation.js +75 -46
- package/dist/TextSegmentation.js.map +1 -1
- package/dist/WordSequence.d.ts +5 -3
- package/dist/WordSequence.js +13 -6
- package/dist/WordSequence.js.map +1 -1
- package/dist/utilities/Utilities.d.ts +1 -0
- package/dist/utilities/Utilities.js +7 -0
- package/dist/utilities/Utilities.js.map +1 -1
- package/package.json +4 -4
- package/src/EastAsianCharacterPatterns.ts +6 -6
- package/src/Patterns.ts +177 -87
- package/src/Suppressions.ts +20 -9
- package/src/Test.ts +18 -5
- package/src/TextSegmentation.ts +98 -58
- package/src/WordSequence.ts +17 -7
- package/src/utilities/Utilities.ts +10 -0
package/dist/WordSequence.js.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"WordSequence.js","sourceRoot":"","sources":["../src/WordSequence.ts"],"names":[],"mappings":"AAAA,MAAM,OAAO,YAAY;IACxB,OAAO,GAAgB,EAAE,CAAA;IAEzB,
|
|
1
|
+
{"version":3,"file":"WordSequence.js","sourceRoot":"","sources":["../src/WordSequence.ts"],"names":[],"mappings":"AAAA,MAAM,OAAO,YAAY;IACxB,OAAO,GAAgB,EAAE,CAAA;IAEzB,OAAO,CAAC,IAAY,EAAE,eAAuB,EAAE,aAAsB;QACpE,MAAM,WAAW,GAAG,eAAe,CAAA;QACnC,MAAM,SAAS,GAAG,WAAW,GAAG,IAAI,CAAC,MAAM,CAAA;QAE3C,IAAI,CAAC,OAAO,CAAC,IAAI,CAAC;YACjB,IAAI;YACJ,WAAW;YACX,SAAS;YACT,aAAa;SACb,CAAC,CAAA;IACH,CAAC;IAED,YAAY,CAAC,UAAkB,EAAE,QAAgB;QAChD,OAAO,IAAI,CAAC,aAAa,CAAC,UAAU,EAAE,QAAQ,CAAC,CAAC,GAAG,CAAC,KAAK,CAAC,EAAE,CAAC,KAAK,CAAC,IAAI,CAAC,CAAA;IACzE,CAAC;IAED,CAAC,gBAAgB,CAAC,UAAkB,EAAE,QAAgB;QACrD,KAAK,IAAI,CAAC,GAAG,UAAU,EAAE,CAAC,GAAG,QAAQ,EAAE,CAAC,EAAE,EAAE,CAAC;YAC5C,MAAM,IAAI,CAAC,SAAS,CAAC,CAAC,CAAC,CAAA;QACxB,CAAC;IACF,CAAC;IAED,SAAS,CAAC,KAAa;QACtB,OAAO,IAAI,CAAC,OAAO,CAAC,KAAK,CAAC,CAAC,IAAI,CAAA;IAChC,CAAC;IAED,aAAa,CAAC,UAAkB,EAAE,QAAgB;QACjD,OAAO,IAAI,CAAC,OAAO,CAAC,KAAK,CAAC,UAAU,EAAE,QAAQ,CAAC,CAAA;IAChD,CAAC;IAED,CAAC,iBAAiB,CAAC,UAAkB,EAAE,QAAgB;QACtD,KAAK,IAAI,CAAC,GAAG,UAAU,EAAE,CAAC,GAAG,QAAQ,EAAE,CAAC,EAAE,EAAE,CAAC;YAC5C,MAAM,IAAI,CAAC,UAAU,CAAC,CAAC,CAAC,CAAA;QACzB,CAAC;IACF,CAAC;IAED,UAAU,CAAC,KAAa;QACvB,OAAO,IAAI,CAAC,OAAO,CAAC,KAAK,CAAC,CAAA;IAC3B,CAAC;IAED,KAAK;QACJ,OAAO,IAAI,CAAC,KAAK,CAAC,CAAC,EAAE,IAAI,CAAC,MAAM,CAAC,CAAA;IAClC,CAAC;IAED,KAAK,CAAC,UAAkB,EAAE,QAAgB;QACzC,MAAM,cAAc,GAAG,IAAI,YAAY,EAAE,CAAA;QAEzC,cAAc,CAAC,OAAO,GAAG,IAAI,CAAC,OAAO,CAAC,KAAK,CAAC,UAAU,EAAE,QAAQ,CAAC,CAAA;QAEjE,OAAO,cAAc,CAAA;IACtB,CAAC;IAED,IAAI,SAAS;QACZ,OAAO,IAAI,CAAC,OAAO,CAAC,GAAG,CAAC,KAAK,CAAC,EAAE,CAAC,KAAK,CAAC,IAAI,CAAC,CAAA;IAC7C,CAAC;IAED,IAAI,SAAS;QACZ,OAAO,IAAI,CAAC,UAAU,EAAE,IAAI,CAAA;IAC7B,CAAC;IAED,IAAI,QAAQ;QACX,OAAO,IAAI,CAAC,SAAS,EAAE,IAAI,CAAA;IAC5B,CAAC;IAED,IAAI,UAAU;QACb,OAAO,IAAI,CAAC,OAAO,CAAC,CAAC,CAAC,CAAA;IACvB,CAAC;IAED,IAAI,SAAS;QACZ,OAAO,IAAI,CAAC,OAAO,CAAC,IAAI,CAAC,MAAM,GAAG,CAAC,CAAC,CAAA;IACrC,CAAC;IAED,IAAI,kBAAkB;QACrB,OAAO,IAAI,CAAC,OAAO,CAAC,MAAM,CAAC,KAAK,CAAC,EAAE,CAAC,KAAK,CAAC,aAAa,KAAK,IAAI,CAAC,CAAA;IAClE,CAAC;IAED,IAAI,gBAAgB;QACnB,OAAO,IAAI,CAAC,OAAO,CAAC,MAAM,CAAC,KAAK,CAAC,EAAE,CAAC,KAAK,CAAC,aAAa,KAAK,IAAI,CAAC,CAAC,GAAG,CAAC,KAAK,CAAC,EAAE,CAAC,KAAK,CAAC,IAAI,CAAC,CAAA;IAC3F,CAAC;IAED,IAAI,qBAAqB;QACxB,OAAO,IAAI,CAAC,OAAO,CAAC,MAAM,CAAC,KAAK,CAAC,EAAE,CAAC,KAAK,CAAC,aAAa,KAAK,KAAK,CAAC,CAAA;IACnE,CAAC;IAED,IAAI,mBAAmB;QACtB,OAAO,IAAI,CAAC,OAAO,CAAC,MAAM,CAAC,KAAK,CAAC,EAAE,CAAC,KAAK,CAAC,aAAa,KAAK,KAAK,CAAC,CAAC,GAAG,CAAC,KAAK,CAAC,EAAE,CAAC,KAAK,CAAC,IAAI,CAAC,CAAA;IAC5F,CAAC;IAED,IAAI,IAAI;QACP,OAAO,IAAI,CAAC,SAAS,CAAC,IAAI,CAAC,EAAE,CAAC,CAAA;IAC/B,CAAC;IAED,IAAI,MAAM;QACT,OAAO,IAAI,CAAC,OAAO,CAAC,MAAM,CAAA;IAC3B,CAAC;CACD"}
|
|
@@ -21,4 +21,11 @@ export function extractSuppressions(entries) {
|
|
|
21
21
|
}
|
|
22
22
|
return suppressions;
|
|
23
23
|
}
|
|
24
|
+
export function getShortLanguageCode(langCode) {
|
|
25
|
+
const dashIndex = langCode.indexOf('-');
|
|
26
|
+
if (dashIndex == -1) {
|
|
27
|
+
return langCode;
|
|
28
|
+
}
|
|
29
|
+
return langCode.substring(0, dashIndex).toLowerCase();
|
|
30
|
+
}
|
|
24
31
|
//# sourceMappingURL=Utilities.js.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"Utilities.js","sourceRoot":"","sources":["../../src/utilities/Utilities.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,WAAW,EAAW,MAAM,iBAAiB,CAAA;AAEtD,MAAM,UAAU,aAAa,CAAC,GAAW,EAAE,MAAM,GAAG,CAAC;IACpD,MAAM,UAAU,GAAG,EAAE,IAAI,MAAM,CAAA;IAE/B,OAAO,IAAI,CAAC,KAAK,CAAC,GAAG,GAAG,UAAU,CAAC,GAAG,UAAU,CAAA;AACjD,CAAC;AAED,MAAM,UAAU,oBAAoB,CAAC,WAAoB;IACxD,MAAM,MAAM,GAAG,WAAW,CAAC,WAAW,CAAC,CAAA;IAEvC,MAAM,aAAa,GAAa,EAAE,CAAA;IAElC,KAAK,IAAI,SAAS,GAAG,CAAC,EAAE,SAAS,GAAG,OAAO,EAAE,SAAS,EAAE,EAAE,CAAC;QAC1D,MAAM,IAAI,GAAG,MAAM,CAAC,aAAa,CAAC,SAAS,CAAC,CAAA;QAE5C,IAAI,MAAM,CAAC,IAAI,CAAC,IAAI,CAAC,EAAE,CAAC;YACvB,aAAa,CAAC,IAAI,CAAC,IAAI,CAAC,CAAA;QACzB,CAAC;IACF,CAAC;IAED,OAAO,aAAa,CAAA;AACrB,CAAC;AAED,MAAM,UAAU,mBAAmB,CAAC,OAAiC;IACpE,MAAM,YAAY,GAAa,EAAE,CAAA;IAEjC,KAAK,MAAM,KAAK,IAAI,OAAO,EAAE,CAAC;QAC7B,YAAY,CAAC,IAAI,CAAC,KAAK,CAAC,WAAW,CAAC,CAAA;IACrC,CAAC;IAED,OAAO,YAAY,CAAA;AACpB,CAAC"}
|
|
1
|
+
{"version":3,"file":"Utilities.js","sourceRoot":"","sources":["../../src/utilities/Utilities.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,WAAW,EAAW,MAAM,iBAAiB,CAAA;AAEtD,MAAM,UAAU,aAAa,CAAC,GAAW,EAAE,MAAM,GAAG,CAAC;IACpD,MAAM,UAAU,GAAG,EAAE,IAAI,MAAM,CAAA;IAE/B,OAAO,IAAI,CAAC,KAAK,CAAC,GAAG,GAAG,UAAU,CAAC,GAAG,UAAU,CAAA;AACjD,CAAC;AAED,MAAM,UAAU,oBAAoB,CAAC,WAAoB;IACxD,MAAM,MAAM,GAAG,WAAW,CAAC,WAAW,CAAC,CAAA;IAEvC,MAAM,aAAa,GAAa,EAAE,CAAA;IAElC,KAAK,IAAI,SAAS,GAAG,CAAC,EAAE,SAAS,GAAG,OAAO,EAAE,SAAS,EAAE,EAAE,CAAC;QAC1D,MAAM,IAAI,GAAG,MAAM,CAAC,aAAa,CAAC,SAAS,CAAC,CAAA;QAE5C,IAAI,MAAM,CAAC,IAAI,CAAC,IAAI,CAAC,EAAE,CAAC;YACvB,aAAa,CAAC,IAAI,CAAC,IAAI,CAAC,CAAA;QACzB,CAAC;IACF,CAAC;IAED,OAAO,aAAa,CAAA;AACrB,CAAC;AAED,MAAM,UAAU,mBAAmB,CAAC,OAAiC;IACpE,MAAM,YAAY,GAAa,EAAE,CAAA;IAEjC,KAAK,MAAM,KAAK,IAAI,OAAO,EAAE,CAAC;QAC7B,YAAY,CAAC,IAAI,CAAC,KAAK,CAAC,WAAW,CAAC,CAAA;IACrC,CAAC;IAED,OAAO,YAAY,CAAA;AACpB,CAAC;AAED,MAAM,UAAU,oBAAoB,CAAC,QAAgB;IACpD,MAAM,SAAS,GAAG,QAAQ,CAAC,OAAO,CAAC,GAAG,CAAC,CAAA;IAEvC,IAAI,SAAS,IAAI,CAAC,CAAC,EAAE,CAAC;QACrB,OAAO,QAAQ,CAAA;IAChB,CAAC;IAED,OAAO,QAAQ,CAAC,SAAS,CAAC,CAAC,EAAE,SAAS,CAAC,CAAC,WAAW,EAAE,CAAA;AACtD,CAAC"}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@echogarden/text-segmentation",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.3.0",
|
|
4
4
|
"description": "A library for multilingual word, phrase and sentence segmentation.",
|
|
5
5
|
"author": "Rotem Dan",
|
|
6
6
|
"license": "MIT",
|
|
@@ -39,10 +39,10 @@
|
|
|
39
39
|
"regexp-composer": "../../regexp-composer"
|
|
40
40
|
},
|
|
41
41
|
"dependencies": {
|
|
42
|
-
"regexp-composer": "^0.
|
|
42
|
+
"regexp-composer": "^0.3.0"
|
|
43
43
|
},
|
|
44
44
|
"peerDependencies": {
|
|
45
|
-
"@echogarden/icu-segmentation-wasm": "^0.1
|
|
45
|
+
"@echogarden/icu-segmentation-wasm": "^0.2.1"
|
|
46
46
|
},
|
|
47
47
|
"peerDependenciesMeta": {
|
|
48
48
|
"@echogarden/icu-segmentation-wasm": {
|
|
@@ -50,6 +50,6 @@
|
|
|
50
50
|
}
|
|
51
51
|
},
|
|
52
52
|
"devDependencies": {
|
|
53
|
-
"@types/node": "^22.15.
|
|
53
|
+
"@types/node": "^22.15.17"
|
|
54
54
|
}
|
|
55
55
|
}
|
|
@@ -1,12 +1,12 @@
|
|
|
1
1
|
import { anyOf, buildRegExp, codepointRange } from 'regexp-composer'
|
|
2
2
|
|
|
3
|
-
|
|
3
|
+
const chineseCharacterRanges = anyOf(
|
|
4
4
|
codepointRange('4E00', '9FFF'), // Main
|
|
5
5
|
codepointRange('3400', '4DBF'), // CJK Unified Ideographs Extension A
|
|
6
6
|
codepointRange('20000', '2A6DF') // CJK Unified Ideographs Extension B
|
|
7
7
|
)
|
|
8
8
|
|
|
9
|
-
|
|
9
|
+
const japaneseHiraganaCharacterRanges = anyOf(
|
|
10
10
|
codepointRange('3040', '309F'), // Hiragana
|
|
11
11
|
codepointRange('1AFF0', '1AFFF'), // Kana Extended-B
|
|
12
12
|
codepointRange('1B000', '1B0FF'), // Kana Supplement
|
|
@@ -14,7 +14,7 @@ export const japaneseHiraganaCharacterRanges = anyOf(
|
|
|
14
14
|
codepointRange('1B130', '1B16F'), // Small Kana Extension
|
|
15
15
|
)
|
|
16
16
|
|
|
17
|
-
|
|
17
|
+
const japaneseKatakanaCharacterRanges = anyOf(
|
|
18
18
|
codepointRange('30A0', '30FF'), // Katakana
|
|
19
19
|
codepointRange('31F0', '31FF'), // Katakana Phonetic Extensions
|
|
20
20
|
codepointRange('3200', '32FF'), // Enclosed CJK Letters and Months
|
|
@@ -25,16 +25,16 @@ export const japaneseKatakanaCharacterRanges = anyOf(
|
|
|
25
25
|
codepointRange('1B130', '1B16F'), // Small Kana Extension
|
|
26
26
|
)
|
|
27
27
|
|
|
28
|
-
|
|
28
|
+
const thaiLetterRanges = anyOf(
|
|
29
29
|
codepointRange('0E00', '0E7F')
|
|
30
30
|
)
|
|
31
31
|
|
|
32
|
-
|
|
32
|
+
const khmerLetterRanges = anyOf(
|
|
33
33
|
codepointRange('1780', '17FF'), // Letters
|
|
34
34
|
codepointRange('19E0', '19FF'), // Symbols
|
|
35
35
|
)
|
|
36
36
|
|
|
37
|
-
|
|
37
|
+
const eastAsianCharRanges = anyOf(
|
|
38
38
|
chineseCharacterRanges,
|
|
39
39
|
japaneseHiraganaCharacterRanges,
|
|
40
40
|
japaneseKatakanaCharacterRanges,
|
package/src/Patterns.ts
CHANGED
|
@@ -1,5 +1,8 @@
|
|
|
1
|
-
import { anyOf, buildRegExp, charRange, inputEnd, inputStart, matches, oneOrMore, possibly, repeated, tab, unicodeProperty, whitespace } from 'regexp-composer'
|
|
1
|
+
import { anyOf, buildRegExp, charRange, inputEnd, inputStart, matches, oneOrMore, possibly, repeated, tab, unicodeProperty, whitespace, zeroOrMore } from 'regexp-composer'
|
|
2
2
|
|
|
3
|
+
////////////////////////////////////////////////////////////////////////////////////////////////
|
|
4
|
+
// Pattern builder methods
|
|
5
|
+
////////////////////////////////////////////////////////////////////////////////////////////////
|
|
3
6
|
export function buildWordOrNumberPattern(suppressions: string[]) {
|
|
4
7
|
return anyOf(
|
|
5
8
|
buildSuppressionPattern(suppressions),
|
|
@@ -18,95 +21,140 @@ export function buildSuppressionPattern(suppressions: string[]) {
|
|
|
18
21
|
}
|
|
19
22
|
|
|
20
23
|
////////////////////////////////////////////////////////////////////////////////////////////////
|
|
21
|
-
//
|
|
24
|
+
// Single character patterns
|
|
22
25
|
////////////////////////////////////////////////////////////////////////////////////////////////
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
+
const letterPattern = unicodeProperty('Letter')
|
|
27
|
+
const markPattern = unicodeProperty('Mark')
|
|
28
|
+
|
|
29
|
+
const letterOrMarkPattern = anyOf(
|
|
30
|
+
letterPattern,
|
|
31
|
+
markPattern,
|
|
32
|
+
)
|
|
33
|
+
|
|
34
|
+
const apostrophPattern = anyOf(`'`, `’`, `‘`)
|
|
35
|
+
|
|
36
|
+
const punctuationPattern = unicodeProperty('Punctuation')
|
|
37
|
+
const digitPattern = unicodeProperty('Decimal_Number')
|
|
38
|
+
const arabicNumeralPattern = charRange('0', '9')
|
|
39
|
+
|
|
40
|
+
const percentageCharacters = ['%']
|
|
41
|
+
const currencyCharacters = ['$', '¥', '€', '£', '₩', '₭', '₽', '₫', '฿', '¢', '₮', '؋', '₦', '₱', '₴', '₪']
|
|
26
42
|
|
|
27
|
-
|
|
43
|
+
const percentageOrCurrencyCharacterPattern = anyOf(...percentageCharacters, ...currencyCharacters)
|
|
44
|
+
|
|
45
|
+
////////////////////////////////////////////////////////////////////////////////////////////////
|
|
46
|
+
// Numeric patterns
|
|
47
|
+
////////////////////////////////////////////////////////////////////////////////////////////////
|
|
48
|
+
const numericSeparatorPattern =
|
|
28
49
|
matches(
|
|
29
50
|
anyOf('.', ',', '٬', '_'), {
|
|
30
51
|
ifPrecededBy: arabicNumeralPattern,
|
|
31
52
|
ifFollowedBy: arabicNumeralPattern
|
|
32
53
|
})
|
|
33
54
|
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
ifPrecededBy: arabicNumeralPattern,
|
|
37
|
-
ifFollowedBy: arabicNumeralPattern
|
|
38
|
-
})
|
|
55
|
+
const dimensionsPattern = matches([
|
|
56
|
+
oneOrMore(arabicNumeralPattern),
|
|
39
57
|
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
arabicNumeralPattern,
|
|
43
|
-
|
|
44
|
-
|
|
58
|
+
oneOrMore([
|
|
59
|
+
'x',
|
|
60
|
+
oneOrMore(arabicNumeralPattern),
|
|
61
|
+
]),
|
|
62
|
+
], {
|
|
63
|
+
ifPrecededBy: anyOf(whitespace, punctuationPattern, inputStart),
|
|
64
|
+
ifFollowedBy: anyOf(whitespace, punctuationPattern, inputEnd)
|
|
65
|
+
})
|
|
45
66
|
|
|
46
|
-
|
|
67
|
+
const spacedThousandsSeparatorPattern =
|
|
47
68
|
matches(
|
|
48
69
|
' ', {
|
|
49
70
|
ifPrecededBy: arabicNumeralPattern,
|
|
50
|
-
ifFollowedBy:
|
|
71
|
+
ifFollowedBy: [
|
|
72
|
+
repeated(3, arabicNumeralPattern),
|
|
73
|
+
anyOf(whitespace, punctuationPattern, inputEnd)
|
|
74
|
+
]
|
|
51
75
|
})
|
|
52
76
|
|
|
53
|
-
|
|
77
|
+
const numericSignPattern =
|
|
54
78
|
matches(
|
|
55
79
|
anyOf('-', '+'), {
|
|
56
|
-
ifPrecededBy: anyOf(whitespace, punctuationPattern),
|
|
80
|
+
ifPrecededBy: anyOf(whitespace, punctuationPattern, inputStart),
|
|
57
81
|
ifFollowedBy: arabicNumeralPattern,
|
|
58
82
|
})
|
|
59
83
|
|
|
60
|
-
|
|
61
|
-
|
|
84
|
+
const numberPattern = [
|
|
85
|
+
possibly(numericSignPattern),
|
|
86
|
+
digitPattern,
|
|
87
|
+
|
|
88
|
+
zeroOrMore(anyOf(
|
|
62
89
|
digitPattern,
|
|
63
90
|
numericSeparatorPattern,
|
|
64
91
|
spacedThousandsSeparatorPattern,
|
|
65
|
-
|
|
66
|
-
)),
|
|
92
|
+
))
|
|
67
93
|
]
|
|
68
94
|
|
|
69
|
-
const
|
|
70
|
-
|
|
95
|
+
const exponentPattern = [
|
|
96
|
+
anyOf('e', 'E'),
|
|
97
|
+
possibly(anyOf('+', '-')),
|
|
98
|
+
oneOrMore(arabicNumeralPattern),
|
|
99
|
+
]
|
|
100
|
+
|
|
101
|
+
const numberPossiblyFollowedByExponentOrLettersPattern = [
|
|
102
|
+
numberPattern,
|
|
71
103
|
|
|
72
|
-
|
|
104
|
+
possibly(anyOf(
|
|
105
|
+
exponentPattern,
|
|
106
|
+
|
|
107
|
+
matches(
|
|
108
|
+
zeroOrMore(unicodeProperty('Letter')), {
|
|
109
|
+
ifNotFollowedBy: digitPattern,
|
|
110
|
+
}),
|
|
111
|
+
))
|
|
112
|
+
]
|
|
73
113
|
|
|
74
|
-
|
|
114
|
+
const precedingPercentageOrCurrencyPattern =
|
|
75
115
|
matches([
|
|
76
|
-
|
|
116
|
+
percentageOrCurrencyCharacterPattern,
|
|
77
117
|
numberPattern,
|
|
78
118
|
], {
|
|
79
119
|
ifNotPrecededBy: digitPattern,
|
|
80
|
-
ifFollowedBy: anyOf(whitespace, punctuationPattern),
|
|
120
|
+
ifFollowedBy: anyOf(whitespace, punctuationPattern, inputEnd),
|
|
81
121
|
})
|
|
82
122
|
|
|
83
|
-
|
|
123
|
+
const followingPercentageOrCurrencyPattern =
|
|
84
124
|
matches([
|
|
85
125
|
numberPattern,
|
|
86
|
-
|
|
126
|
+
percentageOrCurrencyCharacterPattern,
|
|
87
127
|
], {
|
|
88
|
-
ifPrecededBy: anyOf(whitespace, punctuationPattern),
|
|
89
|
-
ifNotFollowedBy: digitPattern,
|
|
128
|
+
ifPrecededBy: anyOf(whitespace, punctuationPattern, inputStart),
|
|
129
|
+
ifNotFollowedBy: anyOf(digitPattern),
|
|
90
130
|
})
|
|
91
131
|
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
132
|
+
const percentageOrCurrencyPattern = anyOf(
|
|
133
|
+
precedingPercentageOrCurrencyPattern,
|
|
134
|
+
followingPercentageOrCurrencyPattern
|
|
95
135
|
)
|
|
96
136
|
|
|
97
137
|
////////////////////////////////////////////////////////////////////////////////////////////////
|
|
98
|
-
//
|
|
138
|
+
// Time patterns
|
|
99
139
|
////////////////////////////////////////////////////////////////////////////////////////////////
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
140
|
+
const timePattern = matches([
|
|
141
|
+
anyOf(arabicNumeralPattern, [charRange('0', '1'), arabicNumeralPattern], ['2', charRange('0', '3')]),
|
|
142
|
+
|
|
143
|
+
repeated([1, 2], [
|
|
144
|
+
':',
|
|
145
|
+
anyOf(arabicNumeralPattern, [charRange('0', '5'), arabicNumeralPattern]),
|
|
146
|
+
])
|
|
147
|
+
], {
|
|
148
|
+
ifPrecededBy: anyOf(whitespace, punctuationPattern, inputStart),
|
|
149
|
+
ifNotPrecededBy: ':',
|
|
150
|
+
ifFollowedBy: anyOf(whitespace, punctuationPattern, inputEnd),
|
|
151
|
+
ifNotFollowedBy: ':',
|
|
152
|
+
})
|
|
108
153
|
|
|
109
|
-
|
|
154
|
+
////////////////////////////////////////////////////////////////////////////////////////////////
|
|
155
|
+
// Letter patterns
|
|
156
|
+
////////////////////////////////////////////////////////////////////////////////////////////////
|
|
157
|
+
const dottedAbbreviationSequencePattern =
|
|
110
158
|
matches(
|
|
111
159
|
anyOf(
|
|
112
160
|
[
|
|
@@ -135,23 +183,14 @@ export const dottedAbbreviationSequencePattern =
|
|
|
135
183
|
ifNotFollowedBy: anyOf(letterOrMarkPattern, digitPattern)
|
|
136
184
|
})
|
|
137
185
|
|
|
138
|
-
|
|
139
|
-
export const wordCharacterPattern =
|
|
186
|
+
const wordCharacterPattern =
|
|
140
187
|
anyOf(
|
|
141
188
|
letterPattern,
|
|
142
189
|
markPattern,
|
|
143
190
|
digitPattern,
|
|
144
191
|
)
|
|
145
192
|
|
|
146
|
-
|
|
147
|
-
matches(
|
|
148
|
-
anyOf('-', '_', '·', '‧', '&'), {
|
|
149
|
-
|
|
150
|
-
ifPrecededBy: letterOrMarkPattern,
|
|
151
|
-
ifFollowedBy: letterOrMarkPattern
|
|
152
|
-
})
|
|
153
|
-
|
|
154
|
-
export const wordInnerApostrophPattern =
|
|
193
|
+
const wordInnerApostrophPattern =
|
|
155
194
|
matches(
|
|
156
195
|
apostrophPattern, {
|
|
157
196
|
|
|
@@ -159,67 +198,118 @@ export const wordInnerApostrophPattern =
|
|
|
159
198
|
ifFollowedBy: letterOrMarkPattern,
|
|
160
199
|
})
|
|
161
200
|
|
|
162
|
-
|
|
201
|
+
const wordStartApostrophPattern =
|
|
163
202
|
matches(
|
|
164
203
|
apostrophPattern, {
|
|
165
204
|
|
|
166
205
|
ifPrecededBy: whitespace,
|
|
167
|
-
ifFollowedBy: [letterOrMarkPattern, letterOrMarkPattern, whitespace]
|
|
206
|
+
ifFollowedBy: [letterOrMarkPattern, letterOrMarkPattern, anyOf(whitespace, inputEnd)]
|
|
168
207
|
})
|
|
169
208
|
|
|
170
|
-
|
|
209
|
+
const basicWordPattern =
|
|
171
210
|
oneOrMore(
|
|
172
211
|
anyOf(
|
|
173
212
|
wordCharacterPattern,
|
|
174
|
-
wordSeparatorPattern,
|
|
175
213
|
wordInnerApostrophPattern,
|
|
176
214
|
),
|
|
177
215
|
)
|
|
178
216
|
|
|
179
|
-
|
|
217
|
+
const hyphenatedWordPattern = [
|
|
218
|
+
basicWordPattern,
|
|
219
|
+
|
|
220
|
+
oneOrMore([
|
|
221
|
+
'-',
|
|
222
|
+
basicWordPattern,
|
|
223
|
+
]),
|
|
224
|
+
]
|
|
225
|
+
|
|
226
|
+
const dotSeparatedWordPattern = matches([
|
|
227
|
+
basicWordPattern,
|
|
228
|
+
|
|
229
|
+
oneOrMore([
|
|
230
|
+
'.',
|
|
231
|
+
basicWordPattern,
|
|
232
|
+
]),
|
|
233
|
+
], {
|
|
234
|
+
//ifPrecededBy: anyOf(whitespace, inputStart),
|
|
235
|
+
//ifFollowedBy: anyOf(whitespace, inputEnd)
|
|
236
|
+
})
|
|
237
|
+
|
|
238
|
+
const interpunctSeparatedWordPattern = [
|
|
239
|
+
basicWordPattern,
|
|
240
|
+
|
|
241
|
+
oneOrMore([
|
|
242
|
+
anyOf('·', '‧'),
|
|
243
|
+
basicWordPattern,
|
|
244
|
+
]),
|
|
245
|
+
]
|
|
246
|
+
|
|
247
|
+
const underscoreSeparatedWordPattern = [
|
|
248
|
+
basicWordPattern,
|
|
249
|
+
|
|
250
|
+
oneOrMore([
|
|
251
|
+
'_',
|
|
252
|
+
basicWordPattern,
|
|
253
|
+
]),
|
|
254
|
+
]
|
|
255
|
+
|
|
256
|
+
const wordSegmentPattern = anyOf(
|
|
257
|
+
timePattern,
|
|
258
|
+
dimensionsPattern,
|
|
259
|
+
percentageOrCurrencyPattern,
|
|
260
|
+
numberPossiblyFollowedByExponentOrLettersPattern,
|
|
261
|
+
|
|
262
|
+
interpunctSeparatedWordPattern,
|
|
263
|
+
dotSeparatedWordPattern,
|
|
180
264
|
dottedAbbreviationSequencePattern,
|
|
181
|
-
|
|
182
|
-
percentagePattern,
|
|
183
|
-
numberPattern,
|
|
265
|
+
underscoreSeparatedWordPattern,
|
|
184
266
|
basicWordPattern,
|
|
185
267
|
)
|
|
186
268
|
|
|
187
269
|
////////////////////////////////////////////////////////////////////////////////////////////////
|
|
188
|
-
//
|
|
270
|
+
// Phrase separation patterns
|
|
189
271
|
////////////////////////////////////////////////////////////////////////////////////////////////
|
|
190
|
-
|
|
272
|
+
const phraseSeparatorCharacters = [',', '、', ',', '،', ';', ';', ':', ':', '—']
|
|
191
273
|
|
|
192
|
-
|
|
274
|
+
const phraseSeparatorPattern = [
|
|
193
275
|
inputStart,
|
|
194
|
-
anyOf(...
|
|
276
|
+
anyOf(...phraseSeparatorCharacters),
|
|
195
277
|
inputEnd
|
|
196
|
-
]
|
|
197
|
-
|
|
198
|
-
export const sentenceSeparators = ['.', '。', '?', '?', '!', '!', '\n']
|
|
278
|
+
]
|
|
199
279
|
|
|
200
|
-
|
|
280
|
+
const phraseSeparatorTrailingPunctuationPattern = [
|
|
201
281
|
inputStart,
|
|
202
|
-
anyOf(...
|
|
282
|
+
anyOf(...phraseSeparatorCharacters, ' ', tab),
|
|
203
283
|
inputEnd
|
|
204
|
-
]
|
|
284
|
+
]
|
|
205
285
|
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
])
|
|
286
|
+
////////////////////////////////////////////////////////////////////////////////////////////////
|
|
287
|
+
// Sentence separation patterns
|
|
288
|
+
////////////////////////////////////////////////////////////////////////////////////////////////
|
|
289
|
+
const sentenceSeparatorCharacters = ['.', '。', '?', '?', '!', '!', '\n']
|
|
211
290
|
|
|
212
|
-
|
|
291
|
+
const sentenceSeparatorPattern = [
|
|
213
292
|
inputStart,
|
|
214
|
-
anyOf(...
|
|
293
|
+
anyOf(...sentenceSeparatorCharacters),
|
|
215
294
|
inputEnd
|
|
216
|
-
]
|
|
295
|
+
]
|
|
217
296
|
|
|
218
|
-
|
|
297
|
+
const sentenceSeparatorTrailingPunctuationPattern = [
|
|
219
298
|
inputStart,
|
|
220
|
-
|
|
299
|
+
anyOf('"', '”', '’', ')', ']', '}', '»', ...sentenceSeparatorCharacters, ...phraseSeparatorCharacters, oneOrMore(whitespace)),
|
|
221
300
|
inputEnd
|
|
222
|
-
]
|
|
301
|
+
]
|
|
223
302
|
|
|
303
|
+
////////////////////////////////////////////////////////////////////////////////////////////////
|
|
304
|
+
// Prebuilt regular expressions
|
|
305
|
+
////////////////////////////////////////////////////////////////////////////////////////////////
|
|
224
306
|
export const wordCharacterRegExp = buildRegExp(wordCharacterPattern)
|
|
225
307
|
export const whitespacePatternRegExp = buildRegExp(whitespace)
|
|
308
|
+
|
|
309
|
+
export const letterPatternGlobalRegExp = buildRegExp(letterPattern, { global: true })
|
|
310
|
+
|
|
311
|
+
export const phraseSeparatorRegExp = buildRegExp(phraseSeparatorPattern)
|
|
312
|
+
export const phraseSeparatorTrailingPunctuationRegExp = buildRegExp(phraseSeparatorTrailingPunctuationPattern)
|
|
313
|
+
|
|
314
|
+
export const sentenceSeparatorRegExp = buildRegExp(sentenceSeparatorPattern)
|
|
315
|
+
export const sentenceSeparatorTrailingPunctuationRegExp = buildRegExp(sentenceSeparatorTrailingPunctuationPattern)
|
package/src/Suppressions.ts
CHANGED
|
@@ -1,16 +1,27 @@
|
|
|
1
1
|
export const cldrSuppressions: Record<string, string[]> = {
|
|
2
|
-
|
|
3
|
-
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
|
|
2
|
+
//en: ['L.P.', 'Alt.', 'Approx.', 'E.G.', 'O.', 'Maj.', 'Misc.', 'P.O.', 'J.D.', 'Jam.', 'Card.', 'Dec.', 'Sept.', 'MR.', 'Long.', 'Hat.', 'G.', 'Link.', 'DC.', 'D.C.', 'M.T.', 'Hz.', 'Mrs.', 'By.', 'Act.', 'Var.', 'N.V.', 'Aug.', 'B.', 'S.A.', 'Up.', 'Job.', 'Num.', 'M.I.T.', 'Ok.', 'Org.', 'Ex.', 'Cont.', 'U.', 'Mart.', 'Fn.', 'Abs.', 'Lt.', 'OK.', 'Z.', 'E.', 'Kb.', 'Est.', 'A.M.', 'L.A.', 'Prof.', 'U.S.', 'Nov.', 'Ph.D.', 'Mar.', 'I.T.', 'exec.', 'Jan.', 'N.Y.', 'X.', 'Md.', 'Op.', 'vs.', 'D.A.', 'A.D.', 'R.L.', 'P.M.', 'Or.', 'M.R.', 'Cap.', 'PC.', 'Feb.', 'Exec.', 'I.e.', 'Sep.', 'Gb.', 'K.', 'U.S.C.', 'Mt.', 'S.', 'A.S.', 'C.O.D.', 'Capt.', 'Col.', 'In.', 'C.F.', 'Adj.', 'AD.', 'I.D.', 'Mgr.', 'R.T.', 'B.V.', 'M.', 'Conn.', 'Yr.', 'Rev.', 'Phys.', 'pp.', 'Ms.', 'To.', 'Sgt.', 'J.K.', 'Nr.', 'Jun.', 'Fri.', 'S.A.R.', 'Lev.', 'Lt.Cdr.', 'Def.', 'F.', 'Do.', 'Joe.', 'Id.', 'Mr.', 'Dept.', 'Is.', 'Pvt.', 'Diff.', 'Hon.B.A.', 'Q.', 'Mb.', 'On.', 'Min.', 'J.B.', 'Ed.', 'AB.', 'A.', 'S.p.A.', 'I.', 'a.m.', 'Comm.', 'Go.', 'VS.', 'L.', 'All.', 'PP.', 'P.V.', 'T.', 'K.R.', 'Etc.', 'D.', 'Adv.', 'Lib.', 'E.g.', 'Pro.', 'U.S.A.', 'S.E.', 'AA.', 'Rep.', 'Sq.', 'As.'],
|
|
3
|
+
en: ['L.P.', 'Alt.', 'Approx.', 'E.G.', 'Maj.', 'Misc.', 'P.O.', 'J.D.', 'Dec.', 'Sept.', 'MR.', 'DC.', 'D.C.', 'M.T.', 'Mrs.', 'N.V.', 'Aug.', 'S.A.', 'Num.', 'M.I.T.', 'Org.', 'Ex.', 'Cont.', 'Mart.', 'Fn.', 'Abs.', 'Lt.', 'Est.', 'A.M.', 'L.A.', 'Prof.', 'U.S.', 'Nov.', 'Ph.D.', 'Mar.', 'I.T.', 'exec.', 'Jan.', 'N.Y.', 'Md.', 'Op.', 'vs.', 'D.A.', 'A.D.', 'R.L.', 'P.M.', 'M.R.', 'Feb.', 'Exec.', 'I.e.', 'Sep.', 'U.S.C.', 'Mt.', 'A.S.', 'C.O.D.', 'Capt.', 'Col.', 'C.F.', 'Adj.', 'I.D.', 'Mgr.', 'R.T.', 'B.V.', 'M.', 'Conn.', 'Yr.', 'Rev.', 'Phys.', 'pp.', 'Ms.', 'Sgt.', 'J.K.', 'Nr.', 'Jun.', 'Fri.', 'S.A.R.', 'Lev.', 'Lt.Cdr.', 'Def.', 'Mr.', 'Dept.', 'Pvt.', 'Diff.', 'Hon.B.A.', 'Mb.', 'Min.', 'J.B.', 'Ed.', 'AB.', 'S.p.A.', 'I.', 'a.m.', 'Comm.', 'VS.', 'PP.', 'P.V.', 'K.R.', 'Etc.', 'Adv.', 'Lib.', 'E.g.', 'U.S.A.', 'S.E.', 'AA.', 'Rep.', 'Sq.'],
|
|
4
|
+
de: ['Port.', 'Alt.', 'Di.', 'Ges.', 'frz.', 'entspr.', 'Gebr.', 'erw.', 'Frl.', 'Inh.', 'k.u.k.', 'Ca.', 'J.D.', 'Ausg.', 'evtl.', 'So.', 'i.B.', 's.a.', 'kgl.', 'Sept.', 'o.B.', 'Sa.', 'ev.', 'Dez.', 'am.', 'i.R.', 'eigtl.', 'i.J.', 'u.U.', 'G.', 'z.Hd.', 'u.A.w.g.', 'Kl.', 'Spezif.', 'Obj.', 'Ing.', 'D. h.', 'Folg.', 'Akt.', 'i.A.', 'Msp.', 'U.U.', 'Chr.', 'R.', 'Einh.', 'schwäb.', 'Vgl.', 'Aug.', 'Dipl.-Ing.', 'W.', 'B.', 'U. U.', 'J.', 'Fa.', 'Mo.', 'n.u.Z.', 'Op.', 'Mrd.', 'e.h.', 'Hr.', 'Hrn.', 'Ztr.', 'k. u. k.', 'Bibl.', 'd.Ä.', 'b.', 'M.', 'i.H.', 'v.R.w.', 'o.A.', 'St.', 'Dr.', 'Fn.', 'Abs.', 'Rd.', 'Dtzd.', 'Jahrh.', 'Z.', 'Std.', 'n. Chr.', 'möbl.', 'tägl.', 'gest.', 'gesch.', 'z.B.', 'Hbf.', 'Abt.', 'A.M.', 'e.Wz.', 'v.T.', 'Nov.', 'z.', 'Prot.', 'U.S.', 'Wg.', 'u.v.a.', 'Adr.', 'App.', 'ggf.', 'ggfs.', 'Jan.', 'O.', 'Rel.', 'od.', 'Pfd.', 'a.a.O.', 'p.Adr.', 'P.', 'Gem.', 'v. Chr.', 'Art.', 'z.Z.', 'S.A.', 'i.V.', 'verh.', 'Ausschl.', 'm.W.', 'Dir.', 'Verf.', 'Sek.', 'r.', 'Chin.', 'Feb.', 'Int.', 'Sep.', 'Gesch.', 'schweiz.', 'Bed.', 'a.Rh.', 'jew.', 'vgl.', 'a.M.', 'Str.', 'exkl.', 'gek.', 'Erf.', 'u.Ä.', 'ehem.', 'näml.', 'u. Z.', 'v. u. Z.', 'sog.', 'C.', 'Dipl.-Kfm.', 'mtl.', 'Hrsg.', 'Qu.', 'röm.', 'u.', 'U.', 'Adj.', 'Kap.', 'hpts.', 'a.D.', 'gedr.', 'Best.', 'N.', 'v.u.Z.', 'Phys.', 'Fr.', 'd.J.', 'Reg.-Bez.', 'm.E.', 'schles.', 'Max.', 'Ltd.', 'südd.', 'inkl.', 'geb.', 'Ggf.', 'Inc.', 'kath.', 'kfm.', 'Nr.', 'Proz.', 'Dim.', 'verw.', 'Reg.', 'Dat.', 'Evtl.', 'led.', 'F.', 'Test.', 'Schr.', 'Do.', 'PIN.', 'Z. Zt.', 'v.Chr.', 'Tägl.', 's.', 'amtl.', 'Temp.', 'Mind.', 'e.V.', 'Abw.', 'P.M.', 'F.f.', 'a.a.S.', 'Mod.', 'Co.', 'Min.', 'Allg.', 'Geograph.', 'Jr.', 'Urspr.', 'Apr.', 'Z. B.', 'v.H.', 'A.', 'einschl.', 'Trans.', 'zzgl.', 'StR.', 'Fam.', 'I.', 'jhrl.', 'u.a.', 'Ben.', 'o.g.', 'Kfm.', 'Konv.', 'Mi.', 'L.', 'beil.', 'T.', 'Ursprüngl.', 'röm.-kath.', 'Okt.', 'u.ä.', 'Tel.', 'D.', 'Ber.', 'Kop.', 'Mio.', 'Y.', 'U.S.A.', 'v. H.', 'Forts. f.', 'Rep.', 'Hptst.', 'österr.'],
|
|
5
|
+
es: ['Rdos.', 'JJ.OO.', 'Sres.', 'fig.', 'may.', 'RR.HH.', 'oct.', 'cap.', 'mié.', 'doc.', 'Excmo.', 'Trab.', 'Excmos.', 'Kit.', 'Inc.', 'FF.CC.', 'DC.', 'ago.', 'trad.', 'SA.', 'Rvdos.', 'ed.', 'Exmo.', 'jul.', 'col.', 'RAM.', 'Srtas.', 'ene.', 'Rol.', 'Fabric.', 'Comm.', 'vid.', 'Da.', 'dic.', 'ss.', 'abr.', 'ntra.', 'Sra.', 'dtor.', 'cf.', 'dom.', 'prov.', 'Emm.', 'Sr.', 'licdo.', 'p.ej.', 'bol.', 'figs.', 'Vda.', 'Dr.', 'ntro.', 'Desv.', 'O.M.', 'Ldo.', 'Drs.', 'sáb.', 'feb.', 'Ltda.', 'Lcda.', 'Exma.', 'C.V.', 'SS.MM.', 'Lda.', 'U.S.', 'hnos.', 'R.D.', 'Korn.', 'v.gr.', 'vs.', 'Ilmas.', 'Rdo.', 'ej.', 'vie.', 'jue.', 'a. C.', 'Ilmos.', 'e. c.', 'Excma.', 'afma.', 'licda.', 'Em.', 'K.', 'sras.', 'MM.', 'fund.', 'Mons.', 'Lcdo.', 'afmo.', 'C.', 'A.C.', 'dptos.', 'Col.', 'Srta.', 'Av.', 'Ant.', 'depto.', 'Var.', 'H.P.', 'D.', 'M.', 'C.P.', 'Rev.', 'Rvdmos.', 'Fr.', 'Ilmo.', 'afmos.', 'Ltd.', 'afmas.', 'prof.', 'lun.', 'SS.AA.', 'Sol.', 'nov.', 'mss.', 'Dña.', 'Seg.', 'mar.', 'Rvdmo.', 'Reg.', 'ms.', 'Sras.', 'sres.', 'U.S.A.', 'Sta.', 'Sdad.', 'Dra.', 'srs.', 'R.U.', 'deptos.', 'dpto.', 'jun.', 'bco.', 'Cía.', 'Id.', 'Mr.', 'e.g.', 'C.S.', 'Excmas.', 'Dª.', 'Rvdo.', 'Lic.', 'cfr.', 'Corp.', 'Dto.', 'Ilma.', 'L.', 'All.', 'PP.', 'd. C.', 'Ltdo.', 'mtro.', 'Mrs.', 'Desc.', 'Avda.', 'Exmas.', 'a. e. c.', 'Bien.', 'Exmos.', 'AA.', 'Sto.', 'CA.', 'sept.', 'Exc.', 'c/c.'],
|
|
6
|
+
fr: ['aux.', 'config.', 'collab.', 'M.', 'dim.', 'imprim.', 'oct.', 'syst.', 'bull.', 'MM.', 'doc.', 'P.O.', 'hôp.', 'Mart.', 'juil.', 'broch.', 'adr.', 'symb.', 'C.', 'anc.', 'voit.', 'Jr.', 'graph.', 'dir.', 'éd.', 'fig.', 'édit.', 'niv.', 'quart.', 'cam.', 'éval.', 'anon.', 'réf.', 'Comm.', 'Prof.', 'févr.', 'indus.', 'DC.', 'équiv.', 'illustr.', 'acoust.', 'nov.', 'L.', 'All.', 'U.S.', 'S.M.A.R.T.', 'sept.', 'avr.', 'jeu.', 'dest.', 'P.-D. G.', 'ill.', 'coll.', 'encycl.', 'mer.', 'Desc.', 'ven.', 'P.', 'lun.', 'Inc.', 'sam.', 'D.', 'append.', 'Var.', 'categ.', 'janv.', 'S.A.', 'imm.', 'U.S.A.', 'mar.', 'exempl.', 'déc.', 'ann.', 'U.', 'synth.', 'dict.', 'av. J.-C.', 'W.', 'Op.', 'ap. J.-C.', 'gouv.', 'trav. publ.'],
|
|
7
|
+
it: ['N.B.', 'div.', 'a.C.', 'fig.', 'd.p.R.', 'c.c.p.', 'Cfr.', 'vol.', 'Geom.', 'O.d.G.', 'S.p.A.', 'ver.', 'N.d.A.', 'dott.', 'arch.', 'd.C.', 'N.d.T.', 'rag.', 'Sig.', 'Mod.', 'pag.', 'dr.', 'tav.', 'N.d.E.', 'DC.', 'mitt.', 'Ing.', 'int.', 'on.', 'C.P.', 'ag.', 'L.', 'U.S.', 'S.M.A.R.T.', 'p.i.', 'tab.', 'Ltd.', 'Liv.', 'D.', 'U.S.A.', 'sez.', 'avv.', 'S.A.R.', 'all.', 'p.'],
|
|
8
|
+
pt: ['psicol.', 'fig.', 'compl.', 'rep.', 'cap.', 'doc.', 'fisiol.', 'dipl.', 'astron.', 'port.', 'eletrôn.', 'geom.', 'mov.', 'ago.', 'trad.', 'arquit.', 'dez.', 'ed.', 'apt.', 'Exmo.', 'col.', 'ff.', 'univ.', 'res.', 'R.', 'transp.', 'D.C', 'l.', 'des.', 'fev.', 'abr.', 'liter.', 'lat.', 'Dir.', 'cf.', 'adm.', 'fot.', 'p.m.', 'P.M.', 'créd.', 'jur.', 'com.', 'anat.', 'dir.', 'end.', 'fís.', 'E.', 'Est.', 'cont.', 'matem.', 'Drs.', 'gên.', 'neol.', 'pág.', 'índ.', 'Ltda.', 'Exma.', 'esp.', 'ingl.', 'tecnol.', 'Mar.', 'símb.', 'Pe.', 'pal.', 'filos.', 'V.T.', 'fasc.', 'vs.', 'mai.', 'S.A.', 'profa.', 'N.Sra.', 'r.s.v.p.', 'cel.', 'mat.', 'abrev.', 'out.', 'long.', 'aux.', 'arit.', 'aer.', 'jul.', 'lin.', 'S.', 'méd.', 'odontol.', 'org.', 'A.C.', 'jun.', 'déb.', 'Av.', 'álg.', 'sup.', 'fl.', 'odont.', 'caps.', 'relat.', 'organiz.', 'hist.', 'Fr.', 'Ilmo.', 'fem.', 'ap.', 'Ltd.', 'pol.', 'séc.', 'prof.', 'cx.', 'nov.', 'quím.', 'mús.', 'agric.', 'mar.', 'W.C.', 'fr.', 'cat.', 'jan.', 'pron.', 'rel.', 'autom.', 'Sta.', 'Dra.', 'p.', 'tel.', 'div.', 'p. ex.', 'a.C.', 'bras.', 'Alm.', 'Dr.', 'comp.', 'pq.', 'arqueol.', 'náut.', 'biogr.', 'f.', 'círc.', 'fac.', 'd.C.', 'apart.', 'ex.', 'Jr.', 'set.', 'tec.', 'sociol.', 'gram.', 'ind.', 'Ilma.', 'vol.', 'eng.', 'rod.', 'Ph.D.', 'Dras.', 'pp.', 'elem.', 'máq.', 'cód.', 'eletr.', 'prod.', 'ref.', 'fil.', 'a.m.', 'A.M', 'obs.', 'N.T.', 'contab.', 'Sto.', 'lit.', 'educ.', 'rementente', 'desc.', 'próx.'],
|
|
9
|
+
ru: ['руб.', 'янв.', 'до н. э.', 'сент.', 'тел.', 'дек.', 'февр.', 'нояб.', 'апр.', 'н. э.', 'окт.', 'тыс.', 'авг.', 'проф.', 'н.э.', 'кв.', 'ул.', 'отд.'],
|
|
10
|
+
}
|
|
11
|
+
|
|
12
|
+
export const additionalSuppressions: Record<string, string[] > = {
|
|
13
|
+
en: [
|
|
14
|
+
'TL;DR'
|
|
15
|
+
],
|
|
9
16
|
}
|
|
10
17
|
|
|
11
18
|
export const leadingApostropheContractionSuppressions: Record<string, string[]> = {
|
|
12
|
-
'en': [
|
|
13
|
-
|
|
19
|
+
'en': [
|
|
20
|
+
`'cause`, `'til`, `'bout`, `'twas`, `'tis`
|
|
21
|
+
],
|
|
22
|
+
'af': [
|
|
23
|
+
`'n`
|
|
24
|
+
]
|
|
14
25
|
}
|
|
15
26
|
|
|
16
27
|
export const nounSuppressions = [
|