@echogarden/text-segmentation 0.2.0 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1 +1 @@
1
- {"version":3,"file":"WordSequence.js","sourceRoot":"","sources":["../src/WordSequence.ts"],"names":[],"mappings":"AAAA,MAAM,OAAO,YAAY;IACxB,OAAO,GAAgB,EAAE,CAAA;IAEzB,YAAY,CAAC,cAAsB,EAAE,WAAmB,EAAE,SAAiB,EAAE,aAAsB;QAClG,MAAM,QAAQ,GAAG,cAAc,CAAC,SAAS,CAAC,WAAW,EAAE,SAAS,CAAC,CAAA;QAEjE,IAAI,CAAC,OAAO,CAAC,IAAI,CAAC;YACjB,IAAI,EAAE,QAAQ;YACd,WAAW;YACX,SAAS;YACT,aAAa;SACb,CAAC,CAAA;IACH,CAAC;IAED,YAAY,CAAC,UAAkB,EAAE,QAAgB;QAChD,OAAO,CAAC,GAAG,IAAI,CAAC,gBAAgB,CAAC,UAAU,EAAE,QAAQ,CAAC,CAAC,CAAA;IACxD,CAAC;IAED,CAAC,gBAAgB,CAAC,UAAkB,EAAE,QAAgB;QACrD,KAAK,IAAI,CAAC,GAAG,UAAU,EAAE,CAAC,GAAG,QAAQ,EAAE,CAAC,EAAE,EAAE,CAAC;YAC5C,MAAM,IAAI,CAAC,SAAS,CAAC,CAAC,CAAC,CAAA;QACxB,CAAC;IACF,CAAC;IAED,SAAS,CAAC,KAAa;QACtB,OAAO,IAAI,CAAC,OAAO,CAAC,KAAK,CAAC,CAAC,IAAI,CAAA;IAChC,CAAC;IAED,aAAa,CAAC,UAAkB,EAAE,QAAgB;QACjD,OAAO,IAAI,CAAC,OAAO,CAAC,KAAK,CAAC,UAAU,EAAE,QAAQ,CAAC,CAAA;IAChD,CAAC;IAED,CAAC,iBAAiB,CAAC,UAAkB,EAAE,QAAgB;QACtD,KAAK,IAAI,CAAC,GAAG,UAAU,EAAE,CAAC,GAAG,QAAQ,EAAE,CAAC,EAAE,EAAE,CAAC;YAC5C,MAAM,IAAI,CAAC,UAAU,CAAC,CAAC,CAAC,CAAA;QACzB,CAAC;IACF,CAAC;IAED,UAAU,CAAC,KAAa;QACvB,OAAO,IAAI,CAAC,OAAO,CAAC,KAAK,CAAC,CAAA;IAC3B,CAAC;IAED,KAAK;QACJ,OAAO,IAAI,CAAC,KAAK,CAAC,CAAC,EAAE,IAAI,CAAC,MAAM,CAAC,CAAA;IAClC,CAAC;IAED,KAAK,CAAC,UAAkB,EAAE,QAAgB;QACzC,MAAM,cAAc,GAAG,IAAI,YAAY,EAAE,CAAA;QAEzC,cAAc,CAAC,OAAO,GAAG,IAAI,CAAC,OAAO,CAAC,KAAK,CAAC,UAAU,EAAE,QAAQ,CAAC,CAAA;QAEjE,OAAO,cAAc,CAAA;IACtB,CAAC;IAED,IAAI,KAAK;QACR,OAAO,IAAI,CAAC,OAAO,CAAC,GAAG,CAAC,KAAK,CAAC,EAAE,CAAC,KAAK,CAAC,IAAI,CAAC,CAAA;IAC7C,CAAC;IAED,IAAI,SAAS;QACZ,OAAO,IAAI,CAAC,UAAU,EAAE,IAAI,CAAA;IAC7B,CAAC;IAED,IAAI,QAAQ;QACX,OAAO,IAAI,CAAC,SAAS,EAAE,IAAI,CAAA;IAC5B,CAAC;IAED,IAAI,UAAU;QACb,OAAO,IAAI,CAAC,OAAO,CAAC,CAAC,CAAC,CAAA;IACvB,CAAC;IAED,IAAI,SAAS;QACZ,OAAO,IAAI,CAAC,OAAO,CAAC,IAAI,CAAC,MAAM,GAAG,CAAC,CAAC,CAAA;IACrC,CAAC;IAED,IAAI,qBAAqB;QACxB,OAAO,IAAI,CAAC,OAAO,CAAC,MAAM,CAAC,KAAK,CAAC,EAAE,CAAC,KAAK,CAAC,aAAa,KAAK,KAAK,CAAC,CAAA;IACnE,CAAC;IAED,IAAI,mBAAmB;QACtB,OAAO,IAAI,CAAC,OAAO,CAAC,MAAM,CAAC,KAAK,CAAC,EAAE,CAAC,KAAK,CAAC,aAAa,KAAK,KAAK,CAAC,CAAC,GAAG,CAAC,KAAK,CAAC,EAAE,CAAC,KAAK,CAAC,IAAI,CAAC,CAAA;IAC5F,CAAC;IAED,IAAI,IAAI;QACP,OAAO,IAAI,CAAC,KAAK,CAAC,IAAI,CAAC,EAAE,CAAC,CAAA;IAC3B,CAAC;IAED,IAAI,MAAM;QACT,OAAO,IAAI,CAAC,OAAO,CAAC,MAAM,CAAA;IAC3B,CAAC;CACD"}
1
+ {"version":3,"file":"WordSequence.js","sourceRoot":"","sources":["../src/WordSequence.ts"],"names":[],"mappings":"AAAA,MAAM,OAAO,YAAY;IACxB,OAAO,GAAgB,EAAE,CAAA;IAEzB,OAAO,CAAC,IAAY,EAAE,eAAuB,EAAE,aAAsB;QACpE,MAAM,WAAW,GAAG,eAAe,CAAA;QACnC,MAAM,SAAS,GAAG,WAAW,GAAG,IAAI,CAAC,MAAM,CAAA;QAE3C,IAAI,CAAC,OAAO,CAAC,IAAI,CAAC;YACjB,IAAI;YACJ,WAAW;YACX,SAAS;YACT,aAAa;SACb,CAAC,CAAA;IACH,CAAC;IAED,YAAY,CAAC,UAAkB,EAAE,QAAgB;QAChD,OAAO,IAAI,CAAC,aAAa,CAAC,UAAU,EAAE,QAAQ,CAAC,CAAC,GAAG,CAAC,KAAK,CAAC,EAAE,CAAC,KAAK,CAAC,IAAI,CAAC,CAAA;IACzE,CAAC;IAED,CAAC,gBAAgB,CAAC,UAAkB,EAAE,QAAgB;QACrD,KAAK,IAAI,CAAC,GAAG,UAAU,EAAE,CAAC,GAAG,QAAQ,EAAE,CAAC,EAAE,EAAE,CAAC;YAC5C,MAAM,IAAI,CAAC,SAAS,CAAC,CAAC,CAAC,CAAA;QACxB,CAAC;IACF,CAAC;IAED,SAAS,CAAC,KAAa;QACtB,OAAO,IAAI,CAAC,OAAO,CAAC,KAAK,CAAC,CAAC,IAAI,CAAA;IAChC,CAAC;IAED,aAAa,CAAC,UAAkB,EAAE,QAAgB;QACjD,OAAO,IAAI,CAAC,OAAO,CAAC,KAAK,CAAC,UAAU,EAAE,QAAQ,CAAC,CAAA;IAChD,CAAC;IAED,CAAC,iBAAiB,CAAC,UAAkB,EAAE,QAAgB;QACtD,KAAK,IAAI,CAAC,GAAG,UAAU,EAAE,CAAC,GAAG,QAAQ,EAAE,CAAC,EAAE,EAAE,CAAC;YAC5C,MAAM,IAAI,CAAC,UAAU,CAAC,CAAC,CAAC,CAAA;QACzB,CAAC;IACF,CAAC;IAED,UAAU,CAAC,KAAa;QACvB,OAAO,IAAI,CAAC,OAAO,CAAC,KAAK,CAAC,CAAA;IAC3B,CAAC;IAED,KAAK;QACJ,OAAO,IAAI,CAAC,KAAK,CAAC,CAAC,EAAE,IAAI,CAAC,MAAM,CAAC,CAAA;IAClC,CAAC;IAED,KAAK,CAAC,UAAkB,EAAE,QAAgB;QACzC,MAAM,cAAc,GAAG,IAAI,YAAY,EAAE,CAAA;QAEzC,cAAc,CAAC,OAAO,GAAG,IAAI,CAAC,OAAO,CAAC,KAAK,CAAC,UAAU,EAAE,QAAQ,CAAC,CAAA;QAEjE,OAAO,cAAc,CAAA;IACtB,CAAC;IAED,IAAI,SAAS;QACZ,OAAO,IAAI,CAAC,OAAO,CAAC,GAAG,CAAC,KAAK,CAAC,EAAE,CAAC,KAAK,CAAC,IAAI,CAAC,CAAA;IAC7C,CAAC;IAED,IAAI,SAAS;QACZ,OAAO,IAAI,CAAC,UAAU,EAAE,IAAI,CAAA;IAC7B,CAAC;IAED,IAAI,QAAQ;QACX,OAAO,IAAI,CAAC,SAAS,EAAE,IAAI,CAAA;IAC5B,CAAC;IAED,IAAI,UAAU;QACb,OAAO,IAAI,CAAC,OAAO,CAAC,CAAC,CAAC,CAAA;IACvB,CAAC;IAED,IAAI,SAAS;QACZ,OAAO,IAAI,CAAC,OAAO,CAAC,IAAI,CAAC,MAAM,GAAG,CAAC,CAAC,CAAA;IACrC,CAAC;IAED,IAAI,kBAAkB;QACrB,OAAO,IAAI,CAAC,OAAO,CAAC,MAAM,CAAC,KAAK,CAAC,EAAE,CAAC,KAAK,CAAC,aAAa,KAAK,IAAI,CAAC,CAAA;IAClE,CAAC;IAED,IAAI,gBAAgB;QACnB,OAAO,IAAI,CAAC,OAAO,CAAC,MAAM,CAAC,KAAK,CAAC,EAAE,CAAC,KAAK,CAAC,aAAa,KAAK,IAAI,CAAC,CAAC,GAAG,CAAC,KAAK,CAAC,EAAE,CAAC,KAAK,CAAC,IAAI,CAAC,CAAA;IAC3F,CAAC;IAED,IAAI,qBAAqB;QACxB,OAAO,IAAI,CAAC,OAAO,CAAC,MAAM,CAAC,KAAK,CAAC,EAAE,CAAC,KAAK,CAAC,aAAa,KAAK,KAAK,CAAC,CAAA;IACnE,CAAC;IAED,IAAI,mBAAmB;QACtB,OAAO,IAAI,CAAC,OAAO,CAAC,MAAM,CAAC,KAAK,CAAC,EAAE,CAAC,KAAK,CAAC,aAAa,KAAK,KAAK,CAAC,CAAC,GAAG,CAAC,KAAK,CAAC,EAAE,CAAC,KAAK,CAAC,IAAI,CAAC,CAAA;IAC5F,CAAC;IAED,IAAI,IAAI;QACP,OAAO,IAAI,CAAC,SAAS,CAAC,IAAI,CAAC,EAAE,CAAC,CAAA;IAC/B,CAAC;IAED,IAAI,MAAM;QACT,OAAO,IAAI,CAAC,OAAO,CAAC,MAAM,CAAA;IAC3B,CAAC;CACD"}
@@ -4,3 +4,4 @@ export declare function listAllCharsMatching(charPattern: Pattern): string[];
4
4
  export declare function extractSuppressions(entries: {
5
5
  suppression: string;
6
6
  }[]): string[];
7
+ export declare function getShortLanguageCode(langCode: string): string;
@@ -21,4 +21,11 @@ export function extractSuppressions(entries) {
21
21
  }
22
22
  return suppressions;
23
23
  }
24
+ export function getShortLanguageCode(langCode) {
25
+ const dashIndex = langCode.indexOf('-');
26
+ if (dashIndex == -1) {
27
+ return langCode;
28
+ }
29
+ return langCode.substring(0, dashIndex).toLowerCase();
30
+ }
24
31
  //# sourceMappingURL=Utilities.js.map
@@ -1 +1 @@
1
- {"version":3,"file":"Utilities.js","sourceRoot":"","sources":["../../src/utilities/Utilities.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,WAAW,EAAW,MAAM,iBAAiB,CAAA;AAEtD,MAAM,UAAU,aAAa,CAAC,GAAW,EAAE,MAAM,GAAG,CAAC;IACpD,MAAM,UAAU,GAAG,EAAE,IAAI,MAAM,CAAA;IAE/B,OAAO,IAAI,CAAC,KAAK,CAAC,GAAG,GAAG,UAAU,CAAC,GAAG,UAAU,CAAA;AACjD,CAAC;AAED,MAAM,UAAU,oBAAoB,CAAC,WAAoB;IACxD,MAAM,MAAM,GAAG,WAAW,CAAC,WAAW,CAAC,CAAA;IAEvC,MAAM,aAAa,GAAa,EAAE,CAAA;IAElC,KAAK,IAAI,SAAS,GAAG,CAAC,EAAE,SAAS,GAAG,OAAO,EAAE,SAAS,EAAE,EAAE,CAAC;QAC1D,MAAM,IAAI,GAAG,MAAM,CAAC,aAAa,CAAC,SAAS,CAAC,CAAA;QAE5C,IAAI,MAAM,CAAC,IAAI,CAAC,IAAI,CAAC,EAAE,CAAC;YACvB,aAAa,CAAC,IAAI,CAAC,IAAI,CAAC,CAAA;QACzB,CAAC;IACF,CAAC;IAED,OAAO,aAAa,CAAA;AACrB,CAAC;AAED,MAAM,UAAU,mBAAmB,CAAC,OAAiC;IACpE,MAAM,YAAY,GAAa,EAAE,CAAA;IAEjC,KAAK,MAAM,KAAK,IAAI,OAAO,EAAE,CAAC;QAC7B,YAAY,CAAC,IAAI,CAAC,KAAK,CAAC,WAAW,CAAC,CAAA;IACrC,CAAC;IAED,OAAO,YAAY,CAAA;AACpB,CAAC"}
1
+ {"version":3,"file":"Utilities.js","sourceRoot":"","sources":["../../src/utilities/Utilities.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,WAAW,EAAW,MAAM,iBAAiB,CAAA;AAEtD,MAAM,UAAU,aAAa,CAAC,GAAW,EAAE,MAAM,GAAG,CAAC;IACpD,MAAM,UAAU,GAAG,EAAE,IAAI,MAAM,CAAA;IAE/B,OAAO,IAAI,CAAC,KAAK,CAAC,GAAG,GAAG,UAAU,CAAC,GAAG,UAAU,CAAA;AACjD,CAAC;AAED,MAAM,UAAU,oBAAoB,CAAC,WAAoB;IACxD,MAAM,MAAM,GAAG,WAAW,CAAC,WAAW,CAAC,CAAA;IAEvC,MAAM,aAAa,GAAa,EAAE,CAAA;IAElC,KAAK,IAAI,SAAS,GAAG,CAAC,EAAE,SAAS,GAAG,OAAO,EAAE,SAAS,EAAE,EAAE,CAAC;QAC1D,MAAM,IAAI,GAAG,MAAM,CAAC,aAAa,CAAC,SAAS,CAAC,CAAA;QAE5C,IAAI,MAAM,CAAC,IAAI,CAAC,IAAI,CAAC,EAAE,CAAC;YACvB,aAAa,CAAC,IAAI,CAAC,IAAI,CAAC,CAAA;QACzB,CAAC;IACF,CAAC;IAED,OAAO,aAAa,CAAA;AACrB,CAAC;AAED,MAAM,UAAU,mBAAmB,CAAC,OAAiC;IACpE,MAAM,YAAY,GAAa,EAAE,CAAA;IAEjC,KAAK,MAAM,KAAK,IAAI,OAAO,EAAE,CAAC;QAC7B,YAAY,CAAC,IAAI,CAAC,KAAK,CAAC,WAAW,CAAC,CAAA;IACrC,CAAC;IAED,OAAO,YAAY,CAAA;AACpB,CAAC;AAED,MAAM,UAAU,oBAAoB,CAAC,QAAgB;IACpD,MAAM,SAAS,GAAG,QAAQ,CAAC,OAAO,CAAC,GAAG,CAAC,CAAA;IAEvC,IAAI,SAAS,IAAI,CAAC,CAAC,EAAE,CAAC;QACrB,OAAO,QAAQ,CAAA;IAChB,CAAC;IAED,OAAO,QAAQ,CAAC,SAAS,CAAC,CAAC,EAAE,SAAS,CAAC,CAAC,WAAW,EAAE,CAAA;AACtD,CAAC"}
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@echogarden/text-segmentation",
3
- "version": "0.2.0",
3
+ "version": "0.3.0",
4
4
  "description": "A library for multilingual word, phrase and sentence segmentation.",
5
5
  "author": "Rotem Dan",
6
6
  "license": "MIT",
@@ -39,10 +39,10 @@
39
39
  "regexp-composer": "../../regexp-composer"
40
40
  },
41
41
  "dependencies": {
42
- "regexp-composer": "^0.2.2"
42
+ "regexp-composer": "^0.3.0"
43
43
  },
44
44
  "peerDependencies": {
45
- "@echogarden/icu-segmentation-wasm": "^0.1.0"
45
+ "@echogarden/icu-segmentation-wasm": "^0.2.1"
46
46
  },
47
47
  "peerDependenciesMeta": {
48
48
  "@echogarden/icu-segmentation-wasm": {
@@ -50,6 +50,6 @@
50
50
  }
51
51
  },
52
52
  "devDependencies": {
53
- "@types/node": "^22.15.16"
53
+ "@types/node": "^22.15.17"
54
54
  }
55
55
  }
@@ -1,12 +1,12 @@
1
1
  import { anyOf, buildRegExp, codepointRange } from 'regexp-composer'
2
2
 
3
- export const chineseCharacterRanges = anyOf(
3
+ const chineseCharacterRanges = anyOf(
4
4
  codepointRange('4E00', '9FFF'), // Main
5
5
  codepointRange('3400', '4DBF'), // CJK Unified Ideographs Extension A
6
6
  codepointRange('20000', '2A6DF') // CJK Unified Ideographs Extension B
7
7
  )
8
8
 
9
- export const japaneseHiraganaCharacterRanges = anyOf(
9
+ const japaneseHiraganaCharacterRanges = anyOf(
10
10
  codepointRange('3040', '309F'), // Hiragana
11
11
  codepointRange('1AFF0', '1AFFF'), // Kana Extended-B
12
12
  codepointRange('1B000', '1B0FF'), // Kana Supplement
@@ -14,7 +14,7 @@ export const japaneseHiraganaCharacterRanges = anyOf(
14
14
  codepointRange('1B130', '1B16F'), // Small Kana Extension
15
15
  )
16
16
 
17
- export const japaneseKatakanaCharacterRanges = anyOf(
17
+ const japaneseKatakanaCharacterRanges = anyOf(
18
18
  codepointRange('30A0', '30FF'), // Katakana
19
19
  codepointRange('31F0', '31FF'), // Katakana Phonetic Extensions
20
20
  codepointRange('3200', '32FF'), // Enclosed CJK Letters and Months
@@ -25,16 +25,16 @@ export const japaneseKatakanaCharacterRanges = anyOf(
25
25
  codepointRange('1B130', '1B16F'), // Small Kana Extension
26
26
  )
27
27
 
28
- export const thaiLetterRanges = anyOf(
28
+ const thaiLetterRanges = anyOf(
29
29
  codepointRange('0E00', '0E7F')
30
30
  )
31
31
 
32
- export const khmerLetterRanges = anyOf(
32
+ const khmerLetterRanges = anyOf(
33
33
  codepointRange('1780', '17FF'), // Letters
34
34
  codepointRange('19E0', '19FF'), // Symbols
35
35
  )
36
36
 
37
- export const eastAsianCharRanges = anyOf(
37
+ const eastAsianCharRanges = anyOf(
38
38
  chineseCharacterRanges,
39
39
  japaneseHiraganaCharacterRanges,
40
40
  japaneseKatakanaCharacterRanges,
package/src/Patterns.ts CHANGED
@@ -1,5 +1,8 @@
1
- import { anyOf, buildRegExp, charRange, inputEnd, inputStart, matches, oneOrMore, possibly, repeated, tab, unicodeProperty, whitespace } from 'regexp-composer'
1
+ import { anyOf, buildRegExp, charRange, inputEnd, inputStart, matches, oneOrMore, possibly, repeated, tab, unicodeProperty, whitespace, zeroOrMore } from 'regexp-composer'
2
2
 
3
+ ////////////////////////////////////////////////////////////////////////////////////////////////
4
+ // Pattern builder methods
5
+ ////////////////////////////////////////////////////////////////////////////////////////////////
3
6
  export function buildWordOrNumberPattern(suppressions: string[]) {
4
7
  return anyOf(
5
8
  buildSuppressionPattern(suppressions),
@@ -18,95 +21,140 @@ export function buildSuppressionPattern(suppressions: string[]) {
18
21
  }
19
22
 
20
23
  ////////////////////////////////////////////////////////////////////////////////////////////////
21
- // Numeric patterns
24
+ // Single character patterns
22
25
  ////////////////////////////////////////////////////////////////////////////////////////////////
23
- export const punctuationPattern = unicodeProperty('Punctuation')
24
- export const digitPattern = unicodeProperty('Decimal_Number')
25
- export const arabicNumeralPattern = charRange('0', '9')
26
+ const letterPattern = unicodeProperty('Letter')
27
+ const markPattern = unicodeProperty('Mark')
28
+
29
+ const letterOrMarkPattern = anyOf(
30
+ letterPattern,
31
+ markPattern,
32
+ )
33
+
34
+ const apostrophPattern = anyOf(`'`, `’`, `‘`)
35
+
36
+ const punctuationPattern = unicodeProperty('Punctuation')
37
+ const digitPattern = unicodeProperty('Decimal_Number')
38
+ const arabicNumeralPattern = charRange('0', '9')
39
+
40
+ const percentageCharacters = ['%']
41
+ const currencyCharacters = ['$', '¥', '€', '£', '₩', '₭', '₽', '₫', '฿', '¢', '₮', '؋', '₦', '₱', '₴', '₪']
26
42
 
27
- export const numericSeparatorPattern =
43
+ const percentageOrCurrencyCharacterPattern = anyOf(...percentageCharacters, ...currencyCharacters)
44
+
45
+ ////////////////////////////////////////////////////////////////////////////////////////////////
46
+ // Numeric patterns
47
+ ////////////////////////////////////////////////////////////////////////////////////////////////
48
+ const numericSeparatorPattern =
28
49
  matches(
29
50
  anyOf('.', ',', '٬', '_'), {
30
51
  ifPrecededBy: arabicNumeralPattern,
31
52
  ifFollowedBy: arabicNumeralPattern
32
53
  })
33
54
 
34
- export const dateTimeSeparatorPattern = matches(
35
- anyOf('/', '-', ':'), {
36
- ifPrecededBy: arabicNumeralPattern,
37
- ifFollowedBy: arabicNumeralPattern
38
- })
55
+ const dimensionsPattern = matches([
56
+ oneOrMore(arabicNumeralPattern),
39
57
 
40
- export const dateTimePattern =
41
- oneOrMore(anyOf(
42
- arabicNumeralPattern,
43
- dateTimeSeparatorPattern,
44
- ))
58
+ oneOrMore([
59
+ 'x',
60
+ oneOrMore(arabicNumeralPattern),
61
+ ]),
62
+ ], {
63
+ ifPrecededBy: anyOf(whitespace, punctuationPattern, inputStart),
64
+ ifFollowedBy: anyOf(whitespace, punctuationPattern, inputEnd)
65
+ })
45
66
 
46
- export const spacedThousandsSeparatorPattern =
67
+ const spacedThousandsSeparatorPattern =
47
68
  matches(
48
69
  ' ', {
49
70
  ifPrecededBy: arabicNumeralPattern,
50
- ifFollowedBy: repeated(2, arabicNumeralPattern)
71
+ ifFollowedBy: [
72
+ repeated(3, arabicNumeralPattern),
73
+ anyOf(whitespace, punctuationPattern, inputEnd)
74
+ ]
51
75
  })
52
76
 
53
- export const numericSignPattern =
77
+ const numericSignPattern =
54
78
  matches(
55
79
  anyOf('-', '+'), {
56
- ifPrecededBy: anyOf(whitespace, punctuationPattern),
80
+ ifPrecededBy: anyOf(whitespace, punctuationPattern, inputStart),
57
81
  ifFollowedBy: arabicNumeralPattern,
58
82
  })
59
83
 
60
- export const numberPattern = [
61
- oneOrMore(anyOf(
84
+ const numberPattern = [
85
+ possibly(numericSignPattern),
86
+ digitPattern,
87
+
88
+ zeroOrMore(anyOf(
62
89
  digitPattern,
63
90
  numericSeparatorPattern,
64
91
  spacedThousandsSeparatorPattern,
65
- numericSignPattern,
66
- )),
92
+ ))
67
93
  ]
68
94
 
69
- const percentageChars = ['%']
70
- const currencySpecialChars = ['$', '¥', '€', '£', '¥', '₩', '₭', '₽', '₫', '฿', '¢', '₮', '؋', '₦', '₱', '₴', '₪']
95
+ const exponentPattern = [
96
+ anyOf('e', 'E'),
97
+ possibly(anyOf('+', '-')),
98
+ oneOrMore(arabicNumeralPattern),
99
+ ]
100
+
101
+ const numberPossiblyFollowedByExponentOrLettersPattern = [
102
+ numberPattern,
71
103
 
72
- const percentageOrCurrencyPattern = anyOf(...percentageChars, ...currencySpecialChars)
104
+ possibly(anyOf(
105
+ exponentPattern,
106
+
107
+ matches(
108
+ zeroOrMore(unicodeProperty('Letter')), {
109
+ ifNotFollowedBy: digitPattern,
110
+ }),
111
+ ))
112
+ ]
73
113
 
74
- export const prefixPercentageOrCurrencyPattern =
114
+ const precedingPercentageOrCurrencyPattern =
75
115
  matches([
76
- percentageOrCurrencyPattern,
116
+ percentageOrCurrencyCharacterPattern,
77
117
  numberPattern,
78
118
  ], {
79
119
  ifNotPrecededBy: digitPattern,
80
- ifFollowedBy: anyOf(whitespace, punctuationPattern),
120
+ ifFollowedBy: anyOf(whitespace, punctuationPattern, inputEnd),
81
121
  })
82
122
 
83
- export const suffixPercentagePattern =
123
+ const followingPercentageOrCurrencyPattern =
84
124
  matches([
85
125
  numberPattern,
86
- percentageOrCurrencyPattern,
126
+ percentageOrCurrencyCharacterPattern,
87
127
  ], {
88
- ifPrecededBy: anyOf(whitespace, punctuationPattern),
89
- ifNotFollowedBy: digitPattern,
128
+ ifPrecededBy: anyOf(whitespace, punctuationPattern, inputStart),
129
+ ifNotFollowedBy: anyOf(digitPattern),
90
130
  })
91
131
 
92
- export const percentagePattern = anyOf(
93
- prefixPercentageOrCurrencyPattern,
94
- suffixPercentagePattern
132
+ const percentageOrCurrencyPattern = anyOf(
133
+ precedingPercentageOrCurrencyPattern,
134
+ followingPercentageOrCurrencyPattern
95
135
  )
96
136
 
97
137
  ////////////////////////////////////////////////////////////////////////////////////////////////
98
- // Letter patterns
138
+ // Time patterns
99
139
  ////////////////////////////////////////////////////////////////////////////////////////////////
100
- export const letterPattern = unicodeProperty('Letter')
101
- export const markPattern = unicodeProperty('Mark')
102
- export const apostrophPattern = anyOf(`'`, `’`, `‘`)
103
-
104
- export const letterOrMarkPattern = anyOf(
105
- letterPattern,
106
- markPattern,
107
- )
140
+ const timePattern = matches([
141
+ anyOf(arabicNumeralPattern, [charRange('0', '1'), arabicNumeralPattern], ['2', charRange('0', '3')]),
142
+
143
+ repeated([1, 2], [
144
+ ':',
145
+ anyOf(arabicNumeralPattern, [charRange('0', '5'), arabicNumeralPattern]),
146
+ ])
147
+ ], {
148
+ ifPrecededBy: anyOf(whitespace, punctuationPattern, inputStart),
149
+ ifNotPrecededBy: ':',
150
+ ifFollowedBy: anyOf(whitespace, punctuationPattern, inputEnd),
151
+ ifNotFollowedBy: ':',
152
+ })
108
153
 
109
- export const dottedAbbreviationSequencePattern =
154
+ ////////////////////////////////////////////////////////////////////////////////////////////////
155
+ // Letter patterns
156
+ ////////////////////////////////////////////////////////////////////////////////////////////////
157
+ const dottedAbbreviationSequencePattern =
110
158
  matches(
111
159
  anyOf(
112
160
  [
@@ -135,23 +183,14 @@ export const dottedAbbreviationSequencePattern =
135
183
  ifNotFollowedBy: anyOf(letterOrMarkPattern, digitPattern)
136
184
  })
137
185
 
138
-
139
- export const wordCharacterPattern =
186
+ const wordCharacterPattern =
140
187
  anyOf(
141
188
  letterPattern,
142
189
  markPattern,
143
190
  digitPattern,
144
191
  )
145
192
 
146
- export const wordSeparatorPattern =
147
- matches(
148
- anyOf('-', '_', '·', '‧', '&'), {
149
-
150
- ifPrecededBy: letterOrMarkPattern,
151
- ifFollowedBy: letterOrMarkPattern
152
- })
153
-
154
- export const wordInnerApostrophPattern =
193
+ const wordInnerApostrophPattern =
155
194
  matches(
156
195
  apostrophPattern, {
157
196
 
@@ -159,67 +198,118 @@ export const wordInnerApostrophPattern =
159
198
  ifFollowedBy: letterOrMarkPattern,
160
199
  })
161
200
 
162
- export const wordStartApostrophPattern =
201
+ const wordStartApostrophPattern =
163
202
  matches(
164
203
  apostrophPattern, {
165
204
 
166
205
  ifPrecededBy: whitespace,
167
- ifFollowedBy: [letterOrMarkPattern, letterOrMarkPattern, whitespace]
206
+ ifFollowedBy: [letterOrMarkPattern, letterOrMarkPattern, anyOf(whitespace, inputEnd)]
168
207
  })
169
208
 
170
- export const basicWordPattern =
209
+ const basicWordPattern =
171
210
  oneOrMore(
172
211
  anyOf(
173
212
  wordCharacterPattern,
174
- wordSeparatorPattern,
175
213
  wordInnerApostrophPattern,
176
214
  ),
177
215
  )
178
216
 
179
- export const wordSegmentPattern = anyOf(
217
+ const hyphenatedWordPattern = [
218
+ basicWordPattern,
219
+
220
+ oneOrMore([
221
+ '-',
222
+ basicWordPattern,
223
+ ]),
224
+ ]
225
+
226
+ const dotSeparatedWordPattern = matches([
227
+ basicWordPattern,
228
+
229
+ oneOrMore([
230
+ '.',
231
+ basicWordPattern,
232
+ ]),
233
+ ], {
234
+ //ifPrecededBy: anyOf(whitespace, inputStart),
235
+ //ifFollowedBy: anyOf(whitespace, inputEnd)
236
+ })
237
+
238
+ const interpunctSeparatedWordPattern = [
239
+ basicWordPattern,
240
+
241
+ oneOrMore([
242
+ anyOf('·', '‧'),
243
+ basicWordPattern,
244
+ ]),
245
+ ]
246
+
247
+ const underscoreSeparatedWordPattern = [
248
+ basicWordPattern,
249
+
250
+ oneOrMore([
251
+ '_',
252
+ basicWordPattern,
253
+ ]),
254
+ ]
255
+
256
+ const wordSegmentPattern = anyOf(
257
+ timePattern,
258
+ dimensionsPattern,
259
+ percentageOrCurrencyPattern,
260
+ numberPossiblyFollowedByExponentOrLettersPattern,
261
+
262
+ interpunctSeparatedWordPattern,
263
+ dotSeparatedWordPattern,
180
264
  dottedAbbreviationSequencePattern,
181
- dateTimeSeparatorPattern,
182
- percentagePattern,
183
- numberPattern,
265
+ underscoreSeparatedWordPattern,
184
266
  basicWordPattern,
185
267
  )
186
268
 
187
269
  ////////////////////////////////////////////////////////////////////////////////////////////////
188
- // Prebuilt regular expressions
270
+ // Phrase separation patterns
189
271
  ////////////////////////////////////////////////////////////////////////////////////////////////
190
- export const phraseSeparators = [',', '、', ',', '،', ';', ';', ':', ':', '—']
272
+ const phraseSeparatorCharacters = [',', '、', ',', '،', ';', ';', ':', ':', '—']
191
273
 
192
- export const phraseSeparatorRegExp = buildRegExp([
274
+ const phraseSeparatorPattern = [
193
275
  inputStart,
194
- anyOf(...phraseSeparators),
276
+ anyOf(...phraseSeparatorCharacters),
195
277
  inputEnd
196
- ])
197
-
198
- export const sentenceSeparators = ['.', '。', '?', '?', '!', '!', '\n']
278
+ ]
199
279
 
200
- export const sentenceSeparatorRegExp = buildRegExp([
280
+ const phraseSeparatorTrailingPunctuationPattern = [
201
281
  inputStart,
202
- anyOf(...sentenceSeparators),
282
+ anyOf(...phraseSeparatorCharacters, ' ', tab),
203
283
  inputEnd
204
- ])
284
+ ]
205
285
 
206
- export const sentenceSeparatorTrailingPunctuationRegExp = buildRegExp([
207
- inputStart,
208
- anyOf('"', '”', '’', ')', ']', '}', '»', ...sentenceSeparators, ...phraseSeparators, oneOrMore(whitespace)),
209
- inputEnd
210
- ])
286
+ ////////////////////////////////////////////////////////////////////////////////////////////////
287
+ // Sentence separation patterns
288
+ ////////////////////////////////////////////////////////////////////////////////////////////////
289
+ const sentenceSeparatorCharacters = ['.', '。', '?', '?', '!', '!', '\n']
211
290
 
212
- export const phraseSeparatorTrailingPunctuationRegExp = buildRegExp([
291
+ const sentenceSeparatorPattern = [
213
292
  inputStart,
214
- anyOf(...phraseSeparators, ' ', tab),
293
+ anyOf(...sentenceSeparatorCharacters),
215
294
  inputEnd
216
- ])
295
+ ]
217
296
 
218
- export const oneOrMoreSpacesRegExp = buildRegExp([
297
+ const sentenceSeparatorTrailingPunctuationPattern = [
219
298
  inputStart,
220
- oneOrMore(' '),
299
+ anyOf('"', '”', '’', ')', ']', '}', '»', ...sentenceSeparatorCharacters, ...phraseSeparatorCharacters, oneOrMore(whitespace)),
221
300
  inputEnd
222
- ])
301
+ ]
223
302
 
303
+ ////////////////////////////////////////////////////////////////////////////////////////////////
304
+ // Prebuilt regular expressions
305
+ ////////////////////////////////////////////////////////////////////////////////////////////////
224
306
  export const wordCharacterRegExp = buildRegExp(wordCharacterPattern)
225
307
  export const whitespacePatternRegExp = buildRegExp(whitespace)
308
+
309
+ export const letterPatternGlobalRegExp = buildRegExp(letterPattern, { global: true })
310
+
311
+ export const phraseSeparatorRegExp = buildRegExp(phraseSeparatorPattern)
312
+ export const phraseSeparatorTrailingPunctuationRegExp = buildRegExp(phraseSeparatorTrailingPunctuationPattern)
313
+
314
+ export const sentenceSeparatorRegExp = buildRegExp(sentenceSeparatorPattern)
315
+ export const sentenceSeparatorTrailingPunctuationRegExp = buildRegExp(sentenceSeparatorTrailingPunctuationPattern)
@@ -1,16 +1,27 @@
1
1
  export const cldrSuppressions: Record<string, string[]> = {
2
- 'en': ['L.P.', 'Alt.', 'Approx.', 'E.G.', 'O.', 'Maj.', 'Misc.', 'P.O.', 'J.D.', 'Jam.', 'Card.', 'Dec.', 'Sept.', 'MR.', 'Long.', 'Hat.', 'G.', 'Link.', 'DC.', 'D.C.', 'M.T.', 'Hz.', 'Mrs.', 'By.', 'Act.', 'Var.', 'N.V.', 'Aug.', 'B.', 'S.A.', 'Up.', 'Job.', 'Num.', 'M.I.T.', 'Ok.', 'Org.', 'Ex.', 'Cont.', 'U.', 'Mart.', 'Fn.', 'Abs.', 'Lt.', 'OK.', 'Z.', 'E.', 'Kb.', 'Est.', 'A.M.', 'L.A.', 'Prof.', 'U.S.', 'Nov.', 'Ph.D.', 'Mar.', 'I.T.', 'exec.', 'Jan.', 'N.Y.', 'X.', 'Md.', 'Op.', 'vs.', 'D.A.', 'A.D.', 'R.L.', 'P.M.', 'Or.', 'M.R.', 'Cap.', 'PC.', 'Feb.', 'Exec.', 'I.e.', 'Sep.', 'Gb.', 'K.', 'U.S.C.', 'Mt.', 'S.', 'A.S.', 'C.O.D.', 'Capt.', 'Col.', 'In.', 'C.F.', 'Adj.', 'AD.', 'I.D.', 'Mgr.', 'R.T.', 'B.V.', 'M.', 'Conn.', 'Yr.', 'Rev.', 'Phys.', 'pp.', 'Ms.', 'To.', 'Sgt.', 'J.K.', 'Nr.', 'Jun.', 'Fri.', 'S.A.R.', 'Lev.', 'Lt.Cdr.', 'Def.', 'F.', 'Do.', 'Joe.', 'Id.', 'Mr.', 'Dept.', 'Is.', 'Pvt.', 'Diff.', 'Hon.B.A.', 'Q.', 'Mb.', 'On.', 'Min.', 'J.B.', 'Ed.', 'AB.', 'A.', 'S.p.A.', 'I.', 'a.m.', 'Comm.', 'Go.', 'VS.', 'L.', 'All.', 'PP.', 'P.V.', 'T.', 'K.R.', 'Etc.', 'D.', 'Adv.', 'Lib.', 'E.g.', 'Pro.', 'U.S.A.', 'S.E.', 'AA.', 'Rep.', 'Sq.', 'As.'],
3
- 'de': ['Port.', 'Alt.', 'Di.', 'Ges.', 'frz.', 'entspr.', 'Gebr.', 'erw.', 'Frl.', 'Inh.', 'k.u.k.', 'Ca.', 'J.D.', 'Ausg.', 'evtl.', 'So.', 'i.B.', 's.a.', 'kgl.', 'Sept.', 'o.B.', 'Sa.', 'ev.', 'Dez.', 'am.', 'i.R.', 'eigtl.', 'i.J.', 'u.U.', 'G.', 'z.Hd.', 'u.A.w.g.', 'Kl.', 'Spezif.', 'Obj.', 'Ing.', 'D. h.', 'Folg.', 'Akt.', 'i.A.', 'Msp.', 'U.U.', 'Chr.', 'R.', 'Einh.', 'schwäb.', 'Vgl.', 'Aug.', 'Dipl.-Ing.', 'W.', 'B.', 'U. U.', 'J.', 'Fa.', 'Mo.', 'n.u.Z.', 'Op.', 'Mrd.', 'e.h.', 'Hr.', 'Hrn.', 'Ztr.', 'k. u. k.', 'Bibl.', 'd.Ä.', 'b.', 'M.', 'i.H.', 'v.R.w.', 'o.A.', 'St.', 'Dr.', 'Fn.', 'Abs.', 'Rd.', 'Dtzd.', 'Jahrh.', 'Z.', 'Std.', 'n. Chr.', 'möbl.', 'tägl.', 'gest.', 'gesch.', 'z.B.', 'Hbf.', 'Abt.', 'A.M.', 'e.Wz.', 'v.T.', 'Nov.', 'z.', 'Prot.', 'U.S.', 'Wg.', 'u.v.a.', 'Adr.', 'App.', 'ggf.', 'ggfs.', 'Jan.', 'O.', 'Rel.', 'od.', 'Pfd.', 'a.a.O.', 'p.Adr.', 'P.', 'Gem.', 'v. Chr.', 'Art.', 'z.Z.', 'S.A.', 'i.V.', 'verh.', 'Ausschl.', 'm.W.', 'Dir.', 'Verf.', 'Sek.', 'r.', 'Chin.', 'Feb.', 'Int.', 'Sep.', 'Gesch.', 'schweiz.', 'Bed.', 'a.Rh.', 'jew.', 'vgl.', 'a.M.', 'Str.', 'exkl.', 'gek.', 'Erf.', 'u.Ä.', 'ehem.', 'näml.', 'u. Z.', 'v. u. Z.', 'sog.', 'C.', 'Dipl.-Kfm.', 'mtl.', 'Hrsg.', 'Qu.', 'röm.', 'u.', 'U.', 'Adj.', 'Kap.', 'hpts.', 'a.D.', 'gedr.', 'Best.', 'N.', 'v.u.Z.', 'Phys.', 'Fr.', 'd.J.', 'Reg.-Bez.', 'm.E.', 'schles.', 'Max.', 'Ltd.', 'südd.', 'inkl.', 'geb.', 'Ggf.', 'Inc.', 'kath.', 'kfm.', 'Nr.', 'Proz.', 'Dim.', 'verw.', 'Reg.', 'Dat.', 'Evtl.', 'led.', 'F.', 'Test.', 'Schr.', 'Do.', 'PIN.', 'Z. Zt.', 'v.Chr.', 'Tägl.', 's.', 'amtl.', 'Temp.', 'Mind.', 'e.V.', 'Abw.', 'P.M.', 'F.f.', 'a.a.S.', 'Mod.', 'Co.', 'Min.', 'Allg.', 'Geograph.', 'Jr.', 'Urspr.', 'Apr.', 'Z. B.', 'v.H.', 'A.', 'einschl.', 'Trans.', 'zzgl.', 'StR.', 'Fam.', 'I.', 'jhrl.', 'u.a.', 'Ben.', 'o.g.', 'Kfm.', 'Konv.', 'Mi.', 'L.', 'beil.', 'T.', 'Ursprüngl.', 'röm.-kath.', 'Okt.', 'u.ä.', 'Tel.', 'D.', 'Ber.', 'Kop.', 'Mio.', 'Y.', 'U.S.A.', 'v. H.', 'Forts. f.', 'Rep.', 'Hptst.', 'österr.'],
4
- 'es': ['Rdos.', 'JJ.OO.', 'Sres.', 'fig.', 'may.', 'RR.HH.', 'oct.', 'cap.', 'mié.', 'doc.', 'Excmo.', 'Trab.', 'Excmos.', 'Kit.', 'Inc.', 'FF.CC.', 'DC.', 'ago.', 'trad.', 'SA.', 'Rvdos.', 'ed.', 'Exmo.', 'jul.', 'col.', 'RAM.', 'Srtas.', 'ene.', 'Rol.', 'Fabric.', 'Comm.', 'vid.', 'Da.', 'dic.', 'ss.', 'abr.', 'ntra.', 'Sra.', 'dtor.', 'cf.', 'dom.', 'prov.', 'Emm.', 'Sr.', 'licdo.', 'p.ej.', 'bol.', 'figs.', 'Vda.', 'Dr.', 'ntro.', 'Desv.', 'O.M.', 'Ldo.', 'Drs.', 'sáb.', 'feb.', 'Ltda.', 'Lcda.', 'Exma.', 'C.V.', 'SS.MM.', 'Lda.', 'U.S.', 'hnos.', 'R.D.', 'Korn.', 'v.gr.', 'vs.', 'Ilmas.', 'Rdo.', 'ej.', 'vie.', 'jue.', 'a. C.', 'Ilmos.', 'e. c.', 'Excma.', 'afma.', 'licda.', 'Em.', 'K.', 'sras.', 'MM.', 'fund.', 'Mons.', 'Lcdo.', 'afmo.', 'C.', 'A.C.', 'dptos.', 'Col.', 'Srta.', 'Av.', 'Ant.', 'depto.', 'Var.', 'H.P.', 'D.', 'M.', 'C.P.', 'Rev.', 'Rvdmos.', 'Fr.', 'Ilmo.', 'afmos.', 'Ltd.', 'afmas.', 'prof.', 'lun.', 'SS.AA.', 'Sol.', 'nov.', 'mss.', 'Dña.', 'Seg.', 'mar.', 'Rvdmo.', 'Reg.', 'ms.', 'Sras.', 'sres.', 'U.S.A.', 'Sta.', 'Sdad.', 'Dra.', 'srs.', 'R.U.', 'deptos.', 'dpto.', 'jun.', 'bco.', 'Cía.', 'Id.', 'Mr.', 'e.g.', 'C.S.', 'Excmas.', 'Dª.', 'Rvdo.', 'Lic.', 'cfr.', 'Corp.', 'Dto.', 'Ilma.', 'L.', 'All.', 'PP.', 'd. C.', 'Ltdo.', 'mtro.', 'Mrs.', 'Desc.', 'Avda.', 'Exmas.', 'a. e. c.', 'Bien.', 'Exmos.', 'AA.', 'Sto.', 'CA.', 'sept.', 'Exc.', 'c/c.'],
5
- 'fr': ['aux.', 'config.', 'collab.', 'M.', 'dim.', 'imprim.', 'oct.', 'syst.', 'bull.', 'MM.', 'doc.', 'P.O.', 'hôp.', 'Mart.', 'juil.', 'broch.', 'adr.', 'symb.', 'C.', 'anc.', 'voit.', 'Jr.', 'graph.', 'dir.', 'éd.', 'fig.', 'édit.', 'niv.', 'quart.', 'cam.', 'éval.', 'anon.', 'réf.', 'Comm.', 'Prof.', 'févr.', 'indus.', 'DC.', 'équiv.', 'illustr.', 'acoust.', 'nov.', 'L.', 'All.', 'U.S.', 'S.M.A.R.T.', 'sept.', 'avr.', 'jeu.', 'dest.', 'P.-D. G.', 'ill.', 'coll.', 'encycl.', 'mer.', 'Desc.', 'ven.', 'P.', 'lun.', 'Inc.', 'sam.', 'D.', 'append.', 'Var.', 'categ.', 'janv.', 'S.A.', 'imm.', 'U.S.A.', 'mar.', 'exempl.', 'déc.', 'ann.', 'U.', 'synth.', 'dict.', 'av. J.-C.', 'W.', 'Op.', 'ap. J.-C.', 'gouv.', 'trav. publ.'],
6
- 'it': ['N.B.', 'div.', 'a.C.', 'fig.', 'd.p.R.', 'c.c.p.', 'Cfr.', 'vol.', 'Geom.', 'O.d.G.', 'S.p.A.', 'ver.', 'N.d.A.', 'dott.', 'arch.', 'd.C.', 'N.d.T.', 'rag.', 'Sig.', 'Mod.', 'pag.', 'dr.', 'tav.', 'N.d.E.', 'DC.', 'mitt.', 'Ing.', 'int.', 'on.', 'C.P.', 'ag.', 'L.', 'U.S.', 'S.M.A.R.T.', 'p.i.', 'tab.', 'Ltd.', 'Liv.', 'D.', 'U.S.A.', 'sez.', 'avv.', 'S.A.R.', 'all.', 'p.'],
7
- 'pt': ['psicol.', 'fig.', 'compl.', 'rep.', 'cap.', 'doc.', 'fisiol.', 'dipl.', 'astron.', 'port.', 'eletrôn.', 'geom.', 'mov.', 'ago.', 'trad.', 'arquit.', 'dez.', 'ed.', 'apt.', 'Exmo.', 'col.', 'ff.', 'univ.', 'res.', 'R.', 'transp.', 'D.C', 'l.', 'des.', 'fev.', 'abr.', 'liter.', 'lat.', 'Dir.', 'cf.', 'adm.', 'fot.', 'p.m.', 'P.M.', 'créd.', 'jur.', 'com.', 'anat.', 'dir.', 'end.', 'fís.', 'E.', 'Est.', 'cont.', 'matem.', 'Drs.', 'gên.', 'neol.', 'pág.', 'índ.', 'Ltda.', 'Exma.', 'esp.', 'ingl.', 'tecnol.', 'Mar.', 'símb.', 'Pe.', 'pal.', 'filos.', 'V.T.', 'fasc.', 'vs.', 'mai.', 'S.A.', 'profa.', 'N.Sra.', 'r.s.v.p.', 'cel.', 'mat.', 'abrev.', 'out.', 'long.', 'aux.', 'arit.', 'aer.', 'jul.', 'lin.', 'S.', 'méd.', 'odontol.', 'org.', 'A.C.', 'jun.', 'déb.', 'Av.', 'álg.', 'sup.', 'fl.', 'odont.', 'caps.', 'relat.', 'organiz.', 'hist.', 'Fr.', 'Ilmo.', 'fem.', 'ap.', 'Ltd.', 'pol.', 'séc.', 'prof.', 'cx.', 'nov.', 'quím.', 'mús.', 'agric.', 'mar.', 'W.C.', 'fr.', 'cat.', 'jan.', 'pron.', 'rel.', 'autom.', 'Sta.', 'Dra.', 'p.', 'tel.', 'div.', 'p. ex.', 'a.C.', 'bras.', 'Alm.', 'Dr.', 'comp.', 'pq.', 'arqueol.', 'náut.', 'biogr.', 'f.', 'círc.', 'fac.', 'd.C.', 'apart.', 'ex.', 'Jr.', 'set.', 'tec.', 'sociol.', 'gram.', 'ind.', 'Ilma.', 'vol.', 'eng.', 'rod.', 'Ph.D.', 'Dras.', 'pp.', 'elem.', 'máq.', 'cód.', 'eletr.', 'prod.', 'ref.', 'fil.', 'a.m.', 'A.M', 'obs.', 'N.T.', 'contab.', 'Sto.', 'lit.', 'educ.', 'rementente', 'desc.', 'próx.'],
8
- 'ru': ['руб.', 'янв.', 'до н. э.', 'сент.', 'тел.', 'дек.', 'февр.', 'нояб.', 'апр.', 'н. э.', 'окт.', 'тыс.', 'авг.', 'проф.', 'н.э.', 'кв.', 'ул.', 'отд.'],
2
+ //en: ['L.P.', 'Alt.', 'Approx.', 'E.G.', 'O.', 'Maj.', 'Misc.', 'P.O.', 'J.D.', 'Jam.', 'Card.', 'Dec.', 'Sept.', 'MR.', 'Long.', 'Hat.', 'G.', 'Link.', 'DC.', 'D.C.', 'M.T.', 'Hz.', 'Mrs.', 'By.', 'Act.', 'Var.', 'N.V.', 'Aug.', 'B.', 'S.A.', 'Up.', 'Job.', 'Num.', 'M.I.T.', 'Ok.', 'Org.', 'Ex.', 'Cont.', 'U.', 'Mart.', 'Fn.', 'Abs.', 'Lt.', 'OK.', 'Z.', 'E.', 'Kb.', 'Est.', 'A.M.', 'L.A.', 'Prof.', 'U.S.', 'Nov.', 'Ph.D.', 'Mar.', 'I.T.', 'exec.', 'Jan.', 'N.Y.', 'X.', 'Md.', 'Op.', 'vs.', 'D.A.', 'A.D.', 'R.L.', 'P.M.', 'Or.', 'M.R.', 'Cap.', 'PC.', 'Feb.', 'Exec.', 'I.e.', 'Sep.', 'Gb.', 'K.', 'U.S.C.', 'Mt.', 'S.', 'A.S.', 'C.O.D.', 'Capt.', 'Col.', 'In.', 'C.F.', 'Adj.', 'AD.', 'I.D.', 'Mgr.', 'R.T.', 'B.V.', 'M.', 'Conn.', 'Yr.', 'Rev.', 'Phys.', 'pp.', 'Ms.', 'To.', 'Sgt.', 'J.K.', 'Nr.', 'Jun.', 'Fri.', 'S.A.R.', 'Lev.', 'Lt.Cdr.', 'Def.', 'F.', 'Do.', 'Joe.', 'Id.', 'Mr.', 'Dept.', 'Is.', 'Pvt.', 'Diff.', 'Hon.B.A.', 'Q.', 'Mb.', 'On.', 'Min.', 'J.B.', 'Ed.', 'AB.', 'A.', 'S.p.A.', 'I.', 'a.m.', 'Comm.', 'Go.', 'VS.', 'L.', 'All.', 'PP.', 'P.V.', 'T.', 'K.R.', 'Etc.', 'D.', 'Adv.', 'Lib.', 'E.g.', 'Pro.', 'U.S.A.', 'S.E.', 'AA.', 'Rep.', 'Sq.', 'As.'],
3
+ en: ['L.P.', 'Alt.', 'Approx.', 'E.G.', 'Maj.', 'Misc.', 'P.O.', 'J.D.', 'Dec.', 'Sept.', 'MR.', 'DC.', 'D.C.', 'M.T.', 'Mrs.', 'N.V.', 'Aug.', 'S.A.', 'Num.', 'M.I.T.', 'Org.', 'Ex.', 'Cont.', 'Mart.', 'Fn.', 'Abs.', 'Lt.', 'Est.', 'A.M.', 'L.A.', 'Prof.', 'U.S.', 'Nov.', 'Ph.D.', 'Mar.', 'I.T.', 'exec.', 'Jan.', 'N.Y.', 'Md.', 'Op.', 'vs.', 'D.A.', 'A.D.', 'R.L.', 'P.M.', 'M.R.', 'Feb.', 'Exec.', 'I.e.', 'Sep.', 'U.S.C.', 'Mt.', 'A.S.', 'C.O.D.', 'Capt.', 'Col.', 'C.F.', 'Adj.', 'I.D.', 'Mgr.', 'R.T.', 'B.V.', 'M.', 'Conn.', 'Yr.', 'Rev.', 'Phys.', 'pp.', 'Ms.', 'Sgt.', 'J.K.', 'Nr.', 'Jun.', 'Fri.', 'S.A.R.', 'Lev.', 'Lt.Cdr.', 'Def.', 'Mr.', 'Dept.', 'Pvt.', 'Diff.', 'Hon.B.A.', 'Mb.', 'Min.', 'J.B.', 'Ed.', 'AB.', 'S.p.A.', 'I.', 'a.m.', 'Comm.', 'VS.', 'PP.', 'P.V.', 'K.R.', 'Etc.', 'Adv.', 'Lib.', 'E.g.', 'U.S.A.', 'S.E.', 'AA.', 'Rep.', 'Sq.'],
4
+ de: ['Port.', 'Alt.', 'Di.', 'Ges.', 'frz.', 'entspr.', 'Gebr.', 'erw.', 'Frl.', 'Inh.', 'k.u.k.', 'Ca.', 'J.D.', 'Ausg.', 'evtl.', 'So.', 'i.B.', 's.a.', 'kgl.', 'Sept.', 'o.B.', 'Sa.', 'ev.', 'Dez.', 'am.', 'i.R.', 'eigtl.', 'i.J.', 'u.U.', 'G.', 'z.Hd.', 'u.A.w.g.', 'Kl.', 'Spezif.', 'Obj.', 'Ing.', 'D. h.', 'Folg.', 'Akt.', 'i.A.', 'Msp.', 'U.U.', 'Chr.', 'R.', 'Einh.', 'schwäb.', 'Vgl.', 'Aug.', 'Dipl.-Ing.', 'W.', 'B.', 'U. U.', 'J.', 'Fa.', 'Mo.', 'n.u.Z.', 'Op.', 'Mrd.', 'e.h.', 'Hr.', 'Hrn.', 'Ztr.', 'k. u. k.', 'Bibl.', 'd.Ä.', 'b.', 'M.', 'i.H.', 'v.R.w.', 'o.A.', 'St.', 'Dr.', 'Fn.', 'Abs.', 'Rd.', 'Dtzd.', 'Jahrh.', 'Z.', 'Std.', 'n. Chr.', 'möbl.', 'tägl.', 'gest.', 'gesch.', 'z.B.', 'Hbf.', 'Abt.', 'A.M.', 'e.Wz.', 'v.T.', 'Nov.', 'z.', 'Prot.', 'U.S.', 'Wg.', 'u.v.a.', 'Adr.', 'App.', 'ggf.', 'ggfs.', 'Jan.', 'O.', 'Rel.', 'od.', 'Pfd.', 'a.a.O.', 'p.Adr.', 'P.', 'Gem.', 'v. Chr.', 'Art.', 'z.Z.', 'S.A.', 'i.V.', 'verh.', 'Ausschl.', 'm.W.', 'Dir.', 'Verf.', 'Sek.', 'r.', 'Chin.', 'Feb.', 'Int.', 'Sep.', 'Gesch.', 'schweiz.', 'Bed.', 'a.Rh.', 'jew.', 'vgl.', 'a.M.', 'Str.', 'exkl.', 'gek.', 'Erf.', 'u.Ä.', 'ehem.', 'näml.', 'u. Z.', 'v. u. Z.', 'sog.', 'C.', 'Dipl.-Kfm.', 'mtl.', 'Hrsg.', 'Qu.', 'röm.', 'u.', 'U.', 'Adj.', 'Kap.', 'hpts.', 'a.D.', 'gedr.', 'Best.', 'N.', 'v.u.Z.', 'Phys.', 'Fr.', 'd.J.', 'Reg.-Bez.', 'm.E.', 'schles.', 'Max.', 'Ltd.', 'südd.', 'inkl.', 'geb.', 'Ggf.', 'Inc.', 'kath.', 'kfm.', 'Nr.', 'Proz.', 'Dim.', 'verw.', 'Reg.', 'Dat.', 'Evtl.', 'led.', 'F.', 'Test.', 'Schr.', 'Do.', 'PIN.', 'Z. Zt.', 'v.Chr.', 'Tägl.', 's.', 'amtl.', 'Temp.', 'Mind.', 'e.V.', 'Abw.', 'P.M.', 'F.f.', 'a.a.S.', 'Mod.', 'Co.', 'Min.', 'Allg.', 'Geograph.', 'Jr.', 'Urspr.', 'Apr.', 'Z. B.', 'v.H.', 'A.', 'einschl.', 'Trans.', 'zzgl.', 'StR.', 'Fam.', 'I.', 'jhrl.', 'u.a.', 'Ben.', 'o.g.', 'Kfm.', 'Konv.', 'Mi.', 'L.', 'beil.', 'T.', 'Ursprüngl.', 'röm.-kath.', 'Okt.', 'u.ä.', 'Tel.', 'D.', 'Ber.', 'Kop.', 'Mio.', 'Y.', 'U.S.A.', 'v. H.', 'Forts. f.', 'Rep.', 'Hptst.', 'österr.'],
5
+ es: ['Rdos.', 'JJ.OO.', 'Sres.', 'fig.', 'may.', 'RR.HH.', 'oct.', 'cap.', 'mié.', 'doc.', 'Excmo.', 'Trab.', 'Excmos.', 'Kit.', 'Inc.', 'FF.CC.', 'DC.', 'ago.', 'trad.', 'SA.', 'Rvdos.', 'ed.', 'Exmo.', 'jul.', 'col.', 'RAM.', 'Srtas.', 'ene.', 'Rol.', 'Fabric.', 'Comm.', 'vid.', 'Da.', 'dic.', 'ss.', 'abr.', 'ntra.', 'Sra.', 'dtor.', 'cf.', 'dom.', 'prov.', 'Emm.', 'Sr.', 'licdo.', 'p.ej.', 'bol.', 'figs.', 'Vda.', 'Dr.', 'ntro.', 'Desv.', 'O.M.', 'Ldo.', 'Drs.', 'sáb.', 'feb.', 'Ltda.', 'Lcda.', 'Exma.', 'C.V.', 'SS.MM.', 'Lda.', 'U.S.', 'hnos.', 'R.D.', 'Korn.', 'v.gr.', 'vs.', 'Ilmas.', 'Rdo.', 'ej.', 'vie.', 'jue.', 'a. C.', 'Ilmos.', 'e. c.', 'Excma.', 'afma.', 'licda.', 'Em.', 'K.', 'sras.', 'MM.', 'fund.', 'Mons.', 'Lcdo.', 'afmo.', 'C.', 'A.C.', 'dptos.', 'Col.', 'Srta.', 'Av.', 'Ant.', 'depto.', 'Var.', 'H.P.', 'D.', 'M.', 'C.P.', 'Rev.', 'Rvdmos.', 'Fr.', 'Ilmo.', 'afmos.', 'Ltd.', 'afmas.', 'prof.', 'lun.', 'SS.AA.', 'Sol.', 'nov.', 'mss.', 'Dña.', 'Seg.', 'mar.', 'Rvdmo.', 'Reg.', 'ms.', 'Sras.', 'sres.', 'U.S.A.', 'Sta.', 'Sdad.', 'Dra.', 'srs.', 'R.U.', 'deptos.', 'dpto.', 'jun.', 'bco.', 'Cía.', 'Id.', 'Mr.', 'e.g.', 'C.S.', 'Excmas.', 'Dª.', 'Rvdo.', 'Lic.', 'cfr.', 'Corp.', 'Dto.', 'Ilma.', 'L.', 'All.', 'PP.', 'd. C.', 'Ltdo.', 'mtro.', 'Mrs.', 'Desc.', 'Avda.', 'Exmas.', 'a. e. c.', 'Bien.', 'Exmos.', 'AA.', 'Sto.', 'CA.', 'sept.', 'Exc.', 'c/c.'],
6
+ fr: ['aux.', 'config.', 'collab.', 'M.', 'dim.', 'imprim.', 'oct.', 'syst.', 'bull.', 'MM.', 'doc.', 'P.O.', 'hôp.', 'Mart.', 'juil.', 'broch.', 'adr.', 'symb.', 'C.', 'anc.', 'voit.', 'Jr.', 'graph.', 'dir.', 'éd.', 'fig.', 'édit.', 'niv.', 'quart.', 'cam.', 'éval.', 'anon.', 'réf.', 'Comm.', 'Prof.', 'févr.', 'indus.', 'DC.', 'équiv.', 'illustr.', 'acoust.', 'nov.', 'L.', 'All.', 'U.S.', 'S.M.A.R.T.', 'sept.', 'avr.', 'jeu.', 'dest.', 'P.-D. G.', 'ill.', 'coll.', 'encycl.', 'mer.', 'Desc.', 'ven.', 'P.', 'lun.', 'Inc.', 'sam.', 'D.', 'append.', 'Var.', 'categ.', 'janv.', 'S.A.', 'imm.', 'U.S.A.', 'mar.', 'exempl.', 'déc.', 'ann.', 'U.', 'synth.', 'dict.', 'av. J.-C.', 'W.', 'Op.', 'ap. J.-C.', 'gouv.', 'trav. publ.'],
7
+ it: ['N.B.', 'div.', 'a.C.', 'fig.', 'd.p.R.', 'c.c.p.', 'Cfr.', 'vol.', 'Geom.', 'O.d.G.', 'S.p.A.', 'ver.', 'N.d.A.', 'dott.', 'arch.', 'd.C.', 'N.d.T.', 'rag.', 'Sig.', 'Mod.', 'pag.', 'dr.', 'tav.', 'N.d.E.', 'DC.', 'mitt.', 'Ing.', 'int.', 'on.', 'C.P.', 'ag.', 'L.', 'U.S.', 'S.M.A.R.T.', 'p.i.', 'tab.', 'Ltd.', 'Liv.', 'D.', 'U.S.A.', 'sez.', 'avv.', 'S.A.R.', 'all.', 'p.'],
8
+ pt: ['psicol.', 'fig.', 'compl.', 'rep.', 'cap.', 'doc.', 'fisiol.', 'dipl.', 'astron.', 'port.', 'eletrôn.', 'geom.', 'mov.', 'ago.', 'trad.', 'arquit.', 'dez.', 'ed.', 'apt.', 'Exmo.', 'col.', 'ff.', 'univ.', 'res.', 'R.', 'transp.', 'D.C', 'l.', 'des.', 'fev.', 'abr.', 'liter.', 'lat.', 'Dir.', 'cf.', 'adm.', 'fot.', 'p.m.', 'P.M.', 'créd.', 'jur.', 'com.', 'anat.', 'dir.', 'end.', 'fís.', 'E.', 'Est.', 'cont.', 'matem.', 'Drs.', 'gên.', 'neol.', 'pág.', 'índ.', 'Ltda.', 'Exma.', 'esp.', 'ingl.', 'tecnol.', 'Mar.', 'símb.', 'Pe.', 'pal.', 'filos.', 'V.T.', 'fasc.', 'vs.', 'mai.', 'S.A.', 'profa.', 'N.Sra.', 'r.s.v.p.', 'cel.', 'mat.', 'abrev.', 'out.', 'long.', 'aux.', 'arit.', 'aer.', 'jul.', 'lin.', 'S.', 'méd.', 'odontol.', 'org.', 'A.C.', 'jun.', 'déb.', 'Av.', 'álg.', 'sup.', 'fl.', 'odont.', 'caps.', 'relat.', 'organiz.', 'hist.', 'Fr.', 'Ilmo.', 'fem.', 'ap.', 'Ltd.', 'pol.', 'séc.', 'prof.', 'cx.', 'nov.', 'quím.', 'mús.', 'agric.', 'mar.', 'W.C.', 'fr.', 'cat.', 'jan.', 'pron.', 'rel.', 'autom.', 'Sta.', 'Dra.', 'p.', 'tel.', 'div.', 'p. ex.', 'a.C.', 'bras.', 'Alm.', 'Dr.', 'comp.', 'pq.', 'arqueol.', 'náut.', 'biogr.', 'f.', 'círc.', 'fac.', 'd.C.', 'apart.', 'ex.', 'Jr.', 'set.', 'tec.', 'sociol.', 'gram.', 'ind.', 'Ilma.', 'vol.', 'eng.', 'rod.', 'Ph.D.', 'Dras.', 'pp.', 'elem.', 'máq.', 'cód.', 'eletr.', 'prod.', 'ref.', 'fil.', 'a.m.', 'A.M', 'obs.', 'N.T.', 'contab.', 'Sto.', 'lit.', 'educ.', 'rementente', 'desc.', 'próx.'],
9
+ ru: ['руб.', 'янв.', 'до н. э.', 'сент.', 'тел.', 'дек.', 'февр.', 'нояб.', 'апр.', 'н. э.', 'окт.', 'тыс.', 'авг.', 'проф.', 'н.э.', 'кв.', 'ул.', 'отд.'],
10
+ }
11
+
12
+ export const additionalSuppressions: Record<string, string[] > = {
13
+ en: [
14
+ 'TL;DR'
15
+ ],
9
16
  }
10
17
 
11
18
  export const leadingApostropheContractionSuppressions: Record<string, string[]> = {
12
- 'en': [`'cause`, `'til`, `'bout`, `'twas`, `'tis`],
13
- 'af': [`'n`]
19
+ 'en': [
20
+ `'cause`, `'til`, `'bout`, `'twas`, `'tis`
21
+ ],
22
+ 'af': [
23
+ `'n`
24
+ ]
14
25
  }
15
26
 
16
27
  export const nounSuppressions = [