ja-compromise 0.0.1 → 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (69) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +367 -6
  3. package/builds/ja-compromise.cjs +4389 -2286
  4. package/builds/ja-compromise.min.js +1 -1
  5. package/builds/ja-compromise.mjs +1 -1
  6. package/package.json +36 -23
  7. package/src/01-one/conjugate/conjugate-adj.js +53 -0
  8. package/src/01-one/conjugate/conjugate-verb.js +145 -0
  9. package/src/01-one/conjugate/deconjugate.js +178 -0
  10. package/src/01-one/conjugate/index.js +13 -0
  11. package/src/01-one/conjugate/kana.js +38 -0
  12. package/src/01-one/conjugate/tags.js +85 -0
  13. package/src/01-one/conjugate/verb-class.js +105 -0
  14. package/src/01-one/lexicon/_data.js +30 -0
  15. package/src/01-one/lexicon/api.js +61 -0
  16. package/src/01-one/lexicon/lexicon.js +163 -0
  17. package/src/01-one/lexicon/misc.js +115 -0
  18. package/src/01-one/lexicon/plugin.js +20 -0
  19. package/src/01-one/numbers/api.js +202 -0
  20. package/src/01-one/numbers/kanji-number.js +69 -0
  21. package/src/01-one/numbers/to-kanji.js +66 -0
  22. package/src/01-one/output/compute/dict.js +13 -0
  23. package/src/01-one/output/compute/english.js +13 -0
  24. package/src/01-one/output/compute/root.js +30 -0
  25. package/src/01-one/output/debug/_color.js +16 -0
  26. package/src/01-one/output/debug/index.js +24 -0
  27. package/src/01-one/output/debug/tags.js +56 -0
  28. package/src/01-one/output/plugin.js +13 -0
  29. package/src/01-one/romanji/api.js +16 -0
  30. package/src/01-one/romanji/compute/index.js +36 -0
  31. package/src/01-one/romanji/compute/kanji-reading/index.js +91 -0
  32. package/src/01-one/romanji/compute/kanji-reading/readings.js +861 -0
  33. package/src/01-one/romanji/compute/kanji-reading/words.js +31 -0
  34. package/src/01-one/romanji/compute/toRomanji/hiragana-map.js +175 -0
  35. package/src/01-one/romanji/compute/toRomanji/index.js +84 -0
  36. package/src/01-one/romanji/compute/toRomanji/katakana-map.js +128 -0
  37. package/src/01-one/romanji/plugin.js +7 -0
  38. package/src/01-one/tokenizer/methods/join-numbers.js +32 -0
  39. package/src/01-one/tokenizer/methods/join-up.js +66 -0
  40. package/src/01-one/tokenizer/methods/lib.js +66 -0
  41. package/src/01-one/tokenizer/methods/naiive-split.js +88 -0
  42. package/src/01-one/tokenizer/methods/okurigana.js +58 -0
  43. package/src/01-one/tokenizer/methods/terms.js +143 -0
  44. package/src/01-one/tokenizer/methods/trie/build.js +13 -0
  45. package/src/01-one/tokenizer/methods/trie/split-up.js +29 -0
  46. package/src/01-one/tokenizer/methods/whitespace.js +44 -0
  47. package/src/01-one/tokenizer/plugin.js +13 -0
  48. package/src/02-two/preTagger/compute/01-script.js +44 -0
  49. package/src/02-two/preTagger/compute/02-particles.js +82 -0
  50. package/src/02-two/preTagger/compute/03-verbs.js +111 -0
  51. package/src/02-two/preTagger/compute/04-adjectives.js +43 -0
  52. package/src/02-two/preTagger/compute/05-people.js +22 -0
  53. package/src/02-two/preTagger/compute/06-numbers.js +80 -0
  54. package/src/02-two/preTagger/compute/07-dates.js +105 -0
  55. package/src/02-two/preTagger/compute/index.js +52 -0
  56. package/src/02-two/preTagger/plugin.js +9 -0
  57. package/src/02-two/tagset/plugin.js +12 -0
  58. package/src/02-two/tagset/tags/dates.js +58 -0
  59. package/src/02-two/tagset/tags/misc.js +91 -0
  60. package/src/02-two/tagset/tags/nouns.js +131 -0
  61. package/src/02-two/tagset/tags/particles.js +50 -0
  62. package/src/02-two/tagset/tags/values.js +68 -0
  63. package/src/02-two/tagset/tags/verbs.js +93 -0
  64. package/src/_lib.js +2 -0
  65. package/src/_version.js +1 -0
  66. package/src/index.js +82 -0
  67. package/types/index.d.ts +268 -0
  68. package/types/japanese.d.ts +159 -0
  69. package/types/misc.d.ts +92 -0
@@ -0,0 +1,31 @@
1
+ // per-character readings can't know that 日本 is にほん and not にちほん.
2
+ // this is a small override-table for the words that get it wrong most often.
3
+ // (a full reading-dictionary is the real fix - see the changelog)
4
+ export default {
5
+ '私': 'わたし', '僕': 'ぼく', '俺': 'おれ', '君': 'きみ', '彼': 'かれ', '彼女': 'かのじょ',
6
+ '日本': 'にほん', '日本語': 'にほんご', '日本人': 'にほんじん', '東京': 'とうきょう',
7
+ '英語': 'えいご', '中国': 'ちゅうごく', '韓国': 'かんこく', '大阪': 'おおさか', '京都': 'きょうと',
8
+ '今日': 'きょう', '明日': 'あした', '昨日': 'きのう', '毎日': 'まいにち', '今朝': 'けさ',
9
+ '一昨日': 'おととい', '明後日': 'あさって', '今年': 'ことし', '去年': 'きょねん', '来年': 'らいねん',
10
+ '時間': 'じかん', '時計': 'とけい', '手紙': 'てがみ', '写真': 'しゃしん', '電話': 'でんわ',
11
+ '学生': 'がくせい', '先生': 'せんせい', '学校': 'がっこう', '大学': 'だいがく', '会社': 'かいしゃ',
12
+ '友達': 'ともだち', '家族': 'かぞく', '子供': 'こども', '大人': 'おとな', '両親': 'りょうしん',
13
+ '本': 'ほん', '人': 'ひと', '男': 'おとこ', '女': 'おんな', '水': 'みず', '火': 'ひ', '木': 'き',
14
+ '空': 'そら', '海': 'うみ', '山': 'やま', '川': 'かわ', '花': 'はな', '犬': 'いぬ', '猫': 'ねこ',
15
+ '車': 'くるま', '家': 'いえ', '店': 'みせ', '駅': 'えき', '道': 'みち', '町': 'まち', '国': 'くに',
16
+ '目': 'め', '耳': 'みみ', '口': 'くち', '手': 'て', '足': 'あし', '頭': 'あたま', '顔': 'かお',
17
+ '朝': 'あさ', '昼': 'ひる', '夜': 'よる', '雨': 'あめ', '雪': 'ゆき', '風': 'かぜ', '天気': 'てんき',
18
+ '食べ物': 'たべもの', '飲み物': 'のみもの', '御飯': 'ごはん', '料理': 'りょうり', '野菜': 'やさい',
19
+ '映画': 'えいが', '音楽': 'おんがく', '仕事': 'しごと', '勉強': 'べんきょう', '質問': 'しつもん',
20
+ '一': 'いち', '二': 'に', '三': 'さん', '四': 'よん', '五': 'ご', '六': 'ろく', '七': 'なな',
21
+ '八': 'はち', '九': 'きゅう', '十': 'じゅう', '百': 'ひゃく', '千': 'せん', '万': 'まん', '円': 'えん',
22
+ // 熟字訓 - readings that belong to the whole word, not its characters
23
+ '美味しい': 'おいしい', '大人': 'おとな', '今日': 'きょう', '一人': 'ひとり', '二人': 'ふたり',
24
+ '大丈夫': 'だいじょうぶ', '上手': 'じょうず', '下手': 'へた', '眼鏡': 'めがね', '果物': 'くだもの',
25
+ '八百屋': 'やおや', '土産': 'みやげ', '相撲': 'すもう', '風邪': 'かぜ', '田舎': 'いなか',
26
+ '素敵': 'すてき', '素晴らしい': 'すばらしい', '綺麗': 'きれい', '沢山': 'たくさん',
27
+ // 来る is irregular in speech as well as in grammar - こ / き / く
28
+ '来る': 'くる', '来ます': 'きます', '来ました': 'きました', '来た': 'きた', '来て': 'きて',
29
+ '来ない': 'こない', '来なかった': 'こなかった', '来い': 'こい', '来られる': 'こられる',
30
+ '食べる': 'たべる', '食べます': 'たべます', '食べた': 'たべた', '食べて': 'たべて',
31
+ }
@@ -0,0 +1,175 @@
1
+ // https://github.com/zkayser/jay_verb/blob/master/lib/japanese/to_romaji.rb
2
+ export default [
3
+ {},
4
+ // single-char
5
+ {
6
+ "あ": "a",
7
+ "い": "i",
8
+ "う": "u",
9
+ "え": "e",
10
+ "お": "o",
11
+ "か": "ka",
12
+ "き": "ki",
13
+ "く": "ku",
14
+ "け": "ke",
15
+ "こ": "ko",
16
+ "さ": "sa",
17
+ "し": "shi",
18
+ "す": "su",
19
+ "せ": "se",
20
+ "そ": "so",
21
+ "た": "ta",
22
+ "ち": "chi",
23
+ "つ": "tsu",
24
+ "て": "te",
25
+ "と": "to",
26
+ "な": "na",
27
+ "に": "ni",
28
+ "ぬ": "nu",
29
+ "ね": "ne",
30
+ "の": "no",
31
+ "は": "ha",
32
+ "ひ": "hi",
33
+ "ふ": "fu",
34
+ "へ": "he",
35
+ "ほ": "ho",
36
+ "ま": "ma",
37
+ "み": "mi",
38
+ "む": "mu",
39
+ "め": "me",
40
+ "も": "mo",
41
+ "や": "ya",
42
+ "ゆ": "yu",
43
+ "よ": "yo",
44
+ "ら": "ra",
45
+ "り": "ri",
46
+ "る": "ru",
47
+ "れ": "re",
48
+ "ろ": "ro",
49
+ "わ": "wa",
50
+ "を": "wo",
51
+ "ん": "n",
52
+ "が": "ga",
53
+ "ぎ": "gi",
54
+ "ぐ": "gu",
55
+ "げ": "ge",
56
+ "ご": "go",
57
+ "ざ": "za",
58
+ "じ": "ji",
59
+ "ず": "zu",
60
+ "ぜ": "ze",
61
+ "ぞ": "zo",
62
+ "だ": "da",
63
+ "ぢ": "dchi",
64
+ "づ": "dzu",
65
+ "で": "de",
66
+ "ど": "do",
67
+ "ば": "ba",
68
+ "び": "bi",
69
+ "ぶ": "bu",
70
+ "べ": "be",
71
+ "ぼ": "bo",
72
+ "ぱ": "pa",
73
+ "ぴ": "pi",
74
+ "ぷ": "pu",
75
+ "ぺ": "pe",
76
+ "ぽ": "po",
77
+ //specials
78
+ "ゃ": "ya",
79
+ "ゅ": "yu",
80
+ "ょ": "yo",
81
+ "っ": "",
82
+ "ぁ": "a",
83
+ "ぃ": "i",
84
+ "ぅ": "u",
85
+ "ぇ": "e",
86
+ "ぉ": "o",
87
+ "。": "."
88
+ },
89
+ // two-char
90
+ {
91
+ "きゃ": "kya",
92
+ "きゅ": "kyu",
93
+ "きょ": "kyo",
94
+ "しゃ": "sha",
95
+ "しゅ": "shu",
96
+ "しょ": "sho",
97
+ "ちゃ": "cha",
98
+ "ちゅ": "chu",
99
+ "ちょ": "cho",
100
+ "にゃ": "nya",
101
+ "にゅ": "nyu",
102
+ "にょ": "nyo",
103
+ "ひゃ": "hya",
104
+ "ひゅ": "hyu",
105
+ "ひょ": "hyo",
106
+ "みゃ": "mya",
107
+ "みゅ": "myu",
108
+ "みょ": "myo",
109
+ "りゃ": "rya",
110
+ "りゅ": "ryu",
111
+ "りょ": "ryo",
112
+ "ぎゃ": "gya",
113
+ "ぎゅ": "gyu",
114
+ "ぎょ": "gyo",
115
+ "じゃ": "ja",
116
+ "じゅ": "ju",
117
+ "じょ": "jo",
118
+ "ぢゃ": "dja",
119
+ "ぢゅ": "dju",
120
+ "ぢょ": "djo",
121
+ "びゃ": "bya",
122
+ "びゅ": "byu",
123
+ "びょ": "byo",
124
+ "ぴゃ": "pya",
125
+ "ぴゅ": "pyu",
126
+ "ぴょ": "pyo",
127
+ "てぃ": "ti",
128
+ "っか": "kka",
129
+ "っき": "kki",
130
+ "っく": "kku",
131
+ "っけ": "kke",
132
+ "っこ": "kko",
133
+ "っさ": "ssa",
134
+ "っし": "sshi",
135
+ "っす": "ssu",
136
+ "っせ": "sse",
137
+ "っそ": "sso",
138
+ "った": "tta",
139
+ "っち": "cchi",
140
+ "っつ": "ttsu",
141
+ "って": "tte",
142
+ "っと": "tto",
143
+ "っば": "bba",
144
+ "っび": "bbi",
145
+ "っぶ": "bbu",
146
+ "っべ": "bbe",
147
+ "っぼ": "bbo",
148
+ "っぱ": "ppa",
149
+ "っぴ": "ppi",
150
+ "っぷ": "ppu",
151
+ "っぺ": "ppe",
152
+ "っぽ": "ppo"
153
+ },
154
+ // three-char
155
+ {
156
+ "っきゃ": "kkya",
157
+ "っきゅ": "kkyu",
158
+ "っきょ": "kkyo",
159
+ "っしゃ": "ssha",
160
+ "っしゅ": "sshu",
161
+ "っしょ": "ssho",
162
+ "っちゃ": "ccha",
163
+ "っちゅ": "cchu",
164
+ "っちょ": "ccho",
165
+ "っじゃ": "jja",
166
+ "っじゅ": "jju",
167
+ "っじょ": "jjo",
168
+ "っびゃ": "bbya",
169
+ "っびゅ": "bbyu",
170
+ "っびょ": "bbyo",
171
+ "っぴゃ": "ppya",
172
+ "っぴゅ": "ppyu",
173
+ "っぴょ": "ppyo"
174
+ }
175
+ ]
@@ -0,0 +1,84 @@
1
+ import hMap from './hiragana-map.js'
2
+
3
+ let hasMulti = new Set(['き', 'ぎ', 'し', 'じ', 'ち', 'ぢ', 'っ', 'て', 'に', 'ひ', 'び', 'ぴ', 'み', 'り'])
4
+
5
+ // katakana and hiragana are the same 46 sounds, 0x60 apart in unicode.
6
+ // fold katakana down so one map handles both scripts.
7
+ const toHiragana = function (str) {
8
+ let out = ''
9
+ for (let i = 0; i < str.length; i += 1) {
10
+ let c = str[i]
11
+ if (c >= 'ァ' && c <= 'ヶ') {
12
+ out += String.fromCharCode(c.charCodeAt(0) - 0x60)
13
+ } else {
14
+ out += c
15
+ }
16
+ }
17
+ return out
18
+ }
19
+
20
+ // there are 46 of these
21
+ const isHiragana = function (ch) {
22
+ return ch >= "\u3040" && ch <= "\u309f";
23
+ }
24
+
25
+ const vowels = 'aiueo'
26
+
27
+ // sound-out japanese script in latin alphabet
28
+ const toRomanji = function (str) {
29
+ let chars = toHiragana(str).split('')
30
+ let out = ''
31
+
32
+ for (let i = 0; i < chars.length; i += 1) {
33
+ let c = chars[i]
34
+ // ー is the 長音符 - it lengthens the vowel of the syllable before it
35
+ if (c === 'ー' || c === 'ー') {
36
+ let last = out[out.length - 1]
37
+ if (last && vowels.includes(last)) {
38
+ out += last
39
+ }
40
+ continue
41
+ }
42
+ // pass non-hiragana right through
43
+ if (!isHiragana(c)) {
44
+ out += c
45
+ continue
46
+ }
47
+ // a lone っ doubles the consonant that follows it, and is silent at the
48
+ // end of a word - it never spells anything by itself
49
+ if (c === 'っ') {
50
+ let next = chars[i + 1] ? hMap[1][chars[i + 1]] : ''
51
+ if (next && !vowels.includes(next[0])) {
52
+ out += next[0]
53
+ }
54
+ continue
55
+ }
56
+ // look ahead at greedy multi-char sequences
57
+ if (hasMulti.has(c)) {
58
+ if (chars[i + 1]) {
59
+ let two = c + chars[i + 1]
60
+ if (hMap[2].hasOwnProperty(two)) {
61
+ out += hMap[2][two]
62
+ i += 1
63
+ continue
64
+ }
65
+ if (chars[i + 2]) {
66
+ let three = c + chars[i + 1] + chars[i + 2]
67
+ if (hMap[3].hasOwnProperty(three)) {
68
+ out += hMap[3][three]
69
+ i += 1
70
+ continue
71
+ }
72
+ }
73
+ }
74
+ }
75
+ // single-char map
76
+ out += hMap[1][c] || c
77
+ }
78
+ return out
79
+ }
80
+
81
+ export default toRomanji
82
+
83
+ // console.log(toRomanji('ひらがな カタカナ'))
84
+ // console.log(toRomanji('あっきょっつああ'))
@@ -0,0 +1,128 @@
1
+ let out = {
2
+ "a": "&#12450;",
3
+ "i": "&#12452;",
4
+ "u": "&#12454;",
5
+ "e": "&#12456;",
6
+ "o": "&#12458;",
7
+ "n": "&#12531;",
8
+ "m": "&#12531;",
9
+ "ka": "&#12459;",
10
+ "ki": "&#12461;",
11
+ "ku": "&#12463;",
12
+ "ke": "&#12465;",
13
+ "ko": "&#12467;",
14
+ "sa": "&#12469;",
15
+ "si": "&#12471;",
16
+ "su": "&#12473;",
17
+ "se": "&#12475;",
18
+ "so": "&#12477;",
19
+ "ta": "&#12479;",
20
+ "ti": "&#12481;",
21
+ "tu": "&#12484;",
22
+ "te": "&#12486;",
23
+ "to": "&#12488;",
24
+ "na": "&#12490;",
25
+ "ni": "&#12491;",
26
+ "nu": "&#12492;",
27
+ "ne": "&#12493;",
28
+ "no": "&#12494;",
29
+ "ha": "&#12495;",
30
+ "hi": "&#12498;",
31
+ "fu": "&#12501;",
32
+ "he": "&#12504;",
33
+ "ho": "&#12507;",
34
+ "ma": "&#12510;",
35
+ "mi": "&#12511;",
36
+ "mu": "&#12512;",
37
+ "me": "&#12513;",
38
+ "mo": "&#12514;",
39
+ "ya": "&#12516;",
40
+ "yu": "&#12518;",
41
+ "yo": "&#12520;",
42
+ "ra": "&#12521;",
43
+ "ri": "&#12522;",
44
+ "ru": "&#12523;",
45
+ "re": "&#12524;",
46
+ "ro": "&#12525;",
47
+ "wa": "&#12527;",
48
+ "wo": "&#12530;",
49
+ "ga": "&#12460;",
50
+ "gi": "&#12462;",
51
+ "gu": "&#12464;",
52
+ "ge": "&#12466;",
53
+ "go": "&#12468;",
54
+ "za": "&#12470;",
55
+ "zi": "&#12472;",
56
+ "zu": "&#12474;",
57
+ "ze": "&#12476;",
58
+ "zo": "&#12478;",
59
+ "da": "&#12480;",
60
+ "di": "&#12482;",
61
+ "du": "&#12485;",
62
+ "de": "&#12487;",
63
+ "do": "&#12489;",
64
+ "ba": "&#12496;",
65
+ "bi": "&#12499;",
66
+ "bu": "&#12502;",
67
+ "be": "&#12505;",
68
+ "bo": "&#12508;",
69
+ "pa": "&#12497;",
70
+ "pi": "&#12500;",
71
+ "pu": "&#12503;",
72
+ "pe": "&#12506;",
73
+ "po": "&#12509;",
74
+ "ja": "&#12472;&#12515;",
75
+ "ju": "&#12472;&#12517;",
76
+ "jo": "&#12472;&#12519;",
77
+ "ji": "&#12472;",
78
+ "vi": "&#12532;&#12451;",
79
+ "kya": "&#12461;&#12515;",
80
+ "kyu": "&#12461;&#12517;",
81
+ "kyo": "&#12461;&#12519;",
82
+ "sha": "&#12471;&#12515;",
83
+ "shu": "&#12471;&#12517;",
84
+ "sho": "&#12471;&#12519;",
85
+ "shi": "&#12471;",
86
+ "tsu": "&#12484;",
87
+ "cha": "&#12481;&#12515;",
88
+ "chu": "&#12481;&#12517;",
89
+ "cho": "&#12481;&#12519;",
90
+ "chi": "&#12481;",
91
+ "nya": "&#12491;&#12515;",
92
+ "nyu": "&#12491;&#12517;",
93
+ "nyo": "&#12491;&#12519;",
94
+ "hya": "&#12498;&#12515;",
95
+ "hyu": "&#12498;&#12517;",
96
+ "hyo": "&#12498;&#12519;",
97
+ "mya": "&#12511;&#12515;",
98
+ "myu": "&#12511;&#12517;",
99
+ "myo": "&#12511;&#12519;",
100
+ "rya": "&#12522;&#12515;",
101
+ "ryu": "&#12522;&#12517;",
102
+ "ryo": "&#12522;&#12519;",
103
+ "gya": "&#12462;&#12515;",
104
+ "gyu": "&#12462;&#12517;",
105
+ "gyo": "&#12462;&#12519;",
106
+ "bya": "&#12499;&#12515;",
107
+ "byu": "&#12499;&#12517;",
108
+ "byo": "&#12499;&#12519;",
109
+ "pya": "&#12500;&#12515;",
110
+ "pyu": "&#12500;&#12517;",
111
+ "pyo": "&#12500;&#12519;",
112
+ "tsu": "&#12483;"
113
+ }
114
+
115
+ const toUtf8 = function (s) {
116
+ return decodeURIComponent(s)
117
+ }
118
+
119
+ Object.entries(out).forEach(a => {
120
+ let num = a[1].replace(/&#/g, '').replace(/;/, '')
121
+ let hex = Number(num).toString(16);
122
+ // console.log('\\u' + hex)
123
+ console.log(toUtf8('\\u' + hex)) //eslint-disable-line
124
+ })
125
+
126
+
127
+ // console.log(toUtf8('\u30a2\u30e1\u30ea\u30ab\u5408\u8846\u56fd')) //アメリカ合衆国 (USA)
128
+ // console.log(toUtf8('\u12496'))
@@ -0,0 +1,7 @@
1
+ import api from './api.js'
2
+ import compute from './compute/index.js'
3
+
4
+ export default {
5
+ compute,
6
+ api
7
+ }
@@ -0,0 +1,32 @@
1
+ import { isNumeral } from '../../numbers/kanji-number.js'
2
+
3
+ const allNumeral = function (str) {
4
+ for (let i = 0; i < str.length; i += 1) {
5
+ if (!isNumeral(str[i])) {
6
+ return false
7
+ }
8
+ }
9
+ return str.length > 0
10
+ }
11
+
12
+ /**
13
+ * 二十三 arrives as 二|十|三, because each of them is a lexicon entry on its own.
14
+ * a run of numerals is one number.
15
+ */
16
+ const joinNumbers = function (arr) {
17
+ let out = []
18
+ for (let i = 0; i < arr.length; i += 1) {
19
+ if (!allNumeral(arr[i])) {
20
+ out.push(arr[i])
21
+ continue
22
+ }
23
+ let run = arr[i]
24
+ while (arr[i + 1] !== undefined && allNumeral(arr[i + 1])) {
25
+ run += arr[i + 1]
26
+ i += 1
27
+ }
28
+ out.push(run)
29
+ }
30
+ return out
31
+ }
32
+ export default joinNumbers
@@ -0,0 +1,66 @@
1
+ import lexicon from '../../lexicon/lexicon.js'
2
+ import { getType, isKanji } from './lib.js'
3
+
4
+ // suffixes that belong to the word in front of them
5
+ const suffixes = new Set(['たち', '達'])
6
+ // honorific prefixes that belong to the word behind them
7
+ const prefixes = new Set(['お', 'ご', '御'])
8
+
9
+ // two unknown characters may only merge if they're the same script.
10
+ // 'kanji then hiragana' used to merge too, which glued particles onto nouns.
11
+ const mergeTypes = function (a, b) {
12
+ return a === b && a !== 'punctuation' && a !== 'other'
13
+ }
14
+
15
+ // an unknown kanji run longer than this is almost certainly two words.
16
+ // katakana has no such limit - ニュージーランド is one word.
17
+ const MAX_RUN = 4
18
+ const runLimit = (type) => (type === 'kanji' || type === 'hiragana' ? MAX_RUN : Infinity)
19
+
20
+ /** glue neighbouring unknown characters into plausible words */
21
+ const joinUp = function (arr) {
22
+ let out = []
23
+ for (let i = 0; i < arr.length; i += 1) {
24
+ let c = arr[i]
25
+ if (c === null || c === '') {
26
+ continue
27
+ }
28
+ let last = out[out.length - 1]
29
+ // 私 + たち, 田中 + さん
30
+ if (suffixes.has(c) && last && !lexicon[last + c]) {
31
+ out[out.length - 1] = last + c
32
+ continue
33
+ }
34
+ // お + 金, ご + 飯
35
+ if (prefixes.has(c) && arr[i + 1] && isKanji(arr[i + 1][0]) && !lexicon[c]) {
36
+ out.push(c + arr[i + 1])
37
+ i += 1
38
+ continue
39
+ }
40
+ if (c.length === 1 && !lexicon[c]) {
41
+ let type = getType(c)
42
+ let run = c
43
+ // race ahead, joining same-script characters.
44
+ // a *known* single kanji may join an already-started run (日+本+人),
45
+ // but never starts one - so 「読んでいる人」 keeps 人 on its own.
46
+ let limit = runLimit(type)
47
+ while (run.length < limit && arr[i + 1] !== undefined) {
48
+ let next = arr[i + 1]
49
+ if (next.length !== 1 || !mergeTypes(type, getType(next))) {
50
+ break
51
+ }
52
+ if (lexicon[next] && type !== 'kanji') {
53
+ break
54
+ }
55
+ run += next
56
+ i += 1
57
+ }
58
+ out.push(run)
59
+ continue
60
+ }
61
+ out.push(c)
62
+ }
63
+ return out
64
+ }
65
+
66
+ export default joinUp
@@ -0,0 +1,66 @@
1
+ // https://github.com/darren-lester/nihongo/blob/master/src/analysers.js
2
+
3
+ const isHiragana = function (ch) {
4
+ return ch >= '぀' && ch <= 'ゟ'
5
+ }
6
+
7
+ const isKatakana = function (ch) {
8
+ // ・ (U+30FB) sits inside the katakana block but is punctuation - it
9
+ // separates the parts of a foreign name, ジョン・スミス
10
+ if (ch === '・' || ch === '゠') {
11
+ return false
12
+ }
13
+ return (ch >= '゠' && ch <= 'ヿ') || (ch >= 'ㇰ' && ch <= 'ㇿ')
14
+ }
15
+
16
+ const isKanji = function (ch) {
17
+ return (
18
+ (ch >= '一' && ch <= '龯') ||
19
+ (ch >= '㐀' && ch <= '䶿') ||
20
+ ch === '々' || // 々 - the repeat-mark, as in 人々
21
+ ch === '〆' ||
22
+ ch === 'ヶ' ||
23
+ ch === '𠮟'
24
+ )
25
+ }
26
+
27
+ const isNumber = function (c) {
28
+ return (c >= '0' && c <= '9') || (c >= '0' && c <= '9') // half & full-width
29
+ }
30
+
31
+ const isAscii = function (c) {
32
+ return /[a-zA-Z]/.test(c) || (c >= 'A' && c <= 'z')
33
+ }
34
+
35
+ // 、。!? and friends
36
+ const isPunctuation = function (c) {
37
+ return /[、。,.!?!?,.::;;・…〜~「」『』()()【】〔〕《》〈〉\s]/.test(c)
38
+ }
39
+
40
+ const getType = function (c) {
41
+ // ー is the 長音符 - it continues whatever script it follows
42
+ if (c === 'ー' || c === 'ー') {
43
+ return 'katakana'
44
+ }
45
+ if (isHiragana(c)) {
46
+ return 'hiragana'
47
+ }
48
+ if (isKatakana(c)) {
49
+ return 'katakana'
50
+ }
51
+ if (isKanji(c)) {
52
+ return 'kanji'
53
+ }
54
+ if (isNumber(c)) {
55
+ return 'number'
56
+ }
57
+ if (isAscii(c)) {
58
+ return 'ascii'
59
+ }
60
+ if (isPunctuation(c)) {
61
+ return 'punctuation'
62
+ }
63
+ return 'other'
64
+ }
65
+
66
+ export { isHiragana, isKatakana, isKanji, isNumber, isAscii, isPunctuation, getType }
@@ -0,0 +1,88 @@
1
+ import { isHiragana, isKatakana, isKanji, isNumber, isAscii } from './lib.js'
2
+
3
+ let markers = [
4
+ // 'は',// - topic marker
5
+ // 'が',// - subject
6
+ // 'を',// - direct object
7
+ // 'の',// - possessive (reverse "of"), question mark (plain)
8
+ // 'な',// - marks an adjective
9
+ //'も',// - "also" (substitutes for wa, ga, or w
10
+ //'で',// - "by means of", "in"/"at" for actio
11
+ //'に',// - indirect object, "in"/"at" for existen
12
+ //'と',// - "and", object of "say" or "thin
13
+ //'や',// - "and" for a li
14
+ //'へ',// - destination "t
15
+ //'か',// - question mark (polit
16
+
17
+ '・', //word-splitter
18
+ '、',//comma
19
+ ':', //colon
20
+ ' ',//space
21
+ '\t',//tab
22
+ '\n',//newline
23
+ ]
24
+
25
+ let always = new Set(markers)
26
+
27
+ const getType = function (c) {
28
+ if (isHiragana(c)) {
29
+ return 'hiragana'
30
+ }
31
+ if (isKatakana(c)) {
32
+ return 'katakana'
33
+ }
34
+ if (isKanji(c)) {
35
+ return 'kanji'
36
+ }
37
+ if (isNumber(c)) {
38
+ return 'number'
39
+ }
40
+ if (isAscii(c)) {
41
+ return 'ascii'
42
+ }
43
+ return null
44
+ }
45
+
46
+ // naiive split by Hiragana/Katakana/Kanji segments
47
+ const tokenize = function (txt) {
48
+ let chars = txt.split('')
49
+ let arr = []
50
+ let past = null
51
+ let run = []
52
+ // loop through every character
53
+ for (let i = 0; i < chars.length; i += 1) {
54
+ let c = chars[i]
55
+ // clear split punctuation
56
+ if (always.has(c)) {
57
+ run.push(c)
58
+ arr.push(run.join(''))
59
+ run = []
60
+ past = null
61
+ continue
62
+ }
63
+ // which script is the character?
64
+ let type = getType(c)
65
+
66
+ if (past == null) {
67
+ past = type
68
+ }
69
+ // keep it going
70
+ if (type === past) {
71
+ run.push(c)
72
+ continue
73
+ }
74
+ // start of a new type
75
+ if (run.length > 0) {
76
+ arr.push(run.join(''))
77
+ }
78
+ run = [c]
79
+ past = type
80
+ }
81
+ if (run.length > 0) {
82
+ arr.push(run.join(''))
83
+ }
84
+ return arr//.filter(s => s.length > 1)
85
+ }
86
+ export default tokenize
87
+ // console.log(tokenize('小さな子供は食料品店に歩いた'))
88
+