ja-compromise 0.0.1 → 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (69) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +367 -6
  3. package/builds/ja-compromise.cjs +4389 -2286
  4. package/builds/ja-compromise.min.js +1 -1
  5. package/builds/ja-compromise.mjs +1 -1
  6. package/package.json +36 -23
  7. package/src/01-one/conjugate/conjugate-adj.js +53 -0
  8. package/src/01-one/conjugate/conjugate-verb.js +145 -0
  9. package/src/01-one/conjugate/deconjugate.js +178 -0
  10. package/src/01-one/conjugate/index.js +13 -0
  11. package/src/01-one/conjugate/kana.js +38 -0
  12. package/src/01-one/conjugate/tags.js +85 -0
  13. package/src/01-one/conjugate/verb-class.js +105 -0
  14. package/src/01-one/lexicon/_data.js +30 -0
  15. package/src/01-one/lexicon/api.js +61 -0
  16. package/src/01-one/lexicon/lexicon.js +163 -0
  17. package/src/01-one/lexicon/misc.js +115 -0
  18. package/src/01-one/lexicon/plugin.js +20 -0
  19. package/src/01-one/numbers/api.js +202 -0
  20. package/src/01-one/numbers/kanji-number.js +69 -0
  21. package/src/01-one/numbers/to-kanji.js +66 -0
  22. package/src/01-one/output/compute/dict.js +13 -0
  23. package/src/01-one/output/compute/english.js +13 -0
  24. package/src/01-one/output/compute/root.js +30 -0
  25. package/src/01-one/output/debug/_color.js +16 -0
  26. package/src/01-one/output/debug/index.js +24 -0
  27. package/src/01-one/output/debug/tags.js +56 -0
  28. package/src/01-one/output/plugin.js +13 -0
  29. package/src/01-one/romanji/api.js +16 -0
  30. package/src/01-one/romanji/compute/index.js +36 -0
  31. package/src/01-one/romanji/compute/kanji-reading/index.js +91 -0
  32. package/src/01-one/romanji/compute/kanji-reading/readings.js +861 -0
  33. package/src/01-one/romanji/compute/kanji-reading/words.js +31 -0
  34. package/src/01-one/romanji/compute/toRomanji/hiragana-map.js +175 -0
  35. package/src/01-one/romanji/compute/toRomanji/index.js +84 -0
  36. package/src/01-one/romanji/compute/toRomanji/katakana-map.js +128 -0
  37. package/src/01-one/romanji/plugin.js +7 -0
  38. package/src/01-one/tokenizer/methods/join-numbers.js +32 -0
  39. package/src/01-one/tokenizer/methods/join-up.js +66 -0
  40. package/src/01-one/tokenizer/methods/lib.js +66 -0
  41. package/src/01-one/tokenizer/methods/naiive-split.js +88 -0
  42. package/src/01-one/tokenizer/methods/okurigana.js +58 -0
  43. package/src/01-one/tokenizer/methods/terms.js +143 -0
  44. package/src/01-one/tokenizer/methods/trie/build.js +13 -0
  45. package/src/01-one/tokenizer/methods/trie/split-up.js +29 -0
  46. package/src/01-one/tokenizer/methods/whitespace.js +44 -0
  47. package/src/01-one/tokenizer/plugin.js +13 -0
  48. package/src/02-two/preTagger/compute/01-script.js +44 -0
  49. package/src/02-two/preTagger/compute/02-particles.js +82 -0
  50. package/src/02-two/preTagger/compute/03-verbs.js +111 -0
  51. package/src/02-two/preTagger/compute/04-adjectives.js +43 -0
  52. package/src/02-two/preTagger/compute/05-people.js +22 -0
  53. package/src/02-two/preTagger/compute/06-numbers.js +80 -0
  54. package/src/02-two/preTagger/compute/07-dates.js +105 -0
  55. package/src/02-two/preTagger/compute/index.js +52 -0
  56. package/src/02-two/preTagger/plugin.js +9 -0
  57. package/src/02-two/tagset/plugin.js +12 -0
  58. package/src/02-two/tagset/tags/dates.js +58 -0
  59. package/src/02-two/tagset/tags/misc.js +91 -0
  60. package/src/02-two/tagset/tags/nouns.js +131 -0
  61. package/src/02-two/tagset/tags/particles.js +50 -0
  62. package/src/02-two/tagset/tags/values.js +68 -0
  63. package/src/02-two/tagset/tags/verbs.js +93 -0
  64. package/src/_lib.js +2 -0
  65. package/src/_version.js +1 -0
  66. package/src/index.js +82 -0
  67. package/types/index.d.ts +268 -0
  68. package/types/japanese.d.ts +159 -0
  69. package/types/misc.d.ts +92 -0
@@ -0,0 +1,202 @@
1
+ import toNumber from './kanji-number.js'
2
+ import toKanji, { toDigits } from './to-kanji.js'
3
+ import counters from '../../../lexicon/counters.js'
4
+
5
+ const isWide = /[0-9]/
6
+ const isKanjiNumeral = /[〇零一二三四五六七八九十百千万億兆壱弐参拾]/
7
+
8
+ // pull the number out of a match, along with how it was written
9
+ const parse = function (view) {
10
+ let terms = view.docs[0] || []
11
+ let term = terms.find(t => typeof t.number === 'number')
12
+ if (!term) {
13
+ // not computed yet - read it off the text
14
+ term = terms[0]
15
+ let num = term ? toNumber(term.text) : null
16
+ return { num, term, kanji: term ? isKanjiNumeral.test(term.text) : false, wide: false }
17
+ }
18
+ return {
19
+ num: term.number,
20
+ term,
21
+ kanji: isKanjiNumeral.test(term.text),
22
+ wide: isWide.test(term.text),
23
+ }
24
+ }
25
+
26
+ // swap a term's text in place. .replaceWith() would put a space between the
27
+ // number and its counter, and japanese doesn't use one - 五冊, never 五 冊
28
+ const setText = function (term, str, num) {
29
+ term.text = str
30
+ term.normal = str.toLowerCase()
31
+ if (typeof num === 'number') {
32
+ term.number = num
33
+ }
34
+ }
35
+
36
+ // write a number back in the shape it was found in
37
+ const write = function (view, num, form) {
38
+ let { term, kanji, wide } = parse(view)
39
+ if (!term || typeof num !== 'number') {
40
+ return view
41
+ }
42
+ let str
43
+ if (form === 'kanji' || (form === undefined && kanji)) {
44
+ str = toKanji(num)
45
+ } else {
46
+ str = toDigits(num, form === undefined ? wide : false)
47
+ }
48
+ if (str === null) {
49
+ return view
50
+ }
51
+ setText(term, str, num)
52
+ return view
53
+ }
54
+
55
+
56
+ const addMethod = function (View) {
57
+ class Numbers extends View {
58
+ constructor(document, pointer, groups) {
59
+ super(document, pointer, groups)
60
+ this.viewType = 'Numbers'
61
+ }
62
+ /** the parsed number, plus how it was written */
63
+ parse(n) {
64
+ return this.getNth(n).map(parse)
65
+ }
66
+ /** the value of each number - 「二十三」 gives 23 */
67
+ get(n) {
68
+ // the second .map is a plain Array.map - compromise's own .map() only
69
+ // unwraps to an array when the callback returns strings or objects
70
+ return this.getNth(n).map(parse).map(o => o.num)
71
+ }
72
+ json(n) {
73
+ let opts = typeof n === 'object' ? n : {}
74
+ return this.getNth(n).map(m => {
75
+ let json = m.toView().json(opts)[0]
76
+ let found = parse(m)
77
+ json.number = { num: found.num, counter: m.units().text() || null }
78
+ return json
79
+ }, [])
80
+ }
81
+ /** the 助数詞 attached to each number - 五冊 gives 冊 */
82
+ units() {
83
+ return this.growRight('#Counter').match('#Counter$')
84
+ }
85
+ /** only the ordinals - 三番目 */
86
+ isOrdinal() {
87
+ return this.if('#Ordinal')
88
+ }
89
+ /** only the cardinals */
90
+ isCardinal() {
91
+ return this.if('#Cardinal')
92
+ }
93
+ /** write each number in digits - 二十三 becomes 23 */
94
+ toNumber() {
95
+ this.forEach(m => {
96
+ let { num } = parse(m)
97
+ if (num !== null) {
98
+ write(m, num, 'digits')
99
+ }
100
+ })
101
+ return this
102
+ }
103
+ /** write each number in kanji - 23 becomes 二十三 */
104
+ toText() {
105
+ this.forEach(m => {
106
+ let { num } = parse(m)
107
+ if (num !== null) {
108
+ write(m, num, 'kanji')
109
+ }
110
+ })
111
+ return this
112
+ }
113
+ /** add thousands-separators - 1234 becomes 1,234 */
114
+ toLocaleString() {
115
+ this.forEach(m => {
116
+ let { num, term } = parse(m)
117
+ if (num === null || !term) {
118
+ return
119
+ }
120
+ setText(term, num.toLocaleString(), num)
121
+ })
122
+ return this
123
+ }
124
+ /** replace each number with this one, keeping its script */
125
+ set(n) {
126
+ if (typeof n !== 'number') {
127
+ return this
128
+ }
129
+ this.forEach(m => write(m, n))
130
+ return this
131
+ }
132
+ add(n) {
133
+ if (typeof n !== 'number') {
134
+ return this
135
+ }
136
+ this.forEach(m => {
137
+ let { num } = parse(m)
138
+ if (num !== null) {
139
+ write(m, num + n)
140
+ }
141
+ })
142
+ return this
143
+ }
144
+ subtract(n) {
145
+ return this.add(-1 * n)
146
+ }
147
+ increment() {
148
+ return this.add(1)
149
+ }
150
+ decrement() {
151
+ return this.add(-1)
152
+ }
153
+ isEqual(n) {
154
+ return this.filter(m => parse(m).num === n)
155
+ }
156
+ greaterThan(n) {
157
+ return this.filter(m => parse(m).num > n)
158
+ }
159
+ lessThan(n) {
160
+ return this.filter(m => parse(m).num < n)
161
+ }
162
+ between(min, max) {
163
+ return this.filter(m => {
164
+ let { num } = parse(m)
165
+ return num > min && num < max
166
+ })
167
+ }
168
+ update(pointer) {
169
+ let m = new Numbers(this.document, pointer)
170
+ m._cache = this._cache
171
+ return m
172
+ }
173
+ }
174
+
175
+ /** every number in the document */
176
+ View.prototype.numbers = function (n) {
177
+ let m = this.match('#Value+')
178
+ m = new Numbers(this.document, m.pointer)
179
+ return typeof n === 'number' ? m.eq(n) : m
180
+ }
181
+ /** .numbers() alias, to match english compromise */
182
+ View.prototype.values = View.prototype.numbers
183
+
184
+ /** every 助数詞 in the document */
185
+ View.prototype.counters = function (n) {
186
+ let m = this.match('#Counter')
187
+ return typeof n === 'number' ? m.eq(n) : m
188
+ }
189
+ /** every 円/ドル amount */
190
+ View.prototype.money = function (n) {
191
+ let m = this.match('#Value+ #Currency')
192
+ return typeof n === 'number' ? m.eq(n) : m
193
+ }
194
+ /** every percentage */
195
+ View.prototype.percentages = function (n) {
196
+ let m = this.match('#Value+ #Percent')
197
+ return typeof n === 'number' ? m.eq(n) : m
198
+ }
199
+ }
200
+
201
+ export default { api: addMethod }
202
+ export { counters }
@@ -0,0 +1,69 @@
1
+ // 漢数字 → a javascript number.
2
+ // japanese numerals are positional-by-power rather than positional-by-digit:
3
+ // 三百二十一 is (3×100) + (2×10) + 1, and 三万 is 3×10,000.
4
+
5
+ const digits = {
6
+ '〇': 0, '零': 0, '0': 0, '0': 0,
7
+ '一': 1, '壱': 1, '1': 1, '1': 1,
8
+ '二': 2, '弐': 2, '2': 2, '2': 2,
9
+ '三': 3, '参': 3, '3': 3, '3': 3,
10
+ '四': 4, '肆': 4, '4': 4, '4': 4,
11
+ '五': 5, '伍': 5, '5': 5, '5': 5,
12
+ '六': 6, '陸': 6, '6': 6, '6': 6,
13
+ '七': 7, '漆': 7, '7': 7, '7': 7,
14
+ '八': 8, '捌': 8, '8': 8, '8': 8,
15
+ '九': 9, '玖': 9, '9': 9, '9': 9,
16
+ }
17
+ // powers that stack inside a group
18
+ const small = { '十': 10, '拾': 10, '百': 100, '佰': 100, '千': 1000, '仟': 1000 }
19
+ // powers that close a group off
20
+ const large = { '万': 1e4, '萬': 1e4, '億': 1e8, '兆': 1e12, '京': 1e16 }
21
+
22
+ const isNumeral = function (c) {
23
+ return digits[c] !== undefined || small[c] !== undefined || large[c] !== undefined
24
+ }
25
+
26
+ /** parse a japanese numeral - returns null if the string isn't one */
27
+ const toNumber = function (str) {
28
+ if (!str) {
29
+ return null
30
+ }
31
+ let total = 0 // everything closed off by 万/億/兆
32
+ let section = 0 // the current group, below 10,000
33
+ let current = 0 // the digits seen since the last power
34
+ let seen = false
35
+
36
+ for (let i = 0; i < str.length; i += 1) {
37
+ let c = str[i]
38
+ if (digits[c] !== undefined) {
39
+ // 15 and 15 are positional, so keep multiplying up
40
+ current = Number(current * 10) + Number(digits[c])
41
+ seen = true
42
+ continue
43
+ }
44
+ if (small[c] !== undefined) {
45
+ // 十 on its own is 10, not 0
46
+ section += (current === 0 ? 1 : current) * small[c]
47
+ current = 0
48
+ seen = true
49
+ continue
50
+ }
51
+ if (large[c] !== undefined) {
52
+ // a bare 万 is 10,000 - but 五十万 is 50×10,000, not 51×10,000
53
+ let group = section + current
54
+ total += (group === 0 ? 1 : group) * large[c]
55
+ section = 0
56
+ current = 0
57
+ seen = true
58
+ continue
59
+ }
60
+ return null // not a numeral
61
+ }
62
+ if (!seen) {
63
+ return null
64
+ }
65
+ return total + section + current
66
+ }
67
+
68
+ export default toNumber
69
+ export { isNumeral, digits, small, large }
@@ -0,0 +1,66 @@
1
+ // a number → 漢数字.
2
+ //
3
+ // japanese numerals are fully compositional - 23 is 二十三, literally
4
+ // 'two-ten-three' - so this is arithmetic rather than a lookup table.
5
+ // the one wrinkle is that japanese groups by 10,000 (万) and not by 1,000.
6
+
7
+ const digit = ['', '一', '二', '三', '四', '五', '六', '七', '八', '九']
8
+ // powers that stack inside a group of four digits
9
+ const small = [['千', 1000], ['百', 100], ['十', 10]]
10
+ // powers that close a group off
11
+ const large = [['京', 1e16], ['兆', 1e12], ['億', 1e8], ['万', 1e4]]
12
+
13
+ // one group of four digits: 千百十 plus the remaining digit
14
+ const toGroup = function (n) {
15
+ let out = ''
16
+ for (let i = 0; i < small.length; i += 1) {
17
+ let [kanji, power] = small[i]
18
+ let d = Math.floor(n / power)
19
+ if (d === 0) {
20
+ continue
21
+ }
22
+ // 一十 and 一百 aren't written - 10 is 十, not 一十
23
+ out += (d === 1 ? '' : digit[d]) + kanji
24
+ n -= d * power
25
+ }
26
+ return out + digit[n]
27
+ }
28
+
29
+ /** 23 → 二十三 */
30
+ const toKanji = function (num) {
31
+ if (typeof num !== 'number' || !Number.isFinite(num)) {
32
+ return null
33
+ }
34
+ if (num < 0) {
35
+ return 'マイナス' + toKanji(Math.abs(num))
36
+ }
37
+ num = Math.round(num)
38
+ if (num === 0) {
39
+ return '〇'
40
+ }
41
+ let out = ''
42
+ for (let i = 0; i < large.length; i += 1) {
43
+ let [kanji, power] = large[i]
44
+ let chunk = Math.floor(num / power)
45
+ if (chunk === 0) {
46
+ continue
47
+ }
48
+ // 万 and up keep their 一 - 10,000 is 一万, never just 万
49
+ out += toGroup(chunk) + kanji
50
+ num -= chunk * power
51
+ }
52
+ return out + toGroup(num)
53
+ }
54
+
55
+ // 23 → '23', using the same width of digit the source used
56
+ const fullWidth = '0123456789'
57
+ const toDigits = function (num, wide) {
58
+ let str = String(num)
59
+ if (!wide) {
60
+ return str
61
+ }
62
+ return str.replace(/[0-9]/g, d => fullWidth[Number(d)])
63
+ }
64
+
65
+ export default toKanji
66
+ export { toDigits }
@@ -0,0 +1,13 @@
1
+ export default {
2
+ は: '-',
3
+ が: '-',
4
+ 目: 'eye',
5
+ '私たち': 'we',
6
+ '彼': 'he',
7
+ '彼女': 'she',
8
+ '泳い': 'swim',
9
+ '泳ぎ': 'swimming',
10
+ '歩い': 'walk',
11
+ 'する': '(do)',
12
+ '家': 'house'
13
+ }
@@ -0,0 +1,13 @@
1
+ import dict from './dict.js'
2
+
3
+ const addEnglish = function (view) {
4
+ view.docs.forEach(terms => {
5
+ terms.forEach(term => {
6
+ if (dict[term.normal]) {
7
+ term.english = dict[term.normal]
8
+ }
9
+ })
10
+ })
11
+ return view
12
+ }
13
+ export default addEnglish
@@ -0,0 +1,30 @@
1
+ import { roots } from '../../lexicon/lexicon.js'
2
+ import { deconjugate } from '../../conjugate/index.js'
3
+
4
+ const isKnown = (w) => lexicon.hasOwnProperty(w)
5
+
6
+ // the dictionary-form of each word - 食べました → 食べる, 高かった → 高い
7
+ const addRoot = function (view) {
8
+ view.docs.forEach(terms => {
9
+ terms.forEach(term => {
10
+ if (term.root) {
11
+ return
12
+ }
13
+ if (roots[term.text] !== undefined) {
14
+ term.root = roots[term.text]
15
+ return
16
+ }
17
+ // an unknown verb can still be walked back to its dictionary-form
18
+ if (term.tags.has('Verb')) {
19
+ let found = deconjugate(term.text, isKnown)
20
+ if (found) {
21
+ term.root = found.root
22
+ return
23
+ }
24
+ }
25
+ term.root = term.text
26
+ })
27
+ })
28
+ return view
29
+ }
30
+ export default addRoot
@@ -0,0 +1,16 @@
1
+ // https://stackoverflow.com/questions/9781218/how-to-change-node-jss-console-font-color
2
+ const reset = '\x1b[0m'
3
+
4
+ //cheaper than requiring chalk
5
+ const cli = {
6
+ green: str => '\x1b[32m' + str + reset,
7
+ red: str => '\x1b[31m' + str + reset,
8
+ blue: str => '\x1b[34m' + str + reset,
9
+ magenta: str => '\x1b[35m' + str + reset,
10
+ cyan: str => '\x1b[36m' + str + reset,
11
+ yellow: str => '\x1b[33m' + str + reset,
12
+ black: str => '\x1b[30m' + str + reset,
13
+ dim: str => '\x1b[2m' + str + reset,
14
+ i: str => '\x1b[3m' + str + reset,
15
+ }
16
+ export default cli
@@ -0,0 +1,24 @@
1
+ /* eslint-disable no-console */
2
+ import showTags from './tags.js'
3
+
4
+ function isClientSide() {
5
+ return typeof window !== 'undefined' && window.document
6
+ }
7
+ //output some helpful stuff to the console
8
+ const debug = function (opts = {}) {
9
+ let view = this
10
+ if (typeof opts === 'string') {
11
+ let tmp = {}
12
+ tmp[opts] = true //allow string input
13
+ opts = tmp
14
+ }
15
+ if (isClientSide()) {
16
+ return view
17
+ }
18
+ if (opts.tags !== false) {
19
+ showTags(view)
20
+ console.log('\n')
21
+ }
22
+ return view
23
+ }
24
+ export default debug
@@ -0,0 +1,56 @@
1
+ /* eslint-disable no-console */
2
+ import cli from './_color.js'
3
+
4
+ const skip = {
5
+ Kanji: 'dim',
6
+ Hiragana: 'dim',
7
+ Katagana: 'dim',
8
+ Ascii: 'dim'
9
+ }
10
+ const tagString = function (tags, model) {
11
+ if (model.one.tagSet) {
12
+ tags = tags.filter(str => !skip[str])
13
+ tags = tags.map(tag => {
14
+ if (!model.one.tagSet.hasOwnProperty(tag)) {
15
+ return tag
16
+ }
17
+ const c = model.one.tagSet[tag].color || skip[tag] || 'blue'
18
+ return cli[c](tag)
19
+ })
20
+ }
21
+ return tags.join(', ')
22
+ }
23
+
24
+ const showTags = function (view) {
25
+ let { docs, model } = view
26
+ if (docs.length === 0) {
27
+ console.log(cli.blue('\n ──────'))
28
+ }
29
+ docs.forEach(terms => {
30
+ console.log(cli.blue('\n ┌─────────'))
31
+ terms.forEach(t => {
32
+ let tags = [...(t.tags || [])]
33
+ let text = t.text || '-'
34
+ if (t.sense) {
35
+ text = `{${t.normal}/${t.sense}}`
36
+ }
37
+ if (t.implicit) {
38
+ text = '[' + t.implicit + ']'
39
+ }
40
+ text = cli.yellow(text)
41
+ let word = "'" + text + "'"
42
+ // word = word.padEnd(15)
43
+ if (t.english) {
44
+ word += cli.i(` {${t.english}}`.padEnd(6))
45
+ }
46
+ if (t.reference) {
47
+ let str = view.update([t.reference]).text('normal')
48
+ word += ` - ${cli.dim(cli.i('[' + str + ']'))}`
49
+ }
50
+ word = word.padEnd(18)
51
+ let str = cli.blue(' │ ') + cli.i(word) + ' - ' + tagString(tags, model)
52
+ console.log(str)
53
+ })
54
+ })
55
+ }
56
+ export default showTags
@@ -0,0 +1,13 @@
1
+ import debug from './debug/index.js'
2
+ import english from './compute/english.js'
3
+ import root from './compute/root.js'
4
+
5
+ const methods = { debug }
6
+
7
+ const api = function (View) {
8
+ Object.assign(View.prototype, methods)
9
+ }
10
+ export default {
11
+ compute: { english, root },
12
+ api
13
+ }
@@ -0,0 +1,16 @@
1
+ export default function (View) {
2
+
3
+ /** */
4
+ View.prototype.romanji = function () {
5
+ this.compute('romanji')
6
+ let out = ''
7
+ this.docs.forEach(terms => {
8
+ terms.forEach(term => {
9
+ out += term.pre + (term.romanji || term.text) + (term.post || ' ')
10
+ })
11
+ })
12
+ // convert inter-bang
13
+ out = out.replace(/・/, ' ')
14
+ return out
15
+ }
16
+ }
@@ -0,0 +1,36 @@
1
+ import toRomanji from './toRomanji/index.js'
2
+ import toReading from './kanji-reading/index.js'
3
+ import { roots } from '../../lexicon/lexicon.js'
4
+
5
+ // は, へ and を are pronounced wa, e and o when they're particles
6
+ const particleSound = { 'は': 'wa', 'へ': 'e', 'を': 'o' }
7
+
8
+ const romanji = function (view) {
9
+ view.document.forEach(terms => {
10
+ terms.forEach(term => {
11
+ if (particleSound[term.text] && term.tags.has('Particle')) {
12
+ term.romanji = particleSound[term.text]
13
+ return
14
+ }
15
+ let word = term.normal
16
+ // any kanji in the word needs sounding-out first
17
+ if (/[一-龯]/.test(word)) {
18
+ word = toReading(word, null, term.root || roots[word])
19
+ }
20
+ term.romanji = toRomanji(word)
21
+ })
22
+ })
23
+ return view
24
+ }
25
+
26
+ const readings = function (view) {
27
+ view.document.forEach(terms => {
28
+ terms.forEach(term => {
29
+ if (/[一-龯]/.test(term.text)) {
30
+ term.reading = toReading(term.normal, null, term.root || roots[term.normal])
31
+ }
32
+ })
33
+ })
34
+ return view
35
+ }
36
+ export default { romanji, readings }
@@ -0,0 +1,91 @@
1
+ import readings from './readings.js'
2
+ import words from './words.js'
3
+
4
+ // the reading-table is written in katakana - fold it to hiragana so the
5
+ // romanizer only has to know one script
6
+ const toHiragana = function (str) {
7
+ let out = ''
8
+ for (let i = 0; i < str.length; i += 1) {
9
+ let c = str[i]
10
+ out += c >= 'ァ' && c <= 'ヶ' ? String.fromCharCode(c.charCodeAt(0) - 0x60) : c
11
+ }
12
+ return out
13
+ }
14
+
15
+ const isKanji = (c) => c >= '一' && c <= '龯'
16
+
17
+ // the kun-reading is written 'い.く' - the part before the dot is what the
18
+ // kanji itself spells; the rest is the okurigana already in the text
19
+ const kunStem = (kun) => kun.split('.')[0]
20
+
21
+ const soundOut = function (char, type) {
22
+ let r = readings[char]
23
+ if (!r) {
24
+ return char
25
+ }
26
+ let [kun, on] = r.split('|')
27
+ let pick = type === 'kun' ? kun || on : on || kun
28
+ return toHiragana(kunStem(pick || char))
29
+ }
30
+
31
+ /**
32
+ * sound-out a word's kanji.
33
+ * a lone kanji followed by okurigana (行き, 読む) is a native verb, so it
34
+ * takes its kun-reading. a run of two or more kanji (勉強) is a sino-japanese
35
+ * compound, and takes the on-reading.
36
+ */
37
+ /**
38
+ * 食べました isn't in the override table, but its dictionary-form 食べる is.
39
+ * the root tells us what its kanji spells (食 → た), and the okurigana is
40
+ * already right there in the text.
41
+ */
42
+ const fromRoot = function (word, root) {
43
+ let reading = words[root]
44
+ if (!reading) {
45
+ return null
46
+ }
47
+ // the kanji sit at the front of both the word and its root
48
+ let stem = root.split('').findIndex(c => !isKanji(c))
49
+ if (stem <= 0) {
50
+ return null
51
+ }
52
+ let okurigana = root.slice(stem)
53
+ if (!reading.endsWith(okurigana)) {
54
+ return null
55
+ }
56
+ let kanjiSound = reading.slice(0, reading.length - okurigana.length)
57
+ return word.startsWith(root.slice(0, stem)) ? kanjiSound + word.slice(stem) : null
58
+ }
59
+
60
+ const spellKanji = function (word, type, root) {
61
+ if (words[word] !== undefined) {
62
+ return words[word]
63
+ }
64
+ if (root && root !== word) {
65
+ let viaRoot = fromRoot(word, root)
66
+ if (viaRoot) {
67
+ return viaRoot
68
+ }
69
+ }
70
+ let out = ''
71
+ let i = 0
72
+ while (i < word.length) {
73
+ if (!isKanji(word[i])) {
74
+ out += word[i]
75
+ i += 1
76
+ continue
77
+ }
78
+ // how long is this run of kanji?
79
+ let run = ''
80
+ while (i < word.length && isKanji(word[i])) {
81
+ run += word[i]
82
+ i += 1
83
+ }
84
+ let use = type || (run.length > 1 ? 'on' : 'kun')
85
+ run.split('').forEach(c => {
86
+ out += soundOut(c, use)
87
+ })
88
+ }
89
+ return out
90
+ }
91
+ export default spellKanji