ja-compromise 0.0.1 → 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +367 -6
- package/builds/ja-compromise.cjs +4389 -2286
- package/builds/ja-compromise.min.js +1 -1
- package/builds/ja-compromise.mjs +1 -1
- package/package.json +36 -23
- package/src/01-one/conjugate/conjugate-adj.js +53 -0
- package/src/01-one/conjugate/conjugate-verb.js +145 -0
- package/src/01-one/conjugate/deconjugate.js +178 -0
- package/src/01-one/conjugate/index.js +13 -0
- package/src/01-one/conjugate/kana.js +38 -0
- package/src/01-one/conjugate/tags.js +85 -0
- package/src/01-one/conjugate/verb-class.js +105 -0
- package/src/01-one/lexicon/_data.js +30 -0
- package/src/01-one/lexicon/api.js +61 -0
- package/src/01-one/lexicon/lexicon.js +163 -0
- package/src/01-one/lexicon/misc.js +115 -0
- package/src/01-one/lexicon/plugin.js +20 -0
- package/src/01-one/numbers/api.js +202 -0
- package/src/01-one/numbers/kanji-number.js +69 -0
- package/src/01-one/numbers/to-kanji.js +66 -0
- package/src/01-one/output/compute/dict.js +13 -0
- package/src/01-one/output/compute/english.js +13 -0
- package/src/01-one/output/compute/root.js +30 -0
- package/src/01-one/output/debug/_color.js +16 -0
- package/src/01-one/output/debug/index.js +24 -0
- package/src/01-one/output/debug/tags.js +56 -0
- package/src/01-one/output/plugin.js +13 -0
- package/src/01-one/romanji/api.js +16 -0
- package/src/01-one/romanji/compute/index.js +36 -0
- package/src/01-one/romanji/compute/kanji-reading/index.js +91 -0
- package/src/01-one/romanji/compute/kanji-reading/readings.js +861 -0
- package/src/01-one/romanji/compute/kanji-reading/words.js +31 -0
- package/src/01-one/romanji/compute/toRomanji/hiragana-map.js +175 -0
- package/src/01-one/romanji/compute/toRomanji/index.js +84 -0
- package/src/01-one/romanji/compute/toRomanji/katakana-map.js +128 -0
- package/src/01-one/romanji/plugin.js +7 -0
- package/src/01-one/tokenizer/methods/join-numbers.js +32 -0
- package/src/01-one/tokenizer/methods/join-up.js +66 -0
- package/src/01-one/tokenizer/methods/lib.js +66 -0
- package/src/01-one/tokenizer/methods/naiive-split.js +88 -0
- package/src/01-one/tokenizer/methods/okurigana.js +58 -0
- package/src/01-one/tokenizer/methods/terms.js +143 -0
- package/src/01-one/tokenizer/methods/trie/build.js +13 -0
- package/src/01-one/tokenizer/methods/trie/split-up.js +29 -0
- package/src/01-one/tokenizer/methods/whitespace.js +44 -0
- package/src/01-one/tokenizer/plugin.js +13 -0
- package/src/02-two/preTagger/compute/01-script.js +44 -0
- package/src/02-two/preTagger/compute/02-particles.js +82 -0
- package/src/02-two/preTagger/compute/03-verbs.js +111 -0
- package/src/02-two/preTagger/compute/04-adjectives.js +43 -0
- package/src/02-two/preTagger/compute/05-people.js +22 -0
- package/src/02-two/preTagger/compute/06-numbers.js +80 -0
- package/src/02-two/preTagger/compute/07-dates.js +105 -0
- package/src/02-two/preTagger/compute/index.js +52 -0
- package/src/02-two/preTagger/plugin.js +9 -0
- package/src/02-two/tagset/plugin.js +12 -0
- package/src/02-two/tagset/tags/dates.js +58 -0
- package/src/02-two/tagset/tags/misc.js +91 -0
- package/src/02-two/tagset/tags/nouns.js +131 -0
- package/src/02-two/tagset/tags/particles.js +50 -0
- package/src/02-two/tagset/tags/values.js +68 -0
- package/src/02-two/tagset/tags/verbs.js +93 -0
- package/src/_lib.js +2 -0
- package/src/_version.js +1 -0
- package/src/index.js +82 -0
- package/types/index.d.ts +268 -0
- package/types/japanese.d.ts +159 -0
- package/types/misc.d.ts +92 -0
|
@@ -0,0 +1,202 @@
|
|
|
1
|
+
import toNumber from './kanji-number.js'
|
|
2
|
+
import toKanji, { toDigits } from './to-kanji.js'
|
|
3
|
+
import counters from '../../../lexicon/counters.js'
|
|
4
|
+
|
|
5
|
+
const isWide = /[0-9]/
|
|
6
|
+
const isKanjiNumeral = /[〇零一二三四五六七八九十百千万億兆壱弐参拾]/
|
|
7
|
+
|
|
8
|
+
// pull the number out of a match, along with how it was written
|
|
9
|
+
const parse = function (view) {
|
|
10
|
+
let terms = view.docs[0] || []
|
|
11
|
+
let term = terms.find(t => typeof t.number === 'number')
|
|
12
|
+
if (!term) {
|
|
13
|
+
// not computed yet - read it off the text
|
|
14
|
+
term = terms[0]
|
|
15
|
+
let num = term ? toNumber(term.text) : null
|
|
16
|
+
return { num, term, kanji: term ? isKanjiNumeral.test(term.text) : false, wide: false }
|
|
17
|
+
}
|
|
18
|
+
return {
|
|
19
|
+
num: term.number,
|
|
20
|
+
term,
|
|
21
|
+
kanji: isKanjiNumeral.test(term.text),
|
|
22
|
+
wide: isWide.test(term.text),
|
|
23
|
+
}
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
// swap a term's text in place. .replaceWith() would put a space between the
|
|
27
|
+
// number and its counter, and japanese doesn't use one - 五冊, never 五 冊
|
|
28
|
+
const setText = function (term, str, num) {
|
|
29
|
+
term.text = str
|
|
30
|
+
term.normal = str.toLowerCase()
|
|
31
|
+
if (typeof num === 'number') {
|
|
32
|
+
term.number = num
|
|
33
|
+
}
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
// write a number back in the shape it was found in
|
|
37
|
+
const write = function (view, num, form) {
|
|
38
|
+
let { term, kanji, wide } = parse(view)
|
|
39
|
+
if (!term || typeof num !== 'number') {
|
|
40
|
+
return view
|
|
41
|
+
}
|
|
42
|
+
let str
|
|
43
|
+
if (form === 'kanji' || (form === undefined && kanji)) {
|
|
44
|
+
str = toKanji(num)
|
|
45
|
+
} else {
|
|
46
|
+
str = toDigits(num, form === undefined ? wide : false)
|
|
47
|
+
}
|
|
48
|
+
if (str === null) {
|
|
49
|
+
return view
|
|
50
|
+
}
|
|
51
|
+
setText(term, str, num)
|
|
52
|
+
return view
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
const addMethod = function (View) {
|
|
57
|
+
class Numbers extends View {
|
|
58
|
+
constructor(document, pointer, groups) {
|
|
59
|
+
super(document, pointer, groups)
|
|
60
|
+
this.viewType = 'Numbers'
|
|
61
|
+
}
|
|
62
|
+
/** the parsed number, plus how it was written */
|
|
63
|
+
parse(n) {
|
|
64
|
+
return this.getNth(n).map(parse)
|
|
65
|
+
}
|
|
66
|
+
/** the value of each number - 「二十三」 gives 23 */
|
|
67
|
+
get(n) {
|
|
68
|
+
// the second .map is a plain Array.map - compromise's own .map() only
|
|
69
|
+
// unwraps to an array when the callback returns strings or objects
|
|
70
|
+
return this.getNth(n).map(parse).map(o => o.num)
|
|
71
|
+
}
|
|
72
|
+
json(n) {
|
|
73
|
+
let opts = typeof n === 'object' ? n : {}
|
|
74
|
+
return this.getNth(n).map(m => {
|
|
75
|
+
let json = m.toView().json(opts)[0]
|
|
76
|
+
let found = parse(m)
|
|
77
|
+
json.number = { num: found.num, counter: m.units().text() || null }
|
|
78
|
+
return json
|
|
79
|
+
}, [])
|
|
80
|
+
}
|
|
81
|
+
/** the 助数詞 attached to each number - 五冊 gives 冊 */
|
|
82
|
+
units() {
|
|
83
|
+
return this.growRight('#Counter').match('#Counter$')
|
|
84
|
+
}
|
|
85
|
+
/** only the ordinals - 三番目 */
|
|
86
|
+
isOrdinal() {
|
|
87
|
+
return this.if('#Ordinal')
|
|
88
|
+
}
|
|
89
|
+
/** only the cardinals */
|
|
90
|
+
isCardinal() {
|
|
91
|
+
return this.if('#Cardinal')
|
|
92
|
+
}
|
|
93
|
+
/** write each number in digits - 二十三 becomes 23 */
|
|
94
|
+
toNumber() {
|
|
95
|
+
this.forEach(m => {
|
|
96
|
+
let { num } = parse(m)
|
|
97
|
+
if (num !== null) {
|
|
98
|
+
write(m, num, 'digits')
|
|
99
|
+
}
|
|
100
|
+
})
|
|
101
|
+
return this
|
|
102
|
+
}
|
|
103
|
+
/** write each number in kanji - 23 becomes 二十三 */
|
|
104
|
+
toText() {
|
|
105
|
+
this.forEach(m => {
|
|
106
|
+
let { num } = parse(m)
|
|
107
|
+
if (num !== null) {
|
|
108
|
+
write(m, num, 'kanji')
|
|
109
|
+
}
|
|
110
|
+
})
|
|
111
|
+
return this
|
|
112
|
+
}
|
|
113
|
+
/** add thousands-separators - 1234 becomes 1,234 */
|
|
114
|
+
toLocaleString() {
|
|
115
|
+
this.forEach(m => {
|
|
116
|
+
let { num, term } = parse(m)
|
|
117
|
+
if (num === null || !term) {
|
|
118
|
+
return
|
|
119
|
+
}
|
|
120
|
+
setText(term, num.toLocaleString(), num)
|
|
121
|
+
})
|
|
122
|
+
return this
|
|
123
|
+
}
|
|
124
|
+
/** replace each number with this one, keeping its script */
|
|
125
|
+
set(n) {
|
|
126
|
+
if (typeof n !== 'number') {
|
|
127
|
+
return this
|
|
128
|
+
}
|
|
129
|
+
this.forEach(m => write(m, n))
|
|
130
|
+
return this
|
|
131
|
+
}
|
|
132
|
+
add(n) {
|
|
133
|
+
if (typeof n !== 'number') {
|
|
134
|
+
return this
|
|
135
|
+
}
|
|
136
|
+
this.forEach(m => {
|
|
137
|
+
let { num } = parse(m)
|
|
138
|
+
if (num !== null) {
|
|
139
|
+
write(m, num + n)
|
|
140
|
+
}
|
|
141
|
+
})
|
|
142
|
+
return this
|
|
143
|
+
}
|
|
144
|
+
subtract(n) {
|
|
145
|
+
return this.add(-1 * n)
|
|
146
|
+
}
|
|
147
|
+
increment() {
|
|
148
|
+
return this.add(1)
|
|
149
|
+
}
|
|
150
|
+
decrement() {
|
|
151
|
+
return this.add(-1)
|
|
152
|
+
}
|
|
153
|
+
isEqual(n) {
|
|
154
|
+
return this.filter(m => parse(m).num === n)
|
|
155
|
+
}
|
|
156
|
+
greaterThan(n) {
|
|
157
|
+
return this.filter(m => parse(m).num > n)
|
|
158
|
+
}
|
|
159
|
+
lessThan(n) {
|
|
160
|
+
return this.filter(m => parse(m).num < n)
|
|
161
|
+
}
|
|
162
|
+
between(min, max) {
|
|
163
|
+
return this.filter(m => {
|
|
164
|
+
let { num } = parse(m)
|
|
165
|
+
return num > min && num < max
|
|
166
|
+
})
|
|
167
|
+
}
|
|
168
|
+
update(pointer) {
|
|
169
|
+
let m = new Numbers(this.document, pointer)
|
|
170
|
+
m._cache = this._cache
|
|
171
|
+
return m
|
|
172
|
+
}
|
|
173
|
+
}
|
|
174
|
+
|
|
175
|
+
/** every number in the document */
|
|
176
|
+
View.prototype.numbers = function (n) {
|
|
177
|
+
let m = this.match('#Value+')
|
|
178
|
+
m = new Numbers(this.document, m.pointer)
|
|
179
|
+
return typeof n === 'number' ? m.eq(n) : m
|
|
180
|
+
}
|
|
181
|
+
/** .numbers() alias, to match english compromise */
|
|
182
|
+
View.prototype.values = View.prototype.numbers
|
|
183
|
+
|
|
184
|
+
/** every 助数詞 in the document */
|
|
185
|
+
View.prototype.counters = function (n) {
|
|
186
|
+
let m = this.match('#Counter')
|
|
187
|
+
return typeof n === 'number' ? m.eq(n) : m
|
|
188
|
+
}
|
|
189
|
+
/** every 円/ドル amount */
|
|
190
|
+
View.prototype.money = function (n) {
|
|
191
|
+
let m = this.match('#Value+ #Currency')
|
|
192
|
+
return typeof n === 'number' ? m.eq(n) : m
|
|
193
|
+
}
|
|
194
|
+
/** every percentage */
|
|
195
|
+
View.prototype.percentages = function (n) {
|
|
196
|
+
let m = this.match('#Value+ #Percent')
|
|
197
|
+
return typeof n === 'number' ? m.eq(n) : m
|
|
198
|
+
}
|
|
199
|
+
}
|
|
200
|
+
|
|
201
|
+
export default { api: addMethod }
|
|
202
|
+
export { counters }
|
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
// 漢数字 → a javascript number.
|
|
2
|
+
// japanese numerals are positional-by-power rather than positional-by-digit:
|
|
3
|
+
// 三百二十一 is (3×100) + (2×10) + 1, and 三万 is 3×10,000.
|
|
4
|
+
|
|
5
|
+
const digits = {
|
|
6
|
+
'〇': 0, '零': 0, '0': 0, '0': 0,
|
|
7
|
+
'一': 1, '壱': 1, '1': 1, '1': 1,
|
|
8
|
+
'二': 2, '弐': 2, '2': 2, '2': 2,
|
|
9
|
+
'三': 3, '参': 3, '3': 3, '3': 3,
|
|
10
|
+
'四': 4, '肆': 4, '4': 4, '4': 4,
|
|
11
|
+
'五': 5, '伍': 5, '5': 5, '5': 5,
|
|
12
|
+
'六': 6, '陸': 6, '6': 6, '6': 6,
|
|
13
|
+
'七': 7, '漆': 7, '7': 7, '7': 7,
|
|
14
|
+
'八': 8, '捌': 8, '8': 8, '8': 8,
|
|
15
|
+
'九': 9, '玖': 9, '9': 9, '9': 9,
|
|
16
|
+
}
|
|
17
|
+
// powers that stack inside a group
|
|
18
|
+
const small = { '十': 10, '拾': 10, '百': 100, '佰': 100, '千': 1000, '仟': 1000 }
|
|
19
|
+
// powers that close a group off
|
|
20
|
+
const large = { '万': 1e4, '萬': 1e4, '億': 1e8, '兆': 1e12, '京': 1e16 }
|
|
21
|
+
|
|
22
|
+
const isNumeral = function (c) {
|
|
23
|
+
return digits[c] !== undefined || small[c] !== undefined || large[c] !== undefined
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
/** parse a japanese numeral - returns null if the string isn't one */
|
|
27
|
+
const toNumber = function (str) {
|
|
28
|
+
if (!str) {
|
|
29
|
+
return null
|
|
30
|
+
}
|
|
31
|
+
let total = 0 // everything closed off by 万/億/兆
|
|
32
|
+
let section = 0 // the current group, below 10,000
|
|
33
|
+
let current = 0 // the digits seen since the last power
|
|
34
|
+
let seen = false
|
|
35
|
+
|
|
36
|
+
for (let i = 0; i < str.length; i += 1) {
|
|
37
|
+
let c = str[i]
|
|
38
|
+
if (digits[c] !== undefined) {
|
|
39
|
+
// 15 and 15 are positional, so keep multiplying up
|
|
40
|
+
current = Number(current * 10) + Number(digits[c])
|
|
41
|
+
seen = true
|
|
42
|
+
continue
|
|
43
|
+
}
|
|
44
|
+
if (small[c] !== undefined) {
|
|
45
|
+
// 十 on its own is 10, not 0
|
|
46
|
+
section += (current === 0 ? 1 : current) * small[c]
|
|
47
|
+
current = 0
|
|
48
|
+
seen = true
|
|
49
|
+
continue
|
|
50
|
+
}
|
|
51
|
+
if (large[c] !== undefined) {
|
|
52
|
+
// a bare 万 is 10,000 - but 五十万 is 50×10,000, not 51×10,000
|
|
53
|
+
let group = section + current
|
|
54
|
+
total += (group === 0 ? 1 : group) * large[c]
|
|
55
|
+
section = 0
|
|
56
|
+
current = 0
|
|
57
|
+
seen = true
|
|
58
|
+
continue
|
|
59
|
+
}
|
|
60
|
+
return null // not a numeral
|
|
61
|
+
}
|
|
62
|
+
if (!seen) {
|
|
63
|
+
return null
|
|
64
|
+
}
|
|
65
|
+
return total + section + current
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
export default toNumber
|
|
69
|
+
export { isNumeral, digits, small, large }
|
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
// a number → 漢数字.
|
|
2
|
+
//
|
|
3
|
+
// japanese numerals are fully compositional - 23 is 二十三, literally
|
|
4
|
+
// 'two-ten-three' - so this is arithmetic rather than a lookup table.
|
|
5
|
+
// the one wrinkle is that japanese groups by 10,000 (万) and not by 1,000.
|
|
6
|
+
|
|
7
|
+
const digit = ['', '一', '二', '三', '四', '五', '六', '七', '八', '九']
|
|
8
|
+
// powers that stack inside a group of four digits
|
|
9
|
+
const small = [['千', 1000], ['百', 100], ['十', 10]]
|
|
10
|
+
// powers that close a group off
|
|
11
|
+
const large = [['京', 1e16], ['兆', 1e12], ['億', 1e8], ['万', 1e4]]
|
|
12
|
+
|
|
13
|
+
// one group of four digits: 千百十 plus the remaining digit
|
|
14
|
+
const toGroup = function (n) {
|
|
15
|
+
let out = ''
|
|
16
|
+
for (let i = 0; i < small.length; i += 1) {
|
|
17
|
+
let [kanji, power] = small[i]
|
|
18
|
+
let d = Math.floor(n / power)
|
|
19
|
+
if (d === 0) {
|
|
20
|
+
continue
|
|
21
|
+
}
|
|
22
|
+
// 一十 and 一百 aren't written - 10 is 十, not 一十
|
|
23
|
+
out += (d === 1 ? '' : digit[d]) + kanji
|
|
24
|
+
n -= d * power
|
|
25
|
+
}
|
|
26
|
+
return out + digit[n]
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
/** 23 → 二十三 */
|
|
30
|
+
const toKanji = function (num) {
|
|
31
|
+
if (typeof num !== 'number' || !Number.isFinite(num)) {
|
|
32
|
+
return null
|
|
33
|
+
}
|
|
34
|
+
if (num < 0) {
|
|
35
|
+
return 'マイナス' + toKanji(Math.abs(num))
|
|
36
|
+
}
|
|
37
|
+
num = Math.round(num)
|
|
38
|
+
if (num === 0) {
|
|
39
|
+
return '〇'
|
|
40
|
+
}
|
|
41
|
+
let out = ''
|
|
42
|
+
for (let i = 0; i < large.length; i += 1) {
|
|
43
|
+
let [kanji, power] = large[i]
|
|
44
|
+
let chunk = Math.floor(num / power)
|
|
45
|
+
if (chunk === 0) {
|
|
46
|
+
continue
|
|
47
|
+
}
|
|
48
|
+
// 万 and up keep their 一 - 10,000 is 一万, never just 万
|
|
49
|
+
out += toGroup(chunk) + kanji
|
|
50
|
+
num -= chunk * power
|
|
51
|
+
}
|
|
52
|
+
return out + toGroup(num)
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
// 23 → '23', using the same width of digit the source used
|
|
56
|
+
const fullWidth = '0123456789'
|
|
57
|
+
const toDigits = function (num, wide) {
|
|
58
|
+
let str = String(num)
|
|
59
|
+
if (!wide) {
|
|
60
|
+
return str
|
|
61
|
+
}
|
|
62
|
+
return str.replace(/[0-9]/g, d => fullWidth[Number(d)])
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
export default toKanji
|
|
66
|
+
export { toDigits }
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
import { roots } from '../../lexicon/lexicon.js'
|
|
2
|
+
import { deconjugate } from '../../conjugate/index.js'
|
|
3
|
+
|
|
4
|
+
const isKnown = (w) => lexicon.hasOwnProperty(w)
|
|
5
|
+
|
|
6
|
+
// the dictionary-form of each word - 食べました → 食べる, 高かった → 高い
|
|
7
|
+
const addRoot = function (view) {
|
|
8
|
+
view.docs.forEach(terms => {
|
|
9
|
+
terms.forEach(term => {
|
|
10
|
+
if (term.root) {
|
|
11
|
+
return
|
|
12
|
+
}
|
|
13
|
+
if (roots[term.text] !== undefined) {
|
|
14
|
+
term.root = roots[term.text]
|
|
15
|
+
return
|
|
16
|
+
}
|
|
17
|
+
// an unknown verb can still be walked back to its dictionary-form
|
|
18
|
+
if (term.tags.has('Verb')) {
|
|
19
|
+
let found = deconjugate(term.text, isKnown)
|
|
20
|
+
if (found) {
|
|
21
|
+
term.root = found.root
|
|
22
|
+
return
|
|
23
|
+
}
|
|
24
|
+
}
|
|
25
|
+
term.root = term.text
|
|
26
|
+
})
|
|
27
|
+
})
|
|
28
|
+
return view
|
|
29
|
+
}
|
|
30
|
+
export default addRoot
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
// https://stackoverflow.com/questions/9781218/how-to-change-node-jss-console-font-color
|
|
2
|
+
const reset = '\x1b[0m'
|
|
3
|
+
|
|
4
|
+
//cheaper than requiring chalk
|
|
5
|
+
const cli = {
|
|
6
|
+
green: str => '\x1b[32m' + str + reset,
|
|
7
|
+
red: str => '\x1b[31m' + str + reset,
|
|
8
|
+
blue: str => '\x1b[34m' + str + reset,
|
|
9
|
+
magenta: str => '\x1b[35m' + str + reset,
|
|
10
|
+
cyan: str => '\x1b[36m' + str + reset,
|
|
11
|
+
yellow: str => '\x1b[33m' + str + reset,
|
|
12
|
+
black: str => '\x1b[30m' + str + reset,
|
|
13
|
+
dim: str => '\x1b[2m' + str + reset,
|
|
14
|
+
i: str => '\x1b[3m' + str + reset,
|
|
15
|
+
}
|
|
16
|
+
export default cli
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
/* eslint-disable no-console */
|
|
2
|
+
import showTags from './tags.js'
|
|
3
|
+
|
|
4
|
+
function isClientSide() {
|
|
5
|
+
return typeof window !== 'undefined' && window.document
|
|
6
|
+
}
|
|
7
|
+
//output some helpful stuff to the console
|
|
8
|
+
const debug = function (opts = {}) {
|
|
9
|
+
let view = this
|
|
10
|
+
if (typeof opts === 'string') {
|
|
11
|
+
let tmp = {}
|
|
12
|
+
tmp[opts] = true //allow string input
|
|
13
|
+
opts = tmp
|
|
14
|
+
}
|
|
15
|
+
if (isClientSide()) {
|
|
16
|
+
return view
|
|
17
|
+
}
|
|
18
|
+
if (opts.tags !== false) {
|
|
19
|
+
showTags(view)
|
|
20
|
+
console.log('\n')
|
|
21
|
+
}
|
|
22
|
+
return view
|
|
23
|
+
}
|
|
24
|
+
export default debug
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
/* eslint-disable no-console */
|
|
2
|
+
import cli from './_color.js'
|
|
3
|
+
|
|
4
|
+
const skip = {
|
|
5
|
+
Kanji: 'dim',
|
|
6
|
+
Hiragana: 'dim',
|
|
7
|
+
Katagana: 'dim',
|
|
8
|
+
Ascii: 'dim'
|
|
9
|
+
}
|
|
10
|
+
const tagString = function (tags, model) {
|
|
11
|
+
if (model.one.tagSet) {
|
|
12
|
+
tags = tags.filter(str => !skip[str])
|
|
13
|
+
tags = tags.map(tag => {
|
|
14
|
+
if (!model.one.tagSet.hasOwnProperty(tag)) {
|
|
15
|
+
return tag
|
|
16
|
+
}
|
|
17
|
+
const c = model.one.tagSet[tag].color || skip[tag] || 'blue'
|
|
18
|
+
return cli[c](tag)
|
|
19
|
+
})
|
|
20
|
+
}
|
|
21
|
+
return tags.join(', ')
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
const showTags = function (view) {
|
|
25
|
+
let { docs, model } = view
|
|
26
|
+
if (docs.length === 0) {
|
|
27
|
+
console.log(cli.blue('\n ──────'))
|
|
28
|
+
}
|
|
29
|
+
docs.forEach(terms => {
|
|
30
|
+
console.log(cli.blue('\n ┌─────────'))
|
|
31
|
+
terms.forEach(t => {
|
|
32
|
+
let tags = [...(t.tags || [])]
|
|
33
|
+
let text = t.text || '-'
|
|
34
|
+
if (t.sense) {
|
|
35
|
+
text = `{${t.normal}/${t.sense}}`
|
|
36
|
+
}
|
|
37
|
+
if (t.implicit) {
|
|
38
|
+
text = '[' + t.implicit + ']'
|
|
39
|
+
}
|
|
40
|
+
text = cli.yellow(text)
|
|
41
|
+
let word = "'" + text + "'"
|
|
42
|
+
// word = word.padEnd(15)
|
|
43
|
+
if (t.english) {
|
|
44
|
+
word += cli.i(` {${t.english}}`.padEnd(6))
|
|
45
|
+
}
|
|
46
|
+
if (t.reference) {
|
|
47
|
+
let str = view.update([t.reference]).text('normal')
|
|
48
|
+
word += ` - ${cli.dim(cli.i('[' + str + ']'))}`
|
|
49
|
+
}
|
|
50
|
+
word = word.padEnd(18)
|
|
51
|
+
let str = cli.blue(' │ ') + cli.i(word) + ' - ' + tagString(tags, model)
|
|
52
|
+
console.log(str)
|
|
53
|
+
})
|
|
54
|
+
})
|
|
55
|
+
}
|
|
56
|
+
export default showTags
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
import debug from './debug/index.js'
|
|
2
|
+
import english from './compute/english.js'
|
|
3
|
+
import root from './compute/root.js'
|
|
4
|
+
|
|
5
|
+
const methods = { debug }
|
|
6
|
+
|
|
7
|
+
const api = function (View) {
|
|
8
|
+
Object.assign(View.prototype, methods)
|
|
9
|
+
}
|
|
10
|
+
export default {
|
|
11
|
+
compute: { english, root },
|
|
12
|
+
api
|
|
13
|
+
}
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
export default function (View) {
|
|
2
|
+
|
|
3
|
+
/** */
|
|
4
|
+
View.prototype.romanji = function () {
|
|
5
|
+
this.compute('romanji')
|
|
6
|
+
let out = ''
|
|
7
|
+
this.docs.forEach(terms => {
|
|
8
|
+
terms.forEach(term => {
|
|
9
|
+
out += term.pre + (term.romanji || term.text) + (term.post || ' ')
|
|
10
|
+
})
|
|
11
|
+
})
|
|
12
|
+
// convert inter-bang
|
|
13
|
+
out = out.replace(/・/, ' ')
|
|
14
|
+
return out
|
|
15
|
+
}
|
|
16
|
+
}
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
import toRomanji from './toRomanji/index.js'
|
|
2
|
+
import toReading from './kanji-reading/index.js'
|
|
3
|
+
import { roots } from '../../lexicon/lexicon.js'
|
|
4
|
+
|
|
5
|
+
// は, へ and を are pronounced wa, e and o when they're particles
|
|
6
|
+
const particleSound = { 'は': 'wa', 'へ': 'e', 'を': 'o' }
|
|
7
|
+
|
|
8
|
+
const romanji = function (view) {
|
|
9
|
+
view.document.forEach(terms => {
|
|
10
|
+
terms.forEach(term => {
|
|
11
|
+
if (particleSound[term.text] && term.tags.has('Particle')) {
|
|
12
|
+
term.romanji = particleSound[term.text]
|
|
13
|
+
return
|
|
14
|
+
}
|
|
15
|
+
let word = term.normal
|
|
16
|
+
// any kanji in the word needs sounding-out first
|
|
17
|
+
if (/[一-龯]/.test(word)) {
|
|
18
|
+
word = toReading(word, null, term.root || roots[word])
|
|
19
|
+
}
|
|
20
|
+
term.romanji = toRomanji(word)
|
|
21
|
+
})
|
|
22
|
+
})
|
|
23
|
+
return view
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
const readings = function (view) {
|
|
27
|
+
view.document.forEach(terms => {
|
|
28
|
+
terms.forEach(term => {
|
|
29
|
+
if (/[一-龯]/.test(term.text)) {
|
|
30
|
+
term.reading = toReading(term.normal, null, term.root || roots[term.normal])
|
|
31
|
+
}
|
|
32
|
+
})
|
|
33
|
+
})
|
|
34
|
+
return view
|
|
35
|
+
}
|
|
36
|
+
export default { romanji, readings }
|
|
@@ -0,0 +1,91 @@
|
|
|
1
|
+
import readings from './readings.js'
|
|
2
|
+
import words from './words.js'
|
|
3
|
+
|
|
4
|
+
// the reading-table is written in katakana - fold it to hiragana so the
|
|
5
|
+
// romanizer only has to know one script
|
|
6
|
+
const toHiragana = function (str) {
|
|
7
|
+
let out = ''
|
|
8
|
+
for (let i = 0; i < str.length; i += 1) {
|
|
9
|
+
let c = str[i]
|
|
10
|
+
out += c >= 'ァ' && c <= 'ヶ' ? String.fromCharCode(c.charCodeAt(0) - 0x60) : c
|
|
11
|
+
}
|
|
12
|
+
return out
|
|
13
|
+
}
|
|
14
|
+
|
|
15
|
+
const isKanji = (c) => c >= '一' && c <= '龯'
|
|
16
|
+
|
|
17
|
+
// the kun-reading is written 'い.く' - the part before the dot is what the
|
|
18
|
+
// kanji itself spells; the rest is the okurigana already in the text
|
|
19
|
+
const kunStem = (kun) => kun.split('.')[0]
|
|
20
|
+
|
|
21
|
+
const soundOut = function (char, type) {
|
|
22
|
+
let r = readings[char]
|
|
23
|
+
if (!r) {
|
|
24
|
+
return char
|
|
25
|
+
}
|
|
26
|
+
let [kun, on] = r.split('|')
|
|
27
|
+
let pick = type === 'kun' ? kun || on : on || kun
|
|
28
|
+
return toHiragana(kunStem(pick || char))
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
/**
|
|
32
|
+
* sound-out a word's kanji.
|
|
33
|
+
* a lone kanji followed by okurigana (行き, 読む) is a native verb, so it
|
|
34
|
+
* takes its kun-reading. a run of two or more kanji (勉強) is a sino-japanese
|
|
35
|
+
* compound, and takes the on-reading.
|
|
36
|
+
*/
|
|
37
|
+
/**
|
|
38
|
+
* 食べました isn't in the override table, but its dictionary-form 食べる is.
|
|
39
|
+
* the root tells us what its kanji spells (食 → た), and the okurigana is
|
|
40
|
+
* already right there in the text.
|
|
41
|
+
*/
|
|
42
|
+
const fromRoot = function (word, root) {
|
|
43
|
+
let reading = words[root]
|
|
44
|
+
if (!reading) {
|
|
45
|
+
return null
|
|
46
|
+
}
|
|
47
|
+
// the kanji sit at the front of both the word and its root
|
|
48
|
+
let stem = root.split('').findIndex(c => !isKanji(c))
|
|
49
|
+
if (stem <= 0) {
|
|
50
|
+
return null
|
|
51
|
+
}
|
|
52
|
+
let okurigana = root.slice(stem)
|
|
53
|
+
if (!reading.endsWith(okurigana)) {
|
|
54
|
+
return null
|
|
55
|
+
}
|
|
56
|
+
let kanjiSound = reading.slice(0, reading.length - okurigana.length)
|
|
57
|
+
return word.startsWith(root.slice(0, stem)) ? kanjiSound + word.slice(stem) : null
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
const spellKanji = function (word, type, root) {
|
|
61
|
+
if (words[word] !== undefined) {
|
|
62
|
+
return words[word]
|
|
63
|
+
}
|
|
64
|
+
if (root && root !== word) {
|
|
65
|
+
let viaRoot = fromRoot(word, root)
|
|
66
|
+
if (viaRoot) {
|
|
67
|
+
return viaRoot
|
|
68
|
+
}
|
|
69
|
+
}
|
|
70
|
+
let out = ''
|
|
71
|
+
let i = 0
|
|
72
|
+
while (i < word.length) {
|
|
73
|
+
if (!isKanji(word[i])) {
|
|
74
|
+
out += word[i]
|
|
75
|
+
i += 1
|
|
76
|
+
continue
|
|
77
|
+
}
|
|
78
|
+
// how long is this run of kanji?
|
|
79
|
+
let run = ''
|
|
80
|
+
while (i < word.length && isKanji(word[i])) {
|
|
81
|
+
run += word[i]
|
|
82
|
+
i += 1
|
|
83
|
+
}
|
|
84
|
+
let use = type || (run.length > 1 ? 'on' : 'kun')
|
|
85
|
+
run.split('').forEach(c => {
|
|
86
|
+
out += soundOut(c, use)
|
|
87
|
+
})
|
|
88
|
+
}
|
|
89
|
+
return out
|
|
90
|
+
}
|
|
91
|
+
export default spellKanji
|