ja-compromise 0.0.1 → 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +367 -6
- package/builds/ja-compromise.cjs +4389 -2286
- package/builds/ja-compromise.min.js +1 -1
- package/builds/ja-compromise.mjs +1 -1
- package/package.json +36 -23
- package/src/01-one/conjugate/conjugate-adj.js +53 -0
- package/src/01-one/conjugate/conjugate-verb.js +145 -0
- package/src/01-one/conjugate/deconjugate.js +178 -0
- package/src/01-one/conjugate/index.js +13 -0
- package/src/01-one/conjugate/kana.js +38 -0
- package/src/01-one/conjugate/tags.js +85 -0
- package/src/01-one/conjugate/verb-class.js +105 -0
- package/src/01-one/lexicon/_data.js +30 -0
- package/src/01-one/lexicon/api.js +61 -0
- package/src/01-one/lexicon/lexicon.js +163 -0
- package/src/01-one/lexicon/misc.js +115 -0
- package/src/01-one/lexicon/plugin.js +20 -0
- package/src/01-one/numbers/api.js +202 -0
- package/src/01-one/numbers/kanji-number.js +69 -0
- package/src/01-one/numbers/to-kanji.js +66 -0
- package/src/01-one/output/compute/dict.js +13 -0
- package/src/01-one/output/compute/english.js +13 -0
- package/src/01-one/output/compute/root.js +30 -0
- package/src/01-one/output/debug/_color.js +16 -0
- package/src/01-one/output/debug/index.js +24 -0
- package/src/01-one/output/debug/tags.js +56 -0
- package/src/01-one/output/plugin.js +13 -0
- package/src/01-one/romanji/api.js +16 -0
- package/src/01-one/romanji/compute/index.js +36 -0
- package/src/01-one/romanji/compute/kanji-reading/index.js +91 -0
- package/src/01-one/romanji/compute/kanji-reading/readings.js +861 -0
- package/src/01-one/romanji/compute/kanji-reading/words.js +31 -0
- package/src/01-one/romanji/compute/toRomanji/hiragana-map.js +175 -0
- package/src/01-one/romanji/compute/toRomanji/index.js +84 -0
- package/src/01-one/romanji/compute/toRomanji/katakana-map.js +128 -0
- package/src/01-one/romanji/plugin.js +7 -0
- package/src/01-one/tokenizer/methods/join-numbers.js +32 -0
- package/src/01-one/tokenizer/methods/join-up.js +66 -0
- package/src/01-one/tokenizer/methods/lib.js +66 -0
- package/src/01-one/tokenizer/methods/naiive-split.js +88 -0
- package/src/01-one/tokenizer/methods/okurigana.js +58 -0
- package/src/01-one/tokenizer/methods/terms.js +143 -0
- package/src/01-one/tokenizer/methods/trie/build.js +13 -0
- package/src/01-one/tokenizer/methods/trie/split-up.js +29 -0
- package/src/01-one/tokenizer/methods/whitespace.js +44 -0
- package/src/01-one/tokenizer/plugin.js +13 -0
- package/src/02-two/preTagger/compute/01-script.js +44 -0
- package/src/02-two/preTagger/compute/02-particles.js +82 -0
- package/src/02-two/preTagger/compute/03-verbs.js +111 -0
- package/src/02-two/preTagger/compute/04-adjectives.js +43 -0
- package/src/02-two/preTagger/compute/05-people.js +22 -0
- package/src/02-two/preTagger/compute/06-numbers.js +80 -0
- package/src/02-two/preTagger/compute/07-dates.js +105 -0
- package/src/02-two/preTagger/compute/index.js +52 -0
- package/src/02-two/preTagger/plugin.js +9 -0
- package/src/02-two/tagset/plugin.js +12 -0
- package/src/02-two/tagset/tags/dates.js +58 -0
- package/src/02-two/tagset/tags/misc.js +91 -0
- package/src/02-two/tagset/tags/nouns.js +131 -0
- package/src/02-two/tagset/tags/particles.js +50 -0
- package/src/02-two/tagset/tags/values.js +68 -0
- package/src/02-two/tagset/tags/verbs.js +93 -0
- package/src/_lib.js +2 -0
- package/src/_version.js +1 -0
- package/src/index.js +82 -0
- package/types/index.d.ts +268 -0
- package/types/japanese.d.ts +159 -0
- package/types/misc.d.ts +92 -0
package/package.json
CHANGED
|
@@ -2,13 +2,19 @@
|
|
|
2
2
|
"author": "Spencer Kelly <spencermountain@gmail.com> (http://spencermounta.in)",
|
|
3
3
|
"name": "ja-compromise",
|
|
4
4
|
"description": "日本語の控えめな自然言語処理",
|
|
5
|
-
"version": "0.0
|
|
6
|
-
"
|
|
5
|
+
"version": "0.1.0",
|
|
6
|
+
"engines": {
|
|
7
|
+
"node": "^22.13.0 || ^24.0.0 || ^26.0.0",
|
|
8
|
+
"pnpm": "^11.5.0"
|
|
9
|
+
},
|
|
10
|
+
"main": "./builds/ja-compromise.mjs",
|
|
11
|
+
"browser": "./builds/ja-compromise.min.js",
|
|
7
12
|
"unpkg": "./builds/ja-compromise.min.js",
|
|
8
13
|
"type": "module",
|
|
9
14
|
"sideEffects": false,
|
|
10
15
|
"exports": {
|
|
11
16
|
".": {
|
|
17
|
+
"types": "./types/index.d.ts",
|
|
12
18
|
"import": "./builds/ja-compromise.mjs",
|
|
13
19
|
"require": "./builds/ja-compromise.cjs"
|
|
14
20
|
}
|
|
@@ -18,34 +24,41 @@
|
|
|
18
24
|
"type": "git",
|
|
19
25
|
"url": "git://github.com/nlp-compromise/ja-compromise.git"
|
|
20
26
|
},
|
|
21
|
-
"scripts": {
|
|
22
|
-
"test": "tape \"./tests/**/*.test.js\" | tap-dancer",
|
|
23
|
-
"testb": "TESTENV=prod npm run test",
|
|
24
|
-
"build": "npm run version && rollup -c --silent",
|
|
25
|
-
"version": "node ./scripts/version.js",
|
|
26
|
-
"pack": "node ./scripts/pack.js",
|
|
27
|
-
"score": " node ./learn/test/index.js",
|
|
28
|
-
"watch": "amble ./scratch.js",
|
|
29
|
-
"stress": "node scripts/stress.js"
|
|
30
|
-
},
|
|
31
27
|
"files": [
|
|
32
28
|
"builds/",
|
|
29
|
+
"types/",
|
|
30
|
+
"src/",
|
|
33
31
|
"docs/"
|
|
34
32
|
],
|
|
35
33
|
"dependencies": {
|
|
36
|
-
"compromise": "14.
|
|
34
|
+
"compromise": "^14.17.0",
|
|
37
35
|
"efrt": "2.7.0",
|
|
38
|
-
"suffix-thumb": "
|
|
36
|
+
"suffix-thumb": "5.0.3"
|
|
39
37
|
},
|
|
40
38
|
"devDependencies": {
|
|
41
|
-
"@
|
|
42
|
-
"@rollup/plugin-node-resolve": "
|
|
43
|
-
"
|
|
44
|
-
"
|
|
45
|
-
"
|
|
39
|
+
"@eslint/js": "^10.0.1",
|
|
40
|
+
"@rollup/plugin-node-resolve": "16.0.3",
|
|
41
|
+
"@rollup/plugin-terser": "1.0.0",
|
|
42
|
+
"cross-env": "10.1.0",
|
|
43
|
+
"eslint": "10.10.0",
|
|
44
|
+
"eslint-plugin-regexp": "3.3.0",
|
|
45
|
+
"rollup": "4.63.1",
|
|
46
46
|
"tap-dancer": "0.3.4",
|
|
47
|
-
"tape": "5.
|
|
48
|
-
"
|
|
47
|
+
"tape": "5.10.2",
|
|
48
|
+
"typescript": "7.0.2"
|
|
49
49
|
},
|
|
50
|
-
"license": "MIT"
|
|
51
|
-
|
|
50
|
+
"license": "MIT",
|
|
51
|
+
"scripts": {
|
|
52
|
+
"typecheck": "tsc -p tsconfig.json",
|
|
53
|
+
"test": "tsc -p tsconfig.json && tape \"./tests/**/*.test.js\" | tap-dancer",
|
|
54
|
+
"testb": "TESTENV=prod pnpm run test",
|
|
55
|
+
"build": "pnpm run version && rollup -c --silent",
|
|
56
|
+
"version": "node ./scripts/version.js",
|
|
57
|
+
"pack": "node ./scripts/pack.js",
|
|
58
|
+
"score": "node ./learn/test/index.js",
|
|
59
|
+
"audit": "node ./learn/verbs/audit.js",
|
|
60
|
+
"lint": "eslint ./src/**/*",
|
|
61
|
+
"watch": "amble ./scratch.js",
|
|
62
|
+
"stress": "node scripts/stress.js"
|
|
63
|
+
}
|
|
64
|
+
}
|
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
// い-adjectives inflect for tense and polarity - they are not 'just adjectives'.
|
|
2
|
+
// 高い → 高くない → 高かった → 高くなかった. the copula does not carry that tense.
|
|
3
|
+
|
|
4
|
+
// いい/良い is the one truly irregular い-adjective - it inflects as よい
|
|
5
|
+
const irregularStem = { 'いい': 'よ', '良い': '良', 'よい': 'よ' }
|
|
6
|
+
|
|
7
|
+
const conjugateAdjective = function (word) {
|
|
8
|
+
if (!word || !word.endsWith('い') || word.length < 2) {
|
|
9
|
+
return null
|
|
10
|
+
}
|
|
11
|
+
let stem = irregularStem[word] !== undefined ? irregularStem[word] : word.slice(0, -1)
|
|
12
|
+
return {
|
|
13
|
+
Infinitive: word, // 高い
|
|
14
|
+
PresentTense: word,
|
|
15
|
+
Negative: stem + 'くない', // 高くない
|
|
16
|
+
PastTense: stem + 'かった', // 高かった
|
|
17
|
+
PastNegative: stem + 'くなかった', // 高くなかった
|
|
18
|
+
Gerund: stem + 'くて', // 高くて
|
|
19
|
+
Adverb: stem + 'く', // 高く
|
|
20
|
+
Provisional: stem + 'ければ', // 高ければ
|
|
21
|
+
Presumptive: stem + 'かろう',
|
|
22
|
+
Polite: word + 'です',
|
|
23
|
+
PolitePast: stem + 'かったです',
|
|
24
|
+
PoliteNegative: stem + 'くないです',
|
|
25
|
+
Superlative: stem + 'すぎる', // 高すぎる
|
|
26
|
+
Impression: stem + 'そう', // 高そう
|
|
27
|
+
Nominal: stem + 'さ', // 高さ
|
|
28
|
+
}
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
// な-adjectives (形容動詞) are stored bare - 静か - and inflect with the copula
|
|
32
|
+
const conjugateNaAdjective = function (word) {
|
|
33
|
+
if (!word) {
|
|
34
|
+
return null
|
|
35
|
+
}
|
|
36
|
+
return {
|
|
37
|
+
Infinitive: word,
|
|
38
|
+
Adnominal: word + 'な', // 静かな
|
|
39
|
+
PresentTense: word + 'だ', // 静かだ
|
|
40
|
+
Negative: word + 'じゃない', // 静かじゃない
|
|
41
|
+
PastTense: word + 'だった', // 静かだった
|
|
42
|
+
PastNegative: word + 'じゃなかった',
|
|
43
|
+
Gerund: word + 'で', // 静かで
|
|
44
|
+
Adverb: word + 'に', // 静かに
|
|
45
|
+
Polite: word + 'です',
|
|
46
|
+
PolitePast: word + 'でした',
|
|
47
|
+
PoliteNegative: word + 'じゃありません',
|
|
48
|
+
Provisional: word + 'なら',
|
|
49
|
+
}
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
export default conjugateAdjective
|
|
53
|
+
export { conjugateNaAdjective }
|
|
@@ -0,0 +1,145 @@
|
|
|
1
|
+
import { aRow, iRow, eRow, oRow, teRow } from './kana.js'
|
|
2
|
+
import verbClass from './verb-class.js'
|
|
3
|
+
|
|
4
|
+
// the five 'bases' every japanese verb form is built out of.
|
|
5
|
+
// get these right and every suffix falls out of them.
|
|
6
|
+
const toBases = function (dict, cls) {
|
|
7
|
+
let stem = dict.slice(0, -1)
|
|
8
|
+
let last = dict[dict.length - 1]
|
|
9
|
+
switch (cls) {
|
|
10
|
+
case 'godan':
|
|
11
|
+
return {
|
|
12
|
+
negStem: stem + aRow[last], // 未然形 - 書か
|
|
13
|
+
stem: stem + iRow[last], // 連用形 - 書き
|
|
14
|
+
cond: stem + eRow[last], // 仮定形 - 書け
|
|
15
|
+
imper: stem + eRow[last], // 命令形 - 書け
|
|
16
|
+
volit: stem + oRow[last] + 'う', // 意向形 - 書こう
|
|
17
|
+
te: stem + teRow[last], // て形 - 書いて
|
|
18
|
+
potential: stem + eRow[last] + 'る',
|
|
19
|
+
passive: stem + aRow[last] + 'れる',
|
|
20
|
+
causative: stem + aRow[last] + 'せる',
|
|
21
|
+
}
|
|
22
|
+
case 'ichidan': {
|
|
23
|
+
let s = dict.slice(0, -1) // 食べ
|
|
24
|
+
// くれる is the one ichidan verb with a bare imperative - くれ, not くれろ
|
|
25
|
+
let imper = /(呉れる|くれる)$/.test(dict) ? s : s + 'ろ'
|
|
26
|
+
return {
|
|
27
|
+
negStem: s, stem: s, cond: s + 'れ', imper: imper,
|
|
28
|
+
volit: s + 'よう', te: s + 'て',
|
|
29
|
+
potential: s + 'られる', passive: s + 'られる', causative: s + 'させる',
|
|
30
|
+
}
|
|
31
|
+
}
|
|
32
|
+
case 'suru': {
|
|
33
|
+
let pre = dict.slice(0, -2) // 勉強
|
|
34
|
+
return {
|
|
35
|
+
negStem: pre + 'し', stem: pre + 'し', cond: pre + 'すれ', imper: pre + 'しろ',
|
|
36
|
+
volit: pre + 'しよう', te: pre + 'して',
|
|
37
|
+
potential: pre === '' ? 'できる' : pre + 'できる',
|
|
38
|
+
passive: pre + 'される', causative: pre + 'させる',
|
|
39
|
+
}
|
|
40
|
+
}
|
|
41
|
+
case 'kuru': {
|
|
42
|
+
let pre = dict.slice(0, -2)
|
|
43
|
+
let kana = dict.endsWith('くる')
|
|
44
|
+
let ko = kana ? 'こ' : '来'
|
|
45
|
+
let ki = kana ? 'き' : '来'
|
|
46
|
+
let ku = kana ? 'く' : '来'
|
|
47
|
+
return {
|
|
48
|
+
negStem: pre + ko, stem: pre + ki, cond: pre + ku + 'れ', imper: pre + ko + 'い',
|
|
49
|
+
volit: pre + ko + 'よう', te: pre + ki + 'て',
|
|
50
|
+
potential: pre + ko + 'られる', passive: pre + ko + 'られる', causative: pre + ko + 'させる',
|
|
51
|
+
}
|
|
52
|
+
}
|
|
53
|
+
case 'aru': {
|
|
54
|
+
let s = dict.slice(0, -1) // あ / 有 / 在
|
|
55
|
+
return {
|
|
56
|
+
negStem: null, // ある has no 未然形 - its negative is just ない
|
|
57
|
+
stem: s + 'り', cond: s + 'れ', imper: s + 'れ', volit: s + 'ろう', te: s + 'って',
|
|
58
|
+
potential: s + 'れる', passive: s + 'られる', causative: s + 'らせる',
|
|
59
|
+
}
|
|
60
|
+
}
|
|
61
|
+
case 'iku': { // 行く takes って, not the regular いて
|
|
62
|
+
let s = dict.slice(0, -1)
|
|
63
|
+
return {
|
|
64
|
+
negStem: s + 'か', stem: s + 'き', cond: s + 'け', imper: s + 'け', volit: s + 'こう',
|
|
65
|
+
te: s + 'って',
|
|
66
|
+
potential: s + 'ける', passive: s + 'かれる', causative: s + 'かせる',
|
|
67
|
+
}
|
|
68
|
+
}
|
|
69
|
+
case 'ou': { // 問う/請う keep うて rather than って
|
|
70
|
+
let s = dict.slice(0, -1)
|
|
71
|
+
return {
|
|
72
|
+
negStem: s + 'わ', stem: s + 'い', cond: s + 'え', imper: s + 'え', volit: s + 'おう',
|
|
73
|
+
te: s + 'うて',
|
|
74
|
+
potential: s + 'える', passive: s + 'われる', causative: s + 'わせる',
|
|
75
|
+
}
|
|
76
|
+
}
|
|
77
|
+
case 'aru5': { // くださる/なさる - godan, but the masu-stem drops the り
|
|
78
|
+
let s = dict.slice(0, -1)
|
|
79
|
+
return {
|
|
80
|
+
negStem: s + 'ら', stem: s + 'い', cond: s + 'れ', imper: s + 'い', volit: s + 'ろう',
|
|
81
|
+
te: s + 'って',
|
|
82
|
+
potential: s + 'れる', passive: s + 'られる', causative: s + 'らせる',
|
|
83
|
+
}
|
|
84
|
+
}
|
|
85
|
+
default:
|
|
86
|
+
return null
|
|
87
|
+
}
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
// て → た, で → だ
|
|
91
|
+
const teToTa = (te) => te.replace(/て$/, 'た').replace(/で$/, 'だ')
|
|
92
|
+
|
|
93
|
+
/** produce the full paradigm of a dictionary-form verb */
|
|
94
|
+
const conjugate = function (dict, hint) {
|
|
95
|
+
let cls = verbClass(dict, hint)
|
|
96
|
+
if (!cls) {
|
|
97
|
+
return null
|
|
98
|
+
}
|
|
99
|
+
let b = toBases(dict, cls)
|
|
100
|
+
if (!b) {
|
|
101
|
+
return null
|
|
102
|
+
}
|
|
103
|
+
// ある is the one verb whose plain negative isn't built on a 未然形
|
|
104
|
+
let negative = b.negStem === null ? 'ない' : b.negStem + 'ない'
|
|
105
|
+
let pastNegative = b.negStem === null ? 'なかった' : b.negStem + 'なかった'
|
|
106
|
+
let te = b.te
|
|
107
|
+
let past = teToTa(te)
|
|
108
|
+
|
|
109
|
+
return {
|
|
110
|
+
class: cls,
|
|
111
|
+
Infinitive: dict, // 辞書形 - 書く
|
|
112
|
+
Stem: b.stem, // 連用形 - 書き
|
|
113
|
+
PresentTense: dict,
|
|
114
|
+
PastTense: past, // 書いた
|
|
115
|
+
Negative: negative, // 書かない
|
|
116
|
+
PastNegative: pastNegative, // 書かなかった
|
|
117
|
+
Gerund: te, // 書いて (て形)
|
|
118
|
+
NegativeGerund: b.negStem === null ? 'なくて' : b.negStem + 'なくて',
|
|
119
|
+
Polite: b.stem + 'ます', // 書きます
|
|
120
|
+
PolitePast: b.stem + 'ました', // 書きました
|
|
121
|
+
PoliteNegative: b.stem + 'ません', // 書きません
|
|
122
|
+
PolitePastNegative: b.stem + 'ませんでした',
|
|
123
|
+
PoliteVolitional: b.stem + 'ましょう',
|
|
124
|
+
Imperative: b.imper, // 書け
|
|
125
|
+
NegativeImperative: dict + 'な', // 書くな
|
|
126
|
+
PoliteImperative: te + 'ください',
|
|
127
|
+
Volitional: b.volit, // 書こう
|
|
128
|
+
Potential: b.potential, // 書ける
|
|
129
|
+
Passive: b.passive, // 書かれる
|
|
130
|
+
Causative: b.causative, // 書かせる
|
|
131
|
+
CausativePassive: b.causative.replace(/せる$/, 'せられる'),
|
|
132
|
+
Conditional: past + 'ら', // 書いたら (たら)
|
|
133
|
+
Provisional: b.cond + 'ば', // 書けば (ば)
|
|
134
|
+
Progressive: te + 'いる', // 書いている
|
|
135
|
+
ProgressivePolite: te + 'います',
|
|
136
|
+
PastProgressive: te + 'いた',
|
|
137
|
+
Desire: b.stem + 'たい', // 書きたい
|
|
138
|
+
Representative: past + 'り', // 書いたり
|
|
139
|
+
Presumptive: dict + 'だろう',
|
|
140
|
+
Continuative: b.stem + 'ながら', // 書きながら
|
|
141
|
+
}
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
export default conjugate
|
|
145
|
+
export { toBases, teToTa }
|
|
@@ -0,0 +1,178 @@
|
|
|
1
|
+
import { aRow, iRow, eRow, oRow } from './kana.js'
|
|
2
|
+
import conjugateVerb from './conjugate-verb.js'
|
|
3
|
+
|
|
4
|
+
// reverse the vowel-row tables, so we can walk a conjugated form back to its dictionary-form
|
|
5
|
+
const invert = (obj) => Object.keys(obj).reduce((h, k) => { h[obj[k]] = k; return h }, {})
|
|
6
|
+
const fromA = invert(aRow)
|
|
7
|
+
const fromI = invert(iRow)
|
|
8
|
+
const fromE = invert(eRow)
|
|
9
|
+
const fromO = invert(oRow)
|
|
10
|
+
|
|
11
|
+
// which 音便 endings can come from which dictionary-endings
|
|
12
|
+
// ordered by how common the dictionary-ending is, since 走って could in
|
|
13
|
+
// principle come from 走う/走つ/走る - only one of which is a real word
|
|
14
|
+
const fromTe = {
|
|
15
|
+
'って': ['る', 'う', 'つ'],
|
|
16
|
+
'んで': ['む', 'ぶ', 'ぬ'],
|
|
17
|
+
'いて': ['く'],
|
|
18
|
+
'いで': ['ぐ'],
|
|
19
|
+
'して': ['す'],
|
|
20
|
+
'て': [], // bare て means an ichidan stem
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
const defaultDict = function (stem, base) {
|
|
24
|
+
let last = stem[stem.length - 1]
|
|
25
|
+
let head = stem.slice(0, -1)
|
|
26
|
+
let out = []
|
|
27
|
+
if (base === 'neg') {
|
|
28
|
+
if (fromA[last]) out.push(head + fromA[last]) // 書か → 書く
|
|
29
|
+
out.push(stem + 'る') // 食べ → 食べる
|
|
30
|
+
} else if (base === 'masu') {
|
|
31
|
+
if (fromI[last]) out.push(head + fromI[last]) // 書き → 書く
|
|
32
|
+
out.push(stem + 'る') // 食べ → 食べる
|
|
33
|
+
} else if (base === 'cond') {
|
|
34
|
+
if (fromE[last]) out.push(head + fromE[last]) // 書け → 書く
|
|
35
|
+
out.push(head + fromE[last] + 'る') // 食べれ → 食べる (via れ)
|
|
36
|
+
} else if (base === 'volit') {
|
|
37
|
+
if (fromO[last]) out.push(head + fromO[last]) // 書こ → 書く
|
|
38
|
+
} else if (base === 'te' || base === 'ta') {
|
|
39
|
+
if (stem === 'し' || stem === 'き' || stem === '来') {
|
|
40
|
+
return [stem === 'し' ? 'する' : '来る']
|
|
41
|
+
}
|
|
42
|
+
// the ending was already consumed, so `stem` still carries the 音便 kana
|
|
43
|
+
let two = stem.slice(-1) // っ or ん or い
|
|
44
|
+
if (two === 'っ') {
|
|
45
|
+
fromTe['って'].forEach((c) => out.push(stem.slice(0, -1) + c))
|
|
46
|
+
// 行く is the one く-verb that takes って - 行った, not 行いた
|
|
47
|
+
out.push(stem.slice(0, -1) + 'く')
|
|
48
|
+
} else if (two === 'ん') {
|
|
49
|
+
fromTe['んで'].forEach((c) => out.push(stem.slice(0, -1) + c))
|
|
50
|
+
} else if (two === 'い') {
|
|
51
|
+
out.push(stem.slice(0, -1) + 'く', stem.slice(0, -1) + 'ぐ')
|
|
52
|
+
} else if (two === 'し') {
|
|
53
|
+
out.push(stem.slice(0, -1) + 'す')
|
|
54
|
+
}
|
|
55
|
+
out.push(stem + 'る') // ichidan: 食べ + た
|
|
56
|
+
}
|
|
57
|
+
return out.filter((s) => s && s.length > 1)
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
// the suffixes we know how to strip, longest-first.
|
|
62
|
+
// `base` says which of the five 活用形 the remaining stem is.
|
|
63
|
+
let suffixes = [
|
|
64
|
+
['ませんでした', ['Verb', 'PastTense', 'Polite', 'Negative'], 'masu'],
|
|
65
|
+
['なかったら', ['Verb', 'ConditionalVerb', 'Negative'], 'neg'],
|
|
66
|
+
['なければ', ['Verb', 'ConditionalVerb', 'Negative'], 'neg'],
|
|
67
|
+
['ましょう', ['Verb', 'Volitional', 'Polite'], 'masu'],
|
|
68
|
+
['なかった', ['Verb', 'PastTense', 'Negative'], 'neg'],
|
|
69
|
+
['ないで', ['Verb', 'Gerund', 'Negative'], 'neg'],
|
|
70
|
+
['なくて', ['Verb', 'Gerund', 'Negative'], 'neg'],
|
|
71
|
+
['ません', ['Verb', 'PresentTense', 'Polite', 'Negative'], 'masu'],
|
|
72
|
+
['ました', ['Verb', 'PastTense', 'Polite'], 'masu'],
|
|
73
|
+
['ながら', ['Verb', 'Continuative'], 'masu'],
|
|
74
|
+
['させる', ['Verb', 'Causative', 'PresentTense'], 'neg'],
|
|
75
|
+
['させられる', ['Verb', 'Causative', 'Passive', 'PresentTense'], 'neg'],
|
|
76
|
+
['られる', ['Verb', 'Passive', 'PresentTense'], 'neg'],
|
|
77
|
+
['たがる', ['Verb', 'Desire', 'PresentTense'], 'masu'],
|
|
78
|
+
['ます', ['Verb', 'PresentTense', 'Polite'], 'masu'],
|
|
79
|
+
['ない', ['Verb', 'PresentTense', 'Negative'], 'neg'],
|
|
80
|
+
['れる', ['Verb', 'Passive', 'PresentTense'], 'neg'],
|
|
81
|
+
['せる', ['Verb', 'Causative', 'PresentTense'], 'neg'],
|
|
82
|
+
['たい', ['Verb', 'Desire'], 'masu'],
|
|
83
|
+
['そう', ['Verb', 'Presumptive'], 'masu'],
|
|
84
|
+
['すぎる', ['Verb', 'PresentTense'], 'masu'],
|
|
85
|
+
['たら', ['Verb', 'ConditionalVerb'], 'ta'],
|
|
86
|
+
['だら', ['Verb', 'ConditionalVerb'], 'ta'],
|
|
87
|
+
['たり', ['Verb', 'Representative'], 'ta'],
|
|
88
|
+
['だり', ['Verb', 'Representative'], 'ta'],
|
|
89
|
+
['ている', ['Verb', 'Progressive', 'PresentTense'], 'te'],
|
|
90
|
+
['でいる', ['Verb', 'Progressive', 'PresentTense'], 'te'],
|
|
91
|
+
['ています', ['Verb', 'Progressive', 'PresentTense', 'Polite'], 'te'],
|
|
92
|
+
['でいます', ['Verb', 'Progressive', 'PresentTense', 'Polite'], 'te'],
|
|
93
|
+
['ていた', ['Verb', 'Progressive', 'PastTense'], 'te'],
|
|
94
|
+
['でいた', ['Verb', 'Progressive', 'PastTense'], 'te'],
|
|
95
|
+
['てる', ['Verb', 'Progressive', 'PresentTense'], 'te'],
|
|
96
|
+
['でる', ['Verb', 'Progressive', 'PresentTense'], 'te'],
|
|
97
|
+
['た', ['Verb', 'PastTense'], 'ta'],
|
|
98
|
+
['だ', ['Verb', 'PastTense'], 'ta'],
|
|
99
|
+
['て', ['Verb', 'Gerund'], 'te'],
|
|
100
|
+
['で', ['Verb', 'Gerund'], 'te'],
|
|
101
|
+
['ば', ['Verb', 'ConditionalVerb'], 'cond'],
|
|
102
|
+
['よう', ['Verb', 'Volitional'], 'neg', 'strict'],
|
|
103
|
+
['う', ['Verb', 'Volitional'], 'volit', 'strict'],
|
|
104
|
+
]
|
|
105
|
+
// always try the longest ending first - ています must win over ます
|
|
106
|
+
suffixes.sort((a, b) => b[0].length - a[0].length)
|
|
107
|
+
|
|
108
|
+
// walk a stem in a given 活用形 back to candidate dictionary-forms
|
|
109
|
+
// する and 来る don't decompose like anything else - their stem changes shape
|
|
110
|
+
const suruStem = new Set(['し', 'さ', 'せ', 'す'])
|
|
111
|
+
const kuruStem = new Set(['き', 'こ', 'く', '来'])
|
|
112
|
+
|
|
113
|
+
const toDict = function (stem, base) {
|
|
114
|
+
if (!stem) {
|
|
115
|
+
return []
|
|
116
|
+
}
|
|
117
|
+
let last = stem[stem.length - 1]
|
|
118
|
+
let pre = stem.slice(0, -1)
|
|
119
|
+
// a bare し/き is する/来る
|
|
120
|
+
if (stem.length === 1) {
|
|
121
|
+
if (suruStem.has(last)) return ['する']
|
|
122
|
+
if (kuruStem.has(last)) return ['来る']
|
|
123
|
+
}
|
|
124
|
+
// 勉強し could be 勉強する or a godan 勉強す - a two-character head is
|
|
125
|
+
// almost always a する-noun (勉強する), a one-character head almost always
|
|
126
|
+
// a godan verb (話す, 出す, 貸す).
|
|
127
|
+
if (suruStem.has(last)) {
|
|
128
|
+
let rest = defaultDict(stem, base)
|
|
129
|
+
return pre.length >= 2 ? [pre + 'する'].concat(rest) : rest.concat([pre + 'する'])
|
|
130
|
+
}
|
|
131
|
+
return defaultDict(stem, base)
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
/**
|
|
135
|
+
* take a conjugated verb and work backwards to its dictionary-form.
|
|
136
|
+
* `isKnown` lets the caller disambiguate 書いた (書く) from a hypothetical 書いる.
|
|
137
|
+
*/
|
|
138
|
+
const deconjugate = function (str, isKnown) {
|
|
139
|
+
if (!str || str.length < 2) {
|
|
140
|
+
return null
|
|
141
|
+
}
|
|
142
|
+
for (let i = 0; i < suffixes.length; i += 1) {
|
|
143
|
+
let [suffix, tags, base, strict] = suffixes[i]
|
|
144
|
+
if (!str.endsWith(suffix) || str.length <= suffix.length) {
|
|
145
|
+
continue
|
|
146
|
+
}
|
|
147
|
+
let stem = str.slice(0, -suffix.length)
|
|
148
|
+
// 'た'/'だ'/'て'/'で' hang off the 音便 kana, which belongs to the stem
|
|
149
|
+
let candidates = toDict(stem, base)
|
|
150
|
+
if (candidates.length === 0) {
|
|
151
|
+
continue
|
|
152
|
+
}
|
|
153
|
+
let hit = isKnown ? candidates.find(isKnown) : null
|
|
154
|
+
// a 'strict' suffix is too common a word-ending to trust on its own -
|
|
155
|
+
// 「たろう」 is a name, not the volitional of 「たる」
|
|
156
|
+
if (strict && !hit) {
|
|
157
|
+
continue
|
|
158
|
+
}
|
|
159
|
+
let root = hit || candidates[0]
|
|
160
|
+
// sanity-check: does re-conjugating the root actually produce this string?
|
|
161
|
+
if (hit) {
|
|
162
|
+
return { root, tags, candidates }
|
|
163
|
+
}
|
|
164
|
+
let verified = candidates.find(c => {
|
|
165
|
+
let forms = conjugateVerb(c)
|
|
166
|
+
return forms && Object.keys(forms).some(k => forms[k] === str)
|
|
167
|
+
})
|
|
168
|
+
if (verified) {
|
|
169
|
+
return { root: verified, tags, candidates }
|
|
170
|
+
}
|
|
171
|
+
return { root, tags, candidates, guess: true }
|
|
172
|
+
}
|
|
173
|
+
return null
|
|
174
|
+
}
|
|
175
|
+
|
|
176
|
+
export default deconjugate
|
|
177
|
+
// exported so tests/tagset.test.js can check these tags are declared
|
|
178
|
+
export { suffixes }
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
import conjugateVerb from './conjugate-verb.js'
|
|
2
|
+
import conjugateAdjective, { conjugateNaAdjective } from './conjugate-adj.js'
|
|
3
|
+
import verbClass from './verb-class.js'
|
|
4
|
+
import deconjugate from './deconjugate.js'
|
|
5
|
+
|
|
6
|
+
export default {
|
|
7
|
+
verb: conjugateVerb,
|
|
8
|
+
adjective: conjugateAdjective,
|
|
9
|
+
naAdjective: conjugateNaAdjective,
|
|
10
|
+
verbClass,
|
|
11
|
+
deconjugate,
|
|
12
|
+
}
|
|
13
|
+
export { conjugateVerb, conjugateAdjective, conjugateNaAdjective, verbClass, deconjugate }
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
// vowel-row transforms for godan (五段) verbs
|
|
2
|
+
// the dictionary-form's final kana tells us the row, and each
|
|
3
|
+
// grammatical 'base' (活用形) shifts it to a different vowel
|
|
4
|
+
|
|
5
|
+
// 未然形 - the 'a' row (takes ない, れる, せる)
|
|
6
|
+
const aRow = {
|
|
7
|
+
'う': 'わ', 'く': 'か', 'ぐ': 'が', 'す': 'さ', 'つ': 'た',
|
|
8
|
+
'ぬ': 'な', 'ぶ': 'ば', 'む': 'ま', 'る': 'ら',
|
|
9
|
+
}
|
|
10
|
+
// 連用形 - the 'i' row (takes ます, たい, ながら) - also the 'masu-stem'
|
|
11
|
+
const iRow = {
|
|
12
|
+
'う': 'い', 'く': 'き', 'ぐ': 'ぎ', 'す': 'し', 'つ': 'ち',
|
|
13
|
+
'ぬ': 'に', 'ぶ': 'び', 'む': 'み', 'る': 'り',
|
|
14
|
+
}
|
|
15
|
+
// 仮定形/命令形 - the 'e' row (takes ば, る for potential)
|
|
16
|
+
const eRow = {
|
|
17
|
+
'う': 'え', 'く': 'け', 'ぐ': 'げ', 'す': 'せ', 'つ': 'て',
|
|
18
|
+
'ぬ': 'ね', 'ぶ': 'べ', 'む': 'め', 'る': 'れ',
|
|
19
|
+
}
|
|
20
|
+
// 意向形 - the 'o' row (takes う for volitional)
|
|
21
|
+
const oRow = {
|
|
22
|
+
'う': 'お', 'く': 'こ', 'ぐ': 'ご', 'す': 'そ', 'つ': 'と',
|
|
23
|
+
'ぬ': 'の', 'ぶ': 'ぼ', 'む': 'も', 'る': 'ろ',
|
|
24
|
+
}
|
|
25
|
+
// 音便 - the sound-change used by the て/た forms
|
|
26
|
+
const teRow = {
|
|
27
|
+
'う': 'って', 'つ': 'って', 'る': 'って',
|
|
28
|
+
'ぬ': 'んで', 'ぶ': 'んで', 'む': 'んで',
|
|
29
|
+
'く': 'いて', 'ぐ': 'いで', 'す': 'して',
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
// kana that can precede る in an ichidan verb (the i-row and e-row)
|
|
33
|
+
const iRowKana = new Set('いきしちにひみりぎじぢびぴ'.split(''))
|
|
34
|
+
const eRowKana = new Set('えけせてねへめれげぜでべぺ'.split(''))
|
|
35
|
+
|
|
36
|
+
const godanEnding = new Set('うくぐすつぬぶむる'.split(''))
|
|
37
|
+
|
|
38
|
+
export { aRow, iRow, eRow, oRow, teRow, iRowKana, eRowKana, godanEnding }
|
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
// which tags each generated verb-form should carry.
|
|
2
|
+
// a form is 'PastTense' *and* 'Polite' *and* 'Negative' all at once - japanese
|
|
3
|
+
// stacks these on one word, so the tag-list has to as well.
|
|
4
|
+
const verbForms = {
|
|
5
|
+
Infinitive: ['Verb', 'Infinitive', 'PresentTense'],
|
|
6
|
+
PresentTense: ['Verb', 'PresentTense'],
|
|
7
|
+
PastTense: ['Verb', 'PastTense'],
|
|
8
|
+
Negative: ['Verb', 'PresentTense', 'Negative'],
|
|
9
|
+
PastNegative: ['Verb', 'PastTense', 'Negative'],
|
|
10
|
+
Gerund: ['Verb', 'Gerund'],
|
|
11
|
+
NegativeGerund: ['Verb', 'Gerund', 'Negative'],
|
|
12
|
+
Polite: ['Verb', 'PresentTense', 'Polite'],
|
|
13
|
+
PolitePast: ['Verb', 'PastTense', 'Polite'],
|
|
14
|
+
PoliteNegative: ['Verb', 'PresentTense', 'Polite', 'Negative'],
|
|
15
|
+
PolitePastNegative: ['Verb', 'PastTense', 'Polite', 'Negative'],
|
|
16
|
+
PoliteVolitional: ['Verb', 'Volitional', 'Polite'],
|
|
17
|
+
Imperative: ['Verb', 'Imperative'],
|
|
18
|
+
NegativeImperative: ['Verb', 'Imperative', 'Negative'],
|
|
19
|
+
Volitional: ['Verb', 'Volitional'],
|
|
20
|
+
Potential: ['Verb', 'Potential', 'PresentTense'],
|
|
21
|
+
Passive: ['Verb', 'Passive', 'PresentTense'],
|
|
22
|
+
Causative: ['Verb', 'Causative', 'PresentTense'],
|
|
23
|
+
CausativePassive: ['Verb', 'Causative', 'Passive', 'PresentTense'],
|
|
24
|
+
Conditional: ['Verb', 'ConditionalVerb'],
|
|
25
|
+
Provisional: ['Verb', 'ConditionalVerb'],
|
|
26
|
+
Progressive: ['Verb', 'Progressive', 'PresentTense'],
|
|
27
|
+
ProgressivePolite: ['Verb', 'Progressive', 'PresentTense', 'Polite'],
|
|
28
|
+
PastProgressive: ['Verb', 'Progressive', 'PastTense'],
|
|
29
|
+
Desire: ['Verb', 'Desire'],
|
|
30
|
+
Representative: ['Verb', 'Representative'],
|
|
31
|
+
Presumptive: ['Verb', 'Presumptive'],
|
|
32
|
+
Continuative: ['Verb', 'Continuative'],
|
|
33
|
+
Stem: ['Verb', 'VerbStem'],
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
const adjForms = {
|
|
37
|
+
Infinitive: ['Adjective', 'IAdjective', 'PresentTense'],
|
|
38
|
+
PresentTense: ['Adjective', 'IAdjective', 'PresentTense'],
|
|
39
|
+
Negative: ['Adjective', 'IAdjective', 'PresentTense', 'Negative'],
|
|
40
|
+
PastTense: ['Adjective', 'IAdjective', 'PastTense'],
|
|
41
|
+
PastNegative: ['Adjective', 'IAdjective', 'PastTense', 'Negative'],
|
|
42
|
+
Gerund: ['Adjective', 'IAdjective', 'Gerund'],
|
|
43
|
+
Adverb: ['Adverb'],
|
|
44
|
+
Provisional: ['Adjective', 'IAdjective', 'ConditionalVerb'],
|
|
45
|
+
Presumptive: ['Adjective', 'IAdjective', 'Presumptive'],
|
|
46
|
+
Superlative: ['Verb', 'PresentTense'],
|
|
47
|
+
Impression: ['Adjective', 'Presumptive'],
|
|
48
|
+
Nominal: ['Noun'],
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
// 食べたい inflects like an い-adjective - 食べたくない, 食べたかった
|
|
52
|
+
const desireForms = {
|
|
53
|
+
Infinitive: ['Verb', 'Desire', 'PresentTense'],
|
|
54
|
+
PresentTense: ['Verb', 'Desire', 'PresentTense'],
|
|
55
|
+
Negative: ['Verb', 'Desire', 'PresentTense', 'Negative'],
|
|
56
|
+
PastTense: ['Verb', 'Desire', 'PastTense'],
|
|
57
|
+
PastNegative: ['Verb', 'Desire', 'PastTense', 'Negative'],
|
|
58
|
+
Gerund: ['Verb', 'Desire', 'Gerund'],
|
|
59
|
+
Provisional: ['Verb', 'Desire', 'ConditionalVerb'],
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
// the passive/potential/causative stems are themselves ichidan verbs, so
|
|
63
|
+
// 褒められる has a past (褒められた) and a polite past (褒められました) of its own
|
|
64
|
+
const derivedForms = {
|
|
65
|
+
PastTense: ['PastTense'],
|
|
66
|
+
Negative: ['PresentTense', 'Negative'],
|
|
67
|
+
PastNegative: ['PastTense', 'Negative'],
|
|
68
|
+
Gerund: ['Gerund'],
|
|
69
|
+
Polite: ['PresentTense', 'Polite'],
|
|
70
|
+
PolitePast: ['PastTense', 'Polite'],
|
|
71
|
+
PoliteNegative: ['PresentTense', 'Polite', 'Negative'],
|
|
72
|
+
PolitePastNegative: ['PastTense', 'Polite', 'Negative'],
|
|
73
|
+
Progressive: ['Progressive', 'PresentTense'],
|
|
74
|
+
Conditional: ['ConditionalVerb'],
|
|
75
|
+
Provisional: ['ConditionalVerb'],
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
// な-adjectives inflect *with the copula* - 静か + でした - and the copula is
|
|
79
|
+
// its own token. only 静かに is a word in its own right.
|
|
80
|
+
const naAdjForms = {
|
|
81
|
+
Infinitive: ['Adjective', 'NaAdjective'],
|
|
82
|
+
Adverb: ['Adverb'],
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
export { verbForms, adjForms, naAdjForms, desireForms, derivedForms }
|