ja-compromise 0.0.1 → 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (69) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +367 -6
  3. package/builds/ja-compromise.cjs +4389 -2286
  4. package/builds/ja-compromise.min.js +1 -1
  5. package/builds/ja-compromise.mjs +1 -1
  6. package/package.json +36 -23
  7. package/src/01-one/conjugate/conjugate-adj.js +53 -0
  8. package/src/01-one/conjugate/conjugate-verb.js +145 -0
  9. package/src/01-one/conjugate/deconjugate.js +178 -0
  10. package/src/01-one/conjugate/index.js +13 -0
  11. package/src/01-one/conjugate/kana.js +38 -0
  12. package/src/01-one/conjugate/tags.js +85 -0
  13. package/src/01-one/conjugate/verb-class.js +105 -0
  14. package/src/01-one/lexicon/_data.js +30 -0
  15. package/src/01-one/lexicon/api.js +61 -0
  16. package/src/01-one/lexicon/lexicon.js +163 -0
  17. package/src/01-one/lexicon/misc.js +115 -0
  18. package/src/01-one/lexicon/plugin.js +20 -0
  19. package/src/01-one/numbers/api.js +202 -0
  20. package/src/01-one/numbers/kanji-number.js +69 -0
  21. package/src/01-one/numbers/to-kanji.js +66 -0
  22. package/src/01-one/output/compute/dict.js +13 -0
  23. package/src/01-one/output/compute/english.js +13 -0
  24. package/src/01-one/output/compute/root.js +30 -0
  25. package/src/01-one/output/debug/_color.js +16 -0
  26. package/src/01-one/output/debug/index.js +24 -0
  27. package/src/01-one/output/debug/tags.js +56 -0
  28. package/src/01-one/output/plugin.js +13 -0
  29. package/src/01-one/romanji/api.js +16 -0
  30. package/src/01-one/romanji/compute/index.js +36 -0
  31. package/src/01-one/romanji/compute/kanji-reading/index.js +91 -0
  32. package/src/01-one/romanji/compute/kanji-reading/readings.js +861 -0
  33. package/src/01-one/romanji/compute/kanji-reading/words.js +31 -0
  34. package/src/01-one/romanji/compute/toRomanji/hiragana-map.js +175 -0
  35. package/src/01-one/romanji/compute/toRomanji/index.js +84 -0
  36. package/src/01-one/romanji/compute/toRomanji/katakana-map.js +128 -0
  37. package/src/01-one/romanji/plugin.js +7 -0
  38. package/src/01-one/tokenizer/methods/join-numbers.js +32 -0
  39. package/src/01-one/tokenizer/methods/join-up.js +66 -0
  40. package/src/01-one/tokenizer/methods/lib.js +66 -0
  41. package/src/01-one/tokenizer/methods/naiive-split.js +88 -0
  42. package/src/01-one/tokenizer/methods/okurigana.js +58 -0
  43. package/src/01-one/tokenizer/methods/terms.js +143 -0
  44. package/src/01-one/tokenizer/methods/trie/build.js +13 -0
  45. package/src/01-one/tokenizer/methods/trie/split-up.js +29 -0
  46. package/src/01-one/tokenizer/methods/whitespace.js +44 -0
  47. package/src/01-one/tokenizer/plugin.js +13 -0
  48. package/src/02-two/preTagger/compute/01-script.js +44 -0
  49. package/src/02-two/preTagger/compute/02-particles.js +82 -0
  50. package/src/02-two/preTagger/compute/03-verbs.js +111 -0
  51. package/src/02-two/preTagger/compute/04-adjectives.js +43 -0
  52. package/src/02-two/preTagger/compute/05-people.js +22 -0
  53. package/src/02-two/preTagger/compute/06-numbers.js +80 -0
  54. package/src/02-two/preTagger/compute/07-dates.js +105 -0
  55. package/src/02-two/preTagger/compute/index.js +52 -0
  56. package/src/02-two/preTagger/plugin.js +9 -0
  57. package/src/02-two/tagset/plugin.js +12 -0
  58. package/src/02-two/tagset/tags/dates.js +58 -0
  59. package/src/02-two/tagset/tags/misc.js +91 -0
  60. package/src/02-two/tagset/tags/nouns.js +131 -0
  61. package/src/02-two/tagset/tags/particles.js +50 -0
  62. package/src/02-two/tagset/tags/values.js +68 -0
  63. package/src/02-two/tagset/tags/verbs.js +93 -0
  64. package/src/_lib.js +2 -0
  65. package/src/_version.js +1 -0
  66. package/src/index.js +82 -0
  67. package/types/index.d.ts +268 -0
  68. package/types/japanese.d.ts +159 -0
  69. package/types/misc.d.ts +92 -0
package/package.json CHANGED
@@ -2,13 +2,19 @@
2
2
  "author": "Spencer Kelly <spencermountain@gmail.com> (http://spencermounta.in)",
3
3
  "name": "ja-compromise",
4
4
  "description": "日本語の控えめな自然言語処理",
5
- "version": "0.0.1",
6
- "main": "./src/index.js",
5
+ "version": "0.1.0",
6
+ "engines": {
7
+ "node": "^22.13.0 || ^24.0.0 || ^26.0.0",
8
+ "pnpm": "^11.5.0"
9
+ },
10
+ "main": "./builds/ja-compromise.mjs",
11
+ "browser": "./builds/ja-compromise.min.js",
7
12
  "unpkg": "./builds/ja-compromise.min.js",
8
13
  "type": "module",
9
14
  "sideEffects": false,
10
15
  "exports": {
11
16
  ".": {
17
+ "types": "./types/index.d.ts",
12
18
  "import": "./builds/ja-compromise.mjs",
13
19
  "require": "./builds/ja-compromise.cjs"
14
20
  }
@@ -18,34 +24,41 @@
18
24
  "type": "git",
19
25
  "url": "git://github.com/nlp-compromise/ja-compromise.git"
20
26
  },
21
- "scripts": {
22
- "test": "tape \"./tests/**/*.test.js\" | tap-dancer",
23
- "testb": "TESTENV=prod npm run test",
24
- "build": "npm run version && rollup -c --silent",
25
- "version": "node ./scripts/version.js",
26
- "pack": "node ./scripts/pack.js",
27
- "score": " node ./learn/test/index.js",
28
- "watch": "amble ./scratch.js",
29
- "stress": "node scripts/stress.js"
30
- },
31
27
  "files": [
32
28
  "builds/",
29
+ "types/",
30
+ "src/",
33
31
  "docs/"
34
32
  ],
35
33
  "dependencies": {
36
- "compromise": "14.8.2",
34
+ "compromise": "^14.17.0",
37
35
  "efrt": "2.7.0",
38
- "suffix-thumb": "4.0.2"
36
+ "suffix-thumb": "5.0.3"
39
37
  },
40
38
  "devDependencies": {
41
- "@rollup/plugin-alias": "3.1.9",
42
- "@rollup/plugin-node-resolve": "13.3.0",
43
- "amble": "1.3.0",
44
- "rollup": "2.75.7",
45
- "rollup-plugin-terser": "7.0.2",
39
+ "@eslint/js": "^10.0.1",
40
+ "@rollup/plugin-node-resolve": "16.0.3",
41
+ "@rollup/plugin-terser": "1.0.0",
42
+ "cross-env": "10.1.0",
43
+ "eslint": "10.10.0",
44
+ "eslint-plugin-regexp": "3.3.0",
45
+ "rollup": "4.63.1",
46
46
  "tap-dancer": "0.3.4",
47
- "tape": "5.6.3",
48
- "xml-stream": "^0.4.5"
47
+ "tape": "5.10.2",
48
+ "typescript": "7.0.2"
49
49
  },
50
- "license": "MIT"
51
- }
50
+ "license": "MIT",
51
+ "scripts": {
52
+ "typecheck": "tsc -p tsconfig.json",
53
+ "test": "tsc -p tsconfig.json && tape \"./tests/**/*.test.js\" | tap-dancer",
54
+ "testb": "TESTENV=prod pnpm run test",
55
+ "build": "pnpm run version && rollup -c --silent",
56
+ "version": "node ./scripts/version.js",
57
+ "pack": "node ./scripts/pack.js",
58
+ "score": "node ./learn/test/index.js",
59
+ "audit": "node ./learn/verbs/audit.js",
60
+ "lint": "eslint ./src/**/*",
61
+ "watch": "amble ./scratch.js",
62
+ "stress": "node scripts/stress.js"
63
+ }
64
+ }
@@ -0,0 +1,53 @@
1
+ // い-adjectives inflect for tense and polarity - they are not 'just adjectives'.
2
+ // 高い → 高くない → 高かった → 高くなかった. the copula does not carry that tense.
3
+
4
+ // いい/良い is the one truly irregular い-adjective - it inflects as よい
5
+ const irregularStem = { 'いい': 'よ', '良い': '良', 'よい': 'よ' }
6
+
7
+ const conjugateAdjective = function (word) {
8
+ if (!word || !word.endsWith('い') || word.length < 2) {
9
+ return null
10
+ }
11
+ let stem = irregularStem[word] !== undefined ? irregularStem[word] : word.slice(0, -1)
12
+ return {
13
+ Infinitive: word, // 高い
14
+ PresentTense: word,
15
+ Negative: stem + 'くない', // 高くない
16
+ PastTense: stem + 'かった', // 高かった
17
+ PastNegative: stem + 'くなかった', // 高くなかった
18
+ Gerund: stem + 'くて', // 高くて
19
+ Adverb: stem + 'く', // 高く
20
+ Provisional: stem + 'ければ', // 高ければ
21
+ Presumptive: stem + 'かろう',
22
+ Polite: word + 'です',
23
+ PolitePast: stem + 'かったです',
24
+ PoliteNegative: stem + 'くないです',
25
+ Superlative: stem + 'すぎる', // 高すぎる
26
+ Impression: stem + 'そう', // 高そう
27
+ Nominal: stem + 'さ', // 高さ
28
+ }
29
+ }
30
+
31
+ // な-adjectives (形容動詞) are stored bare - 静か - and inflect with the copula
32
+ const conjugateNaAdjective = function (word) {
33
+ if (!word) {
34
+ return null
35
+ }
36
+ return {
37
+ Infinitive: word,
38
+ Adnominal: word + 'な', // 静かな
39
+ PresentTense: word + 'だ', // 静かだ
40
+ Negative: word + 'じゃない', // 静かじゃない
41
+ PastTense: word + 'だった', // 静かだった
42
+ PastNegative: word + 'じゃなかった',
43
+ Gerund: word + 'で', // 静かで
44
+ Adverb: word + 'に', // 静かに
45
+ Polite: word + 'です',
46
+ PolitePast: word + 'でした',
47
+ PoliteNegative: word + 'じゃありません',
48
+ Provisional: word + 'なら',
49
+ }
50
+ }
51
+
52
+ export default conjugateAdjective
53
+ export { conjugateNaAdjective }
@@ -0,0 +1,145 @@
1
+ import { aRow, iRow, eRow, oRow, teRow } from './kana.js'
2
+ import verbClass from './verb-class.js'
3
+
4
+ // the five 'bases' every japanese verb form is built out of.
5
+ // get these right and every suffix falls out of them.
6
+ const toBases = function (dict, cls) {
7
+ let stem = dict.slice(0, -1)
8
+ let last = dict[dict.length - 1]
9
+ switch (cls) {
10
+ case 'godan':
11
+ return {
12
+ negStem: stem + aRow[last], // 未然形 - 書か
13
+ stem: stem + iRow[last], // 連用形 - 書き
14
+ cond: stem + eRow[last], // 仮定形 - 書け
15
+ imper: stem + eRow[last], // 命令形 - 書け
16
+ volit: stem + oRow[last] + 'う', // 意向形 - 書こう
17
+ te: stem + teRow[last], // て形 - 書いて
18
+ potential: stem + eRow[last] + 'る',
19
+ passive: stem + aRow[last] + 'れる',
20
+ causative: stem + aRow[last] + 'せる',
21
+ }
22
+ case 'ichidan': {
23
+ let s = dict.slice(0, -1) // 食べ
24
+ // くれる is the one ichidan verb with a bare imperative - くれ, not くれろ
25
+ let imper = /(呉れる|くれる)$/.test(dict) ? s : s + 'ろ'
26
+ return {
27
+ negStem: s, stem: s, cond: s + 'れ', imper: imper,
28
+ volit: s + 'よう', te: s + 'て',
29
+ potential: s + 'られる', passive: s + 'られる', causative: s + 'させる',
30
+ }
31
+ }
32
+ case 'suru': {
33
+ let pre = dict.slice(0, -2) // 勉強
34
+ return {
35
+ negStem: pre + 'し', stem: pre + 'し', cond: pre + 'すれ', imper: pre + 'しろ',
36
+ volit: pre + 'しよう', te: pre + 'して',
37
+ potential: pre === '' ? 'できる' : pre + 'できる',
38
+ passive: pre + 'される', causative: pre + 'させる',
39
+ }
40
+ }
41
+ case 'kuru': {
42
+ let pre = dict.slice(0, -2)
43
+ let kana = dict.endsWith('くる')
44
+ let ko = kana ? 'こ' : '来'
45
+ let ki = kana ? 'き' : '来'
46
+ let ku = kana ? 'く' : '来'
47
+ return {
48
+ negStem: pre + ko, stem: pre + ki, cond: pre + ku + 'れ', imper: pre + ko + 'い',
49
+ volit: pre + ko + 'よう', te: pre + ki + 'て',
50
+ potential: pre + ko + 'られる', passive: pre + ko + 'られる', causative: pre + ko + 'させる',
51
+ }
52
+ }
53
+ case 'aru': {
54
+ let s = dict.slice(0, -1) // あ / 有 / 在
55
+ return {
56
+ negStem: null, // ある has no 未然形 - its negative is just ない
57
+ stem: s + 'り', cond: s + 'れ', imper: s + 'れ', volit: s + 'ろう', te: s + 'って',
58
+ potential: s + 'れる', passive: s + 'られる', causative: s + 'らせる',
59
+ }
60
+ }
61
+ case 'iku': { // 行く takes って, not the regular いて
62
+ let s = dict.slice(0, -1)
63
+ return {
64
+ negStem: s + 'か', stem: s + 'き', cond: s + 'け', imper: s + 'け', volit: s + 'こう',
65
+ te: s + 'って',
66
+ potential: s + 'ける', passive: s + 'かれる', causative: s + 'かせる',
67
+ }
68
+ }
69
+ case 'ou': { // 問う/請う keep うて rather than って
70
+ let s = dict.slice(0, -1)
71
+ return {
72
+ negStem: s + 'わ', stem: s + 'い', cond: s + 'え', imper: s + 'え', volit: s + 'おう',
73
+ te: s + 'うて',
74
+ potential: s + 'える', passive: s + 'われる', causative: s + 'わせる',
75
+ }
76
+ }
77
+ case 'aru5': { // くださる/なさる - godan, but the masu-stem drops the り
78
+ let s = dict.slice(0, -1)
79
+ return {
80
+ negStem: s + 'ら', stem: s + 'い', cond: s + 'れ', imper: s + 'い', volit: s + 'ろう',
81
+ te: s + 'って',
82
+ potential: s + 'れる', passive: s + 'られる', causative: s + 'らせる',
83
+ }
84
+ }
85
+ default:
86
+ return null
87
+ }
88
+ }
89
+
90
+ // て → た, で → だ
91
+ const teToTa = (te) => te.replace(/て$/, 'た').replace(/で$/, 'だ')
92
+
93
+ /** produce the full paradigm of a dictionary-form verb */
94
+ const conjugate = function (dict, hint) {
95
+ let cls = verbClass(dict, hint)
96
+ if (!cls) {
97
+ return null
98
+ }
99
+ let b = toBases(dict, cls)
100
+ if (!b) {
101
+ return null
102
+ }
103
+ // ある is the one verb whose plain negative isn't built on a 未然形
104
+ let negative = b.negStem === null ? 'ない' : b.negStem + 'ない'
105
+ let pastNegative = b.negStem === null ? 'なかった' : b.negStem + 'なかった'
106
+ let te = b.te
107
+ let past = teToTa(te)
108
+
109
+ return {
110
+ class: cls,
111
+ Infinitive: dict, // 辞書形 - 書く
112
+ Stem: b.stem, // 連用形 - 書き
113
+ PresentTense: dict,
114
+ PastTense: past, // 書いた
115
+ Negative: negative, // 書かない
116
+ PastNegative: pastNegative, // 書かなかった
117
+ Gerund: te, // 書いて (て形)
118
+ NegativeGerund: b.negStem === null ? 'なくて' : b.negStem + 'なくて',
119
+ Polite: b.stem + 'ます', // 書きます
120
+ PolitePast: b.stem + 'ました', // 書きました
121
+ PoliteNegative: b.stem + 'ません', // 書きません
122
+ PolitePastNegative: b.stem + 'ませんでした',
123
+ PoliteVolitional: b.stem + 'ましょう',
124
+ Imperative: b.imper, // 書け
125
+ NegativeImperative: dict + 'な', // 書くな
126
+ PoliteImperative: te + 'ください',
127
+ Volitional: b.volit, // 書こう
128
+ Potential: b.potential, // 書ける
129
+ Passive: b.passive, // 書かれる
130
+ Causative: b.causative, // 書かせる
131
+ CausativePassive: b.causative.replace(/せる$/, 'せられる'),
132
+ Conditional: past + 'ら', // 書いたら (たら)
133
+ Provisional: b.cond + 'ば', // 書けば (ば)
134
+ Progressive: te + 'いる', // 書いている
135
+ ProgressivePolite: te + 'います',
136
+ PastProgressive: te + 'いた',
137
+ Desire: b.stem + 'たい', // 書きたい
138
+ Representative: past + 'り', // 書いたり
139
+ Presumptive: dict + 'だろう',
140
+ Continuative: b.stem + 'ながら', // 書きながら
141
+ }
142
+ }
143
+
144
+ export default conjugate
145
+ export { toBases, teToTa }
@@ -0,0 +1,178 @@
1
+ import { aRow, iRow, eRow, oRow } from './kana.js'
2
+ import conjugateVerb from './conjugate-verb.js'
3
+
4
+ // reverse the vowel-row tables, so we can walk a conjugated form back to its dictionary-form
5
+ const invert = (obj) => Object.keys(obj).reduce((h, k) => { h[obj[k]] = k; return h }, {})
6
+ const fromA = invert(aRow)
7
+ const fromI = invert(iRow)
8
+ const fromE = invert(eRow)
9
+ const fromO = invert(oRow)
10
+
11
+ // which 音便 endings can come from which dictionary-endings
12
+ // ordered by how common the dictionary-ending is, since 走って could in
13
+ // principle come from 走う/走つ/走る - only one of which is a real word
14
+ const fromTe = {
15
+ 'って': ['る', 'う', 'つ'],
16
+ 'んで': ['む', 'ぶ', 'ぬ'],
17
+ 'いて': ['く'],
18
+ 'いで': ['ぐ'],
19
+ 'して': ['す'],
20
+ 'て': [], // bare て means an ichidan stem
21
+ }
22
+
23
+ const defaultDict = function (stem, base) {
24
+ let last = stem[stem.length - 1]
25
+ let head = stem.slice(0, -1)
26
+ let out = []
27
+ if (base === 'neg') {
28
+ if (fromA[last]) out.push(head + fromA[last]) // 書か → 書く
29
+ out.push(stem + 'る') // 食べ → 食べる
30
+ } else if (base === 'masu') {
31
+ if (fromI[last]) out.push(head + fromI[last]) // 書き → 書く
32
+ out.push(stem + 'る') // 食べ → 食べる
33
+ } else if (base === 'cond') {
34
+ if (fromE[last]) out.push(head + fromE[last]) // 書け → 書く
35
+ out.push(head + fromE[last] + 'る') // 食べれ → 食べる (via れ)
36
+ } else if (base === 'volit') {
37
+ if (fromO[last]) out.push(head + fromO[last]) // 書こ → 書く
38
+ } else if (base === 'te' || base === 'ta') {
39
+ if (stem === 'し' || stem === 'き' || stem === '来') {
40
+ return [stem === 'し' ? 'する' : '来る']
41
+ }
42
+ // the ending was already consumed, so `stem` still carries the 音便 kana
43
+ let two = stem.slice(-1) // っ or ん or い
44
+ if (two === 'っ') {
45
+ fromTe['って'].forEach((c) => out.push(stem.slice(0, -1) + c))
46
+ // 行く is the one く-verb that takes って - 行った, not 行いた
47
+ out.push(stem.slice(0, -1) + 'く')
48
+ } else if (two === 'ん') {
49
+ fromTe['んで'].forEach((c) => out.push(stem.slice(0, -1) + c))
50
+ } else if (two === 'い') {
51
+ out.push(stem.slice(0, -1) + 'く', stem.slice(0, -1) + 'ぐ')
52
+ } else if (two === 'し') {
53
+ out.push(stem.slice(0, -1) + 'す')
54
+ }
55
+ out.push(stem + 'る') // ichidan: 食べ + た
56
+ }
57
+ return out.filter((s) => s && s.length > 1)
58
+ }
59
+
60
+
61
+ // the suffixes we know how to strip, longest-first.
62
+ // `base` says which of the five 活用形 the remaining stem is.
63
+ let suffixes = [
64
+ ['ませんでした', ['Verb', 'PastTense', 'Polite', 'Negative'], 'masu'],
65
+ ['なかったら', ['Verb', 'ConditionalVerb', 'Negative'], 'neg'],
66
+ ['なければ', ['Verb', 'ConditionalVerb', 'Negative'], 'neg'],
67
+ ['ましょう', ['Verb', 'Volitional', 'Polite'], 'masu'],
68
+ ['なかった', ['Verb', 'PastTense', 'Negative'], 'neg'],
69
+ ['ないで', ['Verb', 'Gerund', 'Negative'], 'neg'],
70
+ ['なくて', ['Verb', 'Gerund', 'Negative'], 'neg'],
71
+ ['ません', ['Verb', 'PresentTense', 'Polite', 'Negative'], 'masu'],
72
+ ['ました', ['Verb', 'PastTense', 'Polite'], 'masu'],
73
+ ['ながら', ['Verb', 'Continuative'], 'masu'],
74
+ ['させる', ['Verb', 'Causative', 'PresentTense'], 'neg'],
75
+ ['させられる', ['Verb', 'Causative', 'Passive', 'PresentTense'], 'neg'],
76
+ ['られる', ['Verb', 'Passive', 'PresentTense'], 'neg'],
77
+ ['たがる', ['Verb', 'Desire', 'PresentTense'], 'masu'],
78
+ ['ます', ['Verb', 'PresentTense', 'Polite'], 'masu'],
79
+ ['ない', ['Verb', 'PresentTense', 'Negative'], 'neg'],
80
+ ['れる', ['Verb', 'Passive', 'PresentTense'], 'neg'],
81
+ ['せる', ['Verb', 'Causative', 'PresentTense'], 'neg'],
82
+ ['たい', ['Verb', 'Desire'], 'masu'],
83
+ ['そう', ['Verb', 'Presumptive'], 'masu'],
84
+ ['すぎる', ['Verb', 'PresentTense'], 'masu'],
85
+ ['たら', ['Verb', 'ConditionalVerb'], 'ta'],
86
+ ['だら', ['Verb', 'ConditionalVerb'], 'ta'],
87
+ ['たり', ['Verb', 'Representative'], 'ta'],
88
+ ['だり', ['Verb', 'Representative'], 'ta'],
89
+ ['ている', ['Verb', 'Progressive', 'PresentTense'], 'te'],
90
+ ['でいる', ['Verb', 'Progressive', 'PresentTense'], 'te'],
91
+ ['ています', ['Verb', 'Progressive', 'PresentTense', 'Polite'], 'te'],
92
+ ['でいます', ['Verb', 'Progressive', 'PresentTense', 'Polite'], 'te'],
93
+ ['ていた', ['Verb', 'Progressive', 'PastTense'], 'te'],
94
+ ['でいた', ['Verb', 'Progressive', 'PastTense'], 'te'],
95
+ ['てる', ['Verb', 'Progressive', 'PresentTense'], 'te'],
96
+ ['でる', ['Verb', 'Progressive', 'PresentTense'], 'te'],
97
+ ['た', ['Verb', 'PastTense'], 'ta'],
98
+ ['だ', ['Verb', 'PastTense'], 'ta'],
99
+ ['て', ['Verb', 'Gerund'], 'te'],
100
+ ['で', ['Verb', 'Gerund'], 'te'],
101
+ ['ば', ['Verb', 'ConditionalVerb'], 'cond'],
102
+ ['よう', ['Verb', 'Volitional'], 'neg', 'strict'],
103
+ ['う', ['Verb', 'Volitional'], 'volit', 'strict'],
104
+ ]
105
+ // always try the longest ending first - ています must win over ます
106
+ suffixes.sort((a, b) => b[0].length - a[0].length)
107
+
108
+ // walk a stem in a given 活用形 back to candidate dictionary-forms
109
+ // する and 来る don't decompose like anything else - their stem changes shape
110
+ const suruStem = new Set(['し', 'さ', 'せ', 'す'])
111
+ const kuruStem = new Set(['き', 'こ', 'く', '来'])
112
+
113
+ const toDict = function (stem, base) {
114
+ if (!stem) {
115
+ return []
116
+ }
117
+ let last = stem[stem.length - 1]
118
+ let pre = stem.slice(0, -1)
119
+ // a bare し/き is する/来る
120
+ if (stem.length === 1) {
121
+ if (suruStem.has(last)) return ['する']
122
+ if (kuruStem.has(last)) return ['来る']
123
+ }
124
+ // 勉強し could be 勉強する or a godan 勉強す - a two-character head is
125
+ // almost always a する-noun (勉強する), a one-character head almost always
126
+ // a godan verb (話す, 出す, 貸す).
127
+ if (suruStem.has(last)) {
128
+ let rest = defaultDict(stem, base)
129
+ return pre.length >= 2 ? [pre + 'する'].concat(rest) : rest.concat([pre + 'する'])
130
+ }
131
+ return defaultDict(stem, base)
132
+ }
133
+
134
+ /**
135
+ * take a conjugated verb and work backwards to its dictionary-form.
136
+ * `isKnown` lets the caller disambiguate 書いた (書く) from a hypothetical 書いる.
137
+ */
138
+ const deconjugate = function (str, isKnown) {
139
+ if (!str || str.length < 2) {
140
+ return null
141
+ }
142
+ for (let i = 0; i < suffixes.length; i += 1) {
143
+ let [suffix, tags, base, strict] = suffixes[i]
144
+ if (!str.endsWith(suffix) || str.length <= suffix.length) {
145
+ continue
146
+ }
147
+ let stem = str.slice(0, -suffix.length)
148
+ // 'た'/'だ'/'て'/'で' hang off the 音便 kana, which belongs to the stem
149
+ let candidates = toDict(stem, base)
150
+ if (candidates.length === 0) {
151
+ continue
152
+ }
153
+ let hit = isKnown ? candidates.find(isKnown) : null
154
+ // a 'strict' suffix is too common a word-ending to trust on its own -
155
+ // 「たろう」 is a name, not the volitional of 「たる」
156
+ if (strict && !hit) {
157
+ continue
158
+ }
159
+ let root = hit || candidates[0]
160
+ // sanity-check: does re-conjugating the root actually produce this string?
161
+ if (hit) {
162
+ return { root, tags, candidates }
163
+ }
164
+ let verified = candidates.find(c => {
165
+ let forms = conjugateVerb(c)
166
+ return forms && Object.keys(forms).some(k => forms[k] === str)
167
+ })
168
+ if (verified) {
169
+ return { root: verified, tags, candidates }
170
+ }
171
+ return { root, tags, candidates, guess: true }
172
+ }
173
+ return null
174
+ }
175
+
176
+ export default deconjugate
177
+ // exported so tests/tagset.test.js can check these tags are declared
178
+ export { suffixes }
@@ -0,0 +1,13 @@
1
+ import conjugateVerb from './conjugate-verb.js'
2
+ import conjugateAdjective, { conjugateNaAdjective } from './conjugate-adj.js'
3
+ import verbClass from './verb-class.js'
4
+ import deconjugate from './deconjugate.js'
5
+
6
+ export default {
7
+ verb: conjugateVerb,
8
+ adjective: conjugateAdjective,
9
+ naAdjective: conjugateNaAdjective,
10
+ verbClass,
11
+ deconjugate,
12
+ }
13
+ export { conjugateVerb, conjugateAdjective, conjugateNaAdjective, verbClass, deconjugate }
@@ -0,0 +1,38 @@
1
+ // vowel-row transforms for godan (五段) verbs
2
+ // the dictionary-form's final kana tells us the row, and each
3
+ // grammatical 'base' (活用形) shifts it to a different vowel
4
+
5
+ // 未然形 - the 'a' row (takes ない, れる, せる)
6
+ const aRow = {
7
+ 'う': 'わ', 'く': 'か', 'ぐ': 'が', 'す': 'さ', 'つ': 'た',
8
+ 'ぬ': 'な', 'ぶ': 'ば', 'む': 'ま', 'る': 'ら',
9
+ }
10
+ // 連用形 - the 'i' row (takes ます, たい, ながら) - also the 'masu-stem'
11
+ const iRow = {
12
+ 'う': 'い', 'く': 'き', 'ぐ': 'ぎ', 'す': 'し', 'つ': 'ち',
13
+ 'ぬ': 'に', 'ぶ': 'び', 'む': 'み', 'る': 'り',
14
+ }
15
+ // 仮定形/命令形 - the 'e' row (takes ば, る for potential)
16
+ const eRow = {
17
+ 'う': 'え', 'く': 'け', 'ぐ': 'げ', 'す': 'せ', 'つ': 'て',
18
+ 'ぬ': 'ね', 'ぶ': 'べ', 'む': 'め', 'る': 'れ',
19
+ }
20
+ // 意向形 - the 'o' row (takes う for volitional)
21
+ const oRow = {
22
+ 'う': 'お', 'く': 'こ', 'ぐ': 'ご', 'す': 'そ', 'つ': 'と',
23
+ 'ぬ': 'の', 'ぶ': 'ぼ', 'む': 'も', 'る': 'ろ',
24
+ }
25
+ // 音便 - the sound-change used by the て/た forms
26
+ const teRow = {
27
+ 'う': 'って', 'つ': 'って', 'る': 'って',
28
+ 'ぬ': 'んで', 'ぶ': 'んで', 'む': 'んで',
29
+ 'く': 'いて', 'ぐ': 'いで', 'す': 'して',
30
+ }
31
+
32
+ // kana that can precede る in an ichidan verb (the i-row and e-row)
33
+ const iRowKana = new Set('いきしちにひみりぎじぢびぴ'.split(''))
34
+ const eRowKana = new Set('えけせてねへめれげぜでべぺ'.split(''))
35
+
36
+ const godanEnding = new Set('うくぐすつぬぶむる'.split(''))
37
+
38
+ export { aRow, iRow, eRow, oRow, teRow, iRowKana, eRowKana, godanEnding }
@@ -0,0 +1,85 @@
1
+ // which tags each generated verb-form should carry.
2
+ // a form is 'PastTense' *and* 'Polite' *and* 'Negative' all at once - japanese
3
+ // stacks these on one word, so the tag-list has to as well.
4
+ const verbForms = {
5
+ Infinitive: ['Verb', 'Infinitive', 'PresentTense'],
6
+ PresentTense: ['Verb', 'PresentTense'],
7
+ PastTense: ['Verb', 'PastTense'],
8
+ Negative: ['Verb', 'PresentTense', 'Negative'],
9
+ PastNegative: ['Verb', 'PastTense', 'Negative'],
10
+ Gerund: ['Verb', 'Gerund'],
11
+ NegativeGerund: ['Verb', 'Gerund', 'Negative'],
12
+ Polite: ['Verb', 'PresentTense', 'Polite'],
13
+ PolitePast: ['Verb', 'PastTense', 'Polite'],
14
+ PoliteNegative: ['Verb', 'PresentTense', 'Polite', 'Negative'],
15
+ PolitePastNegative: ['Verb', 'PastTense', 'Polite', 'Negative'],
16
+ PoliteVolitional: ['Verb', 'Volitional', 'Polite'],
17
+ Imperative: ['Verb', 'Imperative'],
18
+ NegativeImperative: ['Verb', 'Imperative', 'Negative'],
19
+ Volitional: ['Verb', 'Volitional'],
20
+ Potential: ['Verb', 'Potential', 'PresentTense'],
21
+ Passive: ['Verb', 'Passive', 'PresentTense'],
22
+ Causative: ['Verb', 'Causative', 'PresentTense'],
23
+ CausativePassive: ['Verb', 'Causative', 'Passive', 'PresentTense'],
24
+ Conditional: ['Verb', 'ConditionalVerb'],
25
+ Provisional: ['Verb', 'ConditionalVerb'],
26
+ Progressive: ['Verb', 'Progressive', 'PresentTense'],
27
+ ProgressivePolite: ['Verb', 'Progressive', 'PresentTense', 'Polite'],
28
+ PastProgressive: ['Verb', 'Progressive', 'PastTense'],
29
+ Desire: ['Verb', 'Desire'],
30
+ Representative: ['Verb', 'Representative'],
31
+ Presumptive: ['Verb', 'Presumptive'],
32
+ Continuative: ['Verb', 'Continuative'],
33
+ Stem: ['Verb', 'VerbStem'],
34
+ }
35
+
36
+ const adjForms = {
37
+ Infinitive: ['Adjective', 'IAdjective', 'PresentTense'],
38
+ PresentTense: ['Adjective', 'IAdjective', 'PresentTense'],
39
+ Negative: ['Adjective', 'IAdjective', 'PresentTense', 'Negative'],
40
+ PastTense: ['Adjective', 'IAdjective', 'PastTense'],
41
+ PastNegative: ['Adjective', 'IAdjective', 'PastTense', 'Negative'],
42
+ Gerund: ['Adjective', 'IAdjective', 'Gerund'],
43
+ Adverb: ['Adverb'],
44
+ Provisional: ['Adjective', 'IAdjective', 'ConditionalVerb'],
45
+ Presumptive: ['Adjective', 'IAdjective', 'Presumptive'],
46
+ Superlative: ['Verb', 'PresentTense'],
47
+ Impression: ['Adjective', 'Presumptive'],
48
+ Nominal: ['Noun'],
49
+ }
50
+
51
+ // 食べたい inflects like an い-adjective - 食べたくない, 食べたかった
52
+ const desireForms = {
53
+ Infinitive: ['Verb', 'Desire', 'PresentTense'],
54
+ PresentTense: ['Verb', 'Desire', 'PresentTense'],
55
+ Negative: ['Verb', 'Desire', 'PresentTense', 'Negative'],
56
+ PastTense: ['Verb', 'Desire', 'PastTense'],
57
+ PastNegative: ['Verb', 'Desire', 'PastTense', 'Negative'],
58
+ Gerund: ['Verb', 'Desire', 'Gerund'],
59
+ Provisional: ['Verb', 'Desire', 'ConditionalVerb'],
60
+ }
61
+
62
+ // the passive/potential/causative stems are themselves ichidan verbs, so
63
+ // 褒められる has a past (褒められた) and a polite past (褒められました) of its own
64
+ const derivedForms = {
65
+ PastTense: ['PastTense'],
66
+ Negative: ['PresentTense', 'Negative'],
67
+ PastNegative: ['PastTense', 'Negative'],
68
+ Gerund: ['Gerund'],
69
+ Polite: ['PresentTense', 'Polite'],
70
+ PolitePast: ['PastTense', 'Polite'],
71
+ PoliteNegative: ['PresentTense', 'Polite', 'Negative'],
72
+ PolitePastNegative: ['PastTense', 'Polite', 'Negative'],
73
+ Progressive: ['Progressive', 'PresentTense'],
74
+ Conditional: ['ConditionalVerb'],
75
+ Provisional: ['ConditionalVerb'],
76
+ }
77
+
78
+ // な-adjectives inflect *with the copula* - 静か + でした - and the copula is
79
+ // its own token. only 静かに is a word in its own right.
80
+ const naAdjForms = {
81
+ Infinitive: ['Adjective', 'NaAdjective'],
82
+ Adverb: ['Adverb'],
83
+ }
84
+
85
+ export { verbForms, adjForms, naAdjForms, desireForms, derivedForms }