@cognica-io/uqa-darwin-arm64 0.3.0 → 0.3.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
|
@@ -0,0 +1,360 @@
|
|
|
1
|
+
{
|
|
2
|
+
"format": "uqa-kuromoji-neutral",
|
|
3
|
+
"format_version": 1,
|
|
4
|
+
"byte_order": "big",
|
|
5
|
+
"exporter_sha256": "81068689f85ae983fa817af82922f464e70a7f7c41cd9cf8e096beb7499f4c7d",
|
|
6
|
+
"reference": {
|
|
7
|
+
"lucene_version": "10.5.1",
|
|
8
|
+
"lucene_commit": "64ce863a2bea79c69c19c4d56268c26710ff0ff9",
|
|
9
|
+
"docker_image": "eclipse-temurin@sha256:3b0a98dfbdf1067c20a7854cec159551777d2ee1381bc76cd4bd0719f543b148",
|
|
10
|
+
"runtime": {
|
|
11
|
+
"java_version": "21.0.10",
|
|
12
|
+
"java_runtime_version": "21.0.10+7-LTS",
|
|
13
|
+
"java_vendor": "Eclipse Adoptium"
|
|
14
|
+
},
|
|
15
|
+
"jars": [
|
|
16
|
+
{
|
|
17
|
+
"artifact": "lucene-core",
|
|
18
|
+
"url": "https://repo.maven.apache.org/maven2/org/apache/lucene/lucene-core/10.5.1/lucene-core-10.5.1.jar",
|
|
19
|
+
"bytes": 4744152,
|
|
20
|
+
"sha256": "2b4912cc792f462e8e7b350f7c958f538ee7ec42da4cf4b65902fb9d546bba53"
|
|
21
|
+
},
|
|
22
|
+
{
|
|
23
|
+
"artifact": "lucene-analysis-common",
|
|
24
|
+
"url": "https://repo.maven.apache.org/maven2/org/apache/lucene/lucene-analysis-common/10.5.1/lucene-analysis-common-10.5.1.jar",
|
|
25
|
+
"bytes": 1757025,
|
|
26
|
+
"sha256": "f710afd91a820987a48f3a215e076715b9237b324cf5458281109ef0d2a17b73"
|
|
27
|
+
},
|
|
28
|
+
{
|
|
29
|
+
"artifact": "lucene-analysis-kuromoji",
|
|
30
|
+
"url": "https://repo.maven.apache.org/maven2/org/apache/lucene/lucene-analysis-kuromoji/10.5.1/lucene-analysis-kuromoji-10.5.1.jar",
|
|
31
|
+
"bytes": 4720127,
|
|
32
|
+
"sha256": "bbe9a6d52d89d88c68e1c6db8cc2c829774ae5fcee8be1a1fdc3c6bec2ab9503"
|
|
33
|
+
}
|
|
34
|
+
],
|
|
35
|
+
"dictionary_source": {
|
|
36
|
+
"name": "mecab-ipadic-2.7.0-20070801",
|
|
37
|
+
"url": "https://s3.amazonaws.com/lucene-testdata/mecab/mecab-ipadic-2.7.0-20070801.tar.gz",
|
|
38
|
+
"bytes": 12208105,
|
|
39
|
+
"sha256": "b62f527d881c504576baed9c6ef6561554658b175ce6ae0096a60307e49e3523",
|
|
40
|
+
"license_file": "COPYING",
|
|
41
|
+
"license": "LicenseRef-IPADIC",
|
|
42
|
+
"lucene_normalize_entries": false
|
|
43
|
+
},
|
|
44
|
+
"dictionary_patch": {
|
|
45
|
+
"url": "https://raw.githubusercontent.com/apache/lucene/64ce863a2bea79c69c19c4d56268c26710ff0ff9/lucene/analysis/kuromoji/src/tools/patches/Noun.proper.csv.patch",
|
|
46
|
+
"bytes": 1198,
|
|
47
|
+
"sha256": "771970de4df33f53bade6cfaace783bffa46ee41af8958dd7bac2758afd6216b",
|
|
48
|
+
"path": "Noun.proper.csv.patch",
|
|
49
|
+
"target": "Noun.proper.csv",
|
|
50
|
+
"git_blob": "1e0c8d30fca9c4915f66dc54d4cf2017724e79b3"
|
|
51
|
+
},
|
|
52
|
+
"generation_recipe": {
|
|
53
|
+
"url": "https://raw.githubusercontent.com/apache/lucene/64ce863a2bea79c69c19c4d56268c26710ff0ff9/gradle/generation/kuromoji.gradle",
|
|
54
|
+
"bytes": 4701,
|
|
55
|
+
"sha256": "5b59ab1f6a50df18924b2e580096620154949fb34735e45e736db9be8552a26e"
|
|
56
|
+
},
|
|
57
|
+
"dictionary_resources": [
|
|
58
|
+
{
|
|
59
|
+
"path": "org/apache/lucene/analysis/ja/dict/CharacterDefinition.dat",
|
|
60
|
+
"bytes": 65568,
|
|
61
|
+
"sha256": "16f6e8ff819ef4191a9396d994e0f42e7fa20f974966dace22cf9a0e33790af4"
|
|
62
|
+
},
|
|
63
|
+
{
|
|
64
|
+
"path": "org/apache/lucene/analysis/ja/dict/ConnectionCosts.dat",
|
|
65
|
+
"bytes": 2624540,
|
|
66
|
+
"sha256": "eb936c575c80048542165518761689f085d829a1526683aebfb18c46ccfcef8c"
|
|
67
|
+
},
|
|
68
|
+
{
|
|
69
|
+
"path": "org/apache/lucene/analysis/ja/dict/TokenInfoDictionary$buffer.dat",
|
|
70
|
+
"bytes": 4337224,
|
|
71
|
+
"sha256": "9b68b6e21e7a10f02d07e9157855a31017edc5c1e6a5bd61f4afb3585d51bdb9"
|
|
72
|
+
},
|
|
73
|
+
{
|
|
74
|
+
"path": "org/apache/lucene/analysis/ja/dict/TokenInfoDictionary$fst.dat",
|
|
75
|
+
"bytes": 1686421,
|
|
76
|
+
"sha256": "9ef5cfddc264f755703cd4cd925383294ead43f1bc7aa7bf06f20d1e6d91fbad"
|
|
77
|
+
},
|
|
78
|
+
{
|
|
79
|
+
"path": "org/apache/lucene/analysis/ja/dict/TokenInfoDictionary$posDict.dat",
|
|
80
|
+
"bytes": 54870,
|
|
81
|
+
"sha256": "d1296e3cf1a9d7ba6fe696dba91a96864be22ad4e7cb65bfd4218aa422b0d685"
|
|
82
|
+
},
|
|
83
|
+
{
|
|
84
|
+
"path": "org/apache/lucene/analysis/ja/dict/TokenInfoDictionary$targetMap.dat",
|
|
85
|
+
"bytes": 392166,
|
|
86
|
+
"sha256": "206221f9bcea9b0a45f72492355e4822318cb42211734538840bf0a619c2ce79"
|
|
87
|
+
},
|
|
88
|
+
{
|
|
89
|
+
"path": "org/apache/lucene/analysis/ja/dict/UnknownDictionary$buffer.dat",
|
|
90
|
+
"bytes": 311,
|
|
91
|
+
"sha256": "ffe47f1075e9fdeb3b38ba87b737d876558cdea90d2eecb5453e587c2689cbde"
|
|
92
|
+
},
|
|
93
|
+
{
|
|
94
|
+
"path": "org/apache/lucene/analysis/ja/dict/UnknownDictionary$posDict.dat",
|
|
95
|
+
"bytes": 4111,
|
|
96
|
+
"sha256": "8f7aa665f5333b6ac38a656cf4995b516f023e6eaaccb46bcd7ab259c43f5542"
|
|
97
|
+
},
|
|
98
|
+
{
|
|
99
|
+
"path": "org/apache/lucene/analysis/ja/dict/UnknownDictionary$targetMap.dat",
|
|
100
|
+
"bytes": 69,
|
|
101
|
+
"sha256": "f30d68324104cefd6a98337e315c0694f40171524aae00182df6d00a5effc661"
|
|
102
|
+
}
|
|
103
|
+
],
|
|
104
|
+
"analysis_resources": [
|
|
105
|
+
{
|
|
106
|
+
"path": "org/apache/lucene/analysis/ja/completion/romaji_map.txt",
|
|
107
|
+
"bytes": 4885,
|
|
108
|
+
"sha256": "8f63fe9069055e8d59743aa8b166d10acaaa7b4271748ceb02fd59a2b6a4904a"
|
|
109
|
+
},
|
|
110
|
+
{
|
|
111
|
+
"path": "org/apache/lucene/analysis/ja/stoptags.txt",
|
|
112
|
+
"bytes": 16724,
|
|
113
|
+
"sha256": "925c597ef52be9469f30d397b0bd3bc30b9c5b064eb0d02dd0adf8494d1d5437"
|
|
114
|
+
},
|
|
115
|
+
{
|
|
116
|
+
"path": "org/apache/lucene/analysis/ja/stopwords.txt",
|
|
117
|
+
"bytes": 1810,
|
|
118
|
+
"sha256": "05ec2d60bc9a59578d1a233f61620d407c197b1ce659d8eed50bd476f2dea476"
|
|
119
|
+
}
|
|
120
|
+
]
|
|
121
|
+
},
|
|
122
|
+
"model": {
|
|
123
|
+
"runtime": {
|
|
124
|
+
"java_version": "21.0.10",
|
|
125
|
+
"java_runtime_version": "21.0.10+7-LTS",
|
|
126
|
+
"java_vendor": "Eclipse Adoptium"
|
|
127
|
+
},
|
|
128
|
+
"surface_count": 325872,
|
|
129
|
+
"word_count": 392127,
|
|
130
|
+
"unknown_class_count": 12,
|
|
131
|
+
"unknown_word_count": 41,
|
|
132
|
+
"matrix_forward": 1316,
|
|
133
|
+
"matrix_backward": 1316,
|
|
134
|
+
"unicode_count": 1114112,
|
|
135
|
+
"character_count": 65536,
|
|
136
|
+
"morphology_fields": [
|
|
137
|
+
"part_of_speech",
|
|
138
|
+
"base_form",
|
|
139
|
+
"reading",
|
|
140
|
+
"pronunciation",
|
|
141
|
+
"inflection_type",
|
|
142
|
+
"inflection_form"
|
|
143
|
+
],
|
|
144
|
+
"stop_word_count": 109,
|
|
145
|
+
"stop_tag_count": 27,
|
|
146
|
+
"completion_mapping_count": 329,
|
|
147
|
+
"character_classes": [
|
|
148
|
+
"NGRAM",
|
|
149
|
+
"DEFAULT",
|
|
150
|
+
"SPACE",
|
|
151
|
+
"SYMBOL",
|
|
152
|
+
"NUMERIC",
|
|
153
|
+
"ALPHA",
|
|
154
|
+
"CYRILLIC",
|
|
155
|
+
"GREEK",
|
|
156
|
+
"HIRAGANA",
|
|
157
|
+
"KATAKANA",
|
|
158
|
+
"KANJI",
|
|
159
|
+
"KANJINUMERIC"
|
|
160
|
+
],
|
|
161
|
+
"unicode_scripts": [
|
|
162
|
+
"COMMON",
|
|
163
|
+
"LATIN",
|
|
164
|
+
"GREEK",
|
|
165
|
+
"CYRILLIC",
|
|
166
|
+
"ARMENIAN",
|
|
167
|
+
"HEBREW",
|
|
168
|
+
"ARABIC",
|
|
169
|
+
"SYRIAC",
|
|
170
|
+
"THAANA",
|
|
171
|
+
"DEVANAGARI",
|
|
172
|
+
"BENGALI",
|
|
173
|
+
"GURMUKHI",
|
|
174
|
+
"GUJARATI",
|
|
175
|
+
"ORIYA",
|
|
176
|
+
"TAMIL",
|
|
177
|
+
"TELUGU",
|
|
178
|
+
"KANNADA",
|
|
179
|
+
"MALAYALAM",
|
|
180
|
+
"SINHALA",
|
|
181
|
+
"THAI",
|
|
182
|
+
"LAO",
|
|
183
|
+
"TIBETAN",
|
|
184
|
+
"MYANMAR",
|
|
185
|
+
"GEORGIAN",
|
|
186
|
+
"HANGUL",
|
|
187
|
+
"ETHIOPIC",
|
|
188
|
+
"CHEROKEE",
|
|
189
|
+
"CANADIAN_ABORIGINAL",
|
|
190
|
+
"OGHAM",
|
|
191
|
+
"RUNIC",
|
|
192
|
+
"KHMER",
|
|
193
|
+
"MONGOLIAN",
|
|
194
|
+
"HIRAGANA",
|
|
195
|
+
"KATAKANA",
|
|
196
|
+
"BOPOMOFO",
|
|
197
|
+
"HAN",
|
|
198
|
+
"YI",
|
|
199
|
+
"OLD_ITALIC",
|
|
200
|
+
"GOTHIC",
|
|
201
|
+
"DESERET",
|
|
202
|
+
"INHERITED",
|
|
203
|
+
"TAGALOG",
|
|
204
|
+
"HANUNOO",
|
|
205
|
+
"BUHID",
|
|
206
|
+
"TAGBANWA",
|
|
207
|
+
"LIMBU",
|
|
208
|
+
"TAI_LE",
|
|
209
|
+
"LINEAR_B",
|
|
210
|
+
"UGARITIC",
|
|
211
|
+
"SHAVIAN",
|
|
212
|
+
"OSMANYA",
|
|
213
|
+
"CYPRIOT",
|
|
214
|
+
"BRAILLE",
|
|
215
|
+
"BUGINESE",
|
|
216
|
+
"COPTIC",
|
|
217
|
+
"NEW_TAI_LUE",
|
|
218
|
+
"GLAGOLITIC",
|
|
219
|
+
"TIFINAGH",
|
|
220
|
+
"SYLOTI_NAGRI",
|
|
221
|
+
"OLD_PERSIAN",
|
|
222
|
+
"KHAROSHTHI",
|
|
223
|
+
"BALINESE",
|
|
224
|
+
"CUNEIFORM",
|
|
225
|
+
"PHOENICIAN",
|
|
226
|
+
"PHAGS_PA",
|
|
227
|
+
"NKO",
|
|
228
|
+
"SUNDANESE",
|
|
229
|
+
"BATAK",
|
|
230
|
+
"LEPCHA",
|
|
231
|
+
"OL_CHIKI",
|
|
232
|
+
"VAI",
|
|
233
|
+
"SAURASHTRA",
|
|
234
|
+
"KAYAH_LI",
|
|
235
|
+
"REJANG",
|
|
236
|
+
"LYCIAN",
|
|
237
|
+
"CARIAN",
|
|
238
|
+
"LYDIAN",
|
|
239
|
+
"CHAM",
|
|
240
|
+
"TAI_THAM",
|
|
241
|
+
"TAI_VIET",
|
|
242
|
+
"AVESTAN",
|
|
243
|
+
"EGYPTIAN_HIEROGLYPHS",
|
|
244
|
+
"SAMARITAN",
|
|
245
|
+
"MANDAIC",
|
|
246
|
+
"LISU",
|
|
247
|
+
"BAMUM",
|
|
248
|
+
"JAVANESE",
|
|
249
|
+
"MEETEI_MAYEK",
|
|
250
|
+
"IMPERIAL_ARAMAIC",
|
|
251
|
+
"OLD_SOUTH_ARABIAN",
|
|
252
|
+
"INSCRIPTIONAL_PARTHIAN",
|
|
253
|
+
"INSCRIPTIONAL_PAHLAVI",
|
|
254
|
+
"OLD_TURKIC",
|
|
255
|
+
"BRAHMI",
|
|
256
|
+
"KAITHI",
|
|
257
|
+
"MEROITIC_HIEROGLYPHS",
|
|
258
|
+
"MEROITIC_CURSIVE",
|
|
259
|
+
"SORA_SOMPENG",
|
|
260
|
+
"CHAKMA",
|
|
261
|
+
"SHARADA",
|
|
262
|
+
"TAKRI",
|
|
263
|
+
"MIAO",
|
|
264
|
+
"CAUCASIAN_ALBANIAN",
|
|
265
|
+
"BASSA_VAH",
|
|
266
|
+
"DUPLOYAN",
|
|
267
|
+
"ELBASAN",
|
|
268
|
+
"GRANTHA",
|
|
269
|
+
"PAHAWH_HMONG",
|
|
270
|
+
"KHOJKI",
|
|
271
|
+
"LINEAR_A",
|
|
272
|
+
"MAHAJANI",
|
|
273
|
+
"MANICHAEAN",
|
|
274
|
+
"MENDE_KIKAKUI",
|
|
275
|
+
"MODI",
|
|
276
|
+
"MRO",
|
|
277
|
+
"OLD_NORTH_ARABIAN",
|
|
278
|
+
"NABATAEAN",
|
|
279
|
+
"PALMYRENE",
|
|
280
|
+
"PAU_CIN_HAU",
|
|
281
|
+
"OLD_PERMIC",
|
|
282
|
+
"PSALTER_PAHLAVI",
|
|
283
|
+
"SIDDHAM",
|
|
284
|
+
"KHUDAWADI",
|
|
285
|
+
"TIRHUTA",
|
|
286
|
+
"WARANG_CITI",
|
|
287
|
+
"AHOM",
|
|
288
|
+
"ANATOLIAN_HIEROGLYPHS",
|
|
289
|
+
"HATRAN",
|
|
290
|
+
"MULTANI",
|
|
291
|
+
"OLD_HUNGARIAN",
|
|
292
|
+
"SIGNWRITING",
|
|
293
|
+
"ADLAM",
|
|
294
|
+
"BHAIKSUKI",
|
|
295
|
+
"MARCHEN",
|
|
296
|
+
"NEWA",
|
|
297
|
+
"OSAGE",
|
|
298
|
+
"TANGUT",
|
|
299
|
+
"MASARAM_GONDI",
|
|
300
|
+
"NUSHU",
|
|
301
|
+
"SOYOMBO",
|
|
302
|
+
"ZANABAZAR_SQUARE",
|
|
303
|
+
"HANIFI_ROHINGYA",
|
|
304
|
+
"OLD_SOGDIAN",
|
|
305
|
+
"SOGDIAN",
|
|
306
|
+
"DOGRA",
|
|
307
|
+
"GUNJALA_GONDI",
|
|
308
|
+
"MAKASAR",
|
|
309
|
+
"MEDEFAIDRIN",
|
|
310
|
+
"ELYMAIC",
|
|
311
|
+
"NANDINAGARI",
|
|
312
|
+
"NYIAKENG_PUACHUE_HMONG",
|
|
313
|
+
"WANCHO",
|
|
314
|
+
"YEZIDI",
|
|
315
|
+
"CHORASMIAN",
|
|
316
|
+
"DIVES_AKURU",
|
|
317
|
+
"KHITAN_SMALL_SCRIPT",
|
|
318
|
+
"VITHKUQI",
|
|
319
|
+
"OLD_UYGHUR",
|
|
320
|
+
"CYPRO_MINOAN",
|
|
321
|
+
"TANGSA",
|
|
322
|
+
"TOTO",
|
|
323
|
+
"KAWI",
|
|
324
|
+
"NAG_MUNDARI",
|
|
325
|
+
"UNKNOWN"
|
|
326
|
+
]
|
|
327
|
+
},
|
|
328
|
+
"files": [
|
|
329
|
+
{
|
|
330
|
+
"path": "lexicon.bin",
|
|
331
|
+
"bytes": 39666432,
|
|
332
|
+
"sha256": "c120aba24719a287170165dc5ca23aaf6d5179212570d17e8f86eb38a0317eee"
|
|
333
|
+
},
|
|
334
|
+
{
|
|
335
|
+
"path": "unknown.bin",
|
|
336
|
+
"bytes": 2680,
|
|
337
|
+
"sha256": "a75999d86aa3438a844192146a2196c9ed05c10a2c9b74404738bc074d91ab6d"
|
|
338
|
+
},
|
|
339
|
+
{
|
|
340
|
+
"path": "connection_costs.bin",
|
|
341
|
+
"bytes": 3463728,
|
|
342
|
+
"sha256": "a92e7c73b634b26e52e4ba816f2bb202d83b4b83dc879b92fa63b8c2794382d8"
|
|
343
|
+
},
|
|
344
|
+
{
|
|
345
|
+
"path": "characters.bin",
|
|
346
|
+
"bytes": 131100,
|
|
347
|
+
"sha256": "10d1c69f2d8c8b7a3209a97b1ea4e31400e9e2a1851e784c043ca62ccfeed444"
|
|
348
|
+
},
|
|
349
|
+
{
|
|
350
|
+
"path": "unicode.bin",
|
|
351
|
+
"bytes": 8912908,
|
|
352
|
+
"sha256": "0c677a9baf4a4aeb2129ee88d6e7288caebad9f125016975407a9e485dc580b3"
|
|
353
|
+
},
|
|
354
|
+
{
|
|
355
|
+
"path": "analysis.bin",
|
|
356
|
+
"bytes": 10346,
|
|
357
|
+
"sha256": "45e2371d90555488da15336e2b3722b80b087757cb9da3a1ebd4f8fb0096f009"
|
|
358
|
+
}
|
|
359
|
+
]
|
|
360
|
+
}
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
{
|
|
2
|
+
"format": "uqa-kuromoji-bundled-resource",
|
|
3
|
+
"format_version": 1,
|
|
4
|
+
"bundle_format_version": 1,
|
|
5
|
+
"dictionary_id": "dd2691ffee7f5a0d2c29c1a6e66dee249116767208a8cd8e783a6aa479100c8f",
|
|
6
|
+
"packer": "uqa-analysis pack_kuromoji; bundle format 1",
|
|
7
|
+
"compression": "miniz_oxide 0.8.9; zlib level 9",
|
|
8
|
+
"files": [
|
|
9
|
+
{
|
|
10
|
+
"path": "data/kuromoji.uqak",
|
|
11
|
+
"bytes": 6654306,
|
|
12
|
+
"sha256": "bd3dd53f609006e72d0aa6e94ec6067400ac2be328ca0c39263ab36c3197aa97"
|
|
13
|
+
},
|
|
14
|
+
{
|
|
15
|
+
"path": "data/model_manifest.json",
|
|
16
|
+
"bytes": 9984,
|
|
17
|
+
"sha256": "c3fb9ad9f0abf62f13ee508224a70fcdcd367768b916963f11b8861435078609"
|
|
18
|
+
},
|
|
19
|
+
{
|
|
20
|
+
"path": "THIRD-PARTY/JDK-UNICODE.md",
|
|
21
|
+
"bytes": 9078,
|
|
22
|
+
"sha256": "6f72f10d166b2c2e8a395e03e734c5afc852b59aeca73ced124f6b9c96268d53"
|
|
23
|
+
},
|
|
24
|
+
{
|
|
25
|
+
"path": "THIRD-PARTY/LUCENE-LICENSE.txt",
|
|
26
|
+
"bytes": 26085,
|
|
27
|
+
"sha256": "a2521407b3209df7dcebfc12cd6d732b24bfa2fe44982ef613e269666482521d"
|
|
28
|
+
},
|
|
29
|
+
{
|
|
30
|
+
"path": "THIRD-PARTY/LUCENE-NOTICE.txt",
|
|
31
|
+
"bytes": 9782,
|
|
32
|
+
"sha256": "d3b82734d5e181509c4b6b832f5d3a90c8c6a34fc195272fc8957ccb9e8e20a8"
|
|
33
|
+
},
|
|
34
|
+
{
|
|
35
|
+
"path": "THIRD-PARTY/MECAB-IPADIC-COPYING.txt",
|
|
36
|
+
"bytes": 3795,
|
|
37
|
+
"sha256": "fca02d9adb601d101eccdf3131119abfff2d9c825ef5c759d7a89ebbaa972099"
|
|
38
|
+
}
|
|
39
|
+
]
|
|
40
|
+
}
|
|
@@ -4,4 +4,20 @@ The native Nori user-dictionary parser, rolling Viterbi search, token emission,
|
|
|
4
4
|
|
|
5
5
|
The reference files are `lucene/analysis/common/src/java/org/apache/lucene/analysis/morph/Viterbi.java` and the Nori `Viterbi.java`, `KoreanTokenizer.java`, `Token.java`, `DecompoundToken.java`, `dict/UserDictionary.java`, `dict/UserMorphData.java`, `KoreanAnalyzer.java`, `KoreanPartOfSpeechStopFilter.java`, `KoreanReadingFormFilter.java`, and `KoreanNumberFilter.java` under `lucene/analysis/nori/src/java/org/apache/lucene/analysis/ko/`. The shared filter behavior follows `FilteringTokenFilter.java`, `LowerCaseFilter.java`, and `CharacterUtils.java` under `lucene/core/src/java/org/apache/lucene/analysis/`. See the [pinned upstream tree](https://github.com/apache/lucene/tree/64ce863a2bea79c69c19c4d56268c26710ff0ff9/lucene/analysis).
|
|
6
6
|
|
|
7
|
-
Cognica's modifications translate the algorithms to Rust, use the validated portable dictionary representation, retain raw UTF-16 token and morpheme units, make compiled models immutable and shareable, and add explicit resource limits and cancellation. Number arithmetic uses an interruptible exact decimal coefficient and scale; materialized filter streams retain shared terminal-attribute effects.
|
|
7
|
+
Cognica's modifications translate the algorithms to Rust, use the validated portable dictionary representation, retain raw UTF-16 token and morpheme units, make compiled models immutable and shareable, and add explicit resource limits and cancellation. Number arithmetic uses an interruptible exact decimal coefficient and scale in `src/morphology/decimal.rs`, with a private context contract preserving language-owned polling, budgets and typed diagnostics; numeric-prefix traversal and lookahead composition now live in `src/morphology/number`, with `src/nori/number/policy.rs` retaining Korean numeral grammar, attribute accounting, normalization and limits; materialized filter streams retain shared terminal-attribute effects. Rolling position and candidate ownership is shared in `src/morphology/lattice.rs`; `src/morphology/viterbi.rs` shares ordered forward traversal, frontier commits, forced backtraces and EOS choice. Language wrappers preserve their original limits, diagnostics, matching and emission. Shared term/span operations, reserved stream ownership, polling and Java simple lowercase live in `src/morphology/filter.rs` and `src/morphology/filter/`; `src/token/filter.rs` supplies common-token source projection. The Korean kernel in `src/nori/filters/stream.rs` and the language adapter in `src/token/korean.rs` retain optional Korean morphology, profile selection, diagnostics and native mutation behavior. The pinned Docker reference drivers and native differential tests record observable ordering, morphology, graph, offset, and stream-end behavior. The dictionary data and its additional attribution ship separately in `uqa-nori-data`.
|
|
8
|
+
|
|
9
|
+
The Japanese dictionary model in `src/kuromoji/` preserves the public `JaMorphData`, `CharacterDefinition`, `ConnectionCosts`, `JapaneseAnalyzer` default stop sets and `KatakanaRomanizer` mappings from `lucene/analysis/kuromoji/src/java/org/apache/lucene/analysis/ja/` at the same pinned commit. The portable bundle and shared codecs are Cognica implementations; source regeneration and complete exported-model comparisons run against the original Lucene jars in Docker. Exact IPADIC and Lucene data notices are packaged separately in `uqa-kuromoji-data`.
|
|
10
|
+
|
|
11
|
+
The Japanese user-rule compilation and lookup in `src/kuromoji/user_dictionary.rs` and `src/kuromoji/user_dictionary/` port the pinned Japanese `dict/UserDictionary.java`, `dict/UserMorphData.java` and common `org/apache/lucene/analysis/util/CSVUtil.java`. The Rust implementation shares the existing lexicon and preparation limits, retains exact source/model identity, stores phrase POS once rather than duplicating it for every segment, and returns checked errors for absent user morphology fields. Docker differential fixtures preserve the reference CSV quirks, duplicate rejection, UTF-16 segmentation and ordered overlapping lookup behavior.
|
|
12
|
+
|
|
13
|
+
The native Japanese tokenizer in `src/kuromoji/tokenizer.rs` and `src/kuromoji/tokenizer/` ports the pinned `JapaneseTokenizer.java`, Japanese `ViterbiNBest.java` and common `org/apache/lucene/analysis/morph/ViterbiNBest.java`, including matching, resegmentation, alternative costs, stable span deduplication, graph fixups and example probes. Both languages share the ordered rolling search and lattice owners; input encoding/counting is shared in `src/allocation/input.rs`. Japanese policies preserve signed wrapping costs, candidate/alternative ordering, compound positions, punctuation, unknown unigrams, user segments and six independent morphology fields. Rust adds checked model identity, per-call input/lattice/output/resegmentation/alternative bounds, shared example preparation limits, cancellation and retained byte allowances. Pending records defer dictionary attributes until actual emission, matching lazy probe behavior. N-best graphs remain Japanese-owned and are absent from non-positive-cost execution and Nori. Cognica's common native-token bridge in `src/token/native.rs` shares corrected source spans, graph validation and allocation transfer between the Korean and Japanese adapters. Common tokens preserve separate typed optional morphology and raw term units through generic filters. The default Japanese analyzer and its base-form, POS-stop, stopword, Katakana-stem and simple-lowercase stages in `src/kuromoji/analyzer.rs` and `src/kuromoji/filters/` port `JapaneseAnalyzer.java`, `JapaneseBaseFormFilter.java`, `JapanesePartOfSpeechStopFilter.java`, `JapaneseKatakanaStemFilter.java`, common `StopFilter.java` and `LowerCaseFilter.java`. Rust adds bounded, cancellable prepared lookup storage, independent normalization, retained source mapping and private lazy user-field failures; the optional small-kana filters in `src/kuromoji/filters/kana.rs` port `JapaneseHiraganaUppercaseFilter.java` and `JapaneseKatakanaUppercaseFilter.java`, using cancellable reserved term mutation while preserving their exact mappings and contraction behavior. Reading conversion in `src/kuromoji/filters/reading.rs` and `src/kuromoji/filters/reading/` ports `JapaneseReadingFormFilter.java` and the modified-Hepburn method of `dict/ToStringUtil.java`. Rust uses bounded borrowed counting/emission, static sorted syllable replacements and three-unit lookahead, preserving lazy reading failures and raw units; completion romanization remains a separate algorithm. Japanese number normalization and composition in `src/kuromoji/number.rs` and `src/kuromoji/number/policy.rs` port `JapaneseNumberFilter.java`, using the shared exact decimals and prefix/lookahead mechanics. The pinned Japanese and Korean composition methods are identical; Japanese policies retain its numeral table, six independent attributes and typed bounds. Rust preserves prefix fallback, aborted-prefix replay and exhaustion attributes while reserving all coefficients, saved tokens and output buffers, polling throughout, and correcting common-token spans against the original source. The Japanese horizontal iteration stage in `src/char_filter/iteration.rs` ports `JapaneseIterationMarkCharFilter.java`, including the original-source span boundaries, full-stop/surrogate barriers and exact voiced/unvoiced table behavior. Rust uses borrowed UTF-8 iterators in place of the rolling input copy and the common reserved edit/source-map builder for corrected coordinates; lookahead, source selection, emission and coordinate work remain cancellable. Completion romanization in `src/kuromoji/completion.rs` ports `completion/KatakanaRomanizer.java`. Rust reuses the existing lexical-rank owner for longest-prefix matching, counts the complete ordered product before allocation, and emits candidates by mixed-radix traversal with retained buffers, limits and cancellation instead of repeatedly copying partial strings. The completion stream in `src/kuromoji/completion/stream.rs` and completion constructor in `src/kuromoji/analyzer.rs` port `JapaneseCompletionFilter.java`, `completion/CharSequenceUtils.java` and `JapaneseCompletionAnalyzer.java`. Rust caches pending predicates, retains surface/reading buffers, preserves the literal-null reading-builder behavior, clears all token metadata through owning representation adapters, and keeps completion width-only normalization explicit. Native generated tokens have no dictionary origin and their common conversion has no morphology. Upstream terminal attributes survive when no pending token emission clears them. Japanese analyzer catalog integration remains in development.
|
|
14
|
+
|
|
15
|
+
Cognica's scalar normalization owner in `src/normalization/text.rs` shares reserved input conversion, restricted width filtering and scalar output transfer between the native Korean and Japanese adapters. Their existing pinned simple-lowercase mappings and language diagnostics remain in their owning filters. Explicit compiled normalization configurations retain typed immutable profiles and exact descriptor identities; omission preserves existing Korean inference and canonical serialization.
|
|
16
|
+
|
|
17
|
+
The common tokenizer configuration in `src/kuromoji/config.rs` and `src/kuromoji/pipeline.rs` follows the defaults and signed N-best cost combination in the pinned `JapaneseTokenizerFactory.java`. It reuses the native tokenizer, user compiler and example preparation without introducing a filesystem resolver. Rust freezes exact dictionary/user identities and the effective cost in immutable descriptors, restores without repeating probes, and retains lazy attribute errors until filter access or public stream completion.
|
|
18
|
+
|
|
19
|
+
The common simple-lowercase profile configuration in `src/token_filter/unicode.rs` and the prepared stage adapter in `src/kuromoji/pipeline/stages.rs` are Cognica implementations. They retain exact typed dictionary profiles and reuse the existing language-owned filter algorithms without changing the pinned mappings or adding a separate lowercase implementation. Legacy Nori string configuration preserves its original wire identity.
|
|
20
|
+
|
|
21
|
+
Common Japanese base-form, Katakana-stem, small-kana, reading and number filter configurations reuse the same native/common kernels with an optional prepared model. Cognica's adapters omit dictionary ownership for attribute/term-only stages, preserve required defaults/Unicode/completion dependencies, and retain the existing native public signatures, limits, source spans, cancellation and allocation transfer.
|
|
22
|
+
|
|
23
|
+
Common POS/word-stop/completion configurations and the no-I/O constructors in `src/kuromoji/builtins.rs` reuse the native Japanese filters and preserve the pinned `JapaneseAnalyzer.java` and `JapaneseCompletionAnalyzer.java` defaults and stage order, including case-insensitive default stopwords and distinct normalization. Cognica's configuration adapter freezes original-case canonical stop sets, rejects unused dictionary inputs, releases expansion-only models and retains exact Unicode/completion profiles for restoration. No additional language algorithm or upstream data is introduced.
|
|
@@ -0,0 +1,71 @@
|
|
|
1
|
+
Copyright 2000, 2001, 2002, 2003 Nara Institute of Science
|
|
2
|
+
and Technology. All Rights Reserved.
|
|
3
|
+
|
|
4
|
+
Use, reproduction, and distribution of this software is permitted.
|
|
5
|
+
Any copy of this software, whether in its original form or modified,
|
|
6
|
+
must include both the above copyright notice and the following
|
|
7
|
+
paragraphs.
|
|
8
|
+
|
|
9
|
+
Nara Institute of Science and Technology (NAIST),
|
|
10
|
+
the copyright holders, disclaims all warranties with regard to this
|
|
11
|
+
software, including all implied warranties of merchantability and
|
|
12
|
+
fitness, in no event shall NAIST be liable for
|
|
13
|
+
any special, indirect or consequential damages or any damages
|
|
14
|
+
whatsoever resulting from loss of use, data or profits, whether in an
|
|
15
|
+
action of contract, negligence or other tortuous action, arising out
|
|
16
|
+
of or in connection with the use or performance of this software.
|
|
17
|
+
|
|
18
|
+
A large portion of the dictionary entries
|
|
19
|
+
originate from ICOT Free Software. The following conditions for ICOT
|
|
20
|
+
Free Software applies to the current dictionary as well.
|
|
21
|
+
|
|
22
|
+
Each User may also freely distribute the Program, whether in its
|
|
23
|
+
original form or modified, to any third party or parties, PROVIDED
|
|
24
|
+
that the provisions of Section 3 ("NO WARRANTY") will ALWAYS appear
|
|
25
|
+
on, or be attached to, the Program, which is distributed substantially
|
|
26
|
+
in the same form as set out herein and that such intended
|
|
27
|
+
distribution, if actually made, will neither violate or otherwise
|
|
28
|
+
contravene any of the laws and regulations of the countries having
|
|
29
|
+
jurisdiction over the User or the intended distribution itself.
|
|
30
|
+
|
|
31
|
+
NO WARRANTY
|
|
32
|
+
|
|
33
|
+
The program was produced on an experimental basis in the course of the
|
|
34
|
+
research and development conducted during the project and is provided
|
|
35
|
+
to users as so produced on an experimental basis. Accordingly, the
|
|
36
|
+
program is provided without any warranty whatsoever, whether express,
|
|
37
|
+
implied, statutory or otherwise. The term "warranty" used herein
|
|
38
|
+
includes, but is not limited to, any warranty of the quality,
|
|
39
|
+
performance, merchantability and fitness for a particular purpose of
|
|
40
|
+
the program and the nonexistence of any infringement or violation of
|
|
41
|
+
any right of any third party.
|
|
42
|
+
|
|
43
|
+
Each user of the program will agree and understand, and be deemed to
|
|
44
|
+
have agreed and understood, that there is no warranty whatsoever for
|
|
45
|
+
the program and, accordingly, the entire risk arising from or
|
|
46
|
+
otherwise connected with the program is assumed by the user.
|
|
47
|
+
|
|
48
|
+
Therefore, neither ICOT, the copyright holder, or any other
|
|
49
|
+
organization that participated in or was otherwise related to the
|
|
50
|
+
development of the program and their respective officials, directors,
|
|
51
|
+
officers and other employees shall be held liable for any and all
|
|
52
|
+
damages, including, without limitation, general, special, incidental
|
|
53
|
+
and consequential damages, arising out of or otherwise in connection
|
|
54
|
+
with the use or inability to use the program or any product, material
|
|
55
|
+
or result produced or otherwise obtained by using the program,
|
|
56
|
+
regardless of whether they have been advised of, or otherwise had
|
|
57
|
+
knowledge of, the possibility of such damages at any time during the
|
|
58
|
+
project or thereafter. Each user will be deemed to have agreed to the
|
|
59
|
+
foregoing by his or her commencement of use of the program. The term
|
|
60
|
+
"use" as used herein includes, but is not limited to, the use,
|
|
61
|
+
modification, copying and distribution of the program and the
|
|
62
|
+
production of secondary products from the program.
|
|
63
|
+
|
|
64
|
+
In the case where the program, whether in its original form or
|
|
65
|
+
modified, was distributed or delivered to or received by a user from
|
|
66
|
+
any person, organization or entity other than ICOT, unless it makes or
|
|
67
|
+
grants independently of ICOT any specific warranty to the user in
|
|
68
|
+
writing, such person, organization or entity, will also be exempted
|
|
69
|
+
from and not be held liable to the user for any such damages as noted
|
|
70
|
+
above as far as the program is concerned.
|
|
71
|
+
��
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@cognica-io/uqa-darwin-arm64",
|
|
3
|
-
"version": "0.3.
|
|
3
|
+
"version": "0.3.6",
|
|
4
4
|
"description": "Node.js bindings for embedded UQA and local or Cloud HTTP SQL (aarch64-apple-darwin)",
|
|
5
5
|
"license": "AGPL-3.0-only",
|
|
6
6
|
"author": "Jaepil Jeong <jaepil@cognica.io>",
|
|
@@ -22,9 +22,12 @@
|
|
|
22
22
|
"THIRD-PARTY/LUCENE-LICENSE.txt",
|
|
23
23
|
"THIRD-PARTY/LUCENE-NOTICE.txt",
|
|
24
24
|
"THIRD-PARTY/MECAB-COPYING.txt",
|
|
25
|
-
"THIRD-PARTY/LUCENE-SOURCE.md",
|
|
26
25
|
"THIRD-PARTY/NORI-RESOURCE-MANIFEST.json",
|
|
27
|
-
"THIRD-PARTY/NORI-MODEL-MANIFEST.json"
|
|
26
|
+
"THIRD-PARTY/NORI-MODEL-MANIFEST.json",
|
|
27
|
+
"THIRD-PARTY/MECAB-IPADIC-COPYING.txt",
|
|
28
|
+
"THIRD-PARTY/KUROMOJI-RESOURCE-MANIFEST.json",
|
|
29
|
+
"THIRD-PARTY/KUROMOJI-MODEL-MANIFEST.json",
|
|
30
|
+
"THIRD-PARTY/LUCENE-SOURCE.md"
|
|
28
31
|
],
|
|
29
32
|
"os": [
|
|
30
33
|
"darwin"
|
package/uqa.darwin-arm64.node
CHANGED
|
Binary file
|