lindera 6.1.0 → 6.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/Cargo.toml +3 -3
- data/src/tokenizer.rs +40 -12
- data/src/util.rs +78 -9
- metadata +1 -1
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
SHA256:
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
3
|
+
metadata.gz: 529da66cbb1d0fd3372e0ab7695c72ac53c55d8a1a411b99d3db0135ac05be56
|
|
4
|
+
data.tar.gz: '020678eec7845a566b1572f2a61af447252dc0090a5d9c73f183cddf4e656f46'
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: a42a702e498f3690ba0ccc3288d09c0d3e5dcc457905d058c101d112cfe24ecd4fda6cedb0e92a675a519880f6d5383efb4b1519fa1111b0b17ecaa5778e6f68
|
|
7
|
+
data.tar.gz: 57dcbda79af341e03428c7eb963e407e22bdf78a23687e95be022cdb3b91789c9c1a3b97d01b500ff44e8f700a4ea109b5262866b699d8966ae4f2597546f7d9
|
data/Cargo.toml
CHANGED
|
@@ -13,7 +13,7 @@
|
|
|
13
13
|
edition = "2024"
|
|
14
14
|
rust-version = "1.88"
|
|
15
15
|
name = "lindera-ruby"
|
|
16
|
-
version = "6.
|
|
16
|
+
version = "6.2.0"
|
|
17
17
|
build = false
|
|
18
18
|
publish = false
|
|
19
19
|
autolib = false
|
|
@@ -58,10 +58,10 @@ crate-type = [
|
|
|
58
58
|
path = "src/lib.rs"
|
|
59
59
|
|
|
60
60
|
[dependencies.lindera]
|
|
61
|
-
version = "=6.
|
|
61
|
+
version = "=6.2.0"
|
|
62
62
|
|
|
63
63
|
[dependencies.lindera-binding-core]
|
|
64
|
-
version = "=6.
|
|
64
|
+
version = "=6.2.0"
|
|
65
65
|
|
|
66
66
|
[dependencies.magnus]
|
|
67
67
|
version = "0.8.2"
|
data/src/tokenizer.rs
CHANGED
|
@@ -9,6 +9,7 @@ use std::cell::RefCell;
|
|
|
9
9
|
use std::path::Path;
|
|
10
10
|
|
|
11
11
|
use magnus::prelude::*;
|
|
12
|
+
use magnus::scan_args::{get_kwargs, scan_args};
|
|
12
13
|
use magnus::{Error, RArray, RHash, Ruby, Value, function, method};
|
|
13
14
|
|
|
14
15
|
use lindera_binding_core::{CoreTokenizer, CoreTokenizerBuilder};
|
|
@@ -199,28 +200,55 @@ pub struct RbTokenizer {
|
|
|
199
200
|
|
|
200
201
|
/// Creates a new tokenizer with the given dictionary and mode.
|
|
201
202
|
///
|
|
203
|
+
/// Ruby signature: `Tokenizer.new(dictionary, mode = nil, user_dictionary = nil,
|
|
204
|
+
/// space_penalty: nil)`.
|
|
205
|
+
///
|
|
202
206
|
/// # Arguments
|
|
203
207
|
///
|
|
204
|
-
/// * `
|
|
205
|
-
///
|
|
206
|
-
///
|
|
208
|
+
/// * `args` - The Ruby arguments:
|
|
209
|
+
/// * `dictionary` - Dictionary to use.
|
|
210
|
+
/// * `mode` - Tokenization mode ("normal" or "decompose"). `nil` or
|
|
211
|
+
/// omitted means "normal".
|
|
212
|
+
/// * `user_dictionary` - Optional user dictionary.
|
|
213
|
+
/// * `space_penalty:` - The left-space penalty, in the forms
|
|
214
|
+
/// `TokenizerBuilder#set_space_penalty` accepts. `nil` or omitted keeps
|
|
215
|
+
/// the rules the dictionary ships.
|
|
207
216
|
///
|
|
208
217
|
/// # Returns
|
|
209
218
|
///
|
|
210
|
-
/// A new `RbTokenizer` instance
|
|
211
|
-
|
|
212
|
-
|
|
213
|
-
|
|
214
|
-
user_dictionary: Option<&RbUserDictionary>,
|
|
215
|
-
) -> Result<RbTokenizer, Error> {
|
|
219
|
+
/// A new `RbTokenizer` instance, an `ArgumentError` for a wrong number of
|
|
220
|
+
/// arguments or an unknown keyword, or the errors `set_space_penalty` and
|
|
221
|
+
/// `build` raise for an invalid space-penalty setting.
|
|
222
|
+
fn tokenizer_new(args: &[Value]) -> Result<RbTokenizer, Error> {
|
|
216
223
|
let ruby = Ruby::get().expect("Ruby runtime not initialized");
|
|
224
|
+
let args = scan_args::<
|
|
225
|
+
(&RbDictionary,),
|
|
226
|
+
(Option<Option<String>>, Option<Option<&RbUserDictionary>>),
|
|
227
|
+
(),
|
|
228
|
+
(),
|
|
229
|
+
RHash,
|
|
230
|
+
(),
|
|
231
|
+
>(args)?;
|
|
232
|
+
let (dictionary,) = args.required;
|
|
233
|
+
let (mode, user_dictionary) = args.optional;
|
|
234
|
+
let keywords =
|
|
235
|
+
get_kwargs::<_, (), (Option<Value>,), ()>(args.keywords, &[], &["space_penalty"])?;
|
|
236
|
+
let (space_penalty,) = keywords.optional;
|
|
237
|
+
|
|
238
|
+
let mode = mode.flatten();
|
|
239
|
+
let user_dictionary = user_dictionary.flatten();
|
|
217
240
|
let mode_str = mode.as_deref().unwrap_or("normal");
|
|
241
|
+
let space_penalty = match space_penalty {
|
|
242
|
+
Some(value) => rb_value_to_json(&ruby, value)?,
|
|
243
|
+
None => serde_json::Value::Null,
|
|
244
|
+
};
|
|
218
245
|
|
|
219
246
|
let dict = dictionary.inner.clone();
|
|
220
247
|
let user_dict = user_dictionary.map(|d| d.inner.clone());
|
|
221
248
|
|
|
222
|
-
let inner =
|
|
223
|
-
|
|
249
|
+
let inner =
|
|
250
|
+
CoreTokenizer::from_segmenter_with_space_penalty(mode_str, dict, user_dict, &space_penalty)
|
|
251
|
+
.map_err(|err| to_magnus_error(&ruby, err.to_string()))?;
|
|
224
252
|
|
|
225
253
|
Ok(RbTokenizer { inner })
|
|
226
254
|
}
|
|
@@ -361,7 +389,7 @@ pub fn define(ruby: &Ruby, module: &magnus::RModule) -> Result<(), Error> {
|
|
|
361
389
|
builder_class.define_method("build", method!(RbTokenizerBuilder::build, 0))?;
|
|
362
390
|
|
|
363
391
|
let tokenizer_class = module.define_class("Tokenizer", ruby.class_object())?;
|
|
364
|
-
tokenizer_class.define_singleton_method("new", function!(tokenizer_new,
|
|
392
|
+
tokenizer_class.define_singleton_method("new", function!(tokenizer_new, -1))?;
|
|
365
393
|
tokenizer_class.define_method("tokenize", method!(RbTokenizer::tokenize, 1))?;
|
|
366
394
|
tokenizer_class.define_method(
|
|
367
395
|
"tokenize_surfaces",
|
data/src/util.rs
CHANGED
|
@@ -3,6 +3,7 @@
|
|
|
3
3
|
//! This module provides helper functions for converting between Ruby objects
|
|
4
4
|
//! and Rust data structures, particularly for working with JSON-like data.
|
|
5
5
|
|
|
6
|
+
use lindera_binding_core::argument::{MAX_ARGUMENT_DEPTH, argument_too_deep_message};
|
|
6
7
|
use magnus::prelude::*;
|
|
7
8
|
use magnus::{Error, RArray, RHash, Ruby, TryConvert, Value};
|
|
8
9
|
|
|
@@ -20,8 +21,26 @@ use magnus::{Error, RArray, RHash, Ruby, TryConvert, Value};
|
|
|
20
21
|
/// # Errors
|
|
21
22
|
///
|
|
22
23
|
/// Returns a `TypeError` if the Ruby value type is not supported, or if it
|
|
23
|
-
/// is a non-finite `Float` (NaN or Infinity), which JSON cannot represent
|
|
24
|
+
/// is a non-finite `Float` (NaN or Infinity), which JSON cannot represent,
|
|
25
|
+
/// and an `ArgumentError` if it nests arrays and hashes more than
|
|
26
|
+
/// [`MAX_ARGUMENT_DEPTH`] levels deep (as one that contains itself does).
|
|
24
27
|
pub fn rb_value_to_json(ruby: &Ruby, value: Value) -> Result<serde_json::Value, Error> {
|
|
28
|
+
value_at_depth(ruby, value, 0)
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
/// Converts a Ruby value that sits inside `depth` arrays and hashes.
|
|
32
|
+
///
|
|
33
|
+
/// # Arguments
|
|
34
|
+
///
|
|
35
|
+
/// * `ruby` - Ruby runtime handle.
|
|
36
|
+
/// * `value` - Ruby value to convert.
|
|
37
|
+
/// * `depth` - The number of arrays and hashes around `value`.
|
|
38
|
+
///
|
|
39
|
+
/// # Returns
|
|
40
|
+
///
|
|
41
|
+
/// A `serde_json::Value` representing the Ruby value, or the errors of
|
|
42
|
+
/// [`rb_value_to_json`].
|
|
43
|
+
fn value_at_depth(ruby: &Ruby, value: Value, depth: usize) -> Result<serde_json::Value, Error> {
|
|
25
44
|
if value.is_nil() {
|
|
26
45
|
Ok(serde_json::Value::Null)
|
|
27
46
|
} else if value.is_kind_of(ruby.class_true_class())
|
|
@@ -68,9 +87,9 @@ pub fn rb_value_to_json(ruby: &Ruby, value: Value) -> Result<serde_json::Value,
|
|
|
68
87
|
})?;
|
|
69
88
|
Ok(serde_json::Value::String(s))
|
|
70
89
|
} else if let Ok(arr) = RArray::try_convert(value) {
|
|
71
|
-
|
|
90
|
+
array_at_depth(ruby, arr, depth)
|
|
72
91
|
} else if let Ok(hash) = RHash::try_convert(value) {
|
|
73
|
-
|
|
92
|
+
hash_at_depth(ruby, hash, depth)
|
|
74
93
|
} else {
|
|
75
94
|
Err(Error::new(
|
|
76
95
|
ruby.exception_type_error(),
|
|
@@ -81,20 +100,49 @@ pub fn rb_value_to_json(ruby: &Ruby, value: Value) -> Result<serde_json::Value,
|
|
|
81
100
|
}
|
|
82
101
|
}
|
|
83
102
|
|
|
84
|
-
///
|
|
103
|
+
/// Returns the depth of an array or hash that opens inside `depth` others.
|
|
104
|
+
///
|
|
105
|
+
/// # Arguments
|
|
106
|
+
///
|
|
107
|
+
/// * `ruby` - Ruby runtime handle.
|
|
108
|
+
/// * `depth` - The number of arrays and hashes around the new one.
|
|
109
|
+
///
|
|
110
|
+
/// # Returns
|
|
111
|
+
///
|
|
112
|
+
/// The new depth, or an `ArgumentError` when it exceeds
|
|
113
|
+
/// [`MAX_ARGUMENT_DEPTH`].
|
|
114
|
+
fn container_depth(ruby: &Ruby, depth: usize) -> Result<usize, Error> {
|
|
115
|
+
let depth = depth + 1;
|
|
116
|
+
if depth > MAX_ARGUMENT_DEPTH {
|
|
117
|
+
Err(Error::new(
|
|
118
|
+
ruby.exception_arg_error(),
|
|
119
|
+
argument_too_deep_message(),
|
|
120
|
+
))
|
|
121
|
+
} else {
|
|
122
|
+
Ok(depth)
|
|
123
|
+
}
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
/// Converts a Ruby array that sits inside `depth` arrays and hashes.
|
|
127
|
+
///
|
|
128
|
+
/// The depth is counted here rather than in [`value_at_depth`] so that an
|
|
129
|
+
/// array obtained through `to_ary` counts too.
|
|
85
130
|
///
|
|
86
131
|
/// # Arguments
|
|
87
132
|
///
|
|
88
133
|
/// * `ruby` - Ruby runtime handle.
|
|
89
134
|
/// * `array` - Ruby array to convert.
|
|
135
|
+
/// * `depth` - The number of arrays and hashes around `array`.
|
|
90
136
|
///
|
|
91
137
|
/// # Returns
|
|
92
138
|
///
|
|
93
|
-
/// A `serde_json::Value` representing the array
|
|
94
|
-
|
|
139
|
+
/// A `serde_json::Value` representing the array, or the errors of
|
|
140
|
+
/// [`rb_value_to_json`].
|
|
141
|
+
fn array_at_depth(ruby: &Ruby, array: RArray, depth: usize) -> Result<serde_json::Value, Error> {
|
|
142
|
+
let depth = container_depth(ruby, depth)?;
|
|
95
143
|
let mut vec = Vec::new();
|
|
96
144
|
for item in array.into_iter() {
|
|
97
|
-
vec.push(
|
|
145
|
+
vec.push(value_at_depth(ruby, item, depth)?);
|
|
98
146
|
}
|
|
99
147
|
Ok(serde_json::Value::Array(vec))
|
|
100
148
|
}
|
|
@@ -108,11 +156,32 @@ fn rb_array_to_json(ruby: &Ruby, array: RArray) -> Result<serde_json::Value, Err
|
|
|
108
156
|
///
|
|
109
157
|
/// # Returns
|
|
110
158
|
///
|
|
111
|
-
/// A `serde_json::Value` representing the hash
|
|
159
|
+
/// A `serde_json::Value` representing the hash, or the errors of
|
|
160
|
+
/// [`rb_value_to_json`]; the hash itself is the first level.
|
|
112
161
|
pub fn rb_hash_to_json(ruby: &Ruby, hash: RHash) -> Result<serde_json::Value, Error> {
|
|
162
|
+
hash_at_depth(ruby, hash, 0)
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
/// Converts a Ruby hash that sits inside `depth` arrays and hashes.
|
|
166
|
+
///
|
|
167
|
+
/// The depth is counted here rather than in [`value_at_depth`] so that a
|
|
168
|
+
/// hash obtained through `to_hash` counts too.
|
|
169
|
+
///
|
|
170
|
+
/// # Arguments
|
|
171
|
+
///
|
|
172
|
+
/// * `ruby` - Ruby runtime handle.
|
|
173
|
+
/// * `hash` - Ruby hash to convert.
|
|
174
|
+
/// * `depth` - The number of arrays and hashes around `hash`.
|
|
175
|
+
///
|
|
176
|
+
/// # Returns
|
|
177
|
+
///
|
|
178
|
+
/// A `serde_json::Value` representing the hash, or the errors of
|
|
179
|
+
/// [`rb_value_to_json`].
|
|
180
|
+
fn hash_at_depth(ruby: &Ruby, hash: RHash, depth: usize) -> Result<serde_json::Value, Error> {
|
|
181
|
+
let depth = container_depth(ruby, depth)?;
|
|
113
182
|
let mut map = serde_json::Map::new();
|
|
114
183
|
hash.foreach(|key: String, value: Value| {
|
|
115
|
-
let json_value =
|
|
184
|
+
let json_value = value_at_depth(ruby, value, depth)?;
|
|
116
185
|
map.insert(key, json_value);
|
|
117
186
|
Ok(magnus::r_hash::ForEach::Continue)
|
|
118
187
|
})?;
|