avrocadabra 0.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +7 -0
- data/CHANGELOG.md +6 -0
- data/Cargo.lock +923 -0
- data/Cargo.toml +8 -0
- data/LICENSE.txt +21 -0
- data/README.md +131 -0
- data/docs/avro_turf.md +61 -0
- data/docs/licenses/crates/quad-rand-0.2.3/DECLARED-LICENSE.txt +32 -0
- data/docs/licenses/gcc-16.2.0/COPYING +340 -0
- data/docs/licenses/gcc-16.2.0/COPYING.RUNTIME +73 -0
- data/docs/licenses/mingw-w64-14.0.0/AUTHORS +73 -0
- data/docs/licenses/mingw-w64-14.0.0/COPYING +43 -0
- data/docs/licenses/rust-1.99.0/COMPILER-BUILTINS.txt +275 -0
- data/docs/licenses/rust-1.99.0/COPYRIGHT-library.html +29357 -0
- data/docs/licenses/rust-1.99.0/LIBM.txt +258 -0
- data/docs/licenses/rust-1.99.0/LLVM-LIBUNWIND.txt +311 -0
- data/docs/licenses/rust-1.99.0/RUST-LICENSE-APACHE.txt +176 -0
- data/docs/licenses/rust-1.99.0/RUST-LICENSE-MIT.txt +25 -0
- data/docs/licenses/rust-1.99.0/STDLIB-BACKTRACE-MIT.txt +25 -0
- data/docs/licenses/rust-1.99.0/STDLIB-PORTABLE-SIMD-MIT.txt +19 -0
- data/docs/licenses/rust-1.99.0/STDLIB-STDARCH-MIT.txt +25 -0
- data/docs/licenses/rust-1.99.0/licenses/Apache-2.0.txt +73 -0
- data/docs/licenses/rust-1.99.0/licenses/BSD-2-Clause.txt +9 -0
- data/docs/licenses/rust-1.99.0/licenses/CC-BY-SA-4.0.txt +427 -0
- data/docs/licenses/rust-1.99.0/licenses/GCC-exception-3.1.txt +30 -0
- data/docs/licenses/rust-1.99.0/licenses/GPL-2.0-only.txt +133 -0
- data/docs/licenses/rust-1.99.0/licenses/GPL-3.0-or-later.txt +202 -0
- data/docs/licenses/rust-1.99.0/licenses/ISC.txt +7 -0
- data/docs/licenses/rust-1.99.0/licenses/LLVM-exception.txt +15 -0
- data/docs/licenses/rust-1.99.0/licenses/MIT.txt +9 -0
- data/docs/licenses/rust-1.99.0/licenses/NCSA.txt +28 -0
- data/docs/licenses/rust-1.99.0/licenses/OFL-1.1.txt +43 -0
- data/docs/licenses/rust-1.99.0/licenses/Unicode-3.0.txt +39 -0
- data/docs/releasing.md +71 -0
- data/docs/third_party.md +5922 -0
- data/ext/avrocadabra/Cargo.toml +29 -0
- data/ext/avrocadabra/build.rs +4 -0
- data/ext/avrocadabra/extconf.rb +9 -0
- data/ext/avrocadabra/src/big_decimal.rs +107 -0
- data/ext/avrocadabra/src/callback.rs +106 -0
- data/ext/avrocadabra/src/convert.rs +617 -0
- data/ext/avrocadabra/src/decode_value.rs +473 -0
- data/ext/avrocadabra/src/guard.rs +680 -0
- data/ext/avrocadabra/src/lib.rs +670 -0
- data/ext/avrocadabra/src/mapping.rs +63 -0
- data/ext/avrocadabra/src/memory.rs +194 -0
- data/ext/avrocadabra/src/prepare.rs +819 -0
- data/ext/avrocadabra/src/resolution.rs +1932 -0
- data/ext/avrocadabra/src/schema_state.rs +60 -0
- data/ext/avrocadabra/src/validation.rs +174 -0
- data/ext/avrocadabra/src/wire.rs +269 -0
- data/lib/avrocadabra/avro_turf/cache.rb +28 -0
- data/lib/avrocadabra/avro_turf/codec.rb +69 -0
- data/lib/avrocadabra/avro_turf/datum_reader.rb +17 -0
- data/lib/avrocadabra/avro_turf/datum_writer.rb +14 -0
- data/lib/avrocadabra/avro_turf/mapping.rb +48 -0
- data/lib/avrocadabra/avro_turf/messaging.rb +12 -0
- data/lib/avrocadabra/avro_turf/ractor_support.rb +154 -0
- data/lib/avrocadabra/avro_turf/routing.rb +15 -0
- data/lib/avrocadabra/avro_turf/schema_state.rb +70 -0
- data/lib/avrocadabra/avro_turf/validation.rb +33 -0
- data/lib/avrocadabra/avro_turf.rb +25 -0
- data/lib/avrocadabra/logical.rb +71 -0
- data/lib/avrocadabra/schema.rb +48 -0
- data/lib/avrocadabra/version.rb +5 -0
- data/lib/avrocadabra.rb +34 -0
- metadata +183 -0
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
use magnus::{
|
|
2
|
+
Error, RArray, RHash, RString, Symbol, Value, encoding::EncodingCapable, prelude::*,
|
|
3
|
+
r_hash::ForEach, rb_sys::AsRawValue,
|
|
4
|
+
};
|
|
5
|
+
|
|
6
|
+
fn identical(left: Value, right: Value) -> bool {
|
|
7
|
+
left.as_raw() == right.as_raw()
|
|
8
|
+
}
|
|
9
|
+
|
|
10
|
+
pub fn unchanged(attributes: RArray, containers: RArray) -> Result<bool, Error> {
|
|
11
|
+
for index in (0..attributes.len() as isize).step_by(3) {
|
|
12
|
+
let object: Value = attributes.entry(index)?;
|
|
13
|
+
let attribute: Symbol = attributes.entry(index + 1)?;
|
|
14
|
+
let previous: Value = attributes.entry(index + 2)?;
|
|
15
|
+
if !identical(object.funcall(attribute, ())?, previous) {
|
|
16
|
+
return Ok(false);
|
|
17
|
+
}
|
|
18
|
+
}
|
|
19
|
+
for index in (0..containers.len() as isize).step_by(2) {
|
|
20
|
+
let object: Value = containers.entry(index)?;
|
|
21
|
+
let previous: Value = containers.entry(index + 1)?;
|
|
22
|
+
if let Some(string) = RString::from_value(object) {
|
|
23
|
+
let previous = RString::try_convert(previous)?;
|
|
24
|
+
if string.enc_get() != previous.enc_get()
|
|
25
|
+
|| unsafe { string.as_slice() != previous.as_slice() }
|
|
26
|
+
{
|
|
27
|
+
return Ok(false);
|
|
28
|
+
}
|
|
29
|
+
} else if let Some(array) = RArray::from_value(object) {
|
|
30
|
+
let previous = RArray::try_convert(previous)?;
|
|
31
|
+
if array.len() != previous.len()
|
|
32
|
+
|| !unsafe { array.as_slice().iter().zip(previous.as_slice()) }
|
|
33
|
+
.all(|(&left, &right)| identical(left, right))
|
|
34
|
+
{
|
|
35
|
+
return Ok(false);
|
|
36
|
+
}
|
|
37
|
+
} else if let Some(hash) = RHash::from_value(object) {
|
|
38
|
+
let previous = RArray::try_convert(previous)?;
|
|
39
|
+
if hash.len() * 2 != previous.len() {
|
|
40
|
+
return Ok(false);
|
|
41
|
+
}
|
|
42
|
+
let mut index = 0;
|
|
43
|
+
let mut matches = true;
|
|
44
|
+
hash.foreach(|key: Value, value: Value| {
|
|
45
|
+
matches = identical(key, previous.entry(index)?)
|
|
46
|
+
&& identical(value, previous.entry(index + 1)?);
|
|
47
|
+
index += 2;
|
|
48
|
+
Ok(if matches {
|
|
49
|
+
ForEach::Continue
|
|
50
|
+
} else {
|
|
51
|
+
ForEach::Stop
|
|
52
|
+
})
|
|
53
|
+
})?;
|
|
54
|
+
if !matches {
|
|
55
|
+
return Ok(false);
|
|
56
|
+
}
|
|
57
|
+
}
|
|
58
|
+
}
|
|
59
|
+
Ok(true)
|
|
60
|
+
}
|
|
@@ -0,0 +1,174 @@
|
|
|
1
|
+
use apache_avro::{
|
|
2
|
+
AvroResult, Schema,
|
|
3
|
+
error::Details,
|
|
4
|
+
util,
|
|
5
|
+
validator::{
|
|
6
|
+
EnumSymbolNameValidator, RecordFieldNameValidator, SchemaNameValidator,
|
|
7
|
+
SchemaNamespaceValidator, set_enum_symbol_name_validator, set_record_field_name_validator,
|
|
8
|
+
set_schema_name_validator, set_schema_namespace_validator,
|
|
9
|
+
},
|
|
10
|
+
};
|
|
11
|
+
use std::sync::OnceLock;
|
|
12
|
+
|
|
13
|
+
struct Validator;
|
|
14
|
+
|
|
15
|
+
const NAME_PATTERN: &str = r"^((?P<namespace>([A-Za-z_][A-Za-z0-9_]*(\.[A-Za-z_][A-Za-z0-9_]*)*)?)\.)?(?P<name>[A-Za-z_][A-Za-z0-9_]*)$";
|
|
16
|
+
const NAMESPACE_PATTERN: &str = r"^([A-Za-z_][A-Za-z0-9_]*(\.[A-Za-z_][A-Za-z0-9_]*)*)?$";
|
|
17
|
+
|
|
18
|
+
fn identifier(value: &str) -> bool {
|
|
19
|
+
let mut bytes = value.bytes();
|
|
20
|
+
bytes
|
|
21
|
+
.next()
|
|
22
|
+
.is_some_and(|byte| byte.is_ascii_alphabetic() || byte == b'_')
|
|
23
|
+
&& bytes.all(|byte| byte.is_ascii_alphanumeric() || byte == b'_')
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
fn namespace(value: &str) -> bool {
|
|
27
|
+
value.is_empty() || value.split('.').all(identifier)
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
impl SchemaNameValidator for Validator {
|
|
31
|
+
fn validate(&self, value: &str) -> AvroResult<usize> {
|
|
32
|
+
let index = value.rfind('.').map_or(0, |index| index + 1);
|
|
33
|
+
if identifier(&value[index..]) && (index == 0 || namespace(&value[..index - 1])) {
|
|
34
|
+
Ok(index)
|
|
35
|
+
} else {
|
|
36
|
+
Err(Details::InvalidSchemaName(value.into(), NAME_PATTERN).into())
|
|
37
|
+
}
|
|
38
|
+
}
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
impl SchemaNamespaceValidator for Validator {
|
|
42
|
+
fn validate(&self, value: &str) -> AvroResult<()> {
|
|
43
|
+
if namespace(value) {
|
|
44
|
+
Ok(())
|
|
45
|
+
} else {
|
|
46
|
+
Err(Details::InvalidNamespace(value.into(), NAMESPACE_PATTERN).into())
|
|
47
|
+
}
|
|
48
|
+
}
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
impl EnumSymbolNameValidator for Validator {
|
|
52
|
+
fn validate(&self, value: &str) -> AvroResult<()> {
|
|
53
|
+
if identifier(value) {
|
|
54
|
+
Ok(())
|
|
55
|
+
} else {
|
|
56
|
+
Err(Details::EnumSymbolName(value.into()).into())
|
|
57
|
+
}
|
|
58
|
+
}
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
impl RecordFieldNameValidator for Validator {
|
|
62
|
+
fn validate(&self, value: &str) -> AvroResult<()> {
|
|
63
|
+
if identifier(value) {
|
|
64
|
+
Ok(())
|
|
65
|
+
} else {
|
|
66
|
+
Err(Details::FieldName(value.into()).into())
|
|
67
|
+
}
|
|
68
|
+
}
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
pub(crate) fn initialize() -> Result<(), String> {
|
|
72
|
+
static INITIALIZED: OnceLock<Result<(), &'static str>> = OnceLock::new();
|
|
73
|
+
INITIALIZED
|
|
74
|
+
.get_or_init(|| {
|
|
75
|
+
// Apache's default regex caches contain mutexes that can remain locked after fork.
|
|
76
|
+
set_schema_name_validator(Box::new(Validator))
|
|
77
|
+
.map_err(|_| "Apache schema-name validator was initialized before Avrocadabra")?;
|
|
78
|
+
set_schema_namespace_validator(Box::new(Validator))
|
|
79
|
+
.map_err(|_| "Apache namespace validator was initialized before Avrocadabra")?;
|
|
80
|
+
set_enum_symbol_name_validator(Box::new(Validator))
|
|
81
|
+
.map_err(|_| "Apache enum validator was initialized before Avrocadabra")?;
|
|
82
|
+
set_record_field_name_validator(Box::new(Validator))
|
|
83
|
+
.map_err(|_| "Apache field-name validator was initialized before Avrocadabra")?;
|
|
84
|
+
util::max_allocation_bytes(util::DEFAULT_MAX_ALLOCATION_BYTES);
|
|
85
|
+
util::set_serde_human_readable(util::DEFAULT_SERDE_HUMAN_READABLE);
|
|
86
|
+
let _ = Schema::Null == Schema::Int;
|
|
87
|
+
Ok(())
|
|
88
|
+
})
|
|
89
|
+
.map_err(str::to_owned)
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
#[cfg(test)]
|
|
93
|
+
mod tests {
|
|
94
|
+
use super::*;
|
|
95
|
+
|
|
96
|
+
struct Original;
|
|
97
|
+
impl SchemaNameValidator for Original {}
|
|
98
|
+
impl SchemaNamespaceValidator for Original {}
|
|
99
|
+
impl EnumSymbolNameValidator for Original {}
|
|
100
|
+
impl RecordFieldNameValidator for Original {}
|
|
101
|
+
|
|
102
|
+
#[test]
|
|
103
|
+
fn ascii_validation_matches_apache_regex_rules() {
|
|
104
|
+
initialize().unwrap();
|
|
105
|
+
let alphabet = ['A', 'z', '0', '_', '.', '-', 'é', '\n', ' '];
|
|
106
|
+
let mut values = vec![String::new()];
|
|
107
|
+
let mut frontier = vec![String::new()];
|
|
108
|
+
for _ in 0..4 {
|
|
109
|
+
frontier = frontier
|
|
110
|
+
.iter()
|
|
111
|
+
.flat_map(|prefix| {
|
|
112
|
+
alphabet
|
|
113
|
+
.iter()
|
|
114
|
+
.map(move |character| format!("{prefix}{character}"))
|
|
115
|
+
})
|
|
116
|
+
.collect();
|
|
117
|
+
values.extend(frontier.iter().cloned());
|
|
118
|
+
}
|
|
119
|
+
values.extend(
|
|
120
|
+
[
|
|
121
|
+
"example.Name",
|
|
122
|
+
"example.deep.Name",
|
|
123
|
+
".Name",
|
|
124
|
+
"example.",
|
|
125
|
+
".example.Name",
|
|
126
|
+
"__Name9",
|
|
127
|
+
]
|
|
128
|
+
.map(str::to_owned),
|
|
129
|
+
);
|
|
130
|
+
for value in values {
|
|
131
|
+
assert_eq!(
|
|
132
|
+
SchemaNameValidator::validate(&Validator, &value).ok(),
|
|
133
|
+
SchemaNameValidator::validate(&Original, &value).ok(),
|
|
134
|
+
"schema name {value:?}"
|
|
135
|
+
);
|
|
136
|
+
assert_eq!(
|
|
137
|
+
SchemaNamespaceValidator::validate(&Validator, &value).is_ok(),
|
|
138
|
+
SchemaNamespaceValidator::validate(&Original, &value).is_ok(),
|
|
139
|
+
"namespace {value:?}"
|
|
140
|
+
);
|
|
141
|
+
assert_eq!(
|
|
142
|
+
EnumSymbolNameValidator::validate(&Validator, &value).is_ok(),
|
|
143
|
+
EnumSymbolNameValidator::validate(&Original, &value).is_ok(),
|
|
144
|
+
"enum symbol {value:?}"
|
|
145
|
+
);
|
|
146
|
+
assert_eq!(
|
|
147
|
+
RecordFieldNameValidator::validate(&Validator, &value).is_ok(),
|
|
148
|
+
RecordFieldNameValidator::validate(&Original, &value).is_ok(),
|
|
149
|
+
"field name {value:?}"
|
|
150
|
+
);
|
|
151
|
+
}
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
#[test]
|
|
155
|
+
fn initialization_is_idempotent_and_names_keep_enclosing_namespace_rules() {
|
|
156
|
+
initialize().unwrap();
|
|
157
|
+
initialize().unwrap();
|
|
158
|
+
use apache_avro::schema::Name;
|
|
159
|
+
for (name, enclosing, expected) in [
|
|
160
|
+
("Value", None, "Value"),
|
|
161
|
+
("Value", Some(""), "Value"),
|
|
162
|
+
("Value", Some("outer.ns"), "outer.ns.Value"),
|
|
163
|
+
("own.Value", Some("outer.ns"), "own.Value"),
|
|
164
|
+
(".Value", Some("outer.ns"), "Value"),
|
|
165
|
+
] {
|
|
166
|
+
assert_eq!(
|
|
167
|
+
Name::new_with_enclosing_namespace(name, enclosing)
|
|
168
|
+
.unwrap()
|
|
169
|
+
.fullname(None),
|
|
170
|
+
expected
|
|
171
|
+
);
|
|
172
|
+
}
|
|
173
|
+
}
|
|
174
|
+
}
|
|
@@ -0,0 +1,269 @@
|
|
|
1
|
+
use apache_avro::{
|
|
2
|
+
Schema,
|
|
3
|
+
schema::{InnerDecimalSchema, NamesRef, RecordField, RecordSchema, UnionSchema, UuidSchema},
|
|
4
|
+
types::Value,
|
|
5
|
+
};
|
|
6
|
+
use serde::{
|
|
7
|
+
Deserialize, Deserializer, Serialize, Serializer,
|
|
8
|
+
de::{MapAccess, SeqAccess, Visitor},
|
|
9
|
+
ser::{Error, SerializeMap, SerializeSeq, SerializeStruct},
|
|
10
|
+
};
|
|
11
|
+
use std::fmt;
|
|
12
|
+
|
|
13
|
+
// Serde retains map order. A named one-field record represents each enum's wire index
|
|
14
|
+
// because Serde's enum serializer requires static symbol names.
|
|
15
|
+
pub fn schema(schema: &Schema) -> Result<Schema, String> {
|
|
16
|
+
Ok(match schema {
|
|
17
|
+
Schema::Enum(value) => Schema::Record(
|
|
18
|
+
RecordSchema::builder()
|
|
19
|
+
.name(value.name.clone())
|
|
20
|
+
.aliases(value.aliases.clone())
|
|
21
|
+
.fields(vec![
|
|
22
|
+
RecordField::builder()
|
|
23
|
+
.name("index")
|
|
24
|
+
.schema(Schema::Int)
|
|
25
|
+
.build(),
|
|
26
|
+
])
|
|
27
|
+
.build(),
|
|
28
|
+
),
|
|
29
|
+
Schema::Record(value) => {
|
|
30
|
+
let mut value = value.clone();
|
|
31
|
+
for field in &mut value.fields {
|
|
32
|
+
field.schema = self::schema(&field.schema)?;
|
|
33
|
+
field.default = None;
|
|
34
|
+
}
|
|
35
|
+
Schema::Record(value)
|
|
36
|
+
}
|
|
37
|
+
Schema::Array(value) => {
|
|
38
|
+
let mut value = value.clone();
|
|
39
|
+
value.items = Box::new(self::schema(&value.items)?);
|
|
40
|
+
Schema::Array(value)
|
|
41
|
+
}
|
|
42
|
+
Schema::Map(value) => {
|
|
43
|
+
let mut value = value.clone();
|
|
44
|
+
value.types = Box::new(self::schema(&value.types)?);
|
|
45
|
+
Schema::Map(value)
|
|
46
|
+
}
|
|
47
|
+
Schema::Union(value) => Schema::Union(
|
|
48
|
+
UnionSchema::new(
|
|
49
|
+
value
|
|
50
|
+
.variants()
|
|
51
|
+
.iter()
|
|
52
|
+
.map(self::schema)
|
|
53
|
+
.collect::<Result<_, _>>()?,
|
|
54
|
+
)
|
|
55
|
+
.map_err(|error| error.to_string())?,
|
|
56
|
+
),
|
|
57
|
+
Schema::Decimal(value) => match &value.inner {
|
|
58
|
+
InnerDecimalSchema::Bytes => Schema::Bytes,
|
|
59
|
+
InnerDecimalSchema::Fixed(value) => Schema::Fixed(value.clone()),
|
|
60
|
+
},
|
|
61
|
+
Schema::BigDecimal | Schema::Uuid(UuidSchema::Bytes) => Schema::Bytes,
|
|
62
|
+
Schema::Uuid(UuidSchema::String) => Schema::String,
|
|
63
|
+
Schema::Duration(value) | Schema::Uuid(UuidSchema::Fixed(value)) => {
|
|
64
|
+
Schema::Fixed(value.clone())
|
|
65
|
+
}
|
|
66
|
+
Schema::Date | Schema::TimeMillis => Schema::Int,
|
|
67
|
+
Schema::TimeMicros
|
|
68
|
+
| Schema::TimestampMillis
|
|
69
|
+
| Schema::TimestampMicros
|
|
70
|
+
| Schema::TimestampNanos
|
|
71
|
+
| Schema::LocalTimestampMillis
|
|
72
|
+
| Schema::LocalTimestampMicros
|
|
73
|
+
| Schema::LocalTimestampNanos => Schema::Long,
|
|
74
|
+
value => value.clone(),
|
|
75
|
+
})
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
pub struct Datum(pub Value);
|
|
79
|
+
|
|
80
|
+
struct DatumRef<'a>(&'a Value);
|
|
81
|
+
|
|
82
|
+
impl Serialize for Datum {
|
|
83
|
+
fn serialize<S: Serializer>(&self, serializer: S) -> Result<S::Ok, S::Error> {
|
|
84
|
+
DatumRef(&self.0).serialize(serializer)
|
|
85
|
+
}
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
impl Serialize for DatumRef<'_> {
|
|
89
|
+
fn serialize<S: Serializer>(&self, serializer: S) -> Result<S::Ok, S::Error> {
|
|
90
|
+
match self.0 {
|
|
91
|
+
Value::Null => serializer.serialize_unit(),
|
|
92
|
+
Value::Boolean(value) => serializer.serialize_bool(*value),
|
|
93
|
+
Value::Int(value) | Value::Date(value) | Value::TimeMillis(value) => {
|
|
94
|
+
serializer.serialize_i32(*value)
|
|
95
|
+
}
|
|
96
|
+
Value::Long(value)
|
|
97
|
+
| Value::TimeMicros(value)
|
|
98
|
+
| Value::TimestampMillis(value)
|
|
99
|
+
| Value::TimestampMicros(value)
|
|
100
|
+
| Value::TimestampNanos(value)
|
|
101
|
+
| Value::LocalTimestampMillis(value)
|
|
102
|
+
| Value::LocalTimestampMicros(value)
|
|
103
|
+
| Value::LocalTimestampNanos(value) => serializer.serialize_i64(*value),
|
|
104
|
+
Value::Float(value) => serializer.serialize_f32(*value),
|
|
105
|
+
Value::Double(value) => serializer.serialize_f64(*value),
|
|
106
|
+
Value::String(value) => serializer.serialize_str(value),
|
|
107
|
+
Value::Bytes(value) | Value::Fixed(_, value) => serializer.serialize_bytes(value),
|
|
108
|
+
Value::Enum(index, _) => {
|
|
109
|
+
let mut record = serializer.serialize_struct("", 1)?;
|
|
110
|
+
record.serialize_field("index", &(*index as i32))?;
|
|
111
|
+
record.end()
|
|
112
|
+
}
|
|
113
|
+
Value::Union(index, value) => {
|
|
114
|
+
serializer.serialize_newtype_variant("", *index, "", &DatumRef(value))
|
|
115
|
+
}
|
|
116
|
+
Value::Array(values) => {
|
|
117
|
+
let mut seq = serializer.serialize_seq(Some(values.len()))?;
|
|
118
|
+
for value in values {
|
|
119
|
+
seq.serialize_element(&DatumRef(value))?;
|
|
120
|
+
}
|
|
121
|
+
seq.end()
|
|
122
|
+
}
|
|
123
|
+
Value::Record(values) => {
|
|
124
|
+
let mut map = serializer.serialize_map(Some(values.len()))?;
|
|
125
|
+
for (key, value) in values {
|
|
126
|
+
map.serialize_entry(key, &DatumRef(value))?;
|
|
127
|
+
}
|
|
128
|
+
map.end()
|
|
129
|
+
}
|
|
130
|
+
Value::Decimal(value) => value.serialize(serializer),
|
|
131
|
+
Value::BigDecimal(value) => serializer
|
|
132
|
+
.serialize_bytes(&crate::big_decimal::encode(value).map_err(S::Error::custom)?),
|
|
133
|
+
Value::Duration(value) => serializer.serialize_bytes(&<[u8; 12]>::from(*value)),
|
|
134
|
+
Value::Map(_) | Value::Uuid(_) => {
|
|
135
|
+
Err(S::Error::custom("unordered or unconverted datum"))
|
|
136
|
+
}
|
|
137
|
+
}
|
|
138
|
+
}
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
impl<'de> Deserialize<'de> for Datum {
|
|
142
|
+
fn deserialize<D: Deserializer<'de>>(deserializer: D) -> Result<Self, D::Error> {
|
|
143
|
+
deserializer.deserialize_any(DatumVisitor)
|
|
144
|
+
}
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
struct DatumVisitor;
|
|
148
|
+
|
|
149
|
+
struct Key(String);
|
|
150
|
+
|
|
151
|
+
impl<'de> Deserialize<'de> for Key {
|
|
152
|
+
fn deserialize<D: Deserializer<'de>>(deserializer: D) -> Result<Self, D::Error> {
|
|
153
|
+
match deserializer.deserialize_identifier(DatumVisitor)? {
|
|
154
|
+
Datum(Value::String(key)) => Ok(Self(key)),
|
|
155
|
+
_ => Err(serde::de::Error::custom("expected string key")),
|
|
156
|
+
}
|
|
157
|
+
}
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
impl<'de> Visitor<'de> for DatumVisitor {
|
|
161
|
+
type Value = Datum;
|
|
162
|
+
|
|
163
|
+
fn expecting(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result {
|
|
164
|
+
formatter.write_str("an Avro datum")
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
fn visit_unit<E>(self) -> Result<Datum, E> {
|
|
168
|
+
Ok(Datum(Value::Null))
|
|
169
|
+
}
|
|
170
|
+
fn visit_bool<E>(self, value: bool) -> Result<Datum, E> {
|
|
171
|
+
Ok(Datum(Value::Boolean(value)))
|
|
172
|
+
}
|
|
173
|
+
fn visit_i32<E>(self, value: i32) -> Result<Datum, E> {
|
|
174
|
+
Ok(Datum(Value::Int(value)))
|
|
175
|
+
}
|
|
176
|
+
fn visit_i64<E>(self, value: i64) -> Result<Datum, E> {
|
|
177
|
+
Ok(Datum(Value::Long(value)))
|
|
178
|
+
}
|
|
179
|
+
fn visit_f32<E>(self, value: f32) -> Result<Datum, E> {
|
|
180
|
+
Ok(Datum(Value::Float(value)))
|
|
181
|
+
}
|
|
182
|
+
fn visit_f64<E>(self, value: f64) -> Result<Datum, E> {
|
|
183
|
+
Ok(Datum(Value::Double(value)))
|
|
184
|
+
}
|
|
185
|
+
fn visit_string<E>(self, value: String) -> Result<Datum, E> {
|
|
186
|
+
Ok(Datum(Value::String(value)))
|
|
187
|
+
}
|
|
188
|
+
fn visit_str<E>(self, value: &str) -> Result<Datum, E> {
|
|
189
|
+
Ok(Datum(Value::String(value.to_owned())))
|
|
190
|
+
}
|
|
191
|
+
fn visit_byte_buf<E>(self, value: Vec<u8>) -> Result<Datum, E> {
|
|
192
|
+
Ok(Datum(Value::Bytes(value)))
|
|
193
|
+
}
|
|
194
|
+
fn visit_bytes<E>(self, value: &[u8]) -> Result<Datum, E> {
|
|
195
|
+
Ok(Datum(Value::Bytes(value.to_owned())))
|
|
196
|
+
}
|
|
197
|
+
|
|
198
|
+
fn visit_seq<A: SeqAccess<'de>>(self, mut seq: A) -> Result<Datum, A::Error> {
|
|
199
|
+
let mut values = Vec::new();
|
|
200
|
+
while let Some(Datum(value)) = seq.next_element()? {
|
|
201
|
+
values.push(value);
|
|
202
|
+
}
|
|
203
|
+
Ok(Datum(Value::Array(values)))
|
|
204
|
+
}
|
|
205
|
+
|
|
206
|
+
fn visit_map<A: MapAccess<'de>>(self, mut map: A) -> Result<Datum, A::Error> {
|
|
207
|
+
let mut values = Vec::new();
|
|
208
|
+
while let Some((Key(key), Datum(value))) = map.next_entry()? {
|
|
209
|
+
values.push((key, value));
|
|
210
|
+
}
|
|
211
|
+
Ok(Datum(Value::Record(values)))
|
|
212
|
+
}
|
|
213
|
+
}
|
|
214
|
+
|
|
215
|
+
pub fn materialize(
|
|
216
|
+
value: Value,
|
|
217
|
+
schema: &Schema,
|
|
218
|
+
names: &NamesRef<'_>,
|
|
219
|
+
unions: &mut impl Iterator<Item = u32>,
|
|
220
|
+
) -> Result<Value, String> {
|
|
221
|
+
match (schema, value) {
|
|
222
|
+
(Schema::Ref { name }, value) => materialize(
|
|
223
|
+
value,
|
|
224
|
+
names.get(name).ok_or("unresolved wire schema")?,
|
|
225
|
+
names,
|
|
226
|
+
unions,
|
|
227
|
+
),
|
|
228
|
+
(Schema::Union(union), value) => {
|
|
229
|
+
let index = unions.next().ok_or("missing union index")?;
|
|
230
|
+
let branch = union
|
|
231
|
+
.variants()
|
|
232
|
+
.get(index as usize)
|
|
233
|
+
.ok_or("invalid union index")?;
|
|
234
|
+
Ok(Value::Union(
|
|
235
|
+
index,
|
|
236
|
+
Box::new(materialize(value, branch, names, unions)?),
|
|
237
|
+
))
|
|
238
|
+
}
|
|
239
|
+
(Schema::Record(record), Value::Record(values)) => values
|
|
240
|
+
.into_iter()
|
|
241
|
+
.zip(&record.fields)
|
|
242
|
+
.map(|((key, value), field)| {
|
|
243
|
+
Ok((key, materialize(value, &field.schema, names, unions)?))
|
|
244
|
+
})
|
|
245
|
+
.collect::<Result<_, _>>()
|
|
246
|
+
.map(Value::Record),
|
|
247
|
+
(Schema::Map(map), Value::Record(values)) => values
|
|
248
|
+
.into_iter()
|
|
249
|
+
.map(|(key, value)| Ok((key, materialize(value, &map.types, names, unions)?)))
|
|
250
|
+
.collect::<Result<_, _>>()
|
|
251
|
+
.map(Value::Record),
|
|
252
|
+
(Schema::Array(array), Value::Array(values)) => values
|
|
253
|
+
.into_iter()
|
|
254
|
+
.map(|value| materialize(value, &array.items, names, unions))
|
|
255
|
+
.collect::<Result<_, _>>()
|
|
256
|
+
.map(Value::Array),
|
|
257
|
+
(Schema::Enum(enumeration), Value::Record(mut values)) => {
|
|
258
|
+
let Some((_, Value::Int(index))) = values.pop() else {
|
|
259
|
+
return Err("invalid enum index".into());
|
|
260
|
+
};
|
|
261
|
+
let symbol = enumeration
|
|
262
|
+
.symbols
|
|
263
|
+
.get(index as usize)
|
|
264
|
+
.ok_or("invalid enum index")?;
|
|
265
|
+
Ok(Value::Enum(index as u32, symbol.clone()))
|
|
266
|
+
}
|
|
267
|
+
(schema, value) => crate::resolution::logical_value(value, schema),
|
|
268
|
+
}
|
|
269
|
+
}
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Avrocadabra
|
|
4
|
+
module AvroTurf
|
|
5
|
+
class Cache
|
|
6
|
+
LIMIT = 128
|
|
7
|
+
|
|
8
|
+
def initialize
|
|
9
|
+
@entries = {}.compare_by_identity
|
|
10
|
+
@mutex = Mutex.new
|
|
11
|
+
end
|
|
12
|
+
|
|
13
|
+
def fetch(schema)
|
|
14
|
+
codec = @mutex.synchronize { @entries[schema] }
|
|
15
|
+
return codec if codec&.current?
|
|
16
|
+
|
|
17
|
+
@mutex.synchronize do
|
|
18
|
+
codec = @entries[schema]
|
|
19
|
+
return codec if codec&.current?
|
|
20
|
+
|
|
21
|
+
codec = Codec.new(schema)
|
|
22
|
+
@entries.shift if !@entries.key?(schema) && @entries.size >= LIMIT
|
|
23
|
+
@entries[schema] = codec
|
|
24
|
+
end
|
|
25
|
+
end
|
|
26
|
+
end
|
|
27
|
+
end
|
|
28
|
+
end
|
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Avrocadabra
|
|
4
|
+
module AvroTurf
|
|
5
|
+
class Codec
|
|
6
|
+
attr_reader :source
|
|
7
|
+
|
|
8
|
+
def initialize(source)
|
|
9
|
+
@source = source
|
|
10
|
+
@state = SchemaState.new(source)
|
|
11
|
+
@mapping = Mapping.new(source, @state.schemas)
|
|
12
|
+
document = source.to_avro
|
|
13
|
+
prepare_document(source, document)
|
|
14
|
+
@native = Schema.new(JSON.generate(document)).__send__(:native)
|
|
15
|
+
rescue SchemaError => e
|
|
16
|
+
raise ::Avro::SchemaParseError, e.message
|
|
17
|
+
end
|
|
18
|
+
|
|
19
|
+
def encode(datum)
|
|
20
|
+
@native.encode(datum, false, @mapping.native)
|
|
21
|
+
rescue EncodeError => e
|
|
22
|
+
raise ::Avro::IO::AvroTypeError.new(@source, datum), cause: e
|
|
23
|
+
end
|
|
24
|
+
|
|
25
|
+
def current?
|
|
26
|
+
@state.current?
|
|
27
|
+
end
|
|
28
|
+
|
|
29
|
+
def read(decoder, reader)
|
|
30
|
+
@native.decode(decoder.reader, reader.native, false, false, reader.mapping.native)
|
|
31
|
+
rescue ResolutionError => e
|
|
32
|
+
if e.message.include?("reader field is absent from writer schema and has no default")
|
|
33
|
+
raise ::Avro::AvroError, e.message
|
|
34
|
+
end
|
|
35
|
+
|
|
36
|
+
raise ::Avro::IO::SchemaMatchException.new(source, reader.source), cause: e
|
|
37
|
+
rescue DecodeError => e
|
|
38
|
+
raise EOFError, e.message if e.message.include?("truncated")
|
|
39
|
+
|
|
40
|
+
raise ::Avro::AvroError, e.message
|
|
41
|
+
end
|
|
42
|
+
|
|
43
|
+
private
|
|
44
|
+
|
|
45
|
+
def prepare_document(source, document)
|
|
46
|
+
document.delete("logicalType") if document.is_a?(Hash)
|
|
47
|
+
case source.type_sym
|
|
48
|
+
when :record, :error
|
|
49
|
+
return if document.is_a?(String)
|
|
50
|
+
|
|
51
|
+
source.fields.zip(document.fetch("fields")) do |field, entry|
|
|
52
|
+
entry["aliases"] = field.aliases if field.aliases
|
|
53
|
+
prepare_document(field.type, entry.fetch("type"))
|
|
54
|
+
end
|
|
55
|
+
when :array
|
|
56
|
+
prepare_document(source.items, document.fetch("items"))
|
|
57
|
+
when :map
|
|
58
|
+
prepare_document(source.values, document.fetch("values"))
|
|
59
|
+
when :union
|
|
60
|
+
source.schemas.zip(document) { |branch, entry| prepare_document(branch, entry) }
|
|
61
|
+
end
|
|
62
|
+
end
|
|
63
|
+
|
|
64
|
+
protected
|
|
65
|
+
|
|
66
|
+
attr_reader :mapping, :native
|
|
67
|
+
end
|
|
68
|
+
end
|
|
69
|
+
end
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Avrocadabra
|
|
4
|
+
module AvroTurf
|
|
5
|
+
module DatumReader
|
|
6
|
+
def read(decoder)
|
|
7
|
+
cache = Thread.current[:avrocadabra_codecs]
|
|
8
|
+
return super unless cache && decoder.instance_of?(::Avro::IO::BinaryDecoder) && decoder.reader.is_a?(StringIO)
|
|
9
|
+
|
|
10
|
+
self.readers_schema = writers_schema unless readers_schema
|
|
11
|
+
writer = cache.fetch(writers_schema)
|
|
12
|
+
reader = readers_schema.equal?(writers_schema) ? writer : cache.fetch(readers_schema)
|
|
13
|
+
writer.read(decoder, reader)
|
|
14
|
+
end
|
|
15
|
+
end
|
|
16
|
+
end
|
|
17
|
+
end
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Avrocadabra
|
|
4
|
+
module AvroTurf
|
|
5
|
+
module DatumWriter
|
|
6
|
+
def write(datum, encoder)
|
|
7
|
+
cache = Thread.current[:avrocadabra_codecs]
|
|
8
|
+
return super unless cache && encoder.instance_of?(::Avro::IO::BinaryEncoder)
|
|
9
|
+
|
|
10
|
+
encoder.write(cache.fetch(writers_schema).encode(datum))
|
|
11
|
+
end
|
|
12
|
+
end
|
|
13
|
+
end
|
|
14
|
+
end
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Avrocadabra
|
|
4
|
+
module AvroTurf
|
|
5
|
+
class Mapping
|
|
6
|
+
attr_reader :native
|
|
7
|
+
|
|
8
|
+
def initialize(source, schemas)
|
|
9
|
+
nodes = {}.compare_by_identity
|
|
10
|
+
schemas.each do |schema|
|
|
11
|
+
adapter = schema.type_adapter
|
|
12
|
+
nodes[schema] = [schema, adapter == ::Avro::LogicalTypes::Identity ? nil : adapter]
|
|
13
|
+
end
|
|
14
|
+
nodes.each do |schema, node|
|
|
15
|
+
node.push(children(schema).map { nodes.fetch(it) }.freeze).freeze
|
|
16
|
+
end
|
|
17
|
+
@native = [self, nodes.fetch(source)].freeze
|
|
18
|
+
@reader = ::Avro::IO::DatumReader.new
|
|
19
|
+
end
|
|
20
|
+
|
|
21
|
+
def union_index(schema, value, budget)
|
|
22
|
+
schema.schemas.index { ::Avro::Schema.validate(it, value, avrocadabra_budget: budget) } ||
|
|
23
|
+
raise(encoding_error(schema, value))
|
|
24
|
+
end
|
|
25
|
+
|
|
26
|
+
def default_value(schema, name)
|
|
27
|
+
field = schema.fields_hash.fetch(name)
|
|
28
|
+
@reader.read_default_value(field.type, field.default)
|
|
29
|
+
end
|
|
30
|
+
|
|
31
|
+
def encoding_error(schema, value)
|
|
32
|
+
::Avro::IO::AvroTypeError.new(schema, value)
|
|
33
|
+
end
|
|
34
|
+
|
|
35
|
+
private
|
|
36
|
+
|
|
37
|
+
def children(schema)
|
|
38
|
+
case schema.type_sym
|
|
39
|
+
when :record, :error then schema.fields.map(&:type)
|
|
40
|
+
when :array then [schema.items]
|
|
41
|
+
when :map then [schema.values]
|
|
42
|
+
when :union then schema.schemas
|
|
43
|
+
else []
|
|
44
|
+
end
|
|
45
|
+
end
|
|
46
|
+
end
|
|
47
|
+
end
|
|
48
|
+
end
|