corvus_json_schema 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +7 -0
- data/Cargo.lock +390 -0
- data/Cargo.toml +8 -0
- data/LICENSE +201 -0
- data/README.md +111 -0
- data/VERSIONHISTORY.md +7 -0
- data/ext/corvus_json_schema/Cargo.toml +18 -0
- data/ext/corvus_json_schema/extconf.rb +6 -0
- data/ext/corvus_json_schema/rustfmt.toml +2 -0
- data/ext/corvus_json_schema/src/lib.rs +643 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/Cargo.toml +35 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/LICENSE +201 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/README.md +139 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/compiler.rs +1118 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/dialect.rs +108 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/document.rs +905 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/eval/plan/fused.rs +1187 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/eval/plan.rs +1728 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/eval.rs +1451 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/formats.rs +1021 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/instance.rs +228 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/lib.rs +189 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/loader.rs +403 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/metaschemas.rs +29 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/node.rs +417 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/numbers.rs +244 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/options.rs +108 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/pattern.rs +1591 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/results.rs +316 -0
- data/ext/corvus_json_schema/vendor/corvus-json-schema/src/uri.rs +274 -0
- data/lib/corvus_json_schema/version.rb +5 -0
- data/lib/corvus_json_schema.rb +65 -0
- metadata +94 -0
|
@@ -0,0 +1,1021 @@
|
|
|
1
|
+
//! Format assertions, applied only when `format` is asserted (the format-assertion vocabulary or `assert_format`).
|
|
2
|
+
//! A port of the TypeScript port's `formats.ts`, itself matching the C# `JsonSchemaEvaluation` format checks.
|
|
3
|
+
|
|
4
|
+
use std::sync::LazyLock;
|
|
5
|
+
|
|
6
|
+
use regex::Regex;
|
|
7
|
+
use serde_json::Number;
|
|
8
|
+
|
|
9
|
+
use crate::dialect::Dialect;
|
|
10
|
+
|
|
11
|
+
/// The format a dialect recognises for a `format` value (`SchemaCompiler.GetFormatKind`).
|
|
12
|
+
#[derive(Clone, Copy, PartialEq, Eq, Debug)]
|
|
13
|
+
pub(crate) enum FormatKind {
|
|
14
|
+
Unknown,
|
|
15
|
+
Date,
|
|
16
|
+
Time,
|
|
17
|
+
DateTime,
|
|
18
|
+
Duration,
|
|
19
|
+
Uuid,
|
|
20
|
+
Ipv4,
|
|
21
|
+
Ipv6,
|
|
22
|
+
Hostname,
|
|
23
|
+
IdnHostname,
|
|
24
|
+
Email,
|
|
25
|
+
IdnEmail,
|
|
26
|
+
Uri,
|
|
27
|
+
UriReference,
|
|
28
|
+
Iri,
|
|
29
|
+
IriReference,
|
|
30
|
+
UriTemplate,
|
|
31
|
+
JsonPointer,
|
|
32
|
+
RelativeJsonPointer,
|
|
33
|
+
Regex,
|
|
34
|
+
// Numeric formats (a Corvus extension).
|
|
35
|
+
Byte,
|
|
36
|
+
UInt16,
|
|
37
|
+
UInt32,
|
|
38
|
+
UInt64,
|
|
39
|
+
UInt128,
|
|
40
|
+
SByte,
|
|
41
|
+
Int16,
|
|
42
|
+
Int32,
|
|
43
|
+
Int64,
|
|
44
|
+
Int128,
|
|
45
|
+
Half,
|
|
46
|
+
Single,
|
|
47
|
+
Double,
|
|
48
|
+
Decimal,
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
impl FormatKind {
|
|
52
|
+
pub fn of(format: &str, dialect: Dialect) -> FormatKind {
|
|
53
|
+
use FormatKind::*;
|
|
54
|
+
let at_least = |d: Dialect, k: FormatKind| if dialect >= d { k } else { Unknown };
|
|
55
|
+
match format {
|
|
56
|
+
"float" | "single" => Single,
|
|
57
|
+
"byte" => Byte,
|
|
58
|
+
"uint16" => UInt16,
|
|
59
|
+
"uint32" => UInt32,
|
|
60
|
+
"uint64" => UInt64,
|
|
61
|
+
"uint128" => UInt128,
|
|
62
|
+
"sbyte" => SByte,
|
|
63
|
+
"int16" => Int16,
|
|
64
|
+
"int32" => Int32,
|
|
65
|
+
"int64" => Int64,
|
|
66
|
+
"int128" => Int128,
|
|
67
|
+
"half" => Half,
|
|
68
|
+
"double" => Double,
|
|
69
|
+
"decimal" => Decimal,
|
|
70
|
+
"date-time" => DateTime,
|
|
71
|
+
"email" => Email,
|
|
72
|
+
"hostname" => Hostname,
|
|
73
|
+
"ipv4" => Ipv4,
|
|
74
|
+
"ipv6" => Ipv6,
|
|
75
|
+
"uri" => Uri,
|
|
76
|
+
"uri-reference" => at_least(Dialect::Draft6, UriReference),
|
|
77
|
+
"uri-template" => at_least(Dialect::Draft6, UriTemplate),
|
|
78
|
+
"json-pointer" => at_least(Dialect::Draft6, JsonPointer),
|
|
79
|
+
"date" => at_least(Dialect::Draft7, Date),
|
|
80
|
+
"time" => at_least(Dialect::Draft7, Time),
|
|
81
|
+
"regex" => at_least(Dialect::Draft7, Regex),
|
|
82
|
+
"relative-json-pointer" => at_least(Dialect::Draft7, RelativeJsonPointer),
|
|
83
|
+
"idn-email" => at_least(Dialect::Draft7, IdnEmail),
|
|
84
|
+
"idn-hostname" => at_least(Dialect::Draft7, IdnHostname),
|
|
85
|
+
"iri" => at_least(Dialect::Draft7, Iri),
|
|
86
|
+
"iri-reference" => at_least(Dialect::Draft7, IriReference),
|
|
87
|
+
"duration" => at_least(Dialect::Draft201909, Duration),
|
|
88
|
+
"uuid" => at_least(Dialect::Draft201909, Uuid),
|
|
89
|
+
_ => Unknown,
|
|
90
|
+
}
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
pub fn is_numeric(self) -> bool {
|
|
94
|
+
use FormatKind::*;
|
|
95
|
+
matches!(
|
|
96
|
+
self,
|
|
97
|
+
Byte | UInt16
|
|
98
|
+
| UInt32
|
|
99
|
+
| UInt64
|
|
100
|
+
| UInt128
|
|
101
|
+
| SByte
|
|
102
|
+
| Int16
|
|
103
|
+
| Int32
|
|
104
|
+
| Int64
|
|
105
|
+
| Int128
|
|
106
|
+
| Half
|
|
107
|
+
| Single
|
|
108
|
+
| Double
|
|
109
|
+
| Decimal
|
|
110
|
+
)
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
/// The canonical name, for messages.
|
|
114
|
+
pub fn name(self) -> &'static str {
|
|
115
|
+
use FormatKind::*;
|
|
116
|
+
match self {
|
|
117
|
+
Unknown => "unknown",
|
|
118
|
+
Date => "date",
|
|
119
|
+
Time => "time",
|
|
120
|
+
DateTime => "date-time",
|
|
121
|
+
Duration => "duration",
|
|
122
|
+
Uuid => "uuid",
|
|
123
|
+
Ipv4 => "ipv4",
|
|
124
|
+
Ipv6 => "ipv6",
|
|
125
|
+
Hostname => "hostname",
|
|
126
|
+
IdnHostname => "idn-hostname",
|
|
127
|
+
Email => "email",
|
|
128
|
+
IdnEmail => "idn-email",
|
|
129
|
+
Uri => "uri",
|
|
130
|
+
UriReference => "uri-reference",
|
|
131
|
+
Iri => "iri",
|
|
132
|
+
IriReference => "iri-reference",
|
|
133
|
+
UriTemplate => "uri-template",
|
|
134
|
+
JsonPointer => "json-pointer",
|
|
135
|
+
RelativeJsonPointer => "relative-json-pointer",
|
|
136
|
+
Regex => "regex",
|
|
137
|
+
Byte => "byte",
|
|
138
|
+
UInt16 => "uint16",
|
|
139
|
+
UInt32 => "uint32",
|
|
140
|
+
UInt64 => "uint64",
|
|
141
|
+
UInt128 => "uint128",
|
|
142
|
+
SByte => "sbyte",
|
|
143
|
+
Int16 => "int16",
|
|
144
|
+
Int32 => "int32",
|
|
145
|
+
Int64 => "int64",
|
|
146
|
+
Int128 => "int128",
|
|
147
|
+
Half => "half",
|
|
148
|
+
Single => "single",
|
|
149
|
+
Double => "double",
|
|
150
|
+
Decimal => "decimal",
|
|
151
|
+
}
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
/// The message for a string format failure (Corvus.Text.Json Strings.resx).
|
|
155
|
+
pub fn message(self) -> Option<&'static str> {
|
|
156
|
+
use FormatKind::*;
|
|
157
|
+
Some(match self {
|
|
158
|
+
Date => "Expected an ISO8601 Date string.",
|
|
159
|
+
DateTime => "Expected an ISO8601 Offset DateTime string.",
|
|
160
|
+
Time => "Expected an ISO8601 Offset Time string.",
|
|
161
|
+
Duration => "Expected an ISO8601 Duration string.",
|
|
162
|
+
Email => "Expected an RFC5321 Section-4.1.2 Email string.",
|
|
163
|
+
IdnEmail => "Expected an RFC6531 IDN Email string.",
|
|
164
|
+
Hostname => "Expected an RFC1035 hostname.",
|
|
165
|
+
IdnHostname => "Expected an RFC5890 Section-2.3.2.3 IDN hostname.",
|
|
166
|
+
Ipv4 => "Expected an RFC2673 IP V4 address.",
|
|
167
|
+
Ipv6 => "Expected an RFC2373 IP V6 address.",
|
|
168
|
+
Uri => "Expected an absolute URI.",
|
|
169
|
+
UriReference => "Expected a URI reference.",
|
|
170
|
+
Iri => "Expected an absolute IRI.",
|
|
171
|
+
IriReference => "Expected an IRI reference.",
|
|
172
|
+
Uuid => "Expected an RFC4122 UUID.",
|
|
173
|
+
UriTemplate => "Expected an RFC6570 URI Template.",
|
|
174
|
+
JsonPointer => "Expected an RFC6901 JSON Pointer.",
|
|
175
|
+
RelativeJsonPointer => {
|
|
176
|
+
"Expected a Relative JSON Pointer. (https://json-schema.org/draft/2020-12/relative-json-pointer)."
|
|
177
|
+
}
|
|
178
|
+
Regex => "Expected a regular expression specification.",
|
|
179
|
+
_ => return None,
|
|
180
|
+
})
|
|
181
|
+
}
|
|
182
|
+
|
|
183
|
+
/// Asserts a string format. `legacy_hostname` selects the RFC 1123 host name rules of draft 4 and 6.
|
|
184
|
+
pub fn check_string(self, s: &str, legacy_hostname: bool) -> bool {
|
|
185
|
+
use FormatKind::*;
|
|
186
|
+
match self {
|
|
187
|
+
Date => date(s),
|
|
188
|
+
Time => time(s),
|
|
189
|
+
DateTime => date_time(s),
|
|
190
|
+
Duration => DURATION_RE.is_match(s),
|
|
191
|
+
Uuid => uuid(s),
|
|
192
|
+
Ipv4 => ipv4(s),
|
|
193
|
+
Ipv6 => ipv6(s),
|
|
194
|
+
Hostname => {
|
|
195
|
+
if legacy_hostname {
|
|
196
|
+
legacy_host_name(s)
|
|
197
|
+
} else {
|
|
198
|
+
hostname(s)
|
|
199
|
+
}
|
|
200
|
+
}
|
|
201
|
+
IdnHostname => idn_hostname(s),
|
|
202
|
+
Email => email(s, false),
|
|
203
|
+
IdnEmail => email(s, true),
|
|
204
|
+
Uri => URI_RE.is_match(s),
|
|
205
|
+
UriReference => URI_REF_RE.is_match(s),
|
|
206
|
+
Iri => IRI_RE.is_match(s),
|
|
207
|
+
IriReference => IRI_REF_RE.is_match(s),
|
|
208
|
+
UriTemplate => URI_TEMPLATE_RE.is_match(s),
|
|
209
|
+
JsonPointer => JSON_POINTER_RE.is_match(s),
|
|
210
|
+
RelativeJsonPointer => RELATIVE_JSON_POINTER_RE.is_match(s),
|
|
211
|
+
Regex => crate::pattern::is_valid_ecma_regex(s),
|
|
212
|
+
_ => true,
|
|
213
|
+
}
|
|
214
|
+
}
|
|
215
|
+
|
|
216
|
+
/// Asserts a numeric format.
|
|
217
|
+
pub fn check_number(self, n: &Number) -> bool {
|
|
218
|
+
use FormatKind::*;
|
|
219
|
+
let int_range = |min: i128, max: i128| -> bool {
|
|
220
|
+
if let Some(i) = n.as_i64() {
|
|
221
|
+
(i as i128) >= min && (i as i128) <= max
|
|
222
|
+
} else if let Some(u) = n.as_u64() {
|
|
223
|
+
(u as i128) >= min && (u as i128) <= max
|
|
224
|
+
} else {
|
|
225
|
+
let f = n.as_f64().unwrap_or(f64::NAN);
|
|
226
|
+
f.is_finite() && f.fract() == 0.0 && f >= min as f64 && f <= max as f64
|
|
227
|
+
}
|
|
228
|
+
};
|
|
229
|
+
let magnitude = |max: f64| n.as_f64().is_some_and(|f| f.is_finite() && f.abs() <= max);
|
|
230
|
+
match self {
|
|
231
|
+
Byte => int_range(0, 255),
|
|
232
|
+
UInt16 => int_range(0, 65535),
|
|
233
|
+
UInt32 => int_range(0, 4294967295),
|
|
234
|
+
UInt64 => int_range(0, u64::MAX as i128),
|
|
235
|
+
UInt128 => {
|
|
236
|
+
if let Some(i) = n.as_i64() {
|
|
237
|
+
i >= 0
|
|
238
|
+
} else if n.as_u64().is_some() {
|
|
239
|
+
true
|
|
240
|
+
} else {
|
|
241
|
+
let f = n.as_f64().unwrap_or(f64::NAN);
|
|
242
|
+
f.is_finite() && f.fract() == 0.0 && (0.0..=3.402_823_669_209_385e38).contains(&f)
|
|
243
|
+
}
|
|
244
|
+
}
|
|
245
|
+
SByte => int_range(-128, 127),
|
|
246
|
+
Int16 => int_range(-32768, 32767),
|
|
247
|
+
Int32 => int_range(-2147483648, 2147483647),
|
|
248
|
+
Int64 => int_range(i64::MIN as i128, i64::MAX as i128),
|
|
249
|
+
Int128 => {
|
|
250
|
+
if n.as_i64().is_some() || n.as_u64().is_some() {
|
|
251
|
+
true
|
|
252
|
+
} else {
|
|
253
|
+
let f = n.as_f64().unwrap_or(f64::NAN);
|
|
254
|
+
f.is_finite() && f.fract() == 0.0 && f.abs() <= 1.701_411_834_604_692_3e38
|
|
255
|
+
}
|
|
256
|
+
}
|
|
257
|
+
Half => magnitude(65504.0),
|
|
258
|
+
Single => magnitude(3.402_823_466_385_288_6e38),
|
|
259
|
+
Double => magnitude(f64::MAX),
|
|
260
|
+
Decimal => magnitude(7.922_816_251_426_434e28),
|
|
261
|
+
_ => true,
|
|
262
|
+
}
|
|
263
|
+
}
|
|
264
|
+
}
|
|
265
|
+
|
|
266
|
+
fn is_leap_year(y: u32) -> bool {
|
|
267
|
+
y % 4 == 0 && (y % 100 != 0 || y % 400 == 0)
|
|
268
|
+
}
|
|
269
|
+
|
|
270
|
+
fn digits(s: &[u8]) -> Option<u32> {
|
|
271
|
+
if s.is_empty() || !s.iter().all(u8::is_ascii_digit) {
|
|
272
|
+
return None;
|
|
273
|
+
}
|
|
274
|
+
Some(s.iter().fold(0u32, |a, &b| a * 10 + (b - b'0') as u32))
|
|
275
|
+
}
|
|
276
|
+
|
|
277
|
+
pub(crate) fn date(s: &str) -> bool {
|
|
278
|
+
let b = s.as_bytes();
|
|
279
|
+
if b.len() != 10 || b[4] != b'-' || b[7] != b'-' {
|
|
280
|
+
return false;
|
|
281
|
+
}
|
|
282
|
+
let (Some(y), Some(m), Some(d)) = (digits(&b[0..4]), digits(&b[5..7]), digits(&b[8..10])) else {
|
|
283
|
+
return false;
|
|
284
|
+
};
|
|
285
|
+
const DAYS: [u32; 13] = [0, 31, 28, 31, 30, 31, 30, 31, 31, 30, 31, 30, 31];
|
|
286
|
+
(1..=12).contains(&m) && d >= 1 && d <= if m == 2 && is_leap_year(y) { 29 } else { DAYS[m as usize] }
|
|
287
|
+
}
|
|
288
|
+
|
|
289
|
+
pub(crate) fn time(s: &str) -> bool {
|
|
290
|
+
let b = s.as_bytes();
|
|
291
|
+
if b.len() < 9 || b[2] != b':' || b[5] != b':' {
|
|
292
|
+
return false;
|
|
293
|
+
}
|
|
294
|
+
let (Some(h), Some(mi), Some(sec)) = (digits(&b[0..2]), digits(&b[3..5]), digits(&b[6..8])) else {
|
|
295
|
+
return false;
|
|
296
|
+
};
|
|
297
|
+
let mut i = 8;
|
|
298
|
+
if b[i] == b'.' {
|
|
299
|
+
i += 1;
|
|
300
|
+
let start = i;
|
|
301
|
+
while i < b.len() && b[i].is_ascii_digit() {
|
|
302
|
+
i += 1;
|
|
303
|
+
}
|
|
304
|
+
if i == start {
|
|
305
|
+
return false;
|
|
306
|
+
}
|
|
307
|
+
}
|
|
308
|
+
if i >= b.len() {
|
|
309
|
+
return false;
|
|
310
|
+
}
|
|
311
|
+
let (mut oh, mut om, mut sign) = (0u32, 0u32, 0i32);
|
|
312
|
+
match b[i] {
|
|
313
|
+
b'z' | b'Z' => {
|
|
314
|
+
if i + 1 != b.len() {
|
|
315
|
+
return false;
|
|
316
|
+
}
|
|
317
|
+
}
|
|
318
|
+
b'+' | b'-' => {
|
|
319
|
+
if b.len() - i != 6 || b[i + 3] != b':' {
|
|
320
|
+
return false;
|
|
321
|
+
}
|
|
322
|
+
sign = if b[i] == b'-' { 1 } else { -1 };
|
|
323
|
+
let (Some(x), Some(y)) = (digits(&b[i + 1..i + 3]), digits(&b[i + 4..i + 6])) else {
|
|
324
|
+
return false;
|
|
325
|
+
};
|
|
326
|
+
if x > 23 || y > 59 {
|
|
327
|
+
return false;
|
|
328
|
+
}
|
|
329
|
+
oh = x;
|
|
330
|
+
om = y;
|
|
331
|
+
}
|
|
332
|
+
_ => return false,
|
|
333
|
+
}
|
|
334
|
+
if h > 23 || mi > 59 || sec > 60 {
|
|
335
|
+
return false;
|
|
336
|
+
}
|
|
337
|
+
if sec == 60 {
|
|
338
|
+
// A leap second is only valid at 23:59:60 UTC.
|
|
339
|
+
let utc = (h * 60 + mi) as i32 + sign * (oh * 60 + om) as i32;
|
|
340
|
+
return utc.rem_euclid(1440) == 23 * 60 + 59;
|
|
341
|
+
}
|
|
342
|
+
true
|
|
343
|
+
}
|
|
344
|
+
|
|
345
|
+
pub(crate) fn date_time(s: &str) -> bool {
|
|
346
|
+
let b = s.as_bytes();
|
|
347
|
+
b.len() > 11 && (b[10] == b'T' || b[10] == b't') && s.is_char_boundary(10) && date(&s[..10]) && time(&s[11..])
|
|
348
|
+
}
|
|
349
|
+
|
|
350
|
+
static DURATION_RE: LazyLock<Regex> = LazyLock::new(|| {
|
|
351
|
+
let t = r"(?:[0-9]+H(?:[0-9]+M(?:[0-9]+S)?)?|[0-9]+M(?:[0-9]+S)?|[0-9]+S)";
|
|
352
|
+
Regex::new(&format!(
|
|
353
|
+
r"^P(?:[0-9]+W|(?:[0-9]+Y(?:[0-9]+M(?:[0-9]+D)?)?|[0-9]+M(?:[0-9]+D)?|[0-9]+D)(?:T{t})?|T{t})$"
|
|
354
|
+
))
|
|
355
|
+
.unwrap()
|
|
356
|
+
});
|
|
357
|
+
|
|
358
|
+
pub(crate) fn uuid(s: &str) -> bool {
|
|
359
|
+
let b = s.as_bytes();
|
|
360
|
+
b.len() == 36
|
|
361
|
+
&& b.iter()
|
|
362
|
+
.enumerate()
|
|
363
|
+
.all(|(i, &c)| if matches!(i, 8 | 13 | 18 | 23) { c == b'-' } else { c.is_ascii_hexdigit() })
|
|
364
|
+
}
|
|
365
|
+
|
|
366
|
+
pub(crate) fn ipv4(s: &str) -> bool {
|
|
367
|
+
let mut parts = 0;
|
|
368
|
+
for p in s.split('.') {
|
|
369
|
+
parts += 1;
|
|
370
|
+
let b = p.as_bytes();
|
|
371
|
+
if parts > 4
|
|
372
|
+
|| b.is_empty()
|
|
373
|
+
|| b.len() > 3
|
|
374
|
+
|| !b.iter().all(u8::is_ascii_digit)
|
|
375
|
+
|| (b.len() > 1 && b[0] == b'0')
|
|
376
|
+
{
|
|
377
|
+
return false;
|
|
378
|
+
}
|
|
379
|
+
if b.iter().fold(0u32, |v, &c| v * 10 + u32::from(c - b'0')) > 255 {
|
|
380
|
+
return false;
|
|
381
|
+
}
|
|
382
|
+
}
|
|
383
|
+
parts == 4
|
|
384
|
+
}
|
|
385
|
+
|
|
386
|
+
pub(crate) fn ipv6(s: &str) -> bool {
|
|
387
|
+
// A valid address is at most 51 characters (an IPv4 tail after six groups), so longer text fails without a copy.
|
|
388
|
+
if s.len() < 2 || s.len() > 64 || !s.bytes().all(|c| c.is_ascii_hexdigit() || c == b':' || c == b'.') {
|
|
389
|
+
return false;
|
|
390
|
+
}
|
|
391
|
+
// An IPv4 tail counts as two groups: check it, then read the address with "0:0" in its place.
|
|
392
|
+
let mut buf = [0u8; 67];
|
|
393
|
+
let tail = if s.contains('.') {
|
|
394
|
+
let Some(last_colon) = s.rfind(':') else {
|
|
395
|
+
return false;
|
|
396
|
+
};
|
|
397
|
+
if !ipv4(&s[last_colon + 1..]) {
|
|
398
|
+
return false;
|
|
399
|
+
}
|
|
400
|
+
let head = &s.as_bytes()[..=last_colon];
|
|
401
|
+
buf[..head.len()].copy_from_slice(head);
|
|
402
|
+
buf[head.len()..head.len() + 3].copy_from_slice(b"0:0");
|
|
403
|
+
std::str::from_utf8(&buf[..head.len() + 3]).unwrap()
|
|
404
|
+
} else {
|
|
405
|
+
s
|
|
406
|
+
};
|
|
407
|
+
let hex_ok = |p: &str| !p.is_empty() && p.len() <= 4 && p.bytes().all(|c| c.is_ascii_hexdigit());
|
|
408
|
+
// The number of groups, when every one is valid.
|
|
409
|
+
let groups = |x: &str| -> Option<usize> {
|
|
410
|
+
if x.is_empty() {
|
|
411
|
+
return Some(0);
|
|
412
|
+
}
|
|
413
|
+
let mut n = 0;
|
|
414
|
+
for p in x.split(':') {
|
|
415
|
+
if !hex_ok(p) {
|
|
416
|
+
return None;
|
|
417
|
+
}
|
|
418
|
+
n += 1;
|
|
419
|
+
}
|
|
420
|
+
Some(n)
|
|
421
|
+
};
|
|
422
|
+
if let Some(dbl) = tail.find("::") {
|
|
423
|
+
if tail[dbl + 1..].contains("::") {
|
|
424
|
+
return false;
|
|
425
|
+
}
|
|
426
|
+
return matches!((groups(&tail[..dbl]), groups(&tail[dbl + 2..])), (Some(l), Some(r)) if l + r < 8);
|
|
427
|
+
}
|
|
428
|
+
groups(tail) == Some(8)
|
|
429
|
+
}
|
|
430
|
+
|
|
431
|
+
fn ldh_label(l: &str) -> bool {
|
|
432
|
+
let b = l.as_bytes();
|
|
433
|
+
!b.is_empty()
|
|
434
|
+
&& b.len() <= 63
|
|
435
|
+
&& b[0].is_ascii_alphanumeric()
|
|
436
|
+
&& b[b.len() - 1].is_ascii_alphanumeric()
|
|
437
|
+
&& b.iter().all(|&c| c.is_ascii_alphanumeric() || c == b'-')
|
|
438
|
+
}
|
|
439
|
+
|
|
440
|
+
fn starts_with_xn(l: &str) -> bool {
|
|
441
|
+
l.len() >= 4 && l.as_bytes()[..4].eq_ignore_ascii_case(b"xn--")
|
|
442
|
+
}
|
|
443
|
+
|
|
444
|
+
fn hostname_label_ok(label: &str) -> bool {
|
|
445
|
+
if !ldh_label(label) {
|
|
446
|
+
return false;
|
|
447
|
+
}
|
|
448
|
+
// "--" in the third and fourth positions is reserved for A-labels (xn--).
|
|
449
|
+
let b = label.as_bytes();
|
|
450
|
+
!(b.len() >= 4 && b[2] == b'-' && b[3] == b'-' && !starts_with_xn(label))
|
|
451
|
+
}
|
|
452
|
+
|
|
453
|
+
/// RFC 1123 host names (draft 4 and 6): no IDNA rules for "--" or A-labels.
|
|
454
|
+
pub(crate) fn legacy_host_name(s: &str) -> bool {
|
|
455
|
+
!s.is_empty() && s.len() <= 253 && s.split('.').all(ldh_label)
|
|
456
|
+
}
|
|
457
|
+
|
|
458
|
+
pub(crate) fn hostname(s: &str) -> bool {
|
|
459
|
+
if s.is_empty() || s.len() > 253 {
|
|
460
|
+
return false;
|
|
461
|
+
}
|
|
462
|
+
s.split('.').all(|l| hostname_label_ok(l) && (!starts_with_xn(l) || punycode_label_ok(&l[4..])))
|
|
463
|
+
}
|
|
464
|
+
|
|
465
|
+
// Code points IDNA2008 disallows that the tests exercise: controls, format characters, spaces, unassigned,
|
|
466
|
+
// uppercase and titlecase letters (mapped away, never PVALID), and symbols.
|
|
467
|
+
static DISALLOWED_RE: LazyLock<Regex> =
|
|
468
|
+
LazyLock::new(|| Regex::new(r"[\p{Cc}\p{Co}\p{Cn}\p{Zs}\p{Zl}\p{Zp}\p{Lu}\p{Lt}\p{Sm}\p{So}\p{P}]").unwrap());
|
|
469
|
+
const DISALLOWED_EXCEPTIONS: [char; 7] =
|
|
470
|
+
['\u{06fd}', '\u{06fe}', '\u{0f0b}', '\u{00b7}', '\u{05f3}', '\u{05f4}', '\u{30fb}'];
|
|
471
|
+
|
|
472
|
+
fn disallowed(label: &str) -> bool {
|
|
473
|
+
let stripped: String = label.chars().filter(|c| !DISALLOWED_EXCEPTIONS.contains(c) && *c != '-').collect();
|
|
474
|
+
DISALLOWED_RE.is_match(&stripped)
|
|
475
|
+
}
|
|
476
|
+
|
|
477
|
+
fn punycode_label_ok(encoded: &str) -> bool {
|
|
478
|
+
let Some(decoded) = punycode_decode(encoded) else {
|
|
479
|
+
return false;
|
|
480
|
+
};
|
|
481
|
+
if decoded.is_empty() || decoded.is_ascii() {
|
|
482
|
+
return false;
|
|
483
|
+
}
|
|
484
|
+
// The encoding must be canonical: re-encoding the U-label gives the same A-label.
|
|
485
|
+
if punycode_encode(&decoded) != encoded.to_ascii_lowercase() {
|
|
486
|
+
return false;
|
|
487
|
+
}
|
|
488
|
+
idn_label_ok(&decoded) && !disallowed(&decoded) && bidi_label_ok(&decoded, bidi_domain(&decoded))
|
|
489
|
+
}
|
|
490
|
+
|
|
491
|
+
const BASE: u32 = 36;
|
|
492
|
+
const T_MIN: u32 = 1;
|
|
493
|
+
const T_MAX: u32 = 26;
|
|
494
|
+
const SKEW: u32 = 38;
|
|
495
|
+
const DAMP: u32 = 700;
|
|
496
|
+
|
|
497
|
+
fn adapt(mut delta: u32, num_points: u32, first_time: bool) -> u32 {
|
|
498
|
+
delta = if first_time { delta / DAMP } else { delta >> 1 };
|
|
499
|
+
delta += delta / num_points;
|
|
500
|
+
let mut k = 0;
|
|
501
|
+
while delta > ((BASE - T_MIN) * T_MAX) >> 1 {
|
|
502
|
+
delta /= BASE - T_MIN;
|
|
503
|
+
k += BASE;
|
|
504
|
+
}
|
|
505
|
+
k + ((BASE - T_MIN + 1) * delta) / (delta + SKEW)
|
|
506
|
+
}
|
|
507
|
+
|
|
508
|
+
// RFC 3492 encoding, for the canonical round-trip and A-label length checks.
|
|
509
|
+
fn punycode_encode(input: &str) -> String {
|
|
510
|
+
let cps: Vec<u32> = input.chars().map(|c| c as u32).collect();
|
|
511
|
+
let mut out: String = input.chars().filter(|c| c.is_ascii()).collect();
|
|
512
|
+
let basic_length = out.len() as u32;
|
|
513
|
+
let mut h = basic_length;
|
|
514
|
+
if basic_length > 0 {
|
|
515
|
+
out.push('-');
|
|
516
|
+
}
|
|
517
|
+
let (mut n, mut delta, mut bias) = (128u32, 0u32, 72u32);
|
|
518
|
+
let digit = |d: u32| (if d < 26 { d + 97 } else { d + 22 }) as u8 as char;
|
|
519
|
+
while (h as usize) < cps.len() {
|
|
520
|
+
let m = cps.iter().copied().filter(|&c| c >= n).min().unwrap();
|
|
521
|
+
delta = delta.saturating_add((m - n).saturating_mul(h + 1));
|
|
522
|
+
n = m;
|
|
523
|
+
for &c in &cps {
|
|
524
|
+
if c < n {
|
|
525
|
+
delta = delta.saturating_add(1);
|
|
526
|
+
}
|
|
527
|
+
if c == n {
|
|
528
|
+
let mut q = delta;
|
|
529
|
+
let mut k = BASE;
|
|
530
|
+
loop {
|
|
531
|
+
let t = if k <= bias {
|
|
532
|
+
T_MIN
|
|
533
|
+
} else if k >= bias + T_MAX {
|
|
534
|
+
T_MAX
|
|
535
|
+
} else {
|
|
536
|
+
k - bias
|
|
537
|
+
};
|
|
538
|
+
if q < t {
|
|
539
|
+
break;
|
|
540
|
+
}
|
|
541
|
+
out.push(digit(t + (q - t) % (BASE - t)));
|
|
542
|
+
q = (q - t) / (BASE - t);
|
|
543
|
+
k += BASE;
|
|
544
|
+
}
|
|
545
|
+
out.push(digit(q));
|
|
546
|
+
bias = adapt(delta, h + 1, h == basic_length);
|
|
547
|
+
delta = 0;
|
|
548
|
+
h += 1;
|
|
549
|
+
}
|
|
550
|
+
}
|
|
551
|
+
delta += 1;
|
|
552
|
+
n += 1;
|
|
553
|
+
}
|
|
554
|
+
out
|
|
555
|
+
}
|
|
556
|
+
|
|
557
|
+
// RFC 3492 decoding, enough to validate A-labels.
|
|
558
|
+
fn punycode_decode(input: &str) -> Option<String> {
|
|
559
|
+
let bytes = input.as_bytes();
|
|
560
|
+
let (mut n, mut i, mut bias) = (128u32, 0u32, 72u32);
|
|
561
|
+
let mut output: Vec<u32> = Vec::new();
|
|
562
|
+
let basic = input.rfind('-').unwrap_or(0);
|
|
563
|
+
for &c in &bytes[..basic] {
|
|
564
|
+
if c >= 0x80 {
|
|
565
|
+
return None;
|
|
566
|
+
}
|
|
567
|
+
output.push(c as u32);
|
|
568
|
+
}
|
|
569
|
+
let mut index = if basic > 0 { basic + 1 } else { 0 };
|
|
570
|
+
while index < bytes.len() {
|
|
571
|
+
let oldi = i;
|
|
572
|
+
let mut w = 1u32;
|
|
573
|
+
let mut k = BASE;
|
|
574
|
+
loop {
|
|
575
|
+
let c = *bytes.get(index)? as u32;
|
|
576
|
+
index += 1;
|
|
577
|
+
let digit = if c.wrapping_sub(48) < 10 {
|
|
578
|
+
c - 22
|
|
579
|
+
} else if c.wrapping_sub(65) < 26 {
|
|
580
|
+
c - 65
|
|
581
|
+
} else if c.wrapping_sub(97) < 26 {
|
|
582
|
+
c - 97
|
|
583
|
+
} else {
|
|
584
|
+
BASE
|
|
585
|
+
};
|
|
586
|
+
if digit >= BASE {
|
|
587
|
+
return None;
|
|
588
|
+
}
|
|
589
|
+
i = i.checked_add(digit.checked_mul(w)?)?;
|
|
590
|
+
let t = if k <= bias {
|
|
591
|
+
T_MIN
|
|
592
|
+
} else if k >= bias + T_MAX {
|
|
593
|
+
T_MAX
|
|
594
|
+
} else {
|
|
595
|
+
k - bias
|
|
596
|
+
};
|
|
597
|
+
if digit < t {
|
|
598
|
+
break;
|
|
599
|
+
}
|
|
600
|
+
w = w.checked_mul(BASE - t)?;
|
|
601
|
+
k += BASE;
|
|
602
|
+
}
|
|
603
|
+
let len = output.len() as u32 + 1;
|
|
604
|
+
bias = adapt(i - oldi, len, oldi == 0);
|
|
605
|
+
n = n.checked_add(i / len)?;
|
|
606
|
+
i %= len;
|
|
607
|
+
if n > 0x10ffff {
|
|
608
|
+
return None;
|
|
609
|
+
}
|
|
610
|
+
output.insert(i as usize, n);
|
|
611
|
+
i += 1;
|
|
612
|
+
}
|
|
613
|
+
output.into_iter().map(char::from_u32).collect()
|
|
614
|
+
}
|
|
615
|
+
|
|
616
|
+
static SCRIPT_GREEK: LazyLock<Regex> = LazyLock::new(|| Regex::new(r"^\p{Greek}$").unwrap());
|
|
617
|
+
static SCRIPT_HEBREW: LazyLock<Regex> = LazyLock::new(|| Regex::new(r"^\p{Hebrew}$").unwrap());
|
|
618
|
+
static KANA_HAN: LazyLock<Regex> = LazyLock::new(|| Regex::new(r"^[\p{Hiragana}\p{Katakana}\p{Han}]$").unwrap());
|
|
619
|
+
static MN: LazyLock<Regex> = LazyLock::new(|| Regex::new(r"^\p{Mn}$").unwrap());
|
|
620
|
+
static MN_ME: LazyLock<Regex> = LazyLock::new(|| Regex::new(r"^[\p{Mn}\p{Me}]$").unwrap());
|
|
621
|
+
static STARTS_WITH_MARK: LazyLock<Regex> = LazyLock::new(|| Regex::new(r"^\p{M}").unwrap());
|
|
622
|
+
static ARABIC_LIKE: LazyLock<Regex> =
|
|
623
|
+
LazyLock::new(|| Regex::new(r"^[\p{Arabic}\p{Syriac}\p{Thaana}\p{Nko}]$").unwrap());
|
|
624
|
+
static LETTER_OR_MC: LazyLock<Regex> = LazyLock::new(|| Regex::new(r"^[\p{L}\p{Mc}]$").unwrap());
|
|
625
|
+
static RTL: LazyLock<Regex> = LazyLock::new(|| {
|
|
626
|
+
Regex::new(r"[\p{Hebrew}\p{Arabic}\p{Syriac}\p{Thaana}\p{Nko}\x{0660}-\x{0669}\x{066b}\x{066c}]").unwrap()
|
|
627
|
+
});
|
|
628
|
+
static CONTROL_FORMAT_SPACE: LazyLock<Regex> = LazyLock::new(|| Regex::new(r"[\p{Cc}\p{Cf}\p{Zs}\p{Cn}]").unwrap());
|
|
629
|
+
|
|
630
|
+
fn is(re: &Regex, c: char) -> bool {
|
|
631
|
+
let mut buf = [0u8; 4];
|
|
632
|
+
re.is_match(c.encode_utf8(&mut buf))
|
|
633
|
+
}
|
|
634
|
+
|
|
635
|
+
#[derive(Clone, Copy, PartialEq, Eq)]
|
|
636
|
+
enum Bidi {
|
|
637
|
+
L,
|
|
638
|
+
R,
|
|
639
|
+
AL,
|
|
640
|
+
AN,
|
|
641
|
+
EN,
|
|
642
|
+
Nsm,
|
|
643
|
+
ON,
|
|
644
|
+
}
|
|
645
|
+
|
|
646
|
+
// RFC 5893 Bidi rule, with Bidi classes approximated by script and general category.
|
|
647
|
+
fn bidi_class(c: char) -> Bidi {
|
|
648
|
+
let cp = c as u32;
|
|
649
|
+
if is(&MN_ME, c) {
|
|
650
|
+
return Bidi::Nsm;
|
|
651
|
+
}
|
|
652
|
+
if (0x660..=0x669).contains(&cp) || cp == 0x66b || cp == 0x66c {
|
|
653
|
+
return Bidi::AN;
|
|
654
|
+
}
|
|
655
|
+
if (0x30..=0x39).contains(&cp) || (0x6f0..=0x6f9).contains(&cp) {
|
|
656
|
+
return Bidi::EN;
|
|
657
|
+
}
|
|
658
|
+
if is(&SCRIPT_HEBREW, c) {
|
|
659
|
+
return Bidi::R;
|
|
660
|
+
}
|
|
661
|
+
if is(&ARABIC_LIKE, c) {
|
|
662
|
+
return Bidi::AL;
|
|
663
|
+
}
|
|
664
|
+
if is(&LETTER_OR_MC, c) {
|
|
665
|
+
return Bidi::L;
|
|
666
|
+
}
|
|
667
|
+
Bidi::ON
|
|
668
|
+
}
|
|
669
|
+
|
|
670
|
+
fn bidi_domain(label: &str) -> bool {
|
|
671
|
+
RTL.is_match(label) && label.chars().any(|c| matches!(bidi_class(c), Bidi::R | Bidi::AL | Bidi::AN))
|
|
672
|
+
}
|
|
673
|
+
|
|
674
|
+
fn bidi_label_ok(label: &str, is_bidi_domain: bool) -> bool {
|
|
675
|
+
if !is_bidi_domain {
|
|
676
|
+
return true;
|
|
677
|
+
}
|
|
678
|
+
let classes: Vec<Bidi> = label.chars().map(bidi_class).collect();
|
|
679
|
+
let Some(&first) = classes.first() else {
|
|
680
|
+
return false;
|
|
681
|
+
};
|
|
682
|
+
let mut last = classes.len() - 1;
|
|
683
|
+
while last > 0 && classes[last] == Bidi::Nsm {
|
|
684
|
+
last -= 1;
|
|
685
|
+
}
|
|
686
|
+
match first {
|
|
687
|
+
Bidi::R | Bidi::AL => {
|
|
688
|
+
!classes.contains(&Bidi::L)
|
|
689
|
+
&& matches!(classes[last], Bidi::R | Bidi::AL | Bidi::EN | Bidi::AN)
|
|
690
|
+
&& !(classes.contains(&Bidi::EN) && classes.contains(&Bidi::AN))
|
|
691
|
+
}
|
|
692
|
+
Bidi::L => {
|
|
693
|
+
!classes.iter().any(|c| matches!(c, Bidi::R | Bidi::AL | Bidi::AN))
|
|
694
|
+
&& matches!(classes[last], Bidi::L | Bidi::EN)
|
|
695
|
+
}
|
|
696
|
+
_ => false,
|
|
697
|
+
}
|
|
698
|
+
}
|
|
699
|
+
|
|
700
|
+
const VIRAMAS: [u32; 33] = [
|
|
701
|
+
0x094d, 0x09cd, 0x0a4d, 0x0acd, 0x0b4d, 0x0bcd, 0x0c4d, 0x0ccd, 0x0d3b, 0x0d3c, 0x0d4d, 0x0dca, 0x0e3a, 0x0eba,
|
|
702
|
+
0x0f84, 0x1039, 0x103a, 0x1714, 0x1734, 0x17d2, 0x1a60, 0x1b44, 0x1baa, 0x1bab, 0x1bf2, 0x1bf3, 0x2d7f, 0xa806,
|
|
703
|
+
0xa8c4, 0xa953, 0xa9c0, 0xaaf6, 0xabed,
|
|
704
|
+
];
|
|
705
|
+
|
|
706
|
+
fn zwnj_joining_context(cps: &[char], i: usize) -> bool {
|
|
707
|
+
// (Joining_Type:{L,D})(Joining_Type:T)*ZWNJ(Joining_Type:T)*(Joining_Type:{R,D}) approximated with Arabic letters.
|
|
708
|
+
let is_joiner = |c: char| {
|
|
709
|
+
let c = c as u32;
|
|
710
|
+
(0x0620..=0x064a).contains(&c) || (0x066e..=0x06d3).contains(&c)
|
|
711
|
+
};
|
|
712
|
+
let mut l = i as isize - 1;
|
|
713
|
+
while l >= 0 && is(&MN, cps[l as usize]) {
|
|
714
|
+
l -= 1;
|
|
715
|
+
}
|
|
716
|
+
let mut r = i + 1;
|
|
717
|
+
while r < cps.len() && is(&MN, cps[r]) {
|
|
718
|
+
r += 1;
|
|
719
|
+
}
|
|
720
|
+
l >= 0 && r < cps.len() && is_joiner(cps[l as usize]) && is_joiner(cps[r])
|
|
721
|
+
}
|
|
722
|
+
|
|
723
|
+
// Contextual and disallowed code points from RFC 5892 that the test suite exercises.
|
|
724
|
+
fn idn_label_ok(label: &str) -> bool {
|
|
725
|
+
if label.is_empty() || label.starts_with('-') || label.ends_with('-') {
|
|
726
|
+
return false;
|
|
727
|
+
}
|
|
728
|
+
let cps: Vec<char> = label.chars().collect();
|
|
729
|
+
if cps.len() >= 4 && cps[2] == '-' && cps[3] == '-' {
|
|
730
|
+
return false;
|
|
731
|
+
}
|
|
732
|
+
if STARTS_WITH_MARK.is_match(label) {
|
|
733
|
+
return false;
|
|
734
|
+
}
|
|
735
|
+
let has = |lo: u32, hi: u32| cps.iter().any(|&c| (lo..=hi).contains(&(c as u32)));
|
|
736
|
+
if has(0x660, 0x669) && has(0x6f0, 0x6f9) {
|
|
737
|
+
return false;
|
|
738
|
+
}
|
|
739
|
+
for i in 0..cps.len() {
|
|
740
|
+
match cps[i] as u32 {
|
|
741
|
+
0x302e | 0x302f | 0x0640 | 0x07fa | 0x3031..=0x3035 | 0x303b => return false,
|
|
742
|
+
0x00b7 => {
|
|
743
|
+
if !(i > 0 && i < cps.len() - 1 && cps[i - 1] == 'l' && cps[i + 1] == 'l') {
|
|
744
|
+
return false;
|
|
745
|
+
}
|
|
746
|
+
}
|
|
747
|
+
0x0375 => {
|
|
748
|
+
if !(i < cps.len() - 1 && is(&SCRIPT_GREEK, cps[i + 1])) {
|
|
749
|
+
return false;
|
|
750
|
+
}
|
|
751
|
+
}
|
|
752
|
+
0x05f3 | 0x05f4 => {
|
|
753
|
+
if !(i > 0 && is(&SCRIPT_HEBREW, cps[i - 1])) {
|
|
754
|
+
return false;
|
|
755
|
+
}
|
|
756
|
+
}
|
|
757
|
+
0x30fb => {
|
|
758
|
+
if !cps.iter().any(|&d| d != '\u{30fb}' && is(&KANA_HAN, d)) {
|
|
759
|
+
return false;
|
|
760
|
+
}
|
|
761
|
+
}
|
|
762
|
+
0x200d => {
|
|
763
|
+
if i == 0 || !VIRAMAS.contains(&(cps[i - 1] as u32)) {
|
|
764
|
+
return false;
|
|
765
|
+
}
|
|
766
|
+
}
|
|
767
|
+
0x200c if (i == 0 || !VIRAMAS.contains(&(cps[i - 1] as u32))) && !zwnj_joining_context(&cps, i) => {
|
|
768
|
+
return false;
|
|
769
|
+
}
|
|
770
|
+
_ => {}
|
|
771
|
+
}
|
|
772
|
+
}
|
|
773
|
+
true
|
|
774
|
+
}
|
|
775
|
+
|
|
776
|
+
pub(crate) fn idn_hostname(s: &str) -> bool {
|
|
777
|
+
if s.is_empty() {
|
|
778
|
+
return false;
|
|
779
|
+
}
|
|
780
|
+
// Label separators: full stop, ideographic full stop, fullwidth full stop, halfwidth ideographic full stop.
|
|
781
|
+
let labels: Vec<&str> = s.split(['.', '\u{3002}', '\u{ff0e}', '\u{ff61}']).collect();
|
|
782
|
+
let unicode_labels: Vec<String> =
|
|
783
|
+
labels
|
|
784
|
+
.iter()
|
|
785
|
+
.map(|l| {
|
|
786
|
+
if starts_with_xn(l) {
|
|
787
|
+
punycode_decode(&l[4..]).unwrap_or_else(|| l.to_string())
|
|
788
|
+
} else {
|
|
789
|
+
l.to_string()
|
|
790
|
+
}
|
|
791
|
+
})
|
|
792
|
+
.collect();
|
|
793
|
+
let is_bidi = unicode_labels.iter().any(|l| bidi_domain(l));
|
|
794
|
+
let mut ascii_length = 0usize;
|
|
795
|
+
for (i, label) in labels.iter().enumerate() {
|
|
796
|
+
if label.is_empty() {
|
|
797
|
+
return false;
|
|
798
|
+
}
|
|
799
|
+
if label.is_ascii() {
|
|
800
|
+
if !hostname_label_ok(label) {
|
|
801
|
+
return false;
|
|
802
|
+
}
|
|
803
|
+
if starts_with_xn(label) && !punycode_label_ok(&label[4..]) {
|
|
804
|
+
return false;
|
|
805
|
+
}
|
|
806
|
+
if !bidi_label_ok(&unicode_labels[i], is_bidi) {
|
|
807
|
+
return false;
|
|
808
|
+
}
|
|
809
|
+
ascii_length += label.len() + 1;
|
|
810
|
+
} else {
|
|
811
|
+
if !idn_label_ok(label) {
|
|
812
|
+
return false;
|
|
813
|
+
}
|
|
814
|
+
let without_joiners: String = label.chars().filter(|&c| c != '\u{200c}' && c != '\u{200d}').collect();
|
|
815
|
+
if CONTROL_FORMAT_SPACE.is_match(&without_joiners) || disallowed(&without_joiners) {
|
|
816
|
+
return false;
|
|
817
|
+
}
|
|
818
|
+
if !bidi_label_ok(label, is_bidi) {
|
|
819
|
+
return false;
|
|
820
|
+
}
|
|
821
|
+
let a_label_len = 4 + punycode_encode(label).len();
|
|
822
|
+
if a_label_len > 63 {
|
|
823
|
+
return false;
|
|
824
|
+
}
|
|
825
|
+
ascii_length += a_label_len + 1;
|
|
826
|
+
}
|
|
827
|
+
}
|
|
828
|
+
ascii_length - 1 <= 253
|
|
829
|
+
}
|
|
830
|
+
|
|
831
|
+
static EMAIL_LOCAL_RE: LazyLock<Regex> = LazyLock::new(|| {
|
|
832
|
+
Regex::new(r#"^(?:[A-Za-z0-9!#$%&'*+/=?^_`{|}~-]+(?:\.[A-Za-z0-9!#$%&'*+/=?^_`{|}~-]+)*|"(?:[^"\\\r\n]|\\.)*")$"#)
|
|
833
|
+
.unwrap()
|
|
834
|
+
});
|
|
835
|
+
static IDN_EMAIL_LOCAL_RE: LazyLock<Regex> = LazyLock::new(|| {
|
|
836
|
+
Regex::new(
|
|
837
|
+
r#"^(?:[\p{L}\p{M}\p{N}!#$%&'*+/=?^_`{|}~-]+(?:\.[\p{L}\p{M}\p{N}!#$%&'*+/=?^_`{|}~-]+)*|"(?:[^"\\\r\n]|\\.)*")$"#,
|
|
838
|
+
)
|
|
839
|
+
.unwrap()
|
|
840
|
+
});
|
|
841
|
+
|
|
842
|
+
fn email(s: &str, idn: bool) -> bool {
|
|
843
|
+
let Some(at) = s.rfind('@') else {
|
|
844
|
+
return false;
|
|
845
|
+
};
|
|
846
|
+
if at == 0 {
|
|
847
|
+
return false;
|
|
848
|
+
}
|
|
849
|
+
let local = &s[..at];
|
|
850
|
+
let domain = &s[at + 1..];
|
|
851
|
+
let local_ok = if idn { IDN_EMAIL_LOCAL_RE.is_match(local) } else { EMAIL_LOCAL_RE.is_match(local) };
|
|
852
|
+
if !local_ok {
|
|
853
|
+
return false;
|
|
854
|
+
}
|
|
855
|
+
if domain.starts_with('[') && domain.ends_with(']') && domain.len() >= 2 {
|
|
856
|
+
let inner = &domain[1..domain.len() - 1];
|
|
857
|
+
if inner.len() >= 5 && inner.as_bytes()[..5].eq_ignore_ascii_case(b"IPv6:") {
|
|
858
|
+
return ipv6(&inner[5..]);
|
|
859
|
+
}
|
|
860
|
+
return ipv4(inner);
|
|
861
|
+
}
|
|
862
|
+
if idn { idn_hostname(domain) } else { hostname(domain) }
|
|
863
|
+
}
|
|
864
|
+
|
|
865
|
+
// RFC 3986 (URI) and RFC 3987 (IRI) grammars.
|
|
866
|
+
fn uri_regex(iri: bool, reference: bool) -> Regex {
|
|
867
|
+
const HEX: &str = "[0-9A-Fa-f]";
|
|
868
|
+
let pct = format!("%{HEX}{{2}}");
|
|
869
|
+
const SUB: &str = r"[!$&'()*+,;=]";
|
|
870
|
+
const UNRESERVED: &str = r"[A-Za-z0-9\-._~]";
|
|
871
|
+
const UCSCHAR: &str = r"[\x{A0}-\x{D7FF}\x{F900}-\x{FDCF}\x{FDF0}-\x{FFEF}\x{10000}-\x{EFFFD}]";
|
|
872
|
+
let unreserved = if iri { format!("(?:{UNRESERVED}|{UCSCHAR})") } else { UNRESERVED.to_string() };
|
|
873
|
+
let pchar = format!("(?:{unreserved}|{pct}|{SUB}|[:@])");
|
|
874
|
+
let query = if iri {
|
|
875
|
+
format!(r"(?:{pchar}|[/?]|[\x{{E000}}-\x{{F8FF}}\x{{F0000}}-\x{{FFFFD}}\x{{100000}}-\x{{10FFFD}}])*")
|
|
876
|
+
} else {
|
|
877
|
+
format!("(?:{pchar}|[/?])*")
|
|
878
|
+
};
|
|
879
|
+
let fragment = format!("(?:{pchar}|[/?])*");
|
|
880
|
+
let dec_octet = "(?:25[0-5]|2[0-4][0-9]|1[0-9][0-9]|[1-9]?[0-9])";
|
|
881
|
+
let ipv4 = format!(r"{dec_octet}(?:\.{dec_octet}){{3}}");
|
|
882
|
+
let h16 = format!("{HEX}{{1,4}}");
|
|
883
|
+
let ls32 = format!("(?:{h16}:{h16}|{ipv4})");
|
|
884
|
+
let ipv6 = format!(
|
|
885
|
+
"(?:(?:{h16}:){{6}}{ls32}|::(?:{h16}:){{5}}{ls32}|(?:{h16})?::(?:{h16}:){{4}}{ls32}|(?:(?:{h16}:){{0,1}}{h16})?::(?:{h16}:){{3}}{ls32}\
|
|
886
|
+
|(?:(?:{h16}:){{0,2}}{h16})?::(?:{h16}:){{2}}{ls32}|(?:(?:{h16}:){{0,3}}{h16})?::{h16}:{ls32}|(?:(?:{h16}:){{0,4}}{h16})?::{ls32}\
|
|
887
|
+
|(?:(?:{h16}:){{0,5}}{h16})?::{h16}|(?:(?:{h16}:){{0,6}}{h16})?::)"
|
|
888
|
+
);
|
|
889
|
+
let ip_literal = format!(r"\[(?:{ipv6}|v{HEX}+\.(?:{UNRESERVED}|{SUB}|:)+)\]");
|
|
890
|
+
let reg_name = format!("(?:{unreserved}|{pct}|{SUB})*");
|
|
891
|
+
let authority = format!("(?:(?:{unreserved}|{pct}|{SUB}|:)*@)?(?:{ip_literal}|{ipv4}|{reg_name})(?::[0-9]*)?");
|
|
892
|
+
let segment = format!("{pchar}*");
|
|
893
|
+
let segment_nz = format!("{pchar}+");
|
|
894
|
+
let segment_nz_nc = format!("(?:{unreserved}|{pct}|{SUB}|@)+");
|
|
895
|
+
let hier_part =
|
|
896
|
+
format!("(?://{authority}(?:/{segment})*|/(?:{segment_nz}(?:/{segment})*)?|{segment_nz}(?:/{segment})*|)");
|
|
897
|
+
let relative_part =
|
|
898
|
+
format!("(?://{authority}(?:/{segment})*|/(?:{segment_nz}(?:/{segment})*)?|{segment_nz_nc}(?:/{segment})*|)");
|
|
899
|
+
let scheme = r"[A-Za-z][A-Za-z0-9+\-.]*";
|
|
900
|
+
let absolute = format!(r"{scheme}:{hier_part}(?:\?{query})?(?:#{fragment})?");
|
|
901
|
+
let relative = format!(r"{relative_part}(?:\?{query})?(?:#{fragment})?");
|
|
902
|
+
let body = if reference { format!("{absolute}|{relative}") } else { absolute };
|
|
903
|
+
Regex::new(&format!("^(?:{body})$")).unwrap()
|
|
904
|
+
}
|
|
905
|
+
|
|
906
|
+
static URI_RE: LazyLock<Regex> = LazyLock::new(|| uri_regex(false, false));
|
|
907
|
+
static URI_REF_RE: LazyLock<Regex> = LazyLock::new(|| uri_regex(false, true));
|
|
908
|
+
static IRI_RE: LazyLock<Regex> = LazyLock::new(|| uri_regex(true, false));
|
|
909
|
+
static IRI_REF_RE: LazyLock<Regex> = LazyLock::new(|| uri_regex(true, true));
|
|
910
|
+
|
|
911
|
+
static URI_TEMPLATE_RE: LazyLock<Regex> = LazyLock::new(|| {
|
|
912
|
+
let var = "(?:[A-Za-z0-9_]|%[0-9A-Fa-f]{2})(?:\\.?(?:[A-Za-z0-9_]|%[0-9A-Fa-f]{2}))*(?::[1-9][0-9]{0,3}|\\*)?";
|
|
913
|
+
Regex::new(&format!(r#"^(?:[^\x00-\x20"'<>\\^`{{|}}]|\{{[+#./;?&=,!@|]?{var}(?:,{var})*\}})*$"#)).unwrap()
|
|
914
|
+
});
|
|
915
|
+
static JSON_POINTER_RE: LazyLock<Regex> = LazyLock::new(|| Regex::new(r"^(?:/(?:[^~/]|~[01])*)*$").unwrap());
|
|
916
|
+
static RELATIVE_JSON_POINTER_RE: LazyLock<Regex> =
|
|
917
|
+
LazyLock::new(|| Regex::new(r"^(?:0|[1-9][0-9]*)(?:#|(?:/(?:[^~/]|~[01])*)*)$").unwrap());
|
|
918
|
+
|
|
919
|
+
#[cfg(test)]
|
|
920
|
+
mod tests {
|
|
921
|
+
use super::*;
|
|
922
|
+
|
|
923
|
+
#[test]
|
|
924
|
+
fn formats() {
|
|
925
|
+
assert!(date("2020-02-29") && !date("2021-02-29"));
|
|
926
|
+
assert!(time("23:59:60Z") && !time("22:59:60Z") && time("08:30:06.283185+01:00"));
|
|
927
|
+
assert!(date_time("1963-06-19T08:30:06.283185Z") && !date_time("1963-06-19 08:30:06Z"));
|
|
928
|
+
assert!(ipv6("::ffff:192.168.0.1") && !ipv6("1:2:3:4:5:6:7:8:9"));
|
|
929
|
+
assert!(hostname("xn--4gbwdl.xn--wgbh1c") && !hostname("-a.com"));
|
|
930
|
+
assert!(idn_hostname("실례.테스트") && !idn_hostname("〮실례.테스트"));
|
|
931
|
+
assert!(URI_RE.is_match("http://example.com/a?b#c") && !URI_RE.is_match("//example.com"));
|
|
932
|
+
assert!(DURATION_RE.is_match("P4DT12H30M5S") && !DURATION_RE.is_match("PT1D"));
|
|
933
|
+
}
|
|
934
|
+
|
|
935
|
+
/// The straightforward (allocating) readings of the IP address formats.
|
|
936
|
+
fn reference_ipv4(s: &str) -> bool {
|
|
937
|
+
let parts: Vec<&str> = s.split('.').collect();
|
|
938
|
+
parts.len() == 4
|
|
939
|
+
&& parts.iter().all(|p| {
|
|
940
|
+
!p.is_empty()
|
|
941
|
+
&& p.len() <= 3
|
|
942
|
+
&& p.bytes().all(|c| c.is_ascii_digit())
|
|
943
|
+
&& (p.len() == 1 || !p.starts_with('0'))
|
|
944
|
+
&& p.parse::<u32>().is_ok_and(|v| v <= 255)
|
|
945
|
+
})
|
|
946
|
+
}
|
|
947
|
+
|
|
948
|
+
fn reference_ipv6(s: &str) -> bool {
|
|
949
|
+
if s.len() < 2 || !s.bytes().all(|c| c.is_ascii_hexdigit() || c == b':' || c == b'.') {
|
|
950
|
+
return false;
|
|
951
|
+
}
|
|
952
|
+
let mut tail = s.to_string();
|
|
953
|
+
if s.contains('.') {
|
|
954
|
+
let Some(last_colon) = s.rfind(':') else { return false };
|
|
955
|
+
if !reference_ipv4(&s[last_colon + 1..]) {
|
|
956
|
+
return false;
|
|
957
|
+
}
|
|
958
|
+
tail = format!("{}0:0", &s[..=last_colon]);
|
|
959
|
+
}
|
|
960
|
+
let hex_ok = |p: &str| !p.is_empty() && p.len() <= 4 && p.bytes().all(|c| c.is_ascii_hexdigit());
|
|
961
|
+
let parts =
|
|
962
|
+
|x: &str| -> Vec<String> { if x.is_empty() { vec![] } else { x.split(':').map(String::from).collect() } };
|
|
963
|
+
if let Some(dbl) = tail.find("::") {
|
|
964
|
+
if tail[dbl + 1..].contains("::") {
|
|
965
|
+
return false;
|
|
966
|
+
}
|
|
967
|
+
let (left, right) = (parts(&tail[..dbl]), parts(&tail[dbl + 2..]));
|
|
968
|
+
return left.iter().all(|p| hex_ok(p)) && right.iter().all(|p| hex_ok(p)) && left.len() + right.len() < 8;
|
|
969
|
+
}
|
|
970
|
+
let all: Vec<&str> = tail.split(':').collect();
|
|
971
|
+
all.len() == 8 && all.iter().all(|p| hex_ok(p))
|
|
972
|
+
}
|
|
973
|
+
|
|
974
|
+
#[test]
|
|
975
|
+
fn ip_addresses_match_the_reference_readings() {
|
|
976
|
+
let seeds = [
|
|
977
|
+
"1.2.3.4",
|
|
978
|
+
"255.255.255.255",
|
|
979
|
+
"0.0.0.0",
|
|
980
|
+
"::",
|
|
981
|
+
"::1",
|
|
982
|
+
"1::",
|
|
983
|
+
"1:2:3:4:5:6:7:8",
|
|
984
|
+
"fe80::1:2:3:4",
|
|
985
|
+
"::ffff:192.168.0.1",
|
|
986
|
+
"1:2:3:4:5:6:1.2.3.4",
|
|
987
|
+
"abcd:ef01:2345:6789:abcd:ef01:2345:6789",
|
|
988
|
+
"1:2:3:4:5:6:7::",
|
|
989
|
+
"::2:3:4:5:6:7:8",
|
|
990
|
+
];
|
|
991
|
+
let alphabet = b"0123456789abcdefABCDEF:.g ";
|
|
992
|
+
let mut state = 0x2545_f491_4f6c_dd1du64;
|
|
993
|
+
let mut next = |n: usize| {
|
|
994
|
+
state ^= state << 13;
|
|
995
|
+
state ^= state >> 7;
|
|
996
|
+
state ^= state << 17;
|
|
997
|
+
(state % n as u64) as usize
|
|
998
|
+
};
|
|
999
|
+
for seed in seeds {
|
|
1000
|
+
for _ in 0..4000 {
|
|
1001
|
+
let mut b = seed.as_bytes().to_vec();
|
|
1002
|
+
for _ in 0..1 + next(3) {
|
|
1003
|
+
let at = next(b.len() + 1);
|
|
1004
|
+
match next(4) {
|
|
1005
|
+
0 if at < b.len() => {
|
|
1006
|
+
b.remove(at);
|
|
1007
|
+
}
|
|
1008
|
+
1 if at < b.len() => b[at] = alphabet[next(alphabet.len())],
|
|
1009
|
+
2 => b.insert(at, alphabet[next(alphabet.len())]),
|
|
1010
|
+
_ => b.extend_from_within(..next(b.len() + 1)),
|
|
1011
|
+
}
|
|
1012
|
+
}
|
|
1013
|
+
let s = std::str::from_utf8(&b).unwrap();
|
|
1014
|
+
assert_eq!(ipv4(s), reference_ipv4(s), "ipv4 {s:?}");
|
|
1015
|
+
assert_eq!(ipv6(s), reference_ipv6(s), "ipv6 {s:?}");
|
|
1016
|
+
}
|
|
1017
|
+
}
|
|
1018
|
+
let long = "1:".repeat(40) + "1";
|
|
1019
|
+
assert_eq!(ipv6(&long), reference_ipv6(&long));
|
|
1020
|
+
}
|
|
1021
|
+
}
|