polyxml 0.31.0__tar.gz → 0.32.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {polyxml-0.31.0 → polyxml-0.32.0}/Cargo.lock +6 -6
- {polyxml-0.31.0 → polyxml-0.32.0}/Cargo.toml +1 -1
- {polyxml-0.31.0 → polyxml-0.32.0}/PKG-INFO +1 -1
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-core/src/codegen/python/aot.rs +2 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-core/src/codegen/rust/mod.rs +155 -9
- polyxml-0.32.0/crates/polyxml-core/src/ir/chunker.rs +458 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-core/src/ir/mod.rs +9 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-core/tests/test_rust_codegen.rs +77 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/pyproject.toml +1 -1
- {polyxml-0.31.0 → polyxml-0.32.0}/README.md +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-core/Cargo.toml +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-core/README.md +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-core/benches/core_benchmarks.rs +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-core/benches/support/tag_dispatch_fixtures.rs +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-core/benches/tag_dispatch.rs +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-core/src/codegen/cpp/mod.rs +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-core/src/codegen/csharp/mod.rs +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-core/src/codegen/go/mod.rs +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-core/src/codegen/java/codec.rs +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-core/src/codegen/java/mod.rs +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-core/src/codegen/java/models.rs +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-core/src/codegen/mod.rs +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-core/src/codegen/python/mod.rs +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-core/src/codegen/typescript/mod.rs +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-core/src/converters.rs +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-core/src/error.rs +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-core/src/ir/tarjan.rs +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-core/src/json.rs +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-core/src/lib.rs +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-core/src/parser.rs +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-core/src/schema.rs +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-core/src/schema_parser/mod.rs +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-core/src/serializer.rs +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-core/src/transcoder.rs +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-core/src/value.rs +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-core/tests/fixtures/lexical_union.xsd +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-core/tests/fixtures/mixed_content.xsd +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-core/tests/fixtures/ts_mixed_content_inheritance.xsd +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-core/tests/test_adversarial_identifiers.rs +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-core/tests/test_conformance.rs +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-core/tests/test_core.rs +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-core/tests/test_cpp_codegen.rs +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-core/tests/test_csharp_codegen.rs +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-core/tests/test_go_codegen.rs +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-core/tests/test_java_codegen.rs +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-core/tests/test_json.rs +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-core/tests/test_lexical_union_codegen.rs +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-core/tests/test_mixed_content.rs +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-core/tests/test_pattern_codegen.rs +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-core/tests/test_python_codegen.rs +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-core/tests/test_rust_codecs.rs +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-core/tests/test_schema_audit.rs +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-core/tests/test_schema_ir.rs +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-core/tests/test_security.rs +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-core/tests/test_transcoder.rs +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-core/tests/test_ts_codegen.rs +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-core/tests/test_unbounded_choice.rs +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-core/tests/test_xsi_type.rs +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-python/Cargo.toml +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-python/README.md +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-python/benches/aot_vs_dataclass.py +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-python/src/lib.rs +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-python/tests/test_conformance.py +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-python/tests/test_generated_models.py +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-python/tests/test_json.py +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-python/tests/test_mixed_content.py +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-python/tests/test_polyxml.py +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-python/tests/test_xsi_type.py +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/python/polyxml/AGENT_GUIDE.md +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/python/polyxml/__init__.py +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/python/polyxml/compat/__init__.py +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/python/polyxml/compat/xsdata.py +0 -0
- {polyxml-0.31.0 → polyxml-0.32.0}/python/polyxml/py.typed +0 -0
|
@@ -608,7 +608,7 @@ dependencies = [
|
|
|
608
608
|
|
|
609
609
|
[[package]]
|
|
610
610
|
name = "polyxml"
|
|
611
|
-
version = "0.
|
|
611
|
+
version = "0.32.0"
|
|
612
612
|
dependencies = [
|
|
613
613
|
"criterion",
|
|
614
614
|
"heck",
|
|
@@ -629,14 +629,14 @@ dependencies = [
|
|
|
629
629
|
|
|
630
630
|
[[package]]
|
|
631
631
|
name = "polyxml-c"
|
|
632
|
-
version = "0.
|
|
632
|
+
version = "0.32.0"
|
|
633
633
|
dependencies = [
|
|
634
634
|
"polyxml",
|
|
635
635
|
]
|
|
636
636
|
|
|
637
637
|
[[package]]
|
|
638
638
|
name = "polyxml-cli"
|
|
639
|
-
version = "0.
|
|
639
|
+
version = "0.32.0"
|
|
640
640
|
dependencies = [
|
|
641
641
|
"clap",
|
|
642
642
|
"glob",
|
|
@@ -650,7 +650,7 @@ dependencies = [
|
|
|
650
650
|
|
|
651
651
|
[[package]]
|
|
652
652
|
name = "polyxml-js"
|
|
653
|
-
version = "0.
|
|
653
|
+
version = "0.32.0"
|
|
654
654
|
dependencies = [
|
|
655
655
|
"napi",
|
|
656
656
|
"napi-build",
|
|
@@ -660,7 +660,7 @@ dependencies = [
|
|
|
660
660
|
|
|
661
661
|
[[package]]
|
|
662
662
|
name = "polyxml-python"
|
|
663
|
-
version = "0.
|
|
663
|
+
version = "0.32.0"
|
|
664
664
|
dependencies = [
|
|
665
665
|
"lexical-core",
|
|
666
666
|
"polyxml",
|
|
@@ -670,7 +670,7 @@ dependencies = [
|
|
|
670
670
|
|
|
671
671
|
[[package]]
|
|
672
672
|
name = "polyxml-wasm"
|
|
673
|
-
version = "0.
|
|
673
|
+
version = "0.32.0"
|
|
674
674
|
dependencies = [
|
|
675
675
|
"polyxml",
|
|
676
676
|
"quick-xml",
|
|
@@ -147,6 +147,8 @@ restored_json = {module_name}.<ModelName>.from_json(json_data)
|
|
|
147
147
|
pyo3: true,
|
|
148
148
|
pyo3_module_name: Some(module_name.clone()),
|
|
149
149
|
custom_header: self.options.custom_header.clone(),
|
|
150
|
+
split_units: None,
|
|
151
|
+
chunk_size: None,
|
|
150
152
|
};
|
|
151
153
|
let rust_codegen = RustCodegen::new(rust_opts);
|
|
152
154
|
let lib_rs = rust_codegen.generate_module(ir);
|
|
@@ -42,6 +42,12 @@ pub struct RustOptions {
|
|
|
42
42
|
pub pyo3_module_name: Option<String>,
|
|
43
43
|
/// Custom header text to prepend to generated files (default: None).
|
|
44
44
|
pub custom_header: Option<String>,
|
|
45
|
+
/// Split oversized modules into bounded topological chunks (default: None, auto-split if > 400 types).
|
|
46
|
+
#[serde(default)]
|
|
47
|
+
pub split_units: Option<bool>,
|
|
48
|
+
/// Target maximum types per compilation unit chunk (default: 250).
|
|
49
|
+
#[serde(default)]
|
|
50
|
+
pub chunk_size: Option<usize>,
|
|
45
51
|
}
|
|
46
52
|
|
|
47
53
|
impl Default for RustOptions {
|
|
@@ -58,6 +64,8 @@ impl Default for RustOptions {
|
|
|
58
64
|
pyo3: false,
|
|
59
65
|
pyo3_module_name: None,
|
|
60
66
|
custom_header: None,
|
|
67
|
+
split_units: None,
|
|
68
|
+
chunk_size: None,
|
|
61
69
|
}
|
|
62
70
|
}
|
|
63
71
|
}
|
|
@@ -263,12 +271,7 @@ impl RustCodegen {
|
|
|
263
271
|
continue;
|
|
264
272
|
}
|
|
265
273
|
out.push('\n');
|
|
266
|
-
|
|
267
|
-
TypeDef::Simple(s) => self.emit_simple_type(&mut out, s, &types_with_lifetime),
|
|
268
|
-
TypeDef::Enum(e) => self.emit_enum(&mut out, e),
|
|
269
|
-
TypeDef::Union(u) => self.emit_union(&mut out, u, &types_with_lifetime, ir),
|
|
270
|
-
TypeDef::Struct(s) => self.emit_struct(&mut out, s, &types_with_lifetime, ir),
|
|
271
|
-
}
|
|
274
|
+
self.emit_single_type(&mut out, type_def, &types_with_lifetime, ir);
|
|
272
275
|
}
|
|
273
276
|
|
|
274
277
|
if self.options.emit_root_aliases {
|
|
@@ -285,6 +288,147 @@ impl RustCodegen {
|
|
|
285
288
|
out.replace("var_r#", "var_")
|
|
286
289
|
}
|
|
287
290
|
|
|
291
|
+
fn emit_single_type(
|
|
292
|
+
&self,
|
|
293
|
+
out: &mut String,
|
|
294
|
+
type_def: &TypeDef,
|
|
295
|
+
types_with_lifetime: &HashSet<QName>,
|
|
296
|
+
ir: &SchemaIR,
|
|
297
|
+
) {
|
|
298
|
+
match type_def {
|
|
299
|
+
TypeDef::Simple(s) => self.emit_simple_type(out, s, types_with_lifetime),
|
|
300
|
+
TypeDef::Enum(e) => self.emit_enum(out, e),
|
|
301
|
+
TypeDef::Union(u) => self.emit_union(out, u, types_with_lifetime, ir),
|
|
302
|
+
TypeDef::Struct(s) => self.emit_struct(out, s, types_with_lifetime, ir),
|
|
303
|
+
}
|
|
304
|
+
}
|
|
305
|
+
|
|
306
|
+
/// Generate a multi-file or single-file module representation.
|
|
307
|
+
/// When `split_units` is requested or the type count exceeds the threshold,
|
|
308
|
+
/// emits topologically sorted, bounded chunk modules (`chunk_00.rs`, `chunk_01.rs`, ...)
|
|
309
|
+
/// and a parent `mod.rs` that re-exports all chunks.
|
|
310
|
+
pub fn generate_files(&self, ir: &SchemaIR, base_name: &str) -> Vec<(String, String)> {
|
|
311
|
+
set_type_name_map(build_type_name_map(ir, |local| {
|
|
312
|
+
AsPascalCase(local).to_string()
|
|
313
|
+
}));
|
|
314
|
+
|
|
315
|
+
let emitted_count = ir.types.keys().filter(|q| !ir.is_external_type(q)).count();
|
|
316
|
+
let should_split = self.options.split_units.unwrap_or(emitted_count > 400);
|
|
317
|
+
let chunk_size = self.options.chunk_size.unwrap_or(250);
|
|
318
|
+
|
|
319
|
+
if !should_split {
|
|
320
|
+
let code = self.generate_module(ir);
|
|
321
|
+
if base_name == "mod" {
|
|
322
|
+
return vec![("mod.rs".to_string(), code)];
|
|
323
|
+
}
|
|
324
|
+
return vec![
|
|
325
|
+
(format!("{}.rs", base_name), code),
|
|
326
|
+
(
|
|
327
|
+
"mod.rs".to_string(),
|
|
328
|
+
format!("pub mod {base_name};\npub use {base_name}::*;\n"),
|
|
329
|
+
),
|
|
330
|
+
];
|
|
331
|
+
}
|
|
332
|
+
|
|
333
|
+
let plan = ir.partition_topological_chunks(chunk_size);
|
|
334
|
+
if plan.chunks.len() <= 1 {
|
|
335
|
+
let code = self.generate_module(ir);
|
|
336
|
+
if base_name == "mod" {
|
|
337
|
+
return vec![("mod.rs".to_string(), code)];
|
|
338
|
+
}
|
|
339
|
+
return vec![
|
|
340
|
+
(format!("{}.rs", base_name), code),
|
|
341
|
+
(
|
|
342
|
+
"mod.rs".to_string(),
|
|
343
|
+
format!("pub mod {base_name};\npub use {base_name}::*;\n"),
|
|
344
|
+
),
|
|
345
|
+
];
|
|
346
|
+
}
|
|
347
|
+
|
|
348
|
+
let types_with_lifetime = if self.options.zero_copy {
|
|
349
|
+
self.compute_types_with_lifetime(ir)
|
|
350
|
+
} else {
|
|
351
|
+
HashSet::new()
|
|
352
|
+
};
|
|
353
|
+
|
|
354
|
+
let mut files = Vec::new();
|
|
355
|
+
|
|
356
|
+
// 1. Chunks
|
|
357
|
+
for chunk in &plan.chunks {
|
|
358
|
+
let mut out = String::new();
|
|
359
|
+
if let Some(ref header) = self.options.custom_header {
|
|
360
|
+
let trimmed = header.trim();
|
|
361
|
+
if !trimmed.is_empty() {
|
|
362
|
+
out.push_str(trimmed);
|
|
363
|
+
out.push_str("\n\n");
|
|
364
|
+
}
|
|
365
|
+
}
|
|
366
|
+
out.push_str(
|
|
367
|
+
"// @generated by PolyXML Compiler (https://github.com/polyxml/PolyXML)\n",
|
|
368
|
+
);
|
|
369
|
+
out.push_str(
|
|
370
|
+
"#![allow(dead_code, unused_imports, unused_mut, unused_variables, unused_assignments, non_camel_case_types, non_snake_case)]\n\n",
|
|
371
|
+
);
|
|
372
|
+
|
|
373
|
+
self.emit_imports(&mut out, !types_with_lifetime.is_empty());
|
|
374
|
+
out.push_str("\nuse super::*;\n");
|
|
375
|
+
|
|
376
|
+
for qname in &chunk.types {
|
|
377
|
+
if let Some(type_def) = ir.types.get(qname) {
|
|
378
|
+
out.push('\n');
|
|
379
|
+
self.emit_single_type(&mut out, type_def, &types_with_lifetime, ir);
|
|
380
|
+
}
|
|
381
|
+
}
|
|
382
|
+
|
|
383
|
+
let code = out.replace("var_r#", "var_");
|
|
384
|
+
files.push((format!("{}.rs", chunk.name), code));
|
|
385
|
+
}
|
|
386
|
+
|
|
387
|
+
// 2. mod.rs
|
|
388
|
+
let mut mod_out = String::new();
|
|
389
|
+
if let Some(ref header) = self.options.custom_header {
|
|
390
|
+
let trimmed = header.trim();
|
|
391
|
+
if !trimmed.is_empty() {
|
|
392
|
+
mod_out.push_str(trimmed);
|
|
393
|
+
mod_out.push_str("\n\n");
|
|
394
|
+
}
|
|
395
|
+
}
|
|
396
|
+
mod_out
|
|
397
|
+
.push_str("// @generated by PolyXML Compiler (https://github.com/polyxml/PolyXML)\n");
|
|
398
|
+
mod_out.push_str(
|
|
399
|
+
"#![allow(dead_code, unused_imports, unused_mut, unused_variables, unused_assignments, non_camel_case_types, non_snake_case)]\n\n",
|
|
400
|
+
);
|
|
401
|
+
|
|
402
|
+
self.emit_imports(&mut mod_out, !types_with_lifetime.is_empty());
|
|
403
|
+
mod_out.push('\n');
|
|
404
|
+
|
|
405
|
+
for chunk in &plan.chunks {
|
|
406
|
+
let _ = writeln!(mod_out, "pub mod {};", chunk.name);
|
|
407
|
+
}
|
|
408
|
+
mod_out.push('\n');
|
|
409
|
+
for chunk in &plan.chunks {
|
|
410
|
+
let _ = writeln!(mod_out, "pub use {}::*;", chunk.name);
|
|
411
|
+
}
|
|
412
|
+
|
|
413
|
+
if self.options.emit_codecs {
|
|
414
|
+
self.emit_codec_helpers(&mut mod_out);
|
|
415
|
+
}
|
|
416
|
+
|
|
417
|
+
if self.options.emit_root_aliases {
|
|
418
|
+
self.emit_root_aliases(&mut mod_out, ir, &types_with_lifetime);
|
|
419
|
+
}
|
|
420
|
+
|
|
421
|
+
if self.options.pyo3 {
|
|
422
|
+
let sorted_types = self.order_types(ir);
|
|
423
|
+
self.emit_pymodule(&mut mod_out, ir, &sorted_types);
|
|
424
|
+
}
|
|
425
|
+
|
|
426
|
+
let mod_code = mod_out.replace("var_r#", "var_");
|
|
427
|
+
files.push(("mod.rs".to_string(), mod_code));
|
|
428
|
+
|
|
429
|
+
files
|
|
430
|
+
}
|
|
431
|
+
|
|
288
432
|
/// Fixed-point analysis determining which types in SchemaIR require a lifetime parameter `<'a>`.
|
|
289
433
|
fn compute_types_with_lifetime(&self, ir: &SchemaIR) -> HashSet<QName> {
|
|
290
434
|
let mut requires_lifetime = HashSet::new();
|
|
@@ -896,7 +1040,9 @@ impl RustCodegen {
|
|
|
896
1040
|
|
|
897
1041
|
fn emit_codec_helpers(&self, out: &mut String) {
|
|
898
1042
|
out.push_str("\n#[allow(dead_code)]\n");
|
|
899
|
-
out.push_str(
|
|
1043
|
+
out.push_str(
|
|
1044
|
+
"pub(crate) fn skip_xml_element(reader: &mut Reader<&[u8]>) -> Result<()> {\n",
|
|
1045
|
+
);
|
|
900
1046
|
out.push_str(" let mut depth = 1;\n");
|
|
901
1047
|
out.push_str(" loop {\n");
|
|
902
1048
|
out.push_str(" match reader.read_event()? {\n");
|
|
@@ -916,7 +1062,7 @@ impl RustCodegen {
|
|
|
916
1062
|
|
|
917
1063
|
if self.options.zero_copy {
|
|
918
1064
|
out.push_str("#[allow(dead_code)]\n");
|
|
919
|
-
out.push_str("fn read_element_text<'a>(reader: &mut Reader<&'a [u8]>, tag_name: &str) -> Result<Cow<'a, str>> {\n");
|
|
1065
|
+
out.push_str("pub(crate) fn read_element_text<'a>(reader: &mut Reader<&'a [u8]>, tag_name: &str) -> Result<Cow<'a, str>> {\n");
|
|
920
1066
|
out.push_str(" let mut text = Cow::Borrowed(\"\");\n");
|
|
921
1067
|
out.push_str(" loop {\n");
|
|
922
1068
|
out.push_str(" match reader.read_event()? {\n");
|
|
@@ -963,7 +1109,7 @@ impl RustCodegen {
|
|
|
963
1109
|
out.push_str("}\n");
|
|
964
1110
|
} else {
|
|
965
1111
|
out.push_str("#[allow(dead_code)]\n");
|
|
966
|
-
out.push_str("fn read_element_text(reader: &mut Reader<&[u8]>, tag_name: &str) -> Result<String> {\n");
|
|
1112
|
+
out.push_str("pub(crate) fn read_element_text(reader: &mut Reader<&[u8]>, tag_name: &str) -> Result<String> {\n");
|
|
967
1113
|
out.push_str(" let mut text = String::new();\n");
|
|
968
1114
|
out.push_str(" loop {\n");
|
|
969
1115
|
out.push_str(" match reader.read_event()? {\n");
|
|
@@ -0,0 +1,458 @@
|
|
|
1
|
+
use std::cmp::Ordering;
|
|
2
|
+
use std::collections::{BTreeSet, BinaryHeap, HashMap, HashSet};
|
|
3
|
+
|
|
4
|
+
use super::{QName, SchemaIR, TypeDef, TypeRef};
|
|
5
|
+
|
|
6
|
+
/// A bounded compilation unit chunk containing a subset of types from the IR.
|
|
7
|
+
#[derive(Debug, Clone, PartialEq, Eq)]
|
|
8
|
+
pub struct TypeChunk {
|
|
9
|
+
/// 0-based chunk index.
|
|
10
|
+
pub index: usize,
|
|
11
|
+
/// Chunk identifier (e.g. "chunk_00").
|
|
12
|
+
pub name: String,
|
|
13
|
+
/// Fully qualified names of all types assigned to this chunk in topological order.
|
|
14
|
+
pub types: Vec<QName>,
|
|
15
|
+
}
|
|
16
|
+
|
|
17
|
+
/// A complete partitioning plan of an IR's types into topologically sorted chunks.
|
|
18
|
+
#[derive(Debug, Clone, PartialEq, Eq)]
|
|
19
|
+
pub struct ChunkPlan {
|
|
20
|
+
pub chunks: Vec<TypeChunk>,
|
|
21
|
+
/// Map from each type QName to its assigned chunk index.
|
|
22
|
+
pub type_to_chunk: HashMap<QName, usize>,
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
impl ChunkPlan {
|
|
26
|
+
/// Return the list of chunk indices that the specified chunk depends on.
|
|
27
|
+
/// By invariant of topological SCC condensation, every dependency index `q`
|
|
28
|
+
/// satisfies `q < chunk_index`.
|
|
29
|
+
pub fn dependencies_of(&self, chunk_index: usize, ir: &SchemaIR) -> BTreeSet<usize> {
|
|
30
|
+
let mut deps = BTreeSet::new();
|
|
31
|
+
let Some(chunk) = self.chunks.get(chunk_index) else {
|
|
32
|
+
return deps;
|
|
33
|
+
};
|
|
34
|
+
|
|
35
|
+
for qname in &chunk.types {
|
|
36
|
+
let type_deps = extract_type_dependencies(qname, ir);
|
|
37
|
+
for dep in type_deps {
|
|
38
|
+
if let Some(&target_chunk) = self.type_to_chunk.get(&dep) {
|
|
39
|
+
if target_chunk != chunk_index {
|
|
40
|
+
deps.insert(target_chunk);
|
|
41
|
+
}
|
|
42
|
+
}
|
|
43
|
+
}
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
deps
|
|
47
|
+
}
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
/// Helper wrapper for deterministic min-heap priority queue ordering.
|
|
51
|
+
#[derive(Eq, PartialEq)]
|
|
52
|
+
struct ReadyComponent {
|
|
53
|
+
min_qname: QName,
|
|
54
|
+
component_idx: usize,
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
impl Ord for ReadyComponent {
|
|
58
|
+
fn cmp(&self, other: &Self) -> Ordering {
|
|
59
|
+
// Reverse for min-heap
|
|
60
|
+
other
|
|
61
|
+
.min_qname
|
|
62
|
+
.cmp(&self.min_qname)
|
|
63
|
+
.then_with(|| other.component_idx.cmp(&self.component_idx))
|
|
64
|
+
}
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
impl PartialOrd for ReadyComponent {
|
|
68
|
+
fn partial_cmp(&self, other: &Self) -> Option<Ordering> {
|
|
69
|
+
Some(self.cmp(other))
|
|
70
|
+
}
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
/// Extract all direct named type dependencies for a given type in the IR.
|
|
74
|
+
pub fn extract_type_dependencies(qname: &QName, ir: &SchemaIR) -> BTreeSet<QName> {
|
|
75
|
+
let mut deps = BTreeSet::new();
|
|
76
|
+
let Some(type_def) = ir.types.get(qname) else {
|
|
77
|
+
return deps;
|
|
78
|
+
};
|
|
79
|
+
|
|
80
|
+
match type_def {
|
|
81
|
+
TypeDef::Struct(s) => {
|
|
82
|
+
if let Some(ref base) = s.base_type {
|
|
83
|
+
if ir.types.contains_key(base) && !ir.is_external_type(base) {
|
|
84
|
+
deps.insert(base.clone());
|
|
85
|
+
}
|
|
86
|
+
}
|
|
87
|
+
for field in &s.fields {
|
|
88
|
+
collect_type_refs(&field.type_ref, ir, &mut deps);
|
|
89
|
+
}
|
|
90
|
+
}
|
|
91
|
+
TypeDef::Union(u) => {
|
|
92
|
+
for branch in &u.branches {
|
|
93
|
+
collect_type_refs(&branch.type_ref, ir, &mut deps);
|
|
94
|
+
}
|
|
95
|
+
}
|
|
96
|
+
TypeDef::Enum(e) => {
|
|
97
|
+
collect_type_refs(&e.base_type, ir, &mut deps);
|
|
98
|
+
}
|
|
99
|
+
TypeDef::Simple(st) => {
|
|
100
|
+
collect_type_refs(&st.base_type, ir, &mut deps);
|
|
101
|
+
}
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
// Do not include self-dependency in external dependency list
|
|
105
|
+
deps.remove(qname);
|
|
106
|
+
deps
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
fn collect_type_refs(type_ref: &TypeRef, ir: &SchemaIR, deps: &mut BTreeSet<QName>) {
|
|
110
|
+
match type_ref {
|
|
111
|
+
TypeRef::Named(target) => {
|
|
112
|
+
if ir.types.contains_key(target) && !ir.is_external_type(target) {
|
|
113
|
+
deps.insert(target.clone());
|
|
114
|
+
}
|
|
115
|
+
}
|
|
116
|
+
TypeRef::Boxed(inner) | TypeRef::List(inner) => {
|
|
117
|
+
collect_type_refs(inner, ir, deps);
|
|
118
|
+
}
|
|
119
|
+
TypeRef::Primitive(_) => {}
|
|
120
|
+
}
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
/// Partition the types of a SchemaIR into topological SCC chunks.
|
|
124
|
+
///
|
|
125
|
+
/// Each chunk contains at most `budget` types (unless a single cyclic SCC
|
|
126
|
+
/// contains more types than `budget`, in which case it stays co-located in one chunk).
|
|
127
|
+
///
|
|
128
|
+
/// Guaranteed Properties:
|
|
129
|
+
/// 1. Cycle Condensation: All mutually recursive types belong to the same chunk.
|
|
130
|
+
/// 2. Topological Order: If type `A` in chunk `p` depends on type `B` in chunk `q`,
|
|
131
|
+
/// then `q <= p`.
|
|
132
|
+
/// 3. Zero Circular Imports: Cross-chunk circular dependencies are mathematically impossible.
|
|
133
|
+
pub fn partition_topological_chunks(ir: &SchemaIR, budget: usize) -> ChunkPlan {
|
|
134
|
+
// Collect all local emitted types (excluding external types)
|
|
135
|
+
let emitted_types: Vec<QName> = ir
|
|
136
|
+
.types
|
|
137
|
+
.keys()
|
|
138
|
+
.filter(|q| !ir.is_external_type(q))
|
|
139
|
+
.cloned()
|
|
140
|
+
.collect();
|
|
141
|
+
|
|
142
|
+
if emitted_types.is_empty() {
|
|
143
|
+
return ChunkPlan {
|
|
144
|
+
chunks: Vec::new(),
|
|
145
|
+
type_to_chunk: HashMap::new(),
|
|
146
|
+
};
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
// If budget is 0 or all types fit in one chunk, return a single chunk
|
|
150
|
+
if budget == 0 || emitted_types.len() <= budget {
|
|
151
|
+
// Still sort topologically within the single chunk for cleanliness
|
|
152
|
+
let sorted = topological_sort_types(&emitted_types, ir);
|
|
153
|
+
let mut type_to_chunk = HashMap::new();
|
|
154
|
+
for q in &sorted {
|
|
155
|
+
type_to_chunk.insert(q.clone(), 0);
|
|
156
|
+
}
|
|
157
|
+
return ChunkPlan {
|
|
158
|
+
chunks: vec![TypeChunk {
|
|
159
|
+
index: 0,
|
|
160
|
+
name: "chunk_00".to_string(),
|
|
161
|
+
types: sorted,
|
|
162
|
+
}],
|
|
163
|
+
type_to_chunk,
|
|
164
|
+
};
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
// 1. Build adjacency list of dependencies: node -> set of nodes it depends on
|
|
168
|
+
let mut adj: HashMap<QName, Vec<QName>> = HashMap::new();
|
|
169
|
+
for qname in &emitted_types {
|
|
170
|
+
let deps = extract_type_dependencies(qname, ir);
|
|
171
|
+
adj.insert(qname.clone(), deps.into_iter().collect());
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
// 2. Run Tarjan's SCC algorithm on the emitted types
|
|
175
|
+
let sccs = run_tarjan_scc(&emitted_types, &adj);
|
|
176
|
+
|
|
177
|
+
// 3. Build Condensation DAG
|
|
178
|
+
// Map each node to its SCC index
|
|
179
|
+
let mut node_to_scc: HashMap<QName, usize> = HashMap::new();
|
|
180
|
+
for (scc_idx, scc) in sccs.iter().enumerate() {
|
|
181
|
+
for qname in scc {
|
|
182
|
+
node_to_scc.insert(qname.clone(), scc_idx);
|
|
183
|
+
}
|
|
184
|
+
}
|
|
185
|
+
|
|
186
|
+
let num_sccs = sccs.len();
|
|
187
|
+
// prereqs[i]: SCCs that i depends on (must come before i)
|
|
188
|
+
let mut prereqs: Vec<BTreeSet<usize>> = vec![BTreeSet::new(); num_sccs];
|
|
189
|
+
// dependents[j]: SCCs that depend on j (come after j)
|
|
190
|
+
let mut dependents: Vec<BTreeSet<usize>> = vec![BTreeSet::new(); num_sccs];
|
|
191
|
+
|
|
192
|
+
for (scc_idx, scc) in sccs.iter().enumerate() {
|
|
193
|
+
for qname in scc {
|
|
194
|
+
if let Some(deps) = adj.get(qname) {
|
|
195
|
+
for dep in deps {
|
|
196
|
+
if let Some(&dep_scc) = node_to_scc.get(dep) {
|
|
197
|
+
if dep_scc != scc_idx {
|
|
198
|
+
prereqs[scc_idx].insert(dep_scc);
|
|
199
|
+
dependents[dep_scc].insert(scc_idx);
|
|
200
|
+
}
|
|
201
|
+
}
|
|
202
|
+
}
|
|
203
|
+
}
|
|
204
|
+
}
|
|
205
|
+
}
|
|
206
|
+
|
|
207
|
+
// 4. Topological Sort of Condensation DAG using Kahn's algorithm
|
|
208
|
+
// In-degree is the number of unsatisfied prerequisites (SCCs that must come before)
|
|
209
|
+
let mut in_degree: Vec<usize> = prereqs.iter().map(|p| p.len()).collect();
|
|
210
|
+
let mut ready_heap = BinaryHeap::new();
|
|
211
|
+
|
|
212
|
+
for (idx, °) in in_degree.iter().enumerate() {
|
|
213
|
+
if deg == 0 {
|
|
214
|
+
let min_q = sccs[idx].iter().min().cloned().unwrap();
|
|
215
|
+
ready_heap.push(ReadyComponent {
|
|
216
|
+
min_qname: min_q,
|
|
217
|
+
component_idx: idx,
|
|
218
|
+
});
|
|
219
|
+
}
|
|
220
|
+
}
|
|
221
|
+
|
|
222
|
+
let mut sorted_sccs: Vec<usize> = Vec::with_capacity(num_sccs);
|
|
223
|
+
while let Some(ReadyComponent { component_idx, .. }) = ready_heap.pop() {
|
|
224
|
+
sorted_sccs.push(component_idx);
|
|
225
|
+
for &dep_idx in &dependents[component_idx] {
|
|
226
|
+
in_degree[dep_idx] -= 1;
|
|
227
|
+
if in_degree[dep_idx] == 0 {
|
|
228
|
+
let min_q = sccs[dep_idx].iter().min().cloned().unwrap();
|
|
229
|
+
ready_heap.push(ReadyComponent {
|
|
230
|
+
min_qname: min_q,
|
|
231
|
+
component_idx: dep_idx,
|
|
232
|
+
});
|
|
233
|
+
}
|
|
234
|
+
}
|
|
235
|
+
}
|
|
236
|
+
|
|
237
|
+
// Safeguard for completeness
|
|
238
|
+
if sorted_sccs.len() < num_sccs {
|
|
239
|
+
for idx in 0..num_sccs {
|
|
240
|
+
if !sorted_sccs.contains(&idx) {
|
|
241
|
+
sorted_sccs.push(idx);
|
|
242
|
+
}
|
|
243
|
+
}
|
|
244
|
+
}
|
|
245
|
+
|
|
246
|
+
// 5. Bounded chunk packing
|
|
247
|
+
let mut chunks: Vec<TypeChunk> = Vec::new();
|
|
248
|
+
let mut current_chunk_types: Vec<QName> = Vec::new();
|
|
249
|
+
|
|
250
|
+
for &scc_idx in &sorted_sccs {
|
|
251
|
+
let scc_types = &sccs[scc_idx];
|
|
252
|
+
if !current_chunk_types.is_empty() && current_chunk_types.len() + scc_types.len() > budget {
|
|
253
|
+
let idx = chunks.len();
|
|
254
|
+
chunks.push(TypeChunk {
|
|
255
|
+
index: idx,
|
|
256
|
+
name: format!("chunk_{:02}", idx),
|
|
257
|
+
types: std::mem::take(&mut current_chunk_types),
|
|
258
|
+
});
|
|
259
|
+
}
|
|
260
|
+
current_chunk_types.extend(scc_types.iter().cloned());
|
|
261
|
+
}
|
|
262
|
+
|
|
263
|
+
if !current_chunk_types.is_empty() {
|
|
264
|
+
let idx = chunks.len();
|
|
265
|
+
chunks.push(TypeChunk {
|
|
266
|
+
index: idx,
|
|
267
|
+
name: format!("chunk_{:02}", idx),
|
|
268
|
+
types: current_chunk_types,
|
|
269
|
+
});
|
|
270
|
+
}
|
|
271
|
+
|
|
272
|
+
let mut type_to_chunk = HashMap::new();
|
|
273
|
+
for chunk in &chunks {
|
|
274
|
+
for q in &chunk.types {
|
|
275
|
+
type_to_chunk.insert(q.clone(), chunk.index);
|
|
276
|
+
}
|
|
277
|
+
}
|
|
278
|
+
|
|
279
|
+
ChunkPlan {
|
|
280
|
+
chunks,
|
|
281
|
+
type_to_chunk,
|
|
282
|
+
}
|
|
283
|
+
}
|
|
284
|
+
|
|
285
|
+
/// Helper to topologically sort a list of types.
|
|
286
|
+
fn topological_sort_types(types: &[QName], ir: &SchemaIR) -> Vec<QName> {
|
|
287
|
+
let mut adj: HashMap<QName, Vec<QName>> = HashMap::new();
|
|
288
|
+
for q in types {
|
|
289
|
+
let deps = extract_type_dependencies(q, ir);
|
|
290
|
+
adj.insert(q.clone(), deps.into_iter().collect());
|
|
291
|
+
}
|
|
292
|
+
let sccs = run_tarjan_scc(types, &adj);
|
|
293
|
+
let mut flat = Vec::new();
|
|
294
|
+
for scc in sccs {
|
|
295
|
+
flat.extend(scc);
|
|
296
|
+
}
|
|
297
|
+
flat
|
|
298
|
+
}
|
|
299
|
+
|
|
300
|
+
/// Tarjan's SCC algorithm implementation returning SCCs with sorted elements.
|
|
301
|
+
fn run_tarjan_scc(nodes: &[QName], adj: &HashMap<QName, Vec<QName>>) -> Vec<Vec<QName>> {
|
|
302
|
+
struct Tarjan<'a> {
|
|
303
|
+
adj: &'a HashMap<QName, Vec<QName>>,
|
|
304
|
+
index: usize,
|
|
305
|
+
indices: HashMap<QName, usize>,
|
|
306
|
+
lowlinks: HashMap<QName, usize>,
|
|
307
|
+
on_stack: HashSet<QName>,
|
|
308
|
+
stack: Vec<QName>,
|
|
309
|
+
sccs: Vec<Vec<QName>>,
|
|
310
|
+
}
|
|
311
|
+
|
|
312
|
+
impl<'a> Tarjan<'a> {
|
|
313
|
+
fn strongconnect(&mut self, node: &QName) {
|
|
314
|
+
self.indices.insert(node.clone(), self.index);
|
|
315
|
+
self.lowlinks.insert(node.clone(), self.index);
|
|
316
|
+
self.index += 1;
|
|
317
|
+
self.stack.push(node.clone());
|
|
318
|
+
self.on_stack.insert(node.clone());
|
|
319
|
+
|
|
320
|
+
if let Some(neighbors) = self.adj.get(node) {
|
|
321
|
+
for neighbor in neighbors {
|
|
322
|
+
if !self.indices.contains_key(neighbor) {
|
|
323
|
+
self.strongconnect(neighbor);
|
|
324
|
+
let n_lowlink = *self.lowlinks.get(neighbor).unwrap();
|
|
325
|
+
let curr_lowlink = self.lowlinks.get_mut(node).unwrap();
|
|
326
|
+
*curr_lowlink = std::cmp::min(*curr_lowlink, n_lowlink);
|
|
327
|
+
} else if self.on_stack.contains(neighbor) {
|
|
328
|
+
let n_index = *self.indices.get(neighbor).unwrap();
|
|
329
|
+
let curr_lowlink = self.lowlinks.get_mut(node).unwrap();
|
|
330
|
+
*curr_lowlink = std::cmp::min(*curr_lowlink, n_index);
|
|
331
|
+
}
|
|
332
|
+
}
|
|
333
|
+
}
|
|
334
|
+
|
|
335
|
+
if self.lowlinks.get(node) == self.indices.get(node) {
|
|
336
|
+
let mut scc = Vec::new();
|
|
337
|
+
while let Some(w) = self.stack.pop() {
|
|
338
|
+
self.on_stack.remove(&w);
|
|
339
|
+
scc.push(w.clone());
|
|
340
|
+
if &w == node {
|
|
341
|
+
break;
|
|
342
|
+
}
|
|
343
|
+
}
|
|
344
|
+
// Sort deterministically within each SCC
|
|
345
|
+
scc.sort();
|
|
346
|
+
self.sccs.push(scc);
|
|
347
|
+
}
|
|
348
|
+
}
|
|
349
|
+
}
|
|
350
|
+
|
|
351
|
+
let mut tarjan = Tarjan {
|
|
352
|
+
adj,
|
|
353
|
+
index: 0,
|
|
354
|
+
indices: HashMap::new(),
|
|
355
|
+
lowlinks: HashMap::new(),
|
|
356
|
+
on_stack: HashSet::new(),
|
|
357
|
+
stack: Vec::new(),
|
|
358
|
+
sccs: Vec::new(),
|
|
359
|
+
};
|
|
360
|
+
|
|
361
|
+
// Deterministic start order
|
|
362
|
+
let mut sorted_nodes = nodes.to_vec();
|
|
363
|
+
sorted_nodes.sort();
|
|
364
|
+
for node in &sorted_nodes {
|
|
365
|
+
if !tarjan.indices.contains_key(node) {
|
|
366
|
+
tarjan.strongconnect(node);
|
|
367
|
+
}
|
|
368
|
+
}
|
|
369
|
+
|
|
370
|
+
tarjan.sccs
|
|
371
|
+
}
|
|
372
|
+
|
|
373
|
+
#[cfg(test)]
|
|
374
|
+
mod tests {
|
|
375
|
+
use super::*;
|
|
376
|
+
use crate::ir::{Cardinality, FieldDef, FieldKind, StructDef, TypeDef};
|
|
377
|
+
|
|
378
|
+
fn make_struct(name: &str, fields: &[(&str, &str)]) -> TypeDef {
|
|
379
|
+
let qname = QName::local(name);
|
|
380
|
+
let field_defs = fields
|
|
381
|
+
.iter()
|
|
382
|
+
.map(|(fname, target)| {
|
|
383
|
+
let mut f = FieldDef::new(
|
|
384
|
+
*fname,
|
|
385
|
+
*fname,
|
|
386
|
+
FieldKind::Element,
|
|
387
|
+
TypeRef::Named(QName::local(*target)),
|
|
388
|
+
);
|
|
389
|
+
f.cardinality = Cardinality::required_one();
|
|
390
|
+
f
|
|
391
|
+
})
|
|
392
|
+
.collect();
|
|
393
|
+
TypeDef::Struct(StructDef {
|
|
394
|
+
qname,
|
|
395
|
+
base_type: None,
|
|
396
|
+
is_abstract: false,
|
|
397
|
+
is_mixed: false,
|
|
398
|
+
fields: field_defs,
|
|
399
|
+
documentation: None,
|
|
400
|
+
})
|
|
401
|
+
}
|
|
402
|
+
|
|
403
|
+
#[test]
|
|
404
|
+
fn test_topological_scc_chunking_dag_invariants() {
|
|
405
|
+
let mut ir = SchemaIR::new();
|
|
406
|
+
// A -> B -> C (linear dependency: A depends on B, B depends on C)
|
|
407
|
+
ir.types.insert(QName::local("C"), make_struct("C", &[]));
|
|
408
|
+
ir.types
|
|
409
|
+
.insert(QName::local("B"), make_struct("B", &[("c", "C")]));
|
|
410
|
+
ir.types
|
|
411
|
+
.insert(QName::local("A"), make_struct("A", &[("b", "B")]));
|
|
412
|
+
|
|
413
|
+
// D and E form a cycle (mutual recursion)
|
|
414
|
+
ir.types
|
|
415
|
+
.insert(QName::local("D"), make_struct("D", &[("e", "E")]));
|
|
416
|
+
ir.types
|
|
417
|
+
.insert(QName::local("E"), make_struct("E", &[("d", "D")]));
|
|
418
|
+
|
|
419
|
+
// F depends on D
|
|
420
|
+
ir.types
|
|
421
|
+
.insert(QName::local("F"), make_struct("F", &[("d", "D")]));
|
|
422
|
+
|
|
423
|
+
// Partition with budget = 2
|
|
424
|
+
let plan = partition_topological_chunks(&ir, 2);
|
|
425
|
+
|
|
426
|
+
// Verify:
|
|
427
|
+
// 1. D and E MUST be in the exact same chunk because they form an SCC!
|
|
428
|
+
let chunk_d = plan.type_to_chunk[&QName::local("D")];
|
|
429
|
+
let chunk_e = plan.type_to_chunk[&QName::local("E")];
|
|
430
|
+
assert_eq!(
|
|
431
|
+
chunk_d, chunk_e,
|
|
432
|
+
"Cyclic types D and E must be in the same chunk"
|
|
433
|
+
);
|
|
434
|
+
|
|
435
|
+
// 2. F depends on D, so chunk_of(F) >= chunk_of(D)
|
|
436
|
+
let chunk_f = plan.type_to_chunk[&QName::local("F")];
|
|
437
|
+
assert!(
|
|
438
|
+
chunk_f >= chunk_d,
|
|
439
|
+
"Dependent F must come in or after D's chunk"
|
|
440
|
+
);
|
|
441
|
+
|
|
442
|
+
// 3. For EVERY chunk and EVERY type, all dependencies must be in <= current chunk
|
|
443
|
+
for (qname, &chunk_idx) in &plan.type_to_chunk {
|
|
444
|
+
let deps = extract_type_dependencies(qname, &ir);
|
|
445
|
+
for dep in deps {
|
|
446
|
+
let dep_chunk = plan.type_to_chunk[&dep];
|
|
447
|
+
assert!(
|
|
448
|
+
dep_chunk <= chunk_idx,
|
|
449
|
+
"Type {} in chunk {} depends on {} in chunk {} (forward reference violation!)",
|
|
450
|
+
qname,
|
|
451
|
+
chunk_idx,
|
|
452
|
+
dep,
|
|
453
|
+
dep_chunk
|
|
454
|
+
);
|
|
455
|
+
}
|
|
456
|
+
}
|
|
457
|
+
}
|
|
458
|
+
}
|
|
@@ -1,5 +1,8 @@
|
|
|
1
|
+
pub mod chunker;
|
|
1
2
|
pub mod tarjan;
|
|
2
3
|
|
|
4
|
+
pub use chunker::{partition_topological_chunks, ChunkPlan, TypeChunk};
|
|
5
|
+
|
|
3
6
|
use serde::{Deserialize, Serialize};
|
|
4
7
|
use std::collections::{BTreeMap, BTreeSet, HashMap};
|
|
5
8
|
use std::fmt;
|
|
@@ -638,6 +641,12 @@ impl SchemaIR {
|
|
|
638
641
|
.values()
|
|
639
642
|
.filter(|ty| !self.is_external_type(ty.qname()))
|
|
640
643
|
}
|
|
644
|
+
|
|
645
|
+
/// Partition local emitted types into topologically sorted SCC chunks.
|
|
646
|
+
pub fn partition_topological_chunks(&self, budget: usize) -> ChunkPlan {
|
|
647
|
+
chunker::partition_topological_chunks(self, budget)
|
|
648
|
+
}
|
|
649
|
+
|
|
641
650
|
pub fn new() -> Self {
|
|
642
651
|
Self::default()
|
|
643
652
|
}
|
|
@@ -1204,3 +1204,80 @@ fn test_rust_simple_type_cycle_resilience() {
|
|
|
1204
1204
|
let code = codegen.generate_module(&ir);
|
|
1205
1205
|
assert!(code.contains("pub struct CyclicStruct"));
|
|
1206
1206
|
}
|
|
1207
|
+
|
|
1208
|
+
#[test]
|
|
1209
|
+
fn test_rust_topological_scc_chunking() {
|
|
1210
|
+
let mut ir = SchemaIR::new();
|
|
1211
|
+
|
|
1212
|
+
// Leaf type in chunk 0
|
|
1213
|
+
ir.add_type(TypeDef::Simple(Box::new(SimpleTypeDef {
|
|
1214
|
+
qname: QName::local("Leaf"),
|
|
1215
|
+
base_type: TypeRef::Primitive(PrimitiveType::String),
|
|
1216
|
+
facets: Default::default(),
|
|
1217
|
+
documentation: None,
|
|
1218
|
+
})));
|
|
1219
|
+
|
|
1220
|
+
// Mid type depending on Leaf
|
|
1221
|
+
ir.add_type(TypeDef::Struct(StructDef {
|
|
1222
|
+
qname: QName::local("Mid"),
|
|
1223
|
+
base_type: None,
|
|
1224
|
+
is_abstract: false,
|
|
1225
|
+
is_mixed: false,
|
|
1226
|
+
fields: vec![FieldDef::new(
|
|
1227
|
+
"leaf",
|
|
1228
|
+
"leaf",
|
|
1229
|
+
FieldKind::Element,
|
|
1230
|
+
TypeRef::Named(QName::local("Leaf")),
|
|
1231
|
+
)],
|
|
1232
|
+
documentation: None,
|
|
1233
|
+
}));
|
|
1234
|
+
|
|
1235
|
+
// Root type depending on Mid
|
|
1236
|
+
ir.add_type(TypeDef::Struct(StructDef {
|
|
1237
|
+
qname: QName::local("Root"),
|
|
1238
|
+
base_type: None,
|
|
1239
|
+
is_abstract: false,
|
|
1240
|
+
is_mixed: false,
|
|
1241
|
+
fields: vec![FieldDef::new(
|
|
1242
|
+
"mid",
|
|
1243
|
+
"mid",
|
|
1244
|
+
FieldKind::Element,
|
|
1245
|
+
TypeRef::Named(QName::local("Mid")),
|
|
1246
|
+
)],
|
|
1247
|
+
documentation: None,
|
|
1248
|
+
}));
|
|
1249
|
+
|
|
1250
|
+
let opts = RustOptions {
|
|
1251
|
+
split_units: Some(true),
|
|
1252
|
+
chunk_size: Some(1), // 1 type per chunk -> 3 chunks + mod.rs
|
|
1253
|
+
..Default::default()
|
|
1254
|
+
};
|
|
1255
|
+
|
|
1256
|
+
let codegen = RustCodegen::new(opts);
|
|
1257
|
+
let files = codegen.generate_files(&ir, "models");
|
|
1258
|
+
|
|
1259
|
+
assert_eq!(files.len(), 4, "Expected 3 chunks + mod.rs");
|
|
1260
|
+
let filenames: Vec<_> = files.iter().map(|(n, _)| n.as_str()).collect();
|
|
1261
|
+
assert_eq!(
|
|
1262
|
+
filenames,
|
|
1263
|
+
vec!["chunk_00.rs", "chunk_01.rs", "chunk_02.rs", "mod.rs"]
|
|
1264
|
+
);
|
|
1265
|
+
|
|
1266
|
+
// chunk_00 must contain Leaf
|
|
1267
|
+
assert!(files[0].1.contains("pub type Leaf"));
|
|
1268
|
+
// chunk_01 must contain Mid and use super::*
|
|
1269
|
+
assert!(files[1].1.contains("pub struct Mid"));
|
|
1270
|
+
assert!(files[1].1.contains("use super::*;"));
|
|
1271
|
+
// chunk_02 must contain Root
|
|
1272
|
+
assert!(files[2].1.contains("pub struct Root"));
|
|
1273
|
+
assert!(files[2].1.contains("use super::*;"));
|
|
1274
|
+
|
|
1275
|
+
// mod.rs must re-export all chunks
|
|
1276
|
+
let mod_rs = &files[3].1;
|
|
1277
|
+
assert!(mod_rs.contains("pub mod chunk_00;"));
|
|
1278
|
+
assert!(mod_rs.contains("pub mod chunk_01;"));
|
|
1279
|
+
assert!(mod_rs.contains("pub mod chunk_02;"));
|
|
1280
|
+
assert!(mod_rs.contains("pub use chunk_00::*;"));
|
|
1281
|
+
assert!(mod_rs.contains("pub use chunk_01::*;"));
|
|
1282
|
+
assert!(mod_rs.contains("pub use chunk_02::*;"));
|
|
1283
|
+
}
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{polyxml-0.31.0 → polyxml-0.32.0}/crates/polyxml-core/benches/support/tag_dispatch_fixtures.rs
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|