moxml 0.5.93 → 0.5.95

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
checksums.yaml CHANGED
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  SHA256:
3
- metadata.gz: 92d24efac79874e02cbfa26e20fc8135e2a714020a727896cdf2a460fa7d0508
4
- data.tar.gz: 91f18637388f6983c2c75c606e95e93f15919b3099fbdc0c8e59ebf9aca93d5e
3
+ metadata.gz: de8e16de76cef81c34b0a4e097d5aa16196fdbc03d9bfe2cbee1e765312f1599
4
+ data.tar.gz: ab5cc00b4c1a4a209065620c89b7541e107ed3d0a80e1e91e4455cb10c17fd98
5
5
  SHA512:
6
- metadata.gz: 91cd7d93def6717a7eedf675e871ff7d05d3dfb5378f8ddbe8f839daa68d0d1dbe0e1bdf6bdb939704bb03f673950c9faa858265321e2a209fbcb2042224a3cd
7
- data.tar.gz: c973d0559eb5fcc42f45dfb824758cdbec8758ec0dc378a4ce4034d918b3b50fcdada1120cd9f18b47c39a42055d4cd810225588dbcf598346c920a2a510e0c7
6
+ metadata.gz: bf21f0b903d142e6ebce40004d4e3bb657ec812396a2e8e6bd49c2eb06b6763fa6917f18c6918a134784d279360092c127fad56b2a75449f21d7e2dc94ae2e3c
7
+ data.tar.gz: 0bc9b8912c18c946101b76c88ce2c5c1d7f3ab2d8f643d06a8760d78332e1fd0d426844caff21efd8bb5241af9b35471478223fff9754b78405c3dda28d1990b
@@ -80,6 +80,24 @@ adapter = element.context.config.adapter
80
80
  element = Moxml::Element.new(native_node, context)
81
81
  ----
82
82
 
83
+ === Parse bookkeeping opt-outs (leptris)
84
+
85
+ Data-pipeline parses that never read diagnostics or source positions can
86
+ skip that bookkeeping. Both are opt-in per call; the default parse is
87
+ byte-identical either way.
88
+
89
+ [source,ruby]
90
+ ----
91
+ # Duplicate attributes admit silently: queries answer first-wins, but
92
+ # the duplicate stays in the tree, so the serialized form can carry it.
93
+ doc = Moxml.new.parse(xml, skip_dup_detection: true)
94
+
95
+ # Source columns degrade; lines still resolve.
96
+ doc = Moxml.new.parse(xml, skip_source_positions: true)
97
+ ----
98
+
99
+ Without the flags, a duplicate attribute records a recover diagnostic in
100
+ `doc.parse_diagnostics` and parses with the first value.
83
101
 
84
102
  See also:
85
103
 
@@ -856,6 +856,17 @@ module Moxml
856
856
  # Nokogiri semantics: blanks kept, no ATTLIST defaults).
857
857
  # noblanks forwards only once the engine flag is libxml2-safe
858
858
  # (see ENGINE_NOBLANKS_SAFE); otherwise it is moxml-side.
859
+ #
860
+ # The SKIP_ bookkeeping opt-outs (leptris-ruby#352, engine
861
+ # 1.9.272): duplicate attributes admit silently (first wins,
862
+ # no recover diagnostic) and source columns degrade (lines
863
+ # still resolve). Caller opt-in only — never defaults. The
864
+ # constant probe is the dependency-boundary check (the gem
865
+ # is zero-dep; user-resolved binding versions predate them).
866
+ NATIVE_PARSE_SKIPS =
867
+ ::Leptris::XML::ParseOptions.const_defined?(:SKIP_DUP_DETECTION) &&
868
+ ::Leptris::XML::ParseOptions.const_defined?(:SKIP_SOURCE_POSITIONS)
869
+
859
870
  def parse_flags(options, context: nil)
860
871
  flags = 0
861
872
  flags |= ::Leptris::XML::ParseOptions::DTDATTR if options[:dtdattr] == true
@@ -863,6 +874,11 @@ module Moxml
863
874
  (context&.config&.entity_mode == :keep ||
864
875
  options[:keep_entity_refs] == true)
865
876
  flags |= ::Leptris::XML::ParseOptions::NOBLANKS if options[:noblanks] == true && ENGINE_NOBLANKS_SAFE
877
+ if NATIVE_PARSE_SKIPS
878
+ po = ::Leptris::XML::ParseOptions
879
+ flags |= po::SKIP_DUP_DETECTION if options[:skip_dup_detection] == true
880
+ flags |= po::SKIP_SOURCE_POSITIONS if options[:skip_source_positions] == true
881
+ end
866
882
  flags.zero? ? nil : ::Leptris::XML::ParseOptions.new(flags)
867
883
  end
868
884
 
@@ -1942,7 +1958,14 @@ module Moxml
1942
1958
  end
1943
1959
 
1944
1960
  def processing_instruction_content(node)
1945
- return NN_CONTENT.bind_call(node) if NATIVE_READ_LAYER && node.is_a?(NN)
1961
+ if NATIVE_READ_LAYER && node.is_a?(NN)
1962
+ content = NN_CONTENT.bind_call(node)
1963
+ # NativeNode#content answers nil for PI-kind nodes too
1964
+ # (leptris-ruby#344 covers comments); bridge for the data.
1965
+ return to_binding(node).content if content.nil?
1966
+
1967
+ return content
1968
+ end
1946
1969
 
1947
1970
  node.content
1948
1971
  end
@@ -59,6 +59,7 @@ module Moxml
59
59
  # (Document/C14n), which is circular while THIS adapter file
60
60
  # is still loading.
61
61
  def native_inclusive10(native)
62
+ return nil if native.nil?
62
63
  return nil unless native_c14n_byte_safe?
63
64
 
64
65
  native.canonicalize(::Nokogiri::XML::XML_C14N_1_0)
@@ -237,7 +237,9 @@ module Moxml
237
237
  # included element renders with its namespaces, attributes and
238
238
  # DIRECT character data, but child ELEMENTS render only if
239
239
  # matched too — an unmatched Signature subtree stays out even
240
- # though its ancestors are matched.
240
+ # though its ancestors are matched. Comments and PIs render
241
+ # only when the expression itself selects them (spec §3: the
242
+ # node-set is literal).
241
243
  def self.mark_subset_paths(root_node, paths)
242
244
  mark_all(root_node, false)
243
245
  paths.each do |path|
@@ -249,7 +251,7 @@ module Moxml
249
251
  node.namespace_nodes.each { |ns| ns.in_node_set = true }
250
252
  node.attribute_nodes.each { |attr| attr.in_node_set = true }
251
253
  node.children.each do |child|
252
- child.in_node_set = true unless child.is_a?(Nodes::ElementNode)
254
+ child.in_node_set = true if child.is_a?(Nodes::TextNode)
253
255
  end
254
256
  else
255
257
  node.in_node_set = true
data/lib/moxml/version.rb CHANGED
@@ -1,5 +1,5 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  module Moxml
4
- VERSION = "0.5.93"
4
+ VERSION = "0.5.95"
5
5
  end
@@ -0,0 +1,58 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "spec_helper"
4
+
5
+ DUP_XML = %(<r><e id="1">text</e><e id="2" id="3"/></r>)
6
+ POS_XML = %(<r><e id="1">text</e></r>)
7
+
8
+ RSpec.describe "Moxml leptris parse bookkeeping opt-outs" do
9
+ # leptris-ruby#352 (binding 1.9.273.1, engine 1.9.272 Door A):
10
+ # SKIP_DUP_DETECTION admits duplicate attributes silently (first
11
+ # wins, no recover diagnostic); SKIP_SOURCE_POSITIONS degrades
12
+ # source columns (lines still resolve). Caller opt-in only.
13
+ let(:ctx) { Moxml.new(:leptris) }
14
+ let(:adapter) { ctx.config.adapter }
15
+
16
+ before do
17
+ skip "binding without the SKIP_ parse flags" unless
18
+ Leptris::XML::ParseOptions.const_defined?(:SKIP_DUP_DETECTION)
19
+ end
20
+
21
+ it "drops the duplicate-attribute diagnostic when requested" do
22
+ doc = ctx.parse(DUP_XML, skip_dup_detection: true)
23
+ expect(adapter.parse_diagnostics(doc.native)).to eq([])
24
+ end
25
+
26
+ it "keeps the duplicate-attribute diagnostic by default" do
27
+ doc = ctx.parse(DUP_XML)
28
+ expect(adapter.parse_diagnostics(doc.native).map { |d| d[:kind] }).to include(:recover)
29
+ end
30
+
31
+ it "first attribute wins either way" do
32
+ expect(ctx.parse(DUP_XML, skip_dup_detection: true).root.children[1][:id])
33
+ .to eq(ctx.parse(DUP_XML).root.children[1][:id])
34
+ end
35
+
36
+ it "keeps lines but degrades columns for skipped positions" do
37
+ full = ctx.parse(POS_XML).root.children[0].source_position
38
+ skipped = ctx.parse(POS_XML, skip_source_positions: true)
39
+ .root.children[0].source_position
40
+ expect(skipped[:line]).to eq(full[:line])
41
+ expect(skipped[:col_start]).to be < full[:col_start]
42
+ end
43
+
44
+ it "keeps the duplicate in the tree — queries still answer first-wins" do
45
+ # The engine admits the duplicate silently: reads see the first
46
+ # value, but the second attribute stays in the tree, so the
47
+ # serialized form can carry it. Callers opting in trade the
48
+ # well-formed-output guarantee for the bookkeeping speed.
49
+ fast = ctx.parse(DUP_XML, skip_dup_detection: true)
50
+ expect(fast.root.children[1][:id]).to eq("2")
51
+ expect(fast.to_xml).to include('id="2" id="3"')
52
+ end
53
+
54
+ it "still raises on fatal errors with skips on" do
55
+ expect { ctx.parse("<r><e></r>", skip_dup_detection: true) }
56
+ .to raise_error(Moxml::ParseError)
57
+ end
58
+ end
@@ -31,12 +31,21 @@ CORPUS = [
31
31
  ].freeze
32
32
 
33
33
  COMMENTS_CASE = [
34
- "with_comments keeps matched comments only",
34
+ "with_comments renders explicitly selected comments only",
35
+ %(<a><!-- keep --><b><!-- in --><c/></b></a>), "//b | //comment()",
36
+ %(<!-- keep -->\n<b><!-- in --></b>)
37
+ ].freeze
38
+
39
+ COMMENTS_EXCLUDED_CASE = [
40
+ "unselected comments stay out",
35
41
  %(<a><!-- keep --><b><!-- in --><c/></b></a>), "//b",
36
- %(<b><!-- in --></b>)
42
+ %(<b></b>)
37
43
  ].freeze
38
44
 
39
45
  COMMENTS_PADDING_LOSS = %i[ox headed_ox].freeze
46
+ # rexml's adapter xpath does not return comment()/text() node results
47
+ # (pre-existing capability gap, same family for both axes).
48
+ COMMENT_SELECTION_GAP = %i[rexml].freeze
40
49
 
41
50
  RSpec.describe "Moxml::C14n subset canonicalization" do
42
51
  # Node-set semantics follow the enveloped-signature interop
@@ -63,8 +72,11 @@ RSpec.describe "Moxml::C14n subset canonicalization" do
63
72
  end
64
73
 
65
74
  # ox/headed_ox strip comment content padding at the parse
66
- # layer (pre-existing, same family as their PI content loss).
67
- unless COMMENTS_PADDING_LOSS.include?(adapter_name)
75
+ # layer (pre-existing, same family as their PI content loss);
76
+ # only the rendered-comment case depends on it. rexml's
77
+ # comment() selection gap also only affects this case.
78
+ unless COMMENTS_PADDING_LOSS.include?(adapter_name) ||
79
+ COMMENT_SELECTION_GAP.include?(adapter_name)
68
80
  it COMMENTS_CASE[0] do
69
81
  doc = ctx.parse(COMMENTS_CASE[1])
70
82
  expect(Moxml::C14n.canonicalize_subset(doc, COMMENTS_CASE[2],
@@ -72,6 +84,13 @@ RSpec.describe "Moxml::C14n subset canonicalization" do
72
84
  .to eq(COMMENTS_CASE[3])
73
85
  end
74
86
  end
87
+
88
+ it COMMENTS_EXCLUDED_CASE[0] do
89
+ doc = ctx.parse(COMMENTS_EXCLUDED_CASE[1])
90
+ expect(Moxml::C14n.canonicalize_subset(doc, COMMENTS_EXCLUDED_CASE[2],
91
+ with_comments: true))
92
+ .to eq(COMMENTS_EXCLUDED_CASE[3])
93
+ end
75
94
  end
76
95
  end
77
96
  end
@@ -0,0 +1,104 @@
1
+ # frozen_string: true
2
+
3
+ require "spec_helper"
4
+ require "moxml/c14n"
5
+
6
+ # Shapes drawn from the W3C Canonical XML 1.0 REC examples (§2–§3):
7
+ # attribute-value whitespace normalization, character-reference CR
8
+ # preservation vs literal-CR folding, CDATA as character data, empty
9
+ # element form, PI/comment placement, and the §3 document-subset
10
+ # whitespace rules. Expected bytes are the forms all conformant
11
+ # adapters answer identically.
12
+ #
13
+ # Documented parse-layer divergences (skipped rows):
14
+ # - ox/headed_ox expand &#xD; to LF, strip inter-element whitespace
15
+ # and comment padding at parse (out of scope here)
16
+ # - oga keeps the target/content separator space in PI data
17
+ W3C_FULL_FORMS = [
18
+ ["attribute-value whitespace collapses to single spaces",
19
+ "<doc>\n <e attr1=\"v1\"\n attr2 = \"v2\"\n > content </e>\n</doc>",
20
+ "<doc>\n <e attr1=\"v1\" attr2=\"v2\"> content </e>\n</doc>"],
21
+ ["character-reference CR stays &#xD;; literal CR folds to LF",
22
+ %(<doc>a&#xD;b\nc&#xA;d</doc>),
23
+ %(<doc>a&#xD;b\nc\nd</doc>)],
24
+ ["CDATA is character data and escapes",
25
+ %(<doc><![CDATA[a<b]]> &amp; &lt; x</doc>),
26
+ %(<doc>a&lt;b &amp; &lt; x</doc>)],
27
+ ["empty elements render as start and end tags",
28
+ %(<doc><e/><f></f></doc>),
29
+ "<doc><e></e><f></f></doc>"],
30
+ ["PIs render in place; comments drop without with_comments",
31
+ %(<doc><?pi c?><!-- c1 --><e/><!-- c2 --></doc>),
32
+ %(<doc><?pi c?><e></e></doc>)],
33
+ ].freeze
34
+
35
+ W3C_SUBSET_FORM = [
36
+ %(<doc>\n <e1 a="1" b="2"/>\n <e2/>\n <e3/>\n <!-- A small comment -->\n <?pi x?>\n</doc>),
37
+ "//e1 | //e2 | //e3",
38
+ %(<e1 a="1" b="2"></e1><e2></e2><e3></e3>),
39
+ ].freeze
40
+
41
+ W3C_SUBSET_WITH_ROOT = [
42
+ %(<doc>\n <e1 a="1" b="2"/>\n <e2/>\n <e3/>\n <!-- A small comment -->\n <?pi x?>\n</doc>),
43
+ "/doc | //e1 | //e2 | //e3",
44
+ %(<doc>\n <e1 a="1" b="2"></e1>\n <e2></e2>\n <e3></e3>\n \n \n</doc>),
45
+ ].freeze
46
+
47
+ CR_SHAPE = 1
48
+ PI_SHAPE = 4
49
+ WHITESPACE_SHAPES = %i[ox headed_ox].freeze
50
+
51
+ RSpec.describe "Moxml::C14n W3C REC-xml-c14n examples" do
52
+ Moxml::Adapter::AVAILABLE_ADAPTERS.each do |adapter_name|
53
+ context "with the #{adapter_name} adapter" do
54
+ let(:ctx) { Moxml.new(adapter_name) }
55
+
56
+ W3C_FULL_FORMS.each_with_index do |(label, xml, expected), idx|
57
+ # ox/headed_ox strip inter-element whitespace, expand &#xD;
58
+ # and pad comments at parse; oga keeps the PI separator space.
59
+ skip_shapes = { ox: [0, CR_SHAPE, PI_SHAPE],
60
+ headed_ox: [0, CR_SHAPE, PI_SHAPE],
61
+ oga: [PI_SHAPE] }[adapter_name] || []
62
+ next if skip_shapes.include?(idx)
63
+
64
+ it "canonicalizes #{label}" do
65
+ doc = ctx.parse(xml)
66
+ expect(Moxml::C14n.canonicalize_inclusive10(doc)).to eq(expected)
67
+ expect(Moxml::C14n.canonicalize(doc.root)).to eq(expected)
68
+ end
69
+ end
70
+ end
71
+ end
72
+
73
+ describe "W3C §3 document-subset whitespace" do
74
+ Moxml::Adapter::AVAILABLE_ADAPTERS.each do |adapter_name|
75
+ next if WHITESPACE_SHAPES.include?(adapter_name)
76
+
77
+ it "renders selected nodes only, whitespace runs in place (#{adapter_name})" do
78
+ doc = Moxml.new(adapter_name).parse(W3C_SUBSET_FORM[0])
79
+ expect(Moxml::C14n.canonicalize_subset(doc, W3C_SUBSET_FORM[1]))
80
+ .to eq(W3C_SUBSET_FORM[2])
81
+ end
82
+
83
+ it "keeps the root's whitespace runs when the root is selected (#{adapter_name})" do
84
+ doc = Moxml.new(adapter_name).parse(W3C_SUBSET_WITH_ROOT[0])
85
+ expect(Moxml::C14n.canonicalize_subset(doc, W3C_SUBSET_WITH_ROOT[1]))
86
+ .to eq(W3C_SUBSET_WITH_ROOT[2])
87
+ end
88
+ end
89
+ end
90
+
91
+ describe "Ruby reference agrees with the native engines" do
92
+ %i[leptris nokogiri].each do |adapter_name|
93
+ next unless Moxml::Adapter::AVAILABLE_ADAPTERS.include?(adapter_name)
94
+
95
+ it "on every full-document shape (#{adapter_name})" do
96
+ W3C_FULL_FORMS.each do |_label, xml, _expected|
97
+ root = Moxml.new(adapter_name).parse(xml).root
98
+ expect(Moxml::C14n.canonicalize(root))
99
+ .to eq(Moxml::C14n.canonicalize_inclusive10(root))
100
+ end
101
+ end
102
+ end
103
+ end
104
+ end
@@ -2,6 +2,10 @@
2
2
 
3
3
  require "spec_helper"
4
4
 
5
+ # The per-item target class for the typed plan: Struct gives the
6
+ # accessors StructPlan assigns the cast slots to.
7
+ TypedItem = Struct.new(:price, :qty, :ok, :label)
8
+
5
9
  # Typed plan scalars (leptris plan ABI, #1269a consumer face):
6
10
  # [slot, type] declarations cast in C — Integer/Float/TrueClass/
7
11
  # FalseClass members with no Ruby String materialized; unparseable
metadata CHANGED
@@ -1,14 +1,14 @@
1
1
  --- !ruby/object:Gem::Specification
2
2
  name: moxml
3
3
  version: !ruby/object:Gem::Version
4
- version: 0.5.93
4
+ version: 0.5.95
5
5
  platform: ruby
6
6
  authors:
7
7
  - Ribose Inc.
8
8
  autorequire:
9
9
  bindir: exe
10
10
  cert_chain: []
11
- date: 2026-09-28 00:00:00.000000000 Z
11
+ date: 2026-09-30 00:00:00.000000000 Z
12
12
  dependencies: []
13
13
  description: |
14
14
  Moxml is a unified XML manipulation library that provides a common API
@@ -438,6 +438,7 @@ files:
438
438
  - spec/moxml/adapter/entity_restoration_spec.rb
439
439
  - spec/moxml/adapter/headed_ox_spec.rb
440
440
  - spec/moxml/adapter/leptris_doc_parts_spec.rb
441
+ - spec/moxml/adapter/leptris_parse_skips_spec.rb
441
442
  - spec/moxml/adapter/leptris_record_driver_spec.rb
442
443
  - spec/moxml/adapter/leptris_sax_dup_attr_spec.rb
443
444
  - spec/moxml/adapter/leptris_sax_records_spec.rb
@@ -464,6 +465,7 @@ files:
464
465
  - spec/moxml/c14n/inclusive10_spec.rb
465
466
  - spec/moxml/c14n/namespace_edge_cases_spec.rb
466
467
  - spec/moxml/c14n/subset_spec.rb
468
+ - spec/moxml/c14n/w3c_examples_spec.rb
467
469
  - spec/moxml/c14n/xml_attributes_spec.rb
468
470
  - spec/moxml/cdata_spec.rb
469
471
  - spec/moxml/comment_spec.rb