moxml 0.5.93 → 0.5.94

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
checksums.yaml CHANGED
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  SHA256:
3
- metadata.gz: 92d24efac79874e02cbfa26e20fc8135e2a714020a727896cdf2a460fa7d0508
4
- data.tar.gz: 91f18637388f6983c2c75c606e95e93f15919b3099fbdc0c8e59ebf9aca93d5e
3
+ metadata.gz: e54a93041f955c95affa2211704839bfaf47dafc0908cd73b58012946ffe99f1
4
+ data.tar.gz: 4d28d6376f61ae7441e13075ac5e9433bd18b6c393cf95afe363a639a4b8d251
5
5
  SHA512:
6
- metadata.gz: 91cd7d93def6717a7eedf675e871ff7d05d3dfb5378f8ddbe8f839daa68d0d1dbe0e1bdf6bdb939704bb03f673950c9faa858265321e2a209fbcb2042224a3cd
7
- data.tar.gz: c973d0559eb5fcc42f45dfb824758cdbec8758ec0dc378a4ce4034d918b3b50fcdada1120cd9f18b47c39a42055d4cd810225588dbcf598346c920a2a510e0c7
6
+ metadata.gz: 3e13a34f69a9e91b18d7fc9cef1e30c40ced24c8c73b322459ecba15af568439b002db8eec51097af2188571d026bfae7861646cc33077127c1fbd570b179863
7
+ data.tar.gz: a873c43e7ee3ccffc1997bf103263b7848d8ed17c61e25322a8d550bdf62a46d2f1e9e4a93f6e2832484a6df9d93350fd53d833a0b48ce04e1247583fc447d50
@@ -1942,7 +1942,14 @@ module Moxml
1942
1942
  end
1943
1943
 
1944
1944
  def processing_instruction_content(node)
1945
- return NN_CONTENT.bind_call(node) if NATIVE_READ_LAYER && node.is_a?(NN)
1945
+ if NATIVE_READ_LAYER && node.is_a?(NN)
1946
+ content = NN_CONTENT.bind_call(node)
1947
+ # NativeNode#content answers nil for PI-kind nodes too
1948
+ # (leptris-ruby#344 covers comments); bridge for the data.
1949
+ return to_binding(node).content if content.nil?
1950
+
1951
+ return content
1952
+ end
1946
1953
 
1947
1954
  node.content
1948
1955
  end
@@ -237,7 +237,9 @@ module Moxml
237
237
  # included element renders with its namespaces, attributes and
238
238
  # DIRECT character data, but child ELEMENTS render only if
239
239
  # matched too — an unmatched Signature subtree stays out even
240
- # though its ancestors are matched.
240
+ # though its ancestors are matched. Comments and PIs render
241
+ # only when the expression itself selects them (spec §3: the
242
+ # node-set is literal).
241
243
  def self.mark_subset_paths(root_node, paths)
242
244
  mark_all(root_node, false)
243
245
  paths.each do |path|
@@ -249,7 +251,7 @@ module Moxml
249
251
  node.namespace_nodes.each { |ns| ns.in_node_set = true }
250
252
  node.attribute_nodes.each { |attr| attr.in_node_set = true }
251
253
  node.children.each do |child|
252
- child.in_node_set = true unless child.is_a?(Nodes::ElementNode)
254
+ child.in_node_set = true if child.is_a?(Nodes::TextNode)
253
255
  end
254
256
  else
255
257
  node.in_node_set = true
data/lib/moxml/version.rb CHANGED
@@ -1,5 +1,5 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  module Moxml
4
- VERSION = "0.5.93"
4
+ VERSION = "0.5.94"
5
5
  end
@@ -31,12 +31,21 @@ CORPUS = [
31
31
  ].freeze
32
32
 
33
33
  COMMENTS_CASE = [
34
- "with_comments keeps matched comments only",
34
+ "with_comments renders explicitly selected comments only",
35
+ %(<a><!-- keep --><b><!-- in --><c/></b></a>), "//b | //comment()",
36
+ %(<!-- keep -->\n<b><!-- in --></b>)
37
+ ].freeze
38
+
39
+ COMMENTS_EXCLUDED_CASE = [
40
+ "unselected comments stay out",
35
41
  %(<a><!-- keep --><b><!-- in --><c/></b></a>), "//b",
36
- %(<b><!-- in --></b>)
42
+ %(<b></b>)
37
43
  ].freeze
38
44
 
39
45
  COMMENTS_PADDING_LOSS = %i[ox headed_ox].freeze
46
+ # rexml's adapter xpath does not return comment()/text() node results
47
+ # (pre-existing capability gap, same family for both axes).
48
+ COMMENT_SELECTION_GAP = %i[rexml].freeze
40
49
 
41
50
  RSpec.describe "Moxml::C14n subset canonicalization" do
42
51
  # Node-set semantics follow the enveloped-signature interop
@@ -63,8 +72,11 @@ RSpec.describe "Moxml::C14n subset canonicalization" do
63
72
  end
64
73
 
65
74
  # ox/headed_ox strip comment content padding at the parse
66
- # layer (pre-existing, same family as their PI content loss).
67
- unless COMMENTS_PADDING_LOSS.include?(adapter_name)
75
+ # layer (pre-existing, same family as their PI content loss);
76
+ # only the rendered-comment case depends on it. rexml's
77
+ # comment() selection gap also only affects this case.
78
+ unless COMMENTS_PADDING_LOSS.include?(adapter_name) ||
79
+ COMMENT_SELECTION_GAP.include?(adapter_name)
68
80
  it COMMENTS_CASE[0] do
69
81
  doc = ctx.parse(COMMENTS_CASE[1])
70
82
  expect(Moxml::C14n.canonicalize_subset(doc, COMMENTS_CASE[2],
@@ -72,6 +84,13 @@ RSpec.describe "Moxml::C14n subset canonicalization" do
72
84
  .to eq(COMMENTS_CASE[3])
73
85
  end
74
86
  end
87
+
88
+ it COMMENTS_EXCLUDED_CASE[0] do
89
+ doc = ctx.parse(COMMENTS_EXCLUDED_CASE[1])
90
+ expect(Moxml::C14n.canonicalize_subset(doc, COMMENTS_EXCLUDED_CASE[2],
91
+ with_comments: true))
92
+ .to eq(COMMENTS_EXCLUDED_CASE[3])
93
+ end
75
94
  end
76
95
  end
77
96
  end
@@ -0,0 +1,104 @@
1
+ # frozen_string: true
2
+
3
+ require "spec_helper"
4
+ require "moxml/c14n"
5
+
6
+ # Shapes drawn from the W3C Canonical XML 1.0 REC examples (§2–§3):
7
+ # attribute-value whitespace normalization, character-reference CR
8
+ # preservation vs literal-CR folding, CDATA as character data, empty
9
+ # element form, PI/comment placement, and the §3 document-subset
10
+ # whitespace rules. Expected bytes are the forms all conformant
11
+ # adapters answer identically.
12
+ #
13
+ # Documented parse-layer divergences (skipped rows):
14
+ # - ox/headed_ox expand &#xD; to LF, strip inter-element whitespace
15
+ # and comment padding at parse (out of scope here)
16
+ # - oga keeps the target/content separator space in PI data
17
+ W3C_FULL_FORMS = [
18
+ ["attribute-value whitespace collapses to single spaces",
19
+ "<doc>\n <e attr1=\"v1\"\n attr2 = \"v2\"\n > content </e>\n</doc>",
20
+ "<doc>\n <e attr1=\"v1\" attr2=\"v2\"> content </e>\n</doc>"],
21
+ ["character-reference CR stays &#xD;; literal CR folds to LF",
22
+ %(<doc>a&#xD;b\nc&#xA;d</doc>),
23
+ %(<doc>a&#xD;b\nc\nd</doc>)],
24
+ ["CDATA is character data and escapes",
25
+ %(<doc><![CDATA[a<b]]> &amp; &lt; x</doc>),
26
+ %(<doc>a&lt;b &amp; &lt; x</doc>)],
27
+ ["empty elements render as start and end tags",
28
+ %(<doc><e/><f></f></doc>),
29
+ "<doc><e></e><f></f></doc>"],
30
+ ["PIs render in place; comments drop without with_comments",
31
+ %(<doc><?pi c?><!-- c1 --><e/><!-- c2 --></doc>),
32
+ %(<doc><?pi c?><e></e></doc>)],
33
+ ].freeze
34
+
35
+ W3C_SUBSET_FORM = [
36
+ %(<doc>\n <e1 a="1" b="2"/>\n <e2/>\n <e3/>\n <!-- A small comment -->\n <?pi x?>\n</doc>),
37
+ "//e1 | //e2 | //e3",
38
+ %(<e1 a="1" b="2"></e1><e2></e2><e3></e3>),
39
+ ].freeze
40
+
41
+ W3C_SUBSET_WITH_ROOT = [
42
+ %(<doc>\n <e1 a="1" b="2"/>\n <e2/>\n <e3/>\n <!-- A small comment -->\n <?pi x?>\n</doc>),
43
+ "/doc | //e1 | //e2 | //e3",
44
+ %(<doc>\n <e1 a="1" b="2"></e1>\n <e2></e2>\n <e3></e3>\n \n \n</doc>),
45
+ ].freeze
46
+
47
+ CR_SHAPE = 1
48
+ PI_SHAPE = 4
49
+ WHITESPACE_SHAPES = %i[ox headed_ox].freeze
50
+
51
+ RSpec.describe "Moxml::C14n W3C REC-xml-c14n examples" do
52
+ Moxml::Adapter::AVAILABLE_ADAPTERS.each do |adapter_name|
53
+ context "with the #{adapter_name} adapter" do
54
+ let(:ctx) { Moxml.new(adapter_name) }
55
+
56
+ W3C_FULL_FORMS.each_with_index do |(label, xml, expected), idx|
57
+ # ox/headed_ox strip inter-element whitespace, expand &#xD;
58
+ # and pad comments at parse; oga keeps the PI separator space.
59
+ skip_shapes = { ox: [0, CR_SHAPE, PI_SHAPE],
60
+ headed_ox: [0, CR_SHAPE, PI_SHAPE],
61
+ oga: [PI_SHAPE] }[adapter_name] || []
62
+ next if skip_shapes.include?(idx)
63
+
64
+ it "canonicalizes #{label}" do
65
+ doc = ctx.parse(xml)
66
+ expect(Moxml::C14n.canonicalize_inclusive10(doc)).to eq(expected)
67
+ expect(Moxml::C14n.canonicalize(doc.root)).to eq(expected)
68
+ end
69
+ end
70
+ end
71
+ end
72
+
73
+ describe "W3C §3 document-subset whitespace" do
74
+ Moxml::Adapter::AVAILABLE_ADAPTERS.each do |adapter_name|
75
+ next if WHITESPACE_SHAPES.include?(adapter_name)
76
+
77
+ it "renders selected nodes only, whitespace runs in place (#{adapter_name})" do
78
+ doc = Moxml.new(adapter_name).parse(W3C_SUBSET_FORM[0])
79
+ expect(Moxml::C14n.canonicalize_subset(doc, W3C_SUBSET_FORM[1]))
80
+ .to eq(W3C_SUBSET_FORM[2])
81
+ end
82
+
83
+ it "keeps the root's whitespace runs when the root is selected (#{adapter_name})" do
84
+ doc = Moxml.new(adapter_name).parse(W3C_SUBSET_WITH_ROOT[0])
85
+ expect(Moxml::C14n.canonicalize_subset(doc, W3C_SUBSET_WITH_ROOT[1]))
86
+ .to eq(W3C_SUBSET_WITH_ROOT[2])
87
+ end
88
+ end
89
+ end
90
+
91
+ describe "Ruby reference agrees with the native engines" do
92
+ %i[leptris nokogiri].each do |adapter_name|
93
+ next unless Moxml::Adapter::AVAILABLE_ADAPTERS.include?(adapter_name)
94
+
95
+ it "on every full-document shape (#{adapter_name})" do
96
+ W3C_FULL_FORMS.each do |_label, xml, _expected|
97
+ root = Moxml.new(adapter_name).parse(xml).root
98
+ expect(Moxml::C14n.canonicalize(root))
99
+ .to eq(Moxml::C14n.canonicalize_inclusive10(root))
100
+ end
101
+ end
102
+ end
103
+ end
104
+ end
metadata CHANGED
@@ -1,14 +1,14 @@
1
1
  --- !ruby/object:Gem::Specification
2
2
  name: moxml
3
3
  version: !ruby/object:Gem::Version
4
- version: 0.5.93
4
+ version: 0.5.94
5
5
  platform: ruby
6
6
  authors:
7
7
  - Ribose Inc.
8
8
  autorequire:
9
9
  bindir: exe
10
10
  cert_chain: []
11
- date: 2026-09-28 00:00:00.000000000 Z
11
+ date: 2026-09-30 00:00:00.000000000 Z
12
12
  dependencies: []
13
13
  description: |
14
14
  Moxml is a unified XML manipulation library that provides a common API
@@ -464,6 +464,7 @@ files:
464
464
  - spec/moxml/c14n/inclusive10_spec.rb
465
465
  - spec/moxml/c14n/namespace_edge_cases_spec.rb
466
466
  - spec/moxml/c14n/subset_spec.rb
467
+ - spec/moxml/c14n/w3c_examples_spec.rb
467
468
  - spec/moxml/c14n/xml_attributes_spec.rb
468
469
  - spec/moxml/cdata_spec.rb
469
470
  - spec/moxml/comment_spec.rb