moxml 0.5.93 → 0.5.94
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/lib/moxml/adapter/leptris.rb +8 -1
- data/lib/moxml/c14n/data_model.rb +4 -2
- data/lib/moxml/version.rb +1 -1
- data/spec/moxml/c14n/subset_spec.rb +23 -4
- data/spec/moxml/c14n/w3c_examples_spec.rb +104 -0
- metadata +3 -2
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
SHA256:
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
3
|
+
metadata.gz: e54a93041f955c95affa2211704839bfaf47dafc0908cd73b58012946ffe99f1
|
|
4
|
+
data.tar.gz: 4d28d6376f61ae7441e13075ac5e9433bd18b6c393cf95afe363a639a4b8d251
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: 3e13a34f69a9e91b18d7fc9cef1e30c40ced24c8c73b322459ecba15af568439b002db8eec51097af2188571d026bfae7861646cc33077127c1fbd570b179863
|
|
7
|
+
data.tar.gz: a873c43e7ee3ccffc1997bf103263b7848d8ed17c61e25322a8d550bdf62a46d2f1e9e4a93f6e2832484a6df9d93350fd53d833a0b48ce04e1247583fc447d50
|
|
@@ -1942,7 +1942,14 @@ module Moxml
|
|
|
1942
1942
|
end
|
|
1943
1943
|
|
|
1944
1944
|
def processing_instruction_content(node)
|
|
1945
|
-
|
|
1945
|
+
if NATIVE_READ_LAYER && node.is_a?(NN)
|
|
1946
|
+
content = NN_CONTENT.bind_call(node)
|
|
1947
|
+
# NativeNode#content answers nil for PI-kind nodes too
|
|
1948
|
+
# (leptris-ruby#344 covers comments); bridge for the data.
|
|
1949
|
+
return to_binding(node).content if content.nil?
|
|
1950
|
+
|
|
1951
|
+
return content
|
|
1952
|
+
end
|
|
1946
1953
|
|
|
1947
1954
|
node.content
|
|
1948
1955
|
end
|
|
@@ -237,7 +237,9 @@ module Moxml
|
|
|
237
237
|
# included element renders with its namespaces, attributes and
|
|
238
238
|
# DIRECT character data, but child ELEMENTS render only if
|
|
239
239
|
# matched too — an unmatched Signature subtree stays out even
|
|
240
|
-
# though its ancestors are matched.
|
|
240
|
+
# though its ancestors are matched. Comments and PIs render
|
|
241
|
+
# only when the expression itself selects them (spec §3: the
|
|
242
|
+
# node-set is literal).
|
|
241
243
|
def self.mark_subset_paths(root_node, paths)
|
|
242
244
|
mark_all(root_node, false)
|
|
243
245
|
paths.each do |path|
|
|
@@ -249,7 +251,7 @@ module Moxml
|
|
|
249
251
|
node.namespace_nodes.each { |ns| ns.in_node_set = true }
|
|
250
252
|
node.attribute_nodes.each { |attr| attr.in_node_set = true }
|
|
251
253
|
node.children.each do |child|
|
|
252
|
-
child.in_node_set = true
|
|
254
|
+
child.in_node_set = true if child.is_a?(Nodes::TextNode)
|
|
253
255
|
end
|
|
254
256
|
else
|
|
255
257
|
node.in_node_set = true
|
data/lib/moxml/version.rb
CHANGED
|
@@ -31,12 +31,21 @@ CORPUS = [
|
|
|
31
31
|
].freeze
|
|
32
32
|
|
|
33
33
|
COMMENTS_CASE = [
|
|
34
|
-
"with_comments
|
|
34
|
+
"with_comments renders explicitly selected comments only",
|
|
35
|
+
%(<a><!-- keep --><b><!-- in --><c/></b></a>), "//b | //comment()",
|
|
36
|
+
%(<!-- keep -->\n<b><!-- in --></b>)
|
|
37
|
+
].freeze
|
|
38
|
+
|
|
39
|
+
COMMENTS_EXCLUDED_CASE = [
|
|
40
|
+
"unselected comments stay out",
|
|
35
41
|
%(<a><!-- keep --><b><!-- in --><c/></b></a>), "//b",
|
|
36
|
-
%(<b
|
|
42
|
+
%(<b></b>)
|
|
37
43
|
].freeze
|
|
38
44
|
|
|
39
45
|
COMMENTS_PADDING_LOSS = %i[ox headed_ox].freeze
|
|
46
|
+
# rexml's adapter xpath does not return comment()/text() node results
|
|
47
|
+
# (pre-existing capability gap, same family for both axes).
|
|
48
|
+
COMMENT_SELECTION_GAP = %i[rexml].freeze
|
|
40
49
|
|
|
41
50
|
RSpec.describe "Moxml::C14n subset canonicalization" do
|
|
42
51
|
# Node-set semantics follow the enveloped-signature interop
|
|
@@ -63,8 +72,11 @@ RSpec.describe "Moxml::C14n subset canonicalization" do
|
|
|
63
72
|
end
|
|
64
73
|
|
|
65
74
|
# ox/headed_ox strip comment content padding at the parse
|
|
66
|
-
# layer (pre-existing, same family as their PI content loss)
|
|
67
|
-
|
|
75
|
+
# layer (pre-existing, same family as their PI content loss);
|
|
76
|
+
# only the rendered-comment case depends on it. rexml's
|
|
77
|
+
# comment() selection gap also only affects this case.
|
|
78
|
+
unless COMMENTS_PADDING_LOSS.include?(adapter_name) ||
|
|
79
|
+
COMMENT_SELECTION_GAP.include?(adapter_name)
|
|
68
80
|
it COMMENTS_CASE[0] do
|
|
69
81
|
doc = ctx.parse(COMMENTS_CASE[1])
|
|
70
82
|
expect(Moxml::C14n.canonicalize_subset(doc, COMMENTS_CASE[2],
|
|
@@ -72,6 +84,13 @@ RSpec.describe "Moxml::C14n subset canonicalization" do
|
|
|
72
84
|
.to eq(COMMENTS_CASE[3])
|
|
73
85
|
end
|
|
74
86
|
end
|
|
87
|
+
|
|
88
|
+
it COMMENTS_EXCLUDED_CASE[0] do
|
|
89
|
+
doc = ctx.parse(COMMENTS_EXCLUDED_CASE[1])
|
|
90
|
+
expect(Moxml::C14n.canonicalize_subset(doc, COMMENTS_EXCLUDED_CASE[2],
|
|
91
|
+
with_comments: true))
|
|
92
|
+
.to eq(COMMENTS_EXCLUDED_CASE[3])
|
|
93
|
+
end
|
|
75
94
|
end
|
|
76
95
|
end
|
|
77
96
|
end
|
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
# frozen_string: true
|
|
2
|
+
|
|
3
|
+
require "spec_helper"
|
|
4
|
+
require "moxml/c14n"
|
|
5
|
+
|
|
6
|
+
# Shapes drawn from the W3C Canonical XML 1.0 REC examples (§2–§3):
|
|
7
|
+
# attribute-value whitespace normalization, character-reference CR
|
|
8
|
+
# preservation vs literal-CR folding, CDATA as character data, empty
|
|
9
|
+
# element form, PI/comment placement, and the §3 document-subset
|
|
10
|
+
# whitespace rules. Expected bytes are the forms all conformant
|
|
11
|
+
# adapters answer identically.
|
|
12
|
+
#
|
|
13
|
+
# Documented parse-layer divergences (skipped rows):
|
|
14
|
+
# - ox/headed_ox expand 
 to LF, strip inter-element whitespace
|
|
15
|
+
# and comment padding at parse (out of scope here)
|
|
16
|
+
# - oga keeps the target/content separator space in PI data
|
|
17
|
+
W3C_FULL_FORMS = [
|
|
18
|
+
["attribute-value whitespace collapses to single spaces",
|
|
19
|
+
"<doc>\n <e attr1=\"v1\"\n attr2 = \"v2\"\n > content </e>\n</doc>",
|
|
20
|
+
"<doc>\n <e attr1=\"v1\" attr2=\"v2\"> content </e>\n</doc>"],
|
|
21
|
+
["character-reference CR stays 
; literal CR folds to LF",
|
|
22
|
+
%(<doc>a
b\nc
d</doc>),
|
|
23
|
+
%(<doc>a
b\nc\nd</doc>)],
|
|
24
|
+
["CDATA is character data and escapes",
|
|
25
|
+
%(<doc><![CDATA[a<b]]> & < x</doc>),
|
|
26
|
+
%(<doc>a<b & < x</doc>)],
|
|
27
|
+
["empty elements render as start and end tags",
|
|
28
|
+
%(<doc><e/><f></f></doc>),
|
|
29
|
+
"<doc><e></e><f></f></doc>"],
|
|
30
|
+
["PIs render in place; comments drop without with_comments",
|
|
31
|
+
%(<doc><?pi c?><!-- c1 --><e/><!-- c2 --></doc>),
|
|
32
|
+
%(<doc><?pi c?><e></e></doc>)],
|
|
33
|
+
].freeze
|
|
34
|
+
|
|
35
|
+
W3C_SUBSET_FORM = [
|
|
36
|
+
%(<doc>\n <e1 a="1" b="2"/>\n <e2/>\n <e3/>\n <!-- A small comment -->\n <?pi x?>\n</doc>),
|
|
37
|
+
"//e1 | //e2 | //e3",
|
|
38
|
+
%(<e1 a="1" b="2"></e1><e2></e2><e3></e3>),
|
|
39
|
+
].freeze
|
|
40
|
+
|
|
41
|
+
W3C_SUBSET_WITH_ROOT = [
|
|
42
|
+
%(<doc>\n <e1 a="1" b="2"/>\n <e2/>\n <e3/>\n <!-- A small comment -->\n <?pi x?>\n</doc>),
|
|
43
|
+
"/doc | //e1 | //e2 | //e3",
|
|
44
|
+
%(<doc>\n <e1 a="1" b="2"></e1>\n <e2></e2>\n <e3></e3>\n \n \n</doc>),
|
|
45
|
+
].freeze
|
|
46
|
+
|
|
47
|
+
CR_SHAPE = 1
|
|
48
|
+
PI_SHAPE = 4
|
|
49
|
+
WHITESPACE_SHAPES = %i[ox headed_ox].freeze
|
|
50
|
+
|
|
51
|
+
RSpec.describe "Moxml::C14n W3C REC-xml-c14n examples" do
|
|
52
|
+
Moxml::Adapter::AVAILABLE_ADAPTERS.each do |adapter_name|
|
|
53
|
+
context "with the #{adapter_name} adapter" do
|
|
54
|
+
let(:ctx) { Moxml.new(adapter_name) }
|
|
55
|
+
|
|
56
|
+
W3C_FULL_FORMS.each_with_index do |(label, xml, expected), idx|
|
|
57
|
+
# ox/headed_ox strip inter-element whitespace, expand 
|
|
58
|
+
# and pad comments at parse; oga keeps the PI separator space.
|
|
59
|
+
skip_shapes = { ox: [0, CR_SHAPE, PI_SHAPE],
|
|
60
|
+
headed_ox: [0, CR_SHAPE, PI_SHAPE],
|
|
61
|
+
oga: [PI_SHAPE] }[adapter_name] || []
|
|
62
|
+
next if skip_shapes.include?(idx)
|
|
63
|
+
|
|
64
|
+
it "canonicalizes #{label}" do
|
|
65
|
+
doc = ctx.parse(xml)
|
|
66
|
+
expect(Moxml::C14n.canonicalize_inclusive10(doc)).to eq(expected)
|
|
67
|
+
expect(Moxml::C14n.canonicalize(doc.root)).to eq(expected)
|
|
68
|
+
end
|
|
69
|
+
end
|
|
70
|
+
end
|
|
71
|
+
end
|
|
72
|
+
|
|
73
|
+
describe "W3C §3 document-subset whitespace" do
|
|
74
|
+
Moxml::Adapter::AVAILABLE_ADAPTERS.each do |adapter_name|
|
|
75
|
+
next if WHITESPACE_SHAPES.include?(adapter_name)
|
|
76
|
+
|
|
77
|
+
it "renders selected nodes only, whitespace runs in place (#{adapter_name})" do
|
|
78
|
+
doc = Moxml.new(adapter_name).parse(W3C_SUBSET_FORM[0])
|
|
79
|
+
expect(Moxml::C14n.canonicalize_subset(doc, W3C_SUBSET_FORM[1]))
|
|
80
|
+
.to eq(W3C_SUBSET_FORM[2])
|
|
81
|
+
end
|
|
82
|
+
|
|
83
|
+
it "keeps the root's whitespace runs when the root is selected (#{adapter_name})" do
|
|
84
|
+
doc = Moxml.new(adapter_name).parse(W3C_SUBSET_WITH_ROOT[0])
|
|
85
|
+
expect(Moxml::C14n.canonicalize_subset(doc, W3C_SUBSET_WITH_ROOT[1]))
|
|
86
|
+
.to eq(W3C_SUBSET_WITH_ROOT[2])
|
|
87
|
+
end
|
|
88
|
+
end
|
|
89
|
+
end
|
|
90
|
+
|
|
91
|
+
describe "Ruby reference agrees with the native engines" do
|
|
92
|
+
%i[leptris nokogiri].each do |adapter_name|
|
|
93
|
+
next unless Moxml::Adapter::AVAILABLE_ADAPTERS.include?(adapter_name)
|
|
94
|
+
|
|
95
|
+
it "on every full-document shape (#{adapter_name})" do
|
|
96
|
+
W3C_FULL_FORMS.each do |_label, xml, _expected|
|
|
97
|
+
root = Moxml.new(adapter_name).parse(xml).root
|
|
98
|
+
expect(Moxml::C14n.canonicalize(root))
|
|
99
|
+
.to eq(Moxml::C14n.canonicalize_inclusive10(root))
|
|
100
|
+
end
|
|
101
|
+
end
|
|
102
|
+
end
|
|
103
|
+
end
|
|
104
|
+
end
|
metadata
CHANGED
|
@@ -1,14 +1,14 @@
|
|
|
1
1
|
--- !ruby/object:Gem::Specification
|
|
2
2
|
name: moxml
|
|
3
3
|
version: !ruby/object:Gem::Version
|
|
4
|
-
version: 0.5.
|
|
4
|
+
version: 0.5.94
|
|
5
5
|
platform: ruby
|
|
6
6
|
authors:
|
|
7
7
|
- Ribose Inc.
|
|
8
8
|
autorequire:
|
|
9
9
|
bindir: exe
|
|
10
10
|
cert_chain: []
|
|
11
|
-
date: 2026-09-
|
|
11
|
+
date: 2026-09-30 00:00:00.000000000 Z
|
|
12
12
|
dependencies: []
|
|
13
13
|
description: |
|
|
14
14
|
Moxml is a unified XML manipulation library that provides a common API
|
|
@@ -464,6 +464,7 @@ files:
|
|
|
464
464
|
- spec/moxml/c14n/inclusive10_spec.rb
|
|
465
465
|
- spec/moxml/c14n/namespace_edge_cases_spec.rb
|
|
466
466
|
- spec/moxml/c14n/subset_spec.rb
|
|
467
|
+
- spec/moxml/c14n/w3c_examples_spec.rb
|
|
467
468
|
- spec/moxml/c14n/xml_attributes_spec.rb
|
|
468
469
|
- spec/moxml/cdata_spec.rb
|
|
469
470
|
- spec/moxml/comment_spec.rb
|