makiri 0.10.0.rc2-aarch64-linux → 0.11.0-aarch64-linux
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/.github/workflows/libfuzzer.yml +1 -1
- data/.github/workflows/release.yml +11 -2
- data/.github/workflows/security.yml +7 -8
- data/CHANGELOG.md +206 -0
- data/NOKOGIRI_DIFFERENCES.md +165 -10
- data/README.md +8 -3
- data/Rakefile +39 -9
- data/lib/makiri/3.2/makiri.so +0 -0
- data/lib/makiri/3.3/makiri.so +0 -0
- data/lib/makiri/3.4/makiri.so +0 -0
- data/lib/makiri/4.0/makiri.so +0 -0
- data/lib/makiri/attr.rb +15 -0
- data/lib/makiri/cdata_section.rb +8 -0
- data/lib/makiri/clone_via_dup.rb +19 -0
- data/lib/makiri/comment.rb +7 -0
- data/lib/makiri/css.rb +5 -4
- data/lib/makiri/document_fragment.rb +5 -2
- data/lib/makiri/document_type.rb +2 -3
- data/lib/makiri/element.rb +9 -0
- data/lib/makiri/error.rb +26 -0
- data/lib/makiri/html/document.rb +19 -14
- data/lib/makiri/html/node_methods.rb +9 -8
- data/lib/makiri/html.rb +3 -3
- data/lib/makiri/node.rb +13 -72
- data/lib/makiri/node_path.rb +95 -0
- data/lib/makiri/node_set.rb +50 -25
- data/lib/makiri/processing_instruction.rb +10 -8
- data/lib/makiri/reader_aliases.rb +28 -0
- data/lib/makiri/text.rb +7 -0
- data/lib/makiri/version.rb +1 -1
- data/lib/makiri/xml/builder.rb +17 -8
- data/lib/makiri/xml/document.rb +23 -3
- data/lib/makiri/xml/node_methods.rb +26 -32
- data/lib/makiri/xml.rb +4 -3
- data/lib/makiri/xpath.rb +8 -8
- data/lib/makiri/xpath_context.rb +1 -1
- data/lib/makiri/xpath_syntax.rb +43 -0
- data/lib/makiri.rb +13 -7
- data/script/api_manifest.rb +25 -3
- data/script/check_alloc_failures.rb +350 -16
- data/script/check_unsafe_boundaries.rb +401 -64
- data/script/leaks_harness.rb +14 -1
- data/suppressions/ruby.supp +10 -8
- metadata +6 -1
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
SHA256:
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
3
|
+
metadata.gz: 5b203c0ffb03e49f10b8e30afd6b070d36fd054867b7b3fd0e1376778b9e4b94
|
|
4
|
+
data.tar.gz: 3a26a99a690546bcc5367fab22645b8c47bd65dec58df75df5a7a93950de3386
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: 5397148284ec99562ffe74e4f633f903405c420cc7593c996b600d13c65e9f94a48bf69a477a45063cd6e2a369b87f4d55f324bce1bcc0087b7f37cc6d73f5d4
|
|
7
|
+
data.tar.gz: 232787e75468ce0da009f1755538c062cabab7bdadd0a3a0eeb8b9e5f388f59eb0f4159e9f2b8152761f7c7c9e6619a3aa5568955358435d26f8a130e31a472e
|
|
@@ -86,9 +86,18 @@ jobs:
|
|
|
86
86
|
# names the GNU triple; this installs the std that goes with it.
|
|
87
87
|
# Empty elsewhere, so the other runners are unaffected.
|
|
88
88
|
targets: ${{ runner.os == 'Windows' && 'x86_64-pc-windows-gnu' || '' }}
|
|
89
|
-
|
|
89
|
+
# clang/lld/llvm are for the vendored Lexbor's LTO, not for bindgen: on
|
|
90
|
+
# Linux the whole chain has to be LLVM or extconf turns LTO off (gcc
|
|
91
|
+
# emits GIMPLE, which rust-lld cannot read - see the note in
|
|
92
|
+
# extconf.rb). It DETECTS them, so a runner without them still builds,
|
|
93
|
+
# just without LTO - which is what this job used to ship, silently,
|
|
94
|
+
# to every user who installs the precompiled gem (see the measured win
|
|
95
|
+
# in CLAUDE.md's "Vendored Lexbor is built with LTO" section: this is
|
|
96
|
+
# the job that produces those gems, and `ci.yml` installing the chain
|
|
97
|
+
# only exercises the fast path in CI, it does not ship it).
|
|
98
|
+
- name: Install libclang (bindgen) and the LLVM chain (Lexbor LTO)
|
|
90
99
|
if: runner.os == 'Linux'
|
|
91
|
-
run: sudo apt-get update && sudo apt-get install -y libclang-dev
|
|
100
|
+
run: sudo apt-get update && sudo apt-get install -y libclang-dev clang lld llvm
|
|
92
101
|
- uses: ruby/setup-ruby@v1
|
|
93
102
|
with:
|
|
94
103
|
ruby-version: ${{ matrix.ruby }}
|
|
@@ -55,7 +55,7 @@ jobs:
|
|
|
55
55
|
bundler-cache: true
|
|
56
56
|
|
|
57
57
|
- name: Run short fuzz under sanitizers
|
|
58
|
-
run: bundle exec rake fuzz:sanitize
|
|
58
|
+
run: FUZZ_TIME=30 bundle exec rake fuzz:sanitize
|
|
59
59
|
|
|
60
60
|
# macOS-only malloc-leak gate: ASan everywhere runs with detect_leaks=0 (Ruby
|
|
61
61
|
# and Lexbor are uninstrumented), so this is the ONLY automated leak check. It
|
|
@@ -98,9 +98,9 @@ jobs:
|
|
|
98
98
|
run: bundle exec rake leaks
|
|
99
99
|
|
|
100
100
|
# OOM-injection sweep: rebuilds with MAKIRI_ALLOC_INJECT=1 and fails each core
|
|
101
|
-
#
|
|
102
|
-
# clean exception or a baseline-identical result,
|
|
103
|
-
# (see script/check_alloc_failures.rb).
|
|
101
|
+
# Rust allocation site in turn, gating that every OOM branch fails closed - a
|
|
102
|
+
# clean exception or a baseline-identical result, with stateful objects reusable
|
|
103
|
+
# after the failure (see script/check_alloc_failures.rb).
|
|
104
104
|
security-alloc-inject:
|
|
105
105
|
name: OOM-injection sweep
|
|
106
106
|
runs-on: ubuntu-latest
|
|
@@ -137,9 +137,8 @@ jobs:
|
|
|
137
137
|
- name: Run the OOM-injection sweep
|
|
138
138
|
run: bundle exec rake oom
|
|
139
139
|
|
|
140
|
-
# Nightly: fuzz EVERY target
|
|
141
|
-
#
|
|
142
|
-
# documented memory-safety risk, so each gets a full 300s run).
|
|
140
|
+
# Nightly: fuzz EVERY target for a longer run. The PR job already exercises
|
|
141
|
+
# every surface for 30s; nightly gives each target a full 300s run.
|
|
143
142
|
security-fuzz-nightly:
|
|
144
143
|
name: Nightly sanitized fuzz (${{ matrix.target }})
|
|
145
144
|
runs-on: ubuntu-latest
|
|
@@ -178,7 +177,7 @@ jobs:
|
|
|
178
177
|
bundler-cache: true
|
|
179
178
|
|
|
180
179
|
- name: Run nightly fuzz under sanitizers
|
|
181
|
-
run:
|
|
180
|
+
run: FUZZ_ARGS="--target ${{ matrix.target }} --time 300" bundle exec rake fuzz:sanitize
|
|
182
181
|
|
|
183
182
|
# Nightly: the whole spec suite with Lexbor ITSELF built under ASan (mraw
|
|
184
183
|
# poisoning on), catching intra-arena overflows that a plain ASan build cannot
|
data/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,211 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## [0.11.0] - 2026-09-30
|
|
4
|
+
|
|
5
|
+
No code changes since 0.11.0.rc2. Coming from 0.10.x, read the rc2 and rc1
|
|
6
|
+
notes below as well.
|
|
7
|
+
|
|
8
|
+
## [0.11.0.rc2] - 2026-09-30
|
|
9
|
+
|
|
10
|
+
### Added
|
|
11
|
+
|
|
12
|
+
* `Makiri::HTML::Document#create_element_ns(namespace_uri, qualified_name)`
|
|
13
|
+
(DOM `createElementNS`). An SVG or MathML element made this way keeps its
|
|
14
|
+
name's case and behaves like a parsed one. An upper-case name in the HTML
|
|
15
|
+
namespace that names a known element (`BR`, `DIV`) raises `Makiri::Error`.
|
|
16
|
+
* `Makiri::XML::Element#set_loose_dom_attribute(qualified_name, value)` (DOM
|
|
17
|
+
`setAttribute`): the attribute is in no namespace and keeps the name as
|
|
18
|
+
given (`xmlns`, `xlink:href`, `v-on:click`). `to_xml` raises while the
|
|
19
|
+
document holds one XML cannot write.
|
|
20
|
+
|
|
21
|
+
### Changed
|
|
22
|
+
|
|
23
|
+
* A NUL in an element, attribute, doctype or PI target name raises
|
|
24
|
+
`ArgumentError`, like any other invalid name. A namespace that does not fit
|
|
25
|
+
the name still raises `Makiri::Error`.
|
|
26
|
+
* HTML type selectors are case-sensitive on SVG and MathML elements, as in
|
|
27
|
+
browsers: `fegaussianblur` no longer matches `feGaussianBlur`.
|
|
28
|
+
* `set_attribute_ns` accepts the XML namespace with any prefix or none again
|
|
29
|
+
(rc1 raised); `to_xml` writes such an attribute as `xml:name`.
|
|
30
|
+
* XML `set_attribute_ns` accepts a namespace declaration XML forbids, such as
|
|
31
|
+
`xmlns:foo=""`. It binds nothing, and `to_xml` raises while it is present.
|
|
32
|
+
* XML `create_document_type` accepts any public and system id, and any name
|
|
33
|
+
without whitespace, NUL or `>`. `to_xml` raises if the doctype cannot be
|
|
34
|
+
written as XML.
|
|
35
|
+
|
|
36
|
+
### Fixed
|
|
37
|
+
|
|
38
|
+
* `import_node` from XML to HTML keeps an upper-case XHTML element name
|
|
39
|
+
(`Foo`) instead of lower-casing it, and raises for one that names a known
|
|
40
|
+
element (`BR`).
|
|
41
|
+
|
|
42
|
+
## [0.11.0.rc1] - 2026-09-29
|
|
43
|
+
|
|
44
|
+
### Changed
|
|
45
|
+
|
|
46
|
+
* HTML `#css` / `#at_css` / `#matches?` now match with Makiri's own Rust
|
|
47
|
+
implementation instead of Lexbor's `lxb_selectors` (selectors are still
|
|
48
|
+
parsed by Lexbor). Where the answers differ:
|
|
49
|
+
* `:nth-child(An+B of S)` / `:nth-last-child(... of S)` count by the CSS
|
|
50
|
+
definition; Lexbor miscounted when `S` was a comma list, held a
|
|
51
|
+
combinator, or carried pseudo-classes such as `:enabled`.
|
|
52
|
+
* `:disabled`, `:enabled` and `:checked` follow the HTML Standard: a
|
|
53
|
+
control inside a `<fieldset disabled>` (outside its first `<legend>`) is
|
|
54
|
+
disabled, `<option>` / `<optgroup>` count, and `:enabled` matches only
|
|
55
|
+
form elements.
|
|
56
|
+
* Attribute selector names are matched by qualified name, and
|
|
57
|
+
case-sensitively on SVG/MathML elements: `[href]` no longer finds
|
|
58
|
+
`xlink:href`, nor `[viewbox]` an SVG `viewBox`.
|
|
59
|
+
* `#id` / `.class` match only the no-namespace `id` / `class` attribute.
|
|
60
|
+
* A selector chain is capped at 64 compounds (as `Makiri::XML` already
|
|
61
|
+
was), and a query at 50M steps of work (XPath's limit); past either,
|
|
62
|
+
`Makiri::Error`.
|
|
63
|
+
* The column combinator `||` raises `Makiri::Error` instead of matching
|
|
64
|
+
nothing.
|
|
65
|
+
|
|
66
|
+
### Removed
|
|
67
|
+
|
|
68
|
+
* `:lexbor-contains()` in HTML `#css` / `#at_css` / `#matches?`: it still
|
|
69
|
+
parses, but raises `Makiri::Error`. `Makiri::XML`'s `#css` keeps it. See
|
|
70
|
+
NOKOGIRI_DIFFERENCES.md.
|
|
71
|
+
* `Node#name=` / `#node_name=`. The DOM cannot rename an element, and renaming
|
|
72
|
+
a Lexbor element in place could crash (`div` to `template`). Create a new
|
|
73
|
+
element and `replace` the old one.
|
|
74
|
+
|
|
75
|
+
### Security
|
|
76
|
+
|
|
77
|
+
* Hardening: every heap block the vendored Lexbor allocates carries 16 bytes
|
|
78
|
+
of slack past its end, so a small overrun cannot reach neighbouring memory
|
|
79
|
+
(not in sanitizer builds).
|
|
80
|
+
* Adding an `id` or `class` attribute beside an existing one
|
|
81
|
+
(`set_attribute_ns(nil, "ID", v)`, a namespaced `id`) no longer frees the
|
|
82
|
+
existing one under a held `Attr`. HTML attribute reads and writes follow the
|
|
83
|
+
DOM: `#[]`, `#key?`, `#[]=` and `#delete` match the qualified name
|
|
84
|
+
(`svg_a["href"]` no longer returns `xlink:href`), copies and imports keep
|
|
85
|
+
every attribute, and a no-namespace name keeps its case.
|
|
86
|
+
* `content=` on an HTML element and `delete(name)` detach the nodes they
|
|
87
|
+
remove instead of freeing them under a live wrapper.
|
|
88
|
+
* `el[name] = value` on an existing attribute no longer leaves a destroyed
|
|
89
|
+
attribute linked when storing the value runs out of memory.
|
|
90
|
+
* A checked argument String is locked while its bytes are in use, so a later
|
|
91
|
+
argument's `#to_s` cannot rewrite it (it raises instead); a receiver frozen
|
|
92
|
+
by an argument's `#to_s` is not edited; and a mutator's argument can no
|
|
93
|
+
longer rebuild the document's indexes in the middle of the edit.
|
|
94
|
+
* HTML element and attribute names follow the DOM's rules:
|
|
95
|
+
`create_element("img src=x onerror=alert(1)")` raises `ArgumentError`
|
|
96
|
+
instead of writing that markup. See NOKOGIRI_DIFFERENCES.md.
|
|
97
|
+
* HTML parsing bounds the tree depth (`max_tree_depth:`, default 400) and the
|
|
98
|
+
`<option>`s one `<select>` receives (10,000), raising `Makiri::Error` past
|
|
99
|
+
either; both made the parse quadratic.
|
|
100
|
+
* `Makiri::Lexbor::CSS.parse_stylesheet` no longer crashes on a
|
|
101
|
+
`:lexbor-contains()` after a string ended by CR, FF or a newline.
|
|
102
|
+
* A CSS query that runs out of memory raises instead of answering with what it
|
|
103
|
+
had collected; a selector-parse OOM is reported as OOM, not
|
|
104
|
+
`CSS::SyntaxError`.
|
|
105
|
+
* Inputs whose cost grew faster than their size are linear or budgeted: many
|
|
106
|
+
namespace bindings, the `preceding` axis and reverse-axis steps, XML
|
|
107
|
+
duplicate-attribute checks, `contains` / `substring-*` / `translate` on long
|
|
108
|
+
strings, XML CSS `:nth-*`, CDATA full of `]]>`, and stylesheets with
|
|
109
|
+
rejected `:lexbor-contains()`.
|
|
110
|
+
* A panic in any Ruby method Makiri defines raises `Makiri::InternalError`
|
|
111
|
+
(rescuable), not `fatal`.
|
|
112
|
+
|
|
113
|
+
### Fixed
|
|
114
|
+
|
|
115
|
+
* A document that grows by editing (appended nodes, `inner_html=`) reports
|
|
116
|
+
its new size to the GC, so memory pressure from it triggers collections; it
|
|
117
|
+
was reported once, at parse time.
|
|
118
|
+
* A node wrapped while memory runs out raises `Makiri::Error` instead of
|
|
119
|
+
coming back as a second Ruby object for the same node, without the first
|
|
120
|
+
one's `freeze`, instance variables or singleton methods.
|
|
121
|
+
* Insertion follows the DOM's pre-insertion rules in HTML and XML alike: no
|
|
122
|
+
cycles through a template's contents (which hung `dup`), no children on
|
|
123
|
+
Text/Comment/PI/DocumentType/Attr, no Text directly under an XML Document,
|
|
124
|
+
and an XML `DocumentFragment` takes children.
|
|
125
|
+
* `inner_html` / `inner_html=` on a `<template>` target its contents, agreeing
|
|
126
|
+
with `#to_html` and `content_fragment`.
|
|
127
|
+
* HTML `#keys` / `#values` raise on out-of-memory instead of returning a
|
|
128
|
+
truncated Array.
|
|
129
|
+
* `dup`, `clone_node` and `import_node` keep an element's name as written
|
|
130
|
+
(`linearGradient`, `q:Bar`), and a copied XML doctype keeps its PUBLIC id.
|
|
131
|
+
* Namespaces across HTML <-> XML `import_node`: attributes keep their own
|
|
132
|
+
namespace, a prefixed element keeps its prefix and case, names XML cannot
|
|
133
|
+
write are refused where they would be written, and copied `xmlns`
|
|
134
|
+
attributes no longer move elements between namespaces.
|
|
135
|
+
* XML namespaces: `set_attribute_ns` on a detached element survives insertion;
|
|
136
|
+
`[]=` on an existing attribute changes only its value; the mutators refuse
|
|
137
|
+
what the parser refuses (forbidden declarations, duplicate expanded names,
|
|
138
|
+
bad doctype names and PUBLIC ids); `canonicalize` raises instead of writing a
|
|
139
|
+
wrong namespace; and a refused declaration says which rule it broke.
|
|
140
|
+
* `to_xml` output re-parses: CDATA holding `]]>`, a SYSTEM id holding `"`, and
|
|
141
|
+
attributes with a namespace but no prefix.
|
|
142
|
+
* XPath: `string()` of a number follows libxml2 (as Nokogiri does);
|
|
143
|
+
`substring()` rounds per spec; node-set vs boolean comparisons follow §3.4;
|
|
144
|
+
the `xml` prefix is always bound; a comparison no longer raises
|
|
145
|
+
`LimitExceeded` because its values added up past 64 MB.
|
|
146
|
+
* `Node#path` round-trips through `#at_xpath` for CDATA, PIs, and namespaced
|
|
147
|
+
nodes (SVG/MathML, XML namespaces, `xlink:href`, Vue/Word-style names); an
|
|
148
|
+
unreachable node answers `"?"` as in Nokogiri.
|
|
149
|
+
* CSS over XML agrees with the HTML matcher on `.x\ y`, `:root`, empty and
|
|
150
|
+
whitespace attribute values, `$=` with non-ASCII, `:empty` beside a comment,
|
|
151
|
+
and `[|a]`; the `s` modifier is accepted.
|
|
152
|
+
* An HTML document refuses a second root element or a text child; an
|
|
153
|
+
attribute's parent is always its owner element.
|
|
154
|
+
* `Node#line` of a node copied from another document is nil.
|
|
155
|
+
* `Makiri::XML` nodes compare with `<=>` in document order (an attribute with
|
|
156
|
+
itself too), and `XML::Document#dup` / `#clone` return a copy.
|
|
157
|
+
* `clone_node` on a Document raises instead of returning the document.
|
|
158
|
+
* `NodeSet#css` / `#xpath` / `#at_css` / `#at_xpath` pass their extra
|
|
159
|
+
arguments through; `#xpath` of a non-node-set expression raises
|
|
160
|
+
`ArgumentError`, as in Nokogiri.
|
|
161
|
+
* `XML::Builder` no longer answers `respond_to?` for `to_ary` and similar.
|
|
162
|
+
* Namespace Hashes and `register_namespace` arguments are read as a Hash and
|
|
163
|
+
with `String()`, with one length cap and one error message.
|
|
164
|
+
* A rejected stylesheet rule's `selector_text` is sliced by Lexbor's offsets.
|
|
165
|
+
|
|
166
|
+
### Performance
|
|
167
|
+
|
|
168
|
+
* `NodeSet#at_css` / `#at_xpath` stop at the first node with a match.
|
|
169
|
+
* `XML::Node#canonicalize` looks namespaces up by prefix in the scope it keeps:
|
|
170
|
+
a 1000-deep document of declarations went from 0.7 s to 0.02 s.
|
|
171
|
+
|
|
172
|
+
## [0.10.0] - 2026-09-22
|
|
173
|
+
|
|
174
|
+
### Fixed
|
|
175
|
+
|
|
176
|
+
* A frozen node now raises `FrozenError` when it is the ARGUMENT of a tree
|
|
177
|
+
mutation, not only the receiver: `b.add_child(a)` relinks `a` exactly as
|
|
178
|
+
`a.remove` does. Both `Makiri::XML` and `Makiri::HTML`. A fragment argument
|
|
179
|
+
splices its children, which have no wrapper of their own to check.
|
|
180
|
+
* A `Makiri::XML` mutation that exceeds the document's own `max_bytes` /
|
|
181
|
+
`max_nodes` now raises `Makiri::XML::LimitExceeded` instead of reporting the
|
|
182
|
+
refusal as out of memory.
|
|
183
|
+
* Inserting a `DocumentFragment` is all or nothing on `add_child`, `before` and
|
|
184
|
+
`after`, as it already was on `replace`: a child the rules refuse no longer
|
|
185
|
+
leaves the earlier ones linked in a document the caller was told had not
|
|
186
|
+
changed.
|
|
187
|
+
* A rejected `Makiri::XML::Document#fragment` no longer charges the document for
|
|
188
|
+
the nodes it discarded. 100k rejected fragments grew a `<r/>` document to
|
|
189
|
+
77 MB; it is now 516 bytes.
|
|
190
|
+
* A failed `Makiri::XML` parse reports the FIRST failure for all four kinds
|
|
191
|
+
(`syntax` was sticky while `limit` and `unsupported` overwrote each other).
|
|
192
|
+
* `Makiri::XML#to_xml`'s serializer allocates fallibly again, so running out of
|
|
193
|
+
memory raises instead of aborting the process.
|
|
194
|
+
* `:lexbor-contains()` now rejects an argument the bundled CSS parser does not
|
|
195
|
+
take, the way any unknown pseudo-class is rejected: `Makiri::CSS::SyntaxError`
|
|
196
|
+
from `#css` / `#at_css` / `#matches?`, and a `:bad_style` rule from
|
|
197
|
+
`Makiri::Lexbor::CSS.parse_stylesheet`. Well-formed uses are unchanged.
|
|
198
|
+
|
|
199
|
+
### Performance
|
|
200
|
+
|
|
201
|
+
* `Makiri::XML#to_xml` plans namespaces from a binding stack instead of
|
|
202
|
+
re-walking each ancestor's attribute list, which cost O(depth^2 x attributes).
|
|
203
|
+
403 KB of nested prefixed attributes took 4.88s and now takes 0.001s. A
|
|
204
|
+
crafted document fails closed with `Makiri::Error` ("namespace planning
|
|
205
|
+
exceeded its step budget") rather than running on.
|
|
206
|
+
* Setting an attribute on a `Makiri::XML` element walks the attribute list once
|
|
207
|
+
instead of twice (4096 attributes: 47ms -> 27ms).
|
|
208
|
+
|
|
3
209
|
## [0.10.0.rc2] - 2026-09-20
|
|
4
210
|
|
|
5
211
|
### Added
|
data/NOKOGIRI_DIFFERENCES.md
CHANGED
|
@@ -26,11 +26,34 @@ what browsers do - rather than libxml2. Detailed, test-backed notes live in
|
|
|
26
26
|
exactly (`//*[@refX]`, not `@refx`). Only ASCII folds: `Ø` still differs from `ø`.
|
|
27
27
|
This holds in `namespace_matching: :lax` too.
|
|
28
28
|
* `Nokogiri::HTML5` is case-sensitive there.
|
|
29
|
+
* `Node#path` names a node by expanded name where a bare name would not reach it
|
|
30
|
+
* An SVG, MathML or namespaced-XML node, and an HTML name that is no plain
|
|
31
|
+
XPath name (`<o:p>`, `xml:lang`, Vue's `@click`), give
|
|
32
|
+
`*[local-name()='path' and namespace-uri()='http://www.w3.org/2000/svg']`,
|
|
33
|
+
which `#at_xpath` evaluates with no prefix registered. Nokogiri writes
|
|
34
|
+
`svg:svg` or `/*/*[2]` for the first kinds; the paths are equivalent, the
|
|
35
|
+
strings are not.
|
|
36
|
+
* A node not attached to its document answers `"?"`, as a doctype does.
|
|
37
|
+
Nokogiri answers `"/div/p"` for a detached `<div><p>`, which is the path of
|
|
38
|
+
the document's own `/div/p` when it has one.
|
|
29
39
|
* A foreign element's namespace declarations are not attributes
|
|
30
40
|
* `<svg xmlns="...">` has no `@xmlns` for `//*[@xmlns]` or `@*`, as in browsers
|
|
31
41
|
and in XPath's data model. An `xmlns` on an HTML element is an ordinary
|
|
32
42
|
attribute and stays visible.
|
|
33
43
|
|
|
44
|
+
* A `StandardError` raised by a custom-function handler becomes `Makiri::Error`
|
|
45
|
+
("handler raised: <message>"), with the handler's exception as its `#cause`
|
|
46
|
+
* Nokogiri re-raises the handler's exception itself. Anything that is not a
|
|
47
|
+
`StandardError` (`Interrupt`, `SystemExit`, Timeout's exception) and a
|
|
48
|
+
`throw` reach the caller unchanged in both.
|
|
49
|
+
|
|
50
|
+
* A number literal is read as the nearest double; libxml2's own reader is not
|
|
51
|
+
correctly rounded for a literal with more digits than a double holds, so
|
|
52
|
+
`string(0.72609133372266155)` is `0.726091333722662` in Makiri and
|
|
53
|
+
`0.726091333722661` in Nokogiri (the digits differ in the last place). The
|
|
54
|
+
number is then written by libxml2's rule in both (`string(1234567890.5)` is
|
|
55
|
+
`1.2345678905e+09`); only the value read differs.
|
|
56
|
+
|
|
34
57
|
## XML
|
|
35
58
|
|
|
36
59
|
* `Makiri::XML` is XML 1.0 (Fifth Edition) only and non-validating.
|
|
@@ -61,6 +84,13 @@ what browsers do - rather than libxml2. Detailed, test-backed notes live in
|
|
|
61
84
|
* `create_processing_instruction("a:b", ...)` succeeds, as DOM
|
|
62
85
|
`createProcessingInstruction` does, but `#to_xml` / `#canonicalize` then raise,
|
|
63
86
|
as DOM Parsing's well-formed serializer does. Nokogiri writes `<?a:b ...?>`.
|
|
87
|
+
* `#freeze` on a node is ENFORCED: a frozen node's mutators raise `FrozenError`,
|
|
88
|
+
and so does passing a frozen node as the argument of an insertion, which
|
|
89
|
+
relinks it. Nokogiri reports `frozen?` but every mutator still mutates. The
|
|
90
|
+
check reaches the nodes the caller named; a fragment argument splices its
|
|
91
|
+
children, and those cannot be checked, because frozen-ness is a property of a
|
|
92
|
+
Ruby object and the arena keeps no map from a node back to its wrapper.
|
|
93
|
+
|
|
64
94
|
* A node's namespace URI is its identity, not something re-derived from the
|
|
65
95
|
declarations around it - the WHATWG DOM model, measured against Chrome 152
|
|
66
96
|
(`DOMParser` + `XMLSerializer`).
|
|
@@ -77,6 +107,19 @@ what browsers do - rather than libxml2. Detailed, test-backed notes live in
|
|
|
77
107
|
invents one (`ns1`, `ns2`, ...) rather than shadow the other, as browsers do.
|
|
78
108
|
* An element in no namespace stays that way under a default namespace,
|
|
79
109
|
serialized as `xmlns=""`.
|
|
110
|
+
* `create_document_type` takes what the DOM's `createDocumentType` takes -
|
|
111
|
+
any id, and a name without whitespace, NUL or `>` - and `to_xml` refuses a
|
|
112
|
+
doctype XML cannot write (`create_document_type("q", "abcde", %(x"'y))`).
|
|
113
|
+
Nokogiri writes that system id as `"x"'y"`, which parses but reads
|
|
114
|
+
back as the id `x"'y` - a literal expands no references.
|
|
115
|
+
* `set_attribute_ns(XMLNS_NS, "xmlns:foo", "")` - a declaration Namespaces in
|
|
116
|
+
XML forbids, which the DOM's `setAttributeNS` accepts - is kept as an
|
|
117
|
+
attribute that binds nothing, and `to_xml` refuses the tree while it is
|
|
118
|
+
there. `root["xmlns:foo"] = ""` still raises.
|
|
119
|
+
* An attribute in the XML namespace is written as `xml:local`, whatever
|
|
120
|
+
prefix it was given (`set_attribute_ns(XML_NS, "a:bb")`, as the DOM
|
|
121
|
+
allows): Namespaces in XML binds that namespace to `xml` alone. It re-reads
|
|
122
|
+
to the same namespace and local name. `canonicalize` writes it so too.
|
|
80
123
|
* Nodes from the factories (`create_element` and friends) still take their
|
|
81
124
|
namespace from the context they are first inserted into, so a subtree can be
|
|
82
125
|
built detached and attached afterwards. Only later moves carry.
|
|
@@ -93,6 +136,14 @@ what browsers do - rather than libxml2. Detailed, test-backed notes live in
|
|
|
93
136
|
the property-based differential), including namespaces, prolog/epilog comments
|
|
94
137
|
and PIs, and adjacent-CDATA coalescing.
|
|
95
138
|
|
|
139
|
+
* `to_xml` keeps an element's namespace when an `xmlns` attribute on it says
|
|
140
|
+
otherwise
|
|
141
|
+
* `root["xmlns"] = "urn:x"` on an element in no namespace is not written: the
|
|
142
|
+
element stays in no namespace when the output is re-read, as the DOM Parsing
|
|
143
|
+
and Serialization spec asks. Nokogiri writes `<r xmlns="urn:x">`, which moves
|
|
144
|
+
the element into `urn:x` on re-parse. Create the element in the namespace
|
|
145
|
+
instead (`create_element("r", "xmlns" => "urn:x")`, or parse it so).
|
|
146
|
+
|
|
96
147
|
## HTML parsing
|
|
97
148
|
|
|
98
149
|
* `<?php ... ?>` in HTML input is a **ProcessingInstruction** node; `#to_html`
|
|
@@ -102,6 +153,79 @@ what browsers do - rather than libxml2. Detailed, test-backed notes live in
|
|
|
102
153
|
still produces the older bogus comment (`<!--?php ... ?-->`), and
|
|
103
154
|
`Nokogiri::HTML` (libxml2) its own comment.
|
|
104
155
|
|
|
156
|
+
* A tree deeper than `max_tree_depth` raises **`Makiri::Error`**, where
|
|
157
|
+
`Nokogiri::HTML5` raises `ArgumentError`; the default (400) and boundary are
|
|
158
|
+
Nokogiri's. `inner_html=`, `outer_html=` and `Node#parse` take no keyword.
|
|
159
|
+
`Nokogiri::HTML` (libxml2) instead stops at depth 256 and returns the
|
|
160
|
+
truncated document.
|
|
161
|
+
* More than 10,000 `<option>`s in one `<select>` raise `Makiri::Error`;
|
|
162
|
+
Nokogiri has no such limit.
|
|
163
|
+
|
|
164
|
+
## HTML mutation
|
|
165
|
+
|
|
166
|
+
* There is no `Node#name=` / `#node_name=` (on HTML or XML nodes). Nokogiri
|
|
167
|
+
renames a node in place and keeps its identity; the DOM has no rename, and
|
|
168
|
+
Lexbor keeps elements such as `<template>`, `<option>` and `<style>` in
|
|
169
|
+
structs of their own, so a node cannot change its tag safely. Create an
|
|
170
|
+
element of the new name, move the attributes and children, and `replace`
|
|
171
|
+
the old one (see CHANGELOG.md).
|
|
172
|
+
* Element and attribute names follow the WHATWG DOM's rules
|
|
173
|
+
* `create_element`, `[]=` and `set_attribute_ns` raise `ArgumentError` for a
|
|
174
|
+
name the DOM refuses - one holding whitespace, `/`, `>` (or `=` for an
|
|
175
|
+
attribute) - where `Nokogiri::HTML5` accepts it and writes it into the
|
|
176
|
+
markup as it stands: `create_element("img src=x onerror=alert(1)")`
|
|
177
|
+
serializes as that tag.
|
|
178
|
+
The names HTML actually uses (`data-*`, `aria-*`, `@click`, `:href`,
|
|
179
|
+
`v-on:x`, custom elements) are accepted.
|
|
180
|
+
* `set_attribute_ns(nil, "x:y")` raises, as the DOM's `setAttributeNS` does:
|
|
181
|
+
a prefix needs a namespace. `[]=` (HTML) and `set_loose_dom_attribute`
|
|
182
|
+
(XML) are the DOM's `setAttribute`, which makes such an attribute.
|
|
183
|
+
* A refused name raises `ArgumentError` - a NUL in it too - and a namespace
|
|
184
|
+
that does not fit the name (`set_attribute_ns`, `create_element_ns`)
|
|
185
|
+
raises `Makiri::Error`. Invalid UTF-8 raises `Makiri::Error` for every
|
|
186
|
+
argument, names included.
|
|
187
|
+
* `create_element_ns` refuses an HTML-namespace name in upper case that
|
|
188
|
+
lower-cases to an element Lexbor knows (`BR`, `DIV`), where the DOM makes
|
|
189
|
+
an unknown element: Lexbor would make that element (`BR` void, its
|
|
190
|
+
children never written). Other names keep their case (`MY-EL`). Nokogiri
|
|
191
|
+
has no `create_element_ns`.
|
|
192
|
+
* An HTML `<template>` follows the WHATWG content model, which Nokogiri does
|
|
193
|
+
not: its parsed contents live in the separate fragment `Element#content_fragment`
|
|
194
|
+
returns, `template.children` is empty, and `inner_html` / `inner_html=`
|
|
195
|
+
special-case the contents (the WHATWG DOM special-cases `innerHTML` alone;
|
|
196
|
+
`append_child`, `content=`, and `children` act on the element's own empty
|
|
197
|
+
children, as the specification says). Nokogiri treats `<template>` as an
|
|
198
|
+
ordinary element, with the parsed nodes as its children, so
|
|
199
|
+
`template.inner_html` and `template.children` answer the other way round and
|
|
200
|
+
there is no `content_fragment`.
|
|
201
|
+
* An HTML document has one root element and no text child, as the DOM requires;
|
|
202
|
+
`doc << element` beside an existing root raises.
|
|
203
|
+
* An insertion the DOM refuses - a child under a text, comment, PI, doctype
|
|
204
|
+
or attribute node, text under an XML Document, a second root - raises
|
|
205
|
+
`Makiri::Error` in both representations. Nokogiri refuses the same ones with
|
|
206
|
+
`ArgumentError` (or `RuntimeError` for a second XML root).
|
|
207
|
+
* Moving HTML into an XML document (`xml_doc.import_node(html_node)`, or
|
|
208
|
+
inserting one) keeps every name's namespace, and refuses what XML cannot
|
|
209
|
+
write that way. An attribute in no namespace whose name has a prefix other
|
|
210
|
+
than `xml` - `v-on:click`, `fb:like`, an `xlink:href` on an HTML (not SVG)
|
|
211
|
+
element - raises `Makiri::Error`: as XML it would be a prefix bound to
|
|
212
|
+
nothing. Nokogiri copies it and writes `v-on:click="..."` into output that is
|
|
213
|
+
not namespace-well-formed. An element named with a colon (`<fb:like>`)
|
|
214
|
+
crosses as a DOM-loose name, which `to_xml` refuses.
|
|
215
|
+
* A known gap, in Lexbor's tag table: an HTML document that already holds a
|
|
216
|
+
parsed element named with a colon (`<x:y>`, one local name) and then
|
|
217
|
+
receives, by `import_node` from another document, a prefixed element
|
|
218
|
+
written the same way (`x:y` from XML: prefix `x`, local name `y`) re-points
|
|
219
|
+
the table's entry for that spelling - the parsed element then no longer
|
|
220
|
+
matches CSS `x\:y`. Copies within one document do not touch the table.
|
|
221
|
+
* An XML element with a prefix, imported into HTML, keeps it (`h:div` in
|
|
222
|
+
XHTML has the local name `div`). Two readers then disagree, as they do in
|
|
223
|
+
browsers: CSS's `div` matches it (Lexbor matches the local name), XPath's
|
|
224
|
+
`//div` does not (an HTML element's name test reads its qualified name). And
|
|
225
|
+
Lexbor's HTML serializer writes the prefix (`<h:div>`), where the HTML
|
|
226
|
+
standard writes the local name, so the HTML does not re-parse to the same
|
|
227
|
+
element.
|
|
228
|
+
|
|
105
229
|
## CSS
|
|
106
230
|
|
|
107
231
|
* Most jQuery/Nokogiri CSS extensions are not supported (`:gt`, `:lt`, `:eq`, `:first`, ...)
|
|
@@ -109,13 +233,18 @@ what browsers do - rather than libxml2. Detailed, test-backed notes live in
|
|
|
109
233
|
text-containment extension. Use XPath (`xpath("//p[contains(., 'x')]")`) or
|
|
110
234
|
Enumerable (`css('li')[1]`) for the rest.
|
|
111
235
|
Standard Level-4 selectors (`:is` / `:where` / `:has`) are supported; some of which Nokogiri rejects.
|
|
112
|
-
* `:lexbor-contains("text")`
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
nodes
|
|
118
|
-
|
|
236
|
+
* `:lexbor-contains("text")` - Lexbor's spelling of the jQuery `:contains()`
|
|
237
|
+
substring filter, matching an element whose text contains the string;
|
|
238
|
+
append ` i` (`:lexbor-contains("text" i)`) for an ASCII case-insensitive
|
|
239
|
+
match. (Nokogiri's name `:contains` is not an alias.) **XML only**: it
|
|
240
|
+
lowers to XPath `child::text()[contains(., "text")]`, testing the
|
|
241
|
+
element's immediate child text nodes, not the deep string-value. HTML no
|
|
242
|
+
longer supports it - `lexbor::css_match`, the safe-Rust port that
|
|
243
|
+
replaced Lexbor's own matcher for HTML, deliberately does not reimplement
|
|
244
|
+
it, so a well-formed `:lexbor-contains()` now raises `Makiri::Error`
|
|
245
|
+
("could not be run") on HTML rather than ever matching (it still PARSES,
|
|
246
|
+
and an actually malformed one still raises `Makiri::CSS::SyntaxError`, as
|
|
247
|
+
before).
|
|
119
248
|
* Untyped `:*-of-type` (`:first-of-type`, `:nth-of-type(an+b)`, ... with no type
|
|
120
249
|
selector) is supported and correct on both HTML and XML - the "type" is the
|
|
121
250
|
element's own expanded name.
|
|
@@ -131,12 +260,35 @@ what browsers do - rather than libxml2. Detailed, test-backed notes live in
|
|
|
131
260
|
prefix IS resolved against the bindings, when the namespace matters.
|
|
132
261
|
* `Makiri::XML` resolves CSS prefixes properly - it lowers the selector to the
|
|
133
262
|
XPath engine, which registers the bindings.
|
|
263
|
+
* The same holds for attribute selectors: HTML `[|href]` (no namespace) and
|
|
264
|
+
plain `[href]` also find an SVG `xlink:href`, which `Nokogiri::HTML5`
|
|
265
|
+
does not.
|
|
266
|
+
`Makiri::XML` reads `[|a]` as the no-namespace attribute.
|
|
267
|
+
* A selector under a node matches the way `Element#querySelectorAll` does in a
|
|
268
|
+
browser, not scoped to that node, on HTML: `at_css("#c").css("div p")` finds a
|
|
269
|
+
`p` inside `#c` when `#c` is itself a `div`, since the selector is matched
|
|
270
|
+
against the whole document and only the results are kept to descendants.
|
|
271
|
+
`Nokogiri::HTML5` and `Nokogiri::XML` scope the selector to the node (`#c`
|
|
272
|
+
cannot be the `div`), and so does `Makiri::XML`, which lowers the selector to
|
|
273
|
+
an XPath from the node. Lexbor has no `:scope`; for a scoped match on HTML,
|
|
274
|
+
use XPath from the node (`xpath(".//div//p")`).
|
|
275
|
+
* The attribute case modifiers (`[a="x" i]`, `[a="x" s]`): HTML supports both,
|
|
276
|
+
through Lexbor's matcher. `Makiri::XML` accepts `s` (case-sensitive, which XML
|
|
277
|
+
values are anyway) and refuses `i` with `Makiri::CSS::SyntaxError`. Nokogiri
|
|
278
|
+
refuses both on either representation.
|
|
134
279
|
* `#matches?` answers for a DETACHED node (`document.create_element("p")
|
|
135
280
|
.matches?("p")` is true, on both representations). Nokogiri raises
|
|
136
281
|
`NoMethodError` there - it implements `#matches?` as a search from
|
|
137
282
|
`ancestors.last`, which a detached node does not have.
|
|
138
|
-
*
|
|
139
|
-
|
|
283
|
+
* Type selectors in an HTML document follow the HTML Standard's
|
|
284
|
+
case-sensitivity rule, as browsers do: lower-cased for an HTML element (`LI`
|
|
285
|
+
matches `<li>`), as written for any other (`feGaussianBlur` matches the SVG
|
|
286
|
+
element, `fegaussianblur` does not). An HTML element named in upper case
|
|
287
|
+
(`create_element_ns(XHTML, "MY-EL")`, which keeps its name as the DOM does)
|
|
288
|
+
therefore matches no type selector.
|
|
289
|
+
* `Nokogiri::HTML5` is case-sensitive on HTML elements too, so `LI` does not
|
|
290
|
+
match `<li>` there. `Makiri::XML`'s `#css` is case-sensitive, as XML names
|
|
291
|
+
are.
|
|
140
292
|
|
|
141
293
|
## Serialization
|
|
142
294
|
|
|
@@ -163,7 +315,10 @@ what browsers do - rather than libxml2. Detailed, test-backed notes live in
|
|
|
163
315
|
and attribute values (`[]=`, `set_attribute_ns`) - and stored/read back
|
|
164
316
|
verbatim, matching the WHATWG DOM / browsers (`document.createTextNode("\0")`).
|
|
165
317
|
It is still rejected in names, tag names, namespaces, PI target/data, CSS
|
|
166
|
-
selectors, and XPath expressions/variable names
|
|
318
|
+
selectors, and XPath expressions/variable names. A NUL in a name given to a
|
|
319
|
+
factory or setter (element, attribute, doctype and PI target names) raises
|
|
320
|
+
`ArgumentError`, as any other refused name does; anywhere else it raises
|
|
321
|
+
`Makiri::Error`.
|
|
167
322
|
* On re-parse, the HTML tokenizer replaces a U+0000 in text/attributes with
|
|
168
323
|
U+FFFD (WHATWG), so a serialized-then-reparsed round-trip is not byte-identical.
|
|
169
324
|
* `Makiri::XML` rejects NUL everywhere: XML 1.0 has no legal U+0000 character,
|
data/README.md
CHANGED
|
@@ -23,11 +23,13 @@ XPath 1.0 evaluation in its own native engine, with no libxml2 dependency.
|
|
|
23
23
|
* Native XML 1.0 parser
|
|
24
24
|
* A strict, non-validating, fail-closed parser with its own node arena (not
|
|
25
25
|
Lexbor's HTML DOM), queried through the same native XPath engine, with
|
|
26
|
-
in-place tree edits (attributes, content,
|
|
26
|
+
in-place tree edits (attributes, content, remove).
|
|
27
27
|
* Conformance is held by the W3C XML Conformance Test Suite, an XPath
|
|
28
28
|
differential, and property-based testing vs Nokogiri (see below).
|
|
29
29
|
* Bounded, fail-closed execution
|
|
30
30
|
* XPath evaluation is bounded by per-evaluation limits on work, memory, and recursion.
|
|
31
|
+
* HTML parsing bounds the tree depth (`max_tree_depth:`, default 400) and the
|
|
32
|
+
`<option>`s per `<select>` (10,000).
|
|
31
33
|
* Ownership and borrowing are kept explicit across layers, with owned/borrowed
|
|
32
34
|
string types and verified text at engine boundaries.
|
|
33
35
|
* Programmatic invalid input, limit violations, allocation failures, and unsupported constructs
|
|
@@ -65,6 +67,10 @@ link.parent.name # => "div"
|
|
|
65
67
|
# Source location (reconstructed from the tokenizer, no Lexbor patches)
|
|
66
68
|
doc.at_css("p").line # => 3
|
|
67
69
|
|
|
70
|
+
# Nesting deeper than 400 elements raises Makiri::Error (Nokogiri's default);
|
|
71
|
+
# max_tree_depth: raises the limit, and a negative value disables it
|
|
72
|
+
Makiri::HTML(deep_html, max_tree_depth: 2000)
|
|
73
|
+
|
|
68
74
|
# Serialization
|
|
69
75
|
doc.at_css("#main").to_html # => "<div id=\"main\" ...>...</div>"
|
|
70
76
|
doc.at_css("#main").inner_html # => "\n <p class=\"lead\">Hello</p>\n..."
|
|
@@ -132,11 +138,10 @@ e = doc.at_xpath("//entry")
|
|
|
132
138
|
e["id"] = "9" # add or replace an attribute (value escaped on output)
|
|
133
139
|
e["dc:k"] = "v" # a prefixed name resolves against the in-scope xmlns
|
|
134
140
|
e.content = "Bye" # replace an element's children with text
|
|
135
|
-
e.name = "post" # rename in place (identity + namespace re-resolved)
|
|
136
141
|
e.delete("id") # remove an attribute
|
|
137
142
|
doc.at_xpath("//draft").remove
|
|
138
143
|
|
|
139
|
-
doc.root.to_xml # => "<feed xmlns:dc=\"urn:dc\"><
|
|
144
|
+
doc.root.to_xml # => "<feed xmlns:dc=\"urn:dc\"><entry dc:k=\"v\">Bye</entry></feed>"
|
|
140
145
|
```
|
|
141
146
|
|
|
142
147
|
XML subtrees can be built using `Document#create_element` and other node factory methods,
|