makiri 0.10.0.rc2-aarch64-linux → 0.11.0-aarch64-linux

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. checksums.yaml +4 -4
  2. data/.github/workflows/libfuzzer.yml +1 -1
  3. data/.github/workflows/release.yml +11 -2
  4. data/.github/workflows/security.yml +7 -8
  5. data/CHANGELOG.md +206 -0
  6. data/NOKOGIRI_DIFFERENCES.md +165 -10
  7. data/README.md +8 -3
  8. data/Rakefile +39 -9
  9. data/lib/makiri/3.2/makiri.so +0 -0
  10. data/lib/makiri/3.3/makiri.so +0 -0
  11. data/lib/makiri/3.4/makiri.so +0 -0
  12. data/lib/makiri/4.0/makiri.so +0 -0
  13. data/lib/makiri/attr.rb +15 -0
  14. data/lib/makiri/cdata_section.rb +8 -0
  15. data/lib/makiri/clone_via_dup.rb +19 -0
  16. data/lib/makiri/comment.rb +7 -0
  17. data/lib/makiri/css.rb +5 -4
  18. data/lib/makiri/document_fragment.rb +5 -2
  19. data/lib/makiri/document_type.rb +2 -3
  20. data/lib/makiri/element.rb +9 -0
  21. data/lib/makiri/error.rb +26 -0
  22. data/lib/makiri/html/document.rb +19 -14
  23. data/lib/makiri/html/node_methods.rb +9 -8
  24. data/lib/makiri/html.rb +3 -3
  25. data/lib/makiri/node.rb +13 -72
  26. data/lib/makiri/node_path.rb +95 -0
  27. data/lib/makiri/node_set.rb +50 -25
  28. data/lib/makiri/processing_instruction.rb +10 -8
  29. data/lib/makiri/reader_aliases.rb +28 -0
  30. data/lib/makiri/text.rb +7 -0
  31. data/lib/makiri/version.rb +1 -1
  32. data/lib/makiri/xml/builder.rb +17 -8
  33. data/lib/makiri/xml/document.rb +23 -3
  34. data/lib/makiri/xml/node_methods.rb +26 -32
  35. data/lib/makiri/xml.rb +4 -3
  36. data/lib/makiri/xpath.rb +8 -8
  37. data/lib/makiri/xpath_context.rb +1 -1
  38. data/lib/makiri/xpath_syntax.rb +43 -0
  39. data/lib/makiri.rb +13 -7
  40. data/script/api_manifest.rb +25 -3
  41. data/script/check_alloc_failures.rb +350 -16
  42. data/script/check_unsafe_boundaries.rb +401 -64
  43. data/script/leaks_harness.rb +14 -1
  44. data/suppressions/ruby.supp +10 -8
  45. metadata +6 -1
checksums.yaml CHANGED
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  SHA256:
3
- metadata.gz: 75d5f9396671a66a5faede037375c56ae2e7920a424c5f857707ac583d09fe6b
4
- data.tar.gz: 5fd67d814657a6a47449d0ee523ae000e0f77b49e5875e69df99fdfcc9c9438f
3
+ metadata.gz: 5b203c0ffb03e49f10b8e30afd6b070d36fd054867b7b3fd0e1376778b9e4b94
4
+ data.tar.gz: 3a26a99a690546bcc5367fab22645b8c47bd65dec58df75df5a7a93950de3386
5
5
  SHA512:
6
- metadata.gz: ffd0bc4de1aa2ab659e536ee69ed43b033cdcbcb01be60ff60e49d59302f3abc428073bf53a62db97f885f23d5b6a61520f8740c5d67a2a4709bb2c303ff1213
7
- data.tar.gz: 3211a8681f03c39bf6f3cfe6369b7bac2cb23ac7a60e6a7e6a1bfe3a2f851843af3a8924bdc2c09d51c79e6a37aae15b4cd027c968f188911b9be2ddbe9ba202
6
+ metadata.gz: 5397148284ec99562ffe74e4f633f903405c420cc7593c996b600d13c65e9f94a48bf69a477a45063cd6e2a369b87f4d55f324bce1bcc0087b7f37cc6d73f5d4
7
+ data.tar.gz: 232787e75468ce0da009f1755538c062cabab7bdadd0a3a0eeb8b9e5f388f59eb0f4159e9f2b8152761f7c7c9e6619a3aa5568955358435d26f8a130e31a472e
@@ -25,7 +25,7 @@ jobs:
25
25
  strategy:
26
26
  fail-fast: false
27
27
  matrix:
28
- target: [xml, xpath, xml_xpath, css, html, html_xpath]
28
+ target: [xml, xpath, xml_xpath, css, html, html_xpath, html_css]
29
29
 
30
30
  steps:
31
31
  - name: Checkout (with vendored Lexbor submodule)
@@ -86,9 +86,18 @@ jobs:
86
86
  # names the GNU triple; this installs the std that goes with it.
87
87
  # Empty elsewhere, so the other runners are unaffected.
88
88
  targets: ${{ runner.os == 'Windows' && 'x86_64-pc-windows-gnu' || '' }}
89
- - name: Install libclang (bindgen, via rb-sys)
89
+ # clang/lld/llvm are for the vendored Lexbor's LTO, not for bindgen: on
90
+ # Linux the whole chain has to be LLVM or extconf turns LTO off (gcc
91
+ # emits GIMPLE, which rust-lld cannot read - see the note in
92
+ # extconf.rb). It DETECTS them, so a runner without them still builds,
93
+ # just without LTO - which is what this job used to ship, silently,
94
+ # to every user who installs the precompiled gem (see the measured win
95
+ # in CLAUDE.md's "Vendored Lexbor is built with LTO" section: this is
96
+ # the job that produces those gems, and `ci.yml` installing the chain
97
+ # only exercises the fast path in CI, it does not ship it).
98
+ - name: Install libclang (bindgen) and the LLVM chain (Lexbor LTO)
90
99
  if: runner.os == 'Linux'
91
- run: sudo apt-get update && sudo apt-get install -y libclang-dev
100
+ run: sudo apt-get update && sudo apt-get install -y libclang-dev clang lld llvm
92
101
  - uses: ruby/setup-ruby@v1
93
102
  with:
94
103
  ruby-version: ${{ matrix.ruby }}
@@ -55,7 +55,7 @@ jobs:
55
55
  bundler-cache: true
56
56
 
57
57
  - name: Run short fuzz under sanitizers
58
- run: bundle exec rake fuzz:sanitize FUZZ_ARGS="--time 30"
58
+ run: FUZZ_TIME=30 bundle exec rake fuzz:sanitize
59
59
 
60
60
  # macOS-only malloc-leak gate: ASan everywhere runs with detect_leaks=0 (Ruby
61
61
  # and Lexbor are uninstrumented), so this is the ONLY automated leak check. It
@@ -98,9 +98,9 @@ jobs:
98
98
  run: bundle exec rake leaks
99
99
 
100
100
  # OOM-injection sweep: rebuilds with MAKIRI_ALLOC_INJECT=1 and fails each core
101
- # C allocation site in turn, gating that every OOM branch fails closed - a
102
- # clean exception or a baseline-identical result, never truncated output
103
- # (see script/check_alloc_failures.rb).
101
+ # Rust allocation site in turn, gating that every OOM branch fails closed - a
102
+ # clean exception or a baseline-identical result, with stateful objects reusable
103
+ # after the failure (see script/check_alloc_failures.rb).
104
104
  security-alloc-inject:
105
105
  name: OOM-injection sweep
106
106
  runs-on: ubuntu-latest
@@ -137,9 +137,8 @@ jobs:
137
137
  - name: Run the OOM-injection sweep
138
138
  run: bundle exec rake oom
139
139
 
140
- # Nightly: fuzz EVERY target (the 30s PR fuzz covers only the default xpath
141
- # target; the CSS engine reuse and the XML parser/mutator are each their own
142
- # documented memory-safety risk, so each gets a full 300s run).
140
+ # Nightly: fuzz EVERY target for a longer run. The PR job already exercises
141
+ # every surface for 30s; nightly gives each target a full 300s run.
143
142
  security-fuzz-nightly:
144
143
  name: Nightly sanitized fuzz (${{ matrix.target }})
145
144
  runs-on: ubuntu-latest
@@ -178,7 +177,7 @@ jobs:
178
177
  bundler-cache: true
179
178
 
180
179
  - name: Run nightly fuzz under sanitizers
181
- run: bundle exec rake fuzz:sanitize FUZZ_ARGS="--target ${{ matrix.target }} --time 300"
180
+ run: FUZZ_ARGS="--target ${{ matrix.target }} --time 300" bundle exec rake fuzz:sanitize
182
181
 
183
182
  # Nightly: the whole spec suite with Lexbor ITSELF built under ASan (mraw
184
183
  # poisoning on), catching intra-arena overflows that a plain ASan build cannot
data/CHANGELOG.md CHANGED
@@ -1,5 +1,211 @@
1
1
  # Changelog
2
2
 
3
+ ## [0.11.0] - 2026-09-30
4
+
5
+ No code changes since 0.11.0.rc2. Coming from 0.10.x, read the rc2 and rc1
6
+ notes below as well.
7
+
8
+ ## [0.11.0.rc2] - 2026-09-30
9
+
10
+ ### Added
11
+
12
+ * `Makiri::HTML::Document#create_element_ns(namespace_uri, qualified_name)`
13
+ (DOM `createElementNS`). An SVG or MathML element made this way keeps its
14
+ name's case and behaves like a parsed one. An upper-case name in the HTML
15
+ namespace that names a known element (`BR`, `DIV`) raises `Makiri::Error`.
16
+ * `Makiri::XML::Element#set_loose_dom_attribute(qualified_name, value)` (DOM
17
+ `setAttribute`): the attribute is in no namespace and keeps the name as
18
+ given (`xmlns`, `xlink:href`, `v-on:click`). `to_xml` raises while the
19
+ document holds one XML cannot write.
20
+
21
+ ### Changed
22
+
23
+ * A NUL in an element, attribute, doctype or PI target name raises
24
+ `ArgumentError`, like any other invalid name. A namespace that does not fit
25
+ the name still raises `Makiri::Error`.
26
+ * HTML type selectors are case-sensitive on SVG and MathML elements, as in
27
+ browsers: `fegaussianblur` no longer matches `feGaussianBlur`.
28
+ * `set_attribute_ns` accepts the XML namespace with any prefix or none again
29
+ (rc1 raised); `to_xml` writes such an attribute as `xml:name`.
30
+ * XML `set_attribute_ns` accepts a namespace declaration XML forbids, such as
31
+ `xmlns:foo=""`. It binds nothing, and `to_xml` raises while it is present.
32
+ * XML `create_document_type` accepts any public and system id, and any name
33
+ without whitespace, NUL or `>`. `to_xml` raises if the doctype cannot be
34
+ written as XML.
35
+
36
+ ### Fixed
37
+
38
+ * `import_node` from XML to HTML keeps an upper-case XHTML element name
39
+ (`Foo`) instead of lower-casing it, and raises for one that names a known
40
+ element (`BR`).
41
+
42
+ ## [0.11.0.rc1] - 2026-09-29
43
+
44
+ ### Changed
45
+
46
+ * HTML `#css` / `#at_css` / `#matches?` now match with Makiri's own Rust
47
+ implementation instead of Lexbor's `lxb_selectors` (selectors are still
48
+ parsed by Lexbor). Where the answers differ:
49
+ * `:nth-child(An+B of S)` / `:nth-last-child(... of S)` count by the CSS
50
+ definition; Lexbor miscounted when `S` was a comma list, held a
51
+ combinator, or carried pseudo-classes such as `:enabled`.
52
+ * `:disabled`, `:enabled` and `:checked` follow the HTML Standard: a
53
+ control inside a `<fieldset disabled>` (outside its first `<legend>`) is
54
+ disabled, `<option>` / `<optgroup>` count, and `:enabled` matches only
55
+ form elements.
56
+ * Attribute selector names are matched by qualified name, and
57
+ case-sensitively on SVG/MathML elements: `[href]` no longer finds
58
+ `xlink:href`, nor `[viewbox]` an SVG `viewBox`.
59
+ * `#id` / `.class` match only the no-namespace `id` / `class` attribute.
60
+ * A selector chain is capped at 64 compounds (as `Makiri::XML` already
61
+ was), and a query at 50M steps of work (XPath's limit); past either,
62
+ `Makiri::Error`.
63
+ * The column combinator `||` raises `Makiri::Error` instead of matching
64
+ nothing.
65
+
66
+ ### Removed
67
+
68
+ * `:lexbor-contains()` in HTML `#css` / `#at_css` / `#matches?`: it still
69
+ parses, but raises `Makiri::Error`. `Makiri::XML`'s `#css` keeps it. See
70
+ NOKOGIRI_DIFFERENCES.md.
71
+ * `Node#name=` / `#node_name=`. The DOM cannot rename an element, and renaming
72
+ a Lexbor element in place could crash (`div` to `template`). Create a new
73
+ element and `replace` the old one.
74
+
75
+ ### Security
76
+
77
+ * Hardening: every heap block the vendored Lexbor allocates carries 16 bytes
78
+ of slack past its end, so a small overrun cannot reach neighbouring memory
79
+ (not in sanitizer builds).
80
+ * Adding an `id` or `class` attribute beside an existing one
81
+ (`set_attribute_ns(nil, "ID", v)`, a namespaced `id`) no longer frees the
82
+ existing one under a held `Attr`. HTML attribute reads and writes follow the
83
+ DOM: `#[]`, `#key?`, `#[]=` and `#delete` match the qualified name
84
+ (`svg_a["href"]` no longer returns `xlink:href`), copies and imports keep
85
+ every attribute, and a no-namespace name keeps its case.
86
+ * `content=` on an HTML element and `delete(name)` detach the nodes they
87
+ remove instead of freeing them under a live wrapper.
88
+ * `el[name] = value` on an existing attribute no longer leaves a destroyed
89
+ attribute linked when storing the value runs out of memory.
90
+ * A checked argument String is locked while its bytes are in use, so a later
91
+ argument's `#to_s` cannot rewrite it (it raises instead); a receiver frozen
92
+ by an argument's `#to_s` is not edited; and a mutator's argument can no
93
+ longer rebuild the document's indexes in the middle of the edit.
94
+ * HTML element and attribute names follow the DOM's rules:
95
+ `create_element("img src=x onerror=alert(1)")` raises `ArgumentError`
96
+ instead of writing that markup. See NOKOGIRI_DIFFERENCES.md.
97
+ * HTML parsing bounds the tree depth (`max_tree_depth:`, default 400) and the
98
+ `<option>`s one `<select>` receives (10,000), raising `Makiri::Error` past
99
+ either; both made the parse quadratic.
100
+ * `Makiri::Lexbor::CSS.parse_stylesheet` no longer crashes on a
101
+ `:lexbor-contains()` after a string ended by CR, FF or a newline.
102
+ * A CSS query that runs out of memory raises instead of answering with what it
103
+ had collected; a selector-parse OOM is reported as OOM, not
104
+ `CSS::SyntaxError`.
105
+ * Inputs whose cost grew faster than their size are linear or budgeted: many
106
+ namespace bindings, the `preceding` axis and reverse-axis steps, XML
107
+ duplicate-attribute checks, `contains` / `substring-*` / `translate` on long
108
+ strings, XML CSS `:nth-*`, CDATA full of `]]>`, and stylesheets with
109
+ rejected `:lexbor-contains()`.
110
+ * A panic in any Ruby method Makiri defines raises `Makiri::InternalError`
111
+ (rescuable), not `fatal`.
112
+
113
+ ### Fixed
114
+
115
+ * A document that grows by editing (appended nodes, `inner_html=`) reports
116
+ its new size to the GC, so memory pressure from it triggers collections; it
117
+ was reported once, at parse time.
118
+ * A node wrapped while memory runs out raises `Makiri::Error` instead of
119
+ coming back as a second Ruby object for the same node, without the first
120
+ one's `freeze`, instance variables or singleton methods.
121
+ * Insertion follows the DOM's pre-insertion rules in HTML and XML alike: no
122
+ cycles through a template's contents (which hung `dup`), no children on
123
+ Text/Comment/PI/DocumentType/Attr, no Text directly under an XML Document,
124
+ and an XML `DocumentFragment` takes children.
125
+ * `inner_html` / `inner_html=` on a `<template>` target its contents, agreeing
126
+ with `#to_html` and `content_fragment`.
127
+ * HTML `#keys` / `#values` raise on out-of-memory instead of returning a
128
+ truncated Array.
129
+ * `dup`, `clone_node` and `import_node` keep an element's name as written
130
+ (`linearGradient`, `q:Bar`), and a copied XML doctype keeps its PUBLIC id.
131
+ * Namespaces across HTML <-> XML `import_node`: attributes keep their own
132
+ namespace, a prefixed element keeps its prefix and case, names XML cannot
133
+ write are refused where they would be written, and copied `xmlns`
134
+ attributes no longer move elements between namespaces.
135
+ * XML namespaces: `set_attribute_ns` on a detached element survives insertion;
136
+ `[]=` on an existing attribute changes only its value; the mutators refuse
137
+ what the parser refuses (forbidden declarations, duplicate expanded names,
138
+ bad doctype names and PUBLIC ids); `canonicalize` raises instead of writing a
139
+ wrong namespace; and a refused declaration says which rule it broke.
140
+ * `to_xml` output re-parses: CDATA holding `]]>`, a SYSTEM id holding `"`, and
141
+ attributes with a namespace but no prefix.
142
+ * XPath: `string()` of a number follows libxml2 (as Nokogiri does);
143
+ `substring()` rounds per spec; node-set vs boolean comparisons follow §3.4;
144
+ the `xml` prefix is always bound; a comparison no longer raises
145
+ `LimitExceeded` because its values added up past 64 MB.
146
+ * `Node#path` round-trips through `#at_xpath` for CDATA, PIs, and namespaced
147
+ nodes (SVG/MathML, XML namespaces, `xlink:href`, Vue/Word-style names); an
148
+ unreachable node answers `"?"` as in Nokogiri.
149
+ * CSS over XML agrees with the HTML matcher on `.x\ y`, `:root`, empty and
150
+ whitespace attribute values, `$=` with non-ASCII, `:empty` beside a comment,
151
+ and `[|a]`; the `s` modifier is accepted.
152
+ * An HTML document refuses a second root element or a text child; an
153
+ attribute's parent is always its owner element.
154
+ * `Node#line` of a node copied from another document is nil.
155
+ * `Makiri::XML` nodes compare with `<=>` in document order (an attribute with
156
+ itself too), and `XML::Document#dup` / `#clone` return a copy.
157
+ * `clone_node` on a Document raises instead of returning the document.
158
+ * `NodeSet#css` / `#xpath` / `#at_css` / `#at_xpath` pass their extra
159
+ arguments through; `#xpath` of a non-node-set expression raises
160
+ `ArgumentError`, as in Nokogiri.
161
+ * `XML::Builder` no longer answers `respond_to?` for `to_ary` and similar.
162
+ * Namespace Hashes and `register_namespace` arguments are read as a Hash and
163
+ with `String()`, with one length cap and one error message.
164
+ * A rejected stylesheet rule's `selector_text` is sliced by Lexbor's offsets.
165
+
166
+ ### Performance
167
+
168
+ * `NodeSet#at_css` / `#at_xpath` stop at the first node with a match.
169
+ * `XML::Node#canonicalize` looks namespaces up by prefix in the scope it keeps:
170
+ a 1000-deep document of declarations went from 0.7 s to 0.02 s.
171
+
172
+ ## [0.10.0] - 2026-09-22
173
+
174
+ ### Fixed
175
+
176
+ * A frozen node now raises `FrozenError` when it is the ARGUMENT of a tree
177
+ mutation, not only the receiver: `b.add_child(a)` relinks `a` exactly as
178
+ `a.remove` does. Both `Makiri::XML` and `Makiri::HTML`. A fragment argument
179
+ splices its children, which have no wrapper of their own to check.
180
+ * A `Makiri::XML` mutation that exceeds the document's own `max_bytes` /
181
+ `max_nodes` now raises `Makiri::XML::LimitExceeded` instead of reporting the
182
+ refusal as out of memory.
183
+ * Inserting a `DocumentFragment` is all or nothing on `add_child`, `before` and
184
+ `after`, as it already was on `replace`: a child the rules refuse no longer
185
+ leaves the earlier ones linked in a document the caller was told had not
186
+ changed.
187
+ * A rejected `Makiri::XML::Document#fragment` no longer charges the document for
188
+ the nodes it discarded. 100k rejected fragments grew a `<r/>` document to
189
+ 77 MB; it is now 516 bytes.
190
+ * A failed `Makiri::XML` parse reports the FIRST failure for all four kinds
191
+ (`syntax` was sticky while `limit` and `unsupported` overwrote each other).
192
+ * `Makiri::XML#to_xml`'s serializer allocates fallibly again, so running out of
193
+ memory raises instead of aborting the process.
194
+ * `:lexbor-contains()` now rejects an argument the bundled CSS parser does not
195
+ take, the way any unknown pseudo-class is rejected: `Makiri::CSS::SyntaxError`
196
+ from `#css` / `#at_css` / `#matches?`, and a `:bad_style` rule from
197
+ `Makiri::Lexbor::CSS.parse_stylesheet`. Well-formed uses are unchanged.
198
+
199
+ ### Performance
200
+
201
+ * `Makiri::XML#to_xml` plans namespaces from a binding stack instead of
202
+ re-walking each ancestor's attribute list, which cost O(depth^2 x attributes).
203
+ 403 KB of nested prefixed attributes took 4.88s and now takes 0.001s. A
204
+ crafted document fails closed with `Makiri::Error` ("namespace planning
205
+ exceeded its step budget") rather than running on.
206
+ * Setting an attribute on a `Makiri::XML` element walks the attribute list once
207
+ instead of twice (4096 attributes: 47ms -> 27ms).
208
+
3
209
  ## [0.10.0.rc2] - 2026-09-20
4
210
 
5
211
  ### Added
@@ -26,11 +26,34 @@ what browsers do - rather than libxml2. Detailed, test-backed notes live in
26
26
  exactly (`//*[@refX]`, not `@refx`). Only ASCII folds: `Ø` still differs from `ø`.
27
27
  This holds in `namespace_matching: :lax` too.
28
28
  * `Nokogiri::HTML5` is case-sensitive there.
29
+ * `Node#path` names a node by expanded name where a bare name would not reach it
30
+ * An SVG, MathML or namespaced-XML node, and an HTML name that is no plain
31
+ XPath name (`<o:p>`, `xml:lang`, Vue's `@click`), give
32
+ `*[local-name()='path' and namespace-uri()='http://www.w3.org/2000/svg']`,
33
+ which `#at_xpath` evaluates with no prefix registered. Nokogiri writes
34
+ `svg:svg` or `/*/*[2]` for the first kinds; the paths are equivalent, the
35
+ strings are not.
36
+ * A node not attached to its document answers `"?"`, as a doctype does.
37
+ Nokogiri answers `"/div/p"` for a detached `<div><p>`, which is the path of
38
+ the document's own `/div/p` when it has one.
29
39
  * A foreign element's namespace declarations are not attributes
30
40
  * `<svg xmlns="...">` has no `@xmlns` for `//*[@xmlns]` or `@*`, as in browsers
31
41
  and in XPath's data model. An `xmlns` on an HTML element is an ordinary
32
42
  attribute and stays visible.
33
43
 
44
+ * A `StandardError` raised by a custom-function handler becomes `Makiri::Error`
45
+ ("handler raised: <message>"), with the handler's exception as its `#cause`
46
+ * Nokogiri re-raises the handler's exception itself. Anything that is not a
47
+ `StandardError` (`Interrupt`, `SystemExit`, Timeout's exception) and a
48
+ `throw` reach the caller unchanged in both.
49
+
50
+ * A number literal is read as the nearest double; libxml2's own reader is not
51
+ correctly rounded for a literal with more digits than a double holds, so
52
+ `string(0.72609133372266155)` is `0.726091333722662` in Makiri and
53
+ `0.726091333722661` in Nokogiri (the digits differ in the last place). The
54
+ number is then written by libxml2's rule in both (`string(1234567890.5)` is
55
+ `1.2345678905e+09`); only the value read differs.
56
+
34
57
  ## XML
35
58
 
36
59
  * `Makiri::XML` is XML 1.0 (Fifth Edition) only and non-validating.
@@ -61,6 +84,13 @@ what browsers do - rather than libxml2. Detailed, test-backed notes live in
61
84
  * `create_processing_instruction("a:b", ...)` succeeds, as DOM
62
85
  `createProcessingInstruction` does, but `#to_xml` / `#canonicalize` then raise,
63
86
  as DOM Parsing's well-formed serializer does. Nokogiri writes `<?a:b ...?>`.
87
+ * `#freeze` on a node is ENFORCED: a frozen node's mutators raise `FrozenError`,
88
+ and so does passing a frozen node as the argument of an insertion, which
89
+ relinks it. Nokogiri reports `frozen?` but every mutator still mutates. The
90
+ check reaches the nodes the caller named; a fragment argument splices its
91
+ children, and those cannot be checked, because frozen-ness is a property of a
92
+ Ruby object and the arena keeps no map from a node back to its wrapper.
93
+
64
94
  * A node's namespace URI is its identity, not something re-derived from the
65
95
  declarations around it - the WHATWG DOM model, measured against Chrome 152
66
96
  (`DOMParser` + `XMLSerializer`).
@@ -77,6 +107,19 @@ what browsers do - rather than libxml2. Detailed, test-backed notes live in
77
107
  invents one (`ns1`, `ns2`, ...) rather than shadow the other, as browsers do.
78
108
  * An element in no namespace stays that way under a default namespace,
79
109
  serialized as `xmlns=""`.
110
+ * `create_document_type` takes what the DOM's `createDocumentType` takes -
111
+ any id, and a name without whitespace, NUL or `>` - and `to_xml` refuses a
112
+ doctype XML cannot write (`create_document_type("q", "abcde", %(x"'y))`).
113
+ Nokogiri writes that system id as `"x&quot;'y"`, which parses but reads
114
+ back as the id `x&quot;'y` - a literal expands no references.
115
+ * `set_attribute_ns(XMLNS_NS, "xmlns:foo", "")` - a declaration Namespaces in
116
+ XML forbids, which the DOM's `setAttributeNS` accepts - is kept as an
117
+ attribute that binds nothing, and `to_xml` refuses the tree while it is
118
+ there. `root["xmlns:foo"] = ""` still raises.
119
+ * An attribute in the XML namespace is written as `xml:local`, whatever
120
+ prefix it was given (`set_attribute_ns(XML_NS, "a:bb")`, as the DOM
121
+ allows): Namespaces in XML binds that namespace to `xml` alone. It re-reads
122
+ to the same namespace and local name. `canonicalize` writes it so too.
80
123
  * Nodes from the factories (`create_element` and friends) still take their
81
124
  namespace from the context they are first inserted into, so a subtree can be
82
125
  built detached and attached afterwards. Only later moves carry.
@@ -93,6 +136,14 @@ what browsers do - rather than libxml2. Detailed, test-backed notes live in
93
136
  the property-based differential), including namespaces, prolog/epilog comments
94
137
  and PIs, and adjacent-CDATA coalescing.
95
138
 
139
+ * `to_xml` keeps an element's namespace when an `xmlns` attribute on it says
140
+ otherwise
141
+ * `root["xmlns"] = "urn:x"` on an element in no namespace is not written: the
142
+ element stays in no namespace when the output is re-read, as the DOM Parsing
143
+ and Serialization spec asks. Nokogiri writes `<r xmlns="urn:x">`, which moves
144
+ the element into `urn:x` on re-parse. Create the element in the namespace
145
+ instead (`create_element("r", "xmlns" => "urn:x")`, or parse it so).
146
+
96
147
  ## HTML parsing
97
148
 
98
149
  * `<?php ... ?>` in HTML input is a **ProcessingInstruction** node; `#to_html`
@@ -102,6 +153,79 @@ what browsers do - rather than libxml2. Detailed, test-backed notes live in
102
153
  still produces the older bogus comment (`<!--?php ... ?-->`), and
103
154
  `Nokogiri::HTML` (libxml2) its own comment.
104
155
 
156
+ * A tree deeper than `max_tree_depth` raises **`Makiri::Error`**, where
157
+ `Nokogiri::HTML5` raises `ArgumentError`; the default (400) and boundary are
158
+ Nokogiri's. `inner_html=`, `outer_html=` and `Node#parse` take no keyword.
159
+ `Nokogiri::HTML` (libxml2) instead stops at depth 256 and returns the
160
+ truncated document.
161
+ * More than 10,000 `<option>`s in one `<select>` raise `Makiri::Error`;
162
+ Nokogiri has no such limit.
163
+
164
+ ## HTML mutation
165
+
166
+ * There is no `Node#name=` / `#node_name=` (on HTML or XML nodes). Nokogiri
167
+ renames a node in place and keeps its identity; the DOM has no rename, and
168
+ Lexbor keeps elements such as `<template>`, `<option>` and `<style>` in
169
+ structs of their own, so a node cannot change its tag safely. Create an
170
+ element of the new name, move the attributes and children, and `replace`
171
+ the old one (see CHANGELOG.md).
172
+ * Element and attribute names follow the WHATWG DOM's rules
173
+ * `create_element`, `[]=` and `set_attribute_ns` raise `ArgumentError` for a
174
+ name the DOM refuses - one holding whitespace, `/`, `>` (or `=` for an
175
+ attribute) - where `Nokogiri::HTML5` accepts it and writes it into the
176
+ markup as it stands: `create_element("img src=x onerror=alert(1)")`
177
+ serializes as that tag.
178
+ The names HTML actually uses (`data-*`, `aria-*`, `@click`, `:href`,
179
+ `v-on:x`, custom elements) are accepted.
180
+ * `set_attribute_ns(nil, "x:y")` raises, as the DOM's `setAttributeNS` does:
181
+ a prefix needs a namespace. `[]=` (HTML) and `set_loose_dom_attribute`
182
+ (XML) are the DOM's `setAttribute`, which makes such an attribute.
183
+ * A refused name raises `ArgumentError` - a NUL in it too - and a namespace
184
+ that does not fit the name (`set_attribute_ns`, `create_element_ns`)
185
+ raises `Makiri::Error`. Invalid UTF-8 raises `Makiri::Error` for every
186
+ argument, names included.
187
+ * `create_element_ns` refuses an HTML-namespace name in upper case that
188
+ lower-cases to an element Lexbor knows (`BR`, `DIV`), where the DOM makes
189
+ an unknown element: Lexbor would make that element (`BR` void, its
190
+ children never written). Other names keep their case (`MY-EL`). Nokogiri
191
+ has no `create_element_ns`.
192
+ * An HTML `<template>` follows the WHATWG content model, which Nokogiri does
193
+ not: its parsed contents live in the separate fragment `Element#content_fragment`
194
+ returns, `template.children` is empty, and `inner_html` / `inner_html=`
195
+ special-case the contents (the WHATWG DOM special-cases `innerHTML` alone;
196
+ `append_child`, `content=`, and `children` act on the element's own empty
197
+ children, as the specification says). Nokogiri treats `<template>` as an
198
+ ordinary element, with the parsed nodes as its children, so
199
+ `template.inner_html` and `template.children` answer the other way round and
200
+ there is no `content_fragment`.
201
+ * An HTML document has one root element and no text child, as the DOM requires;
202
+ `doc << element` beside an existing root raises.
203
+ * An insertion the DOM refuses - a child under a text, comment, PI, doctype
204
+ or attribute node, text under an XML Document, a second root - raises
205
+ `Makiri::Error` in both representations. Nokogiri refuses the same ones with
206
+ `ArgumentError` (or `RuntimeError` for a second XML root).
207
+ * Moving HTML into an XML document (`xml_doc.import_node(html_node)`, or
208
+ inserting one) keeps every name's namespace, and refuses what XML cannot
209
+ write that way. An attribute in no namespace whose name has a prefix other
210
+ than `xml` - `v-on:click`, `fb:like`, an `xlink:href` on an HTML (not SVG)
211
+ element - raises `Makiri::Error`: as XML it would be a prefix bound to
212
+ nothing. Nokogiri copies it and writes `v-on:click="..."` into output that is
213
+ not namespace-well-formed. An element named with a colon (`<fb:like>`)
214
+ crosses as a DOM-loose name, which `to_xml` refuses.
215
+ * A known gap, in Lexbor's tag table: an HTML document that already holds a
216
+ parsed element named with a colon (`<x:y>`, one local name) and then
217
+ receives, by `import_node` from another document, a prefixed element
218
+ written the same way (`x:y` from XML: prefix `x`, local name `y`) re-points
219
+ the table's entry for that spelling - the parsed element then no longer
220
+ matches CSS `x\:y`. Copies within one document do not touch the table.
221
+ * An XML element with a prefix, imported into HTML, keeps it (`h:div` in
222
+ XHTML has the local name `div`). Two readers then disagree, as they do in
223
+ browsers: CSS's `div` matches it (Lexbor matches the local name), XPath's
224
+ `//div` does not (an HTML element's name test reads its qualified name). And
225
+ Lexbor's HTML serializer writes the prefix (`<h:div>`), where the HTML
226
+ standard writes the local name, so the HTML does not re-parse to the same
227
+ element.
228
+
105
229
  ## CSS
106
230
 
107
231
  * Most jQuery/Nokogiri CSS extensions are not supported (`:gt`, `:lt`, `:eq`, `:first`, ...)
@@ -109,13 +233,18 @@ what browsers do - rather than libxml2. Detailed, test-backed notes live in
109
233
  text-containment extension. Use XPath (`xpath("//p[contains(., 'x')]")`) or
110
234
  Enumerable (`css('li')[1]`) for the rest.
111
235
  Standard Level-4 selectors (`:is` / `:where` / `:has`) are supported; some of which Nokogiri rejects.
112
- * `:lexbor-contains("text")` is supported (on both HTML and XML) - Lexbor's
113
- spelling of the jQuery `:contains()` substring filter, matching an element
114
- whose text contains the string; append ` i` (`:lexbor-contains("text" i)`)
115
- for an ASCII case-insensitive match. (Nokogiri's name `:contains` is not an
116
- alias.) Like Lexbor's matcher, it tests the element's immediate child text
117
- nodes (not the deep string-value), so HTML and XML agree; on XML it lowers
118
- to XPath `child::text()[contains(., "text")]`.
236
+ * `:lexbor-contains("text")` - Lexbor's spelling of the jQuery `:contains()`
237
+ substring filter, matching an element whose text contains the string;
238
+ append ` i` (`:lexbor-contains("text" i)`) for an ASCII case-insensitive
239
+ match. (Nokogiri's name `:contains` is not an alias.) **XML only**: it
240
+ lowers to XPath `child::text()[contains(., "text")]`, testing the
241
+ element's immediate child text nodes, not the deep string-value. HTML no
242
+ longer supports it - `lexbor::css_match`, the safe-Rust port that
243
+ replaced Lexbor's own matcher for HTML, deliberately does not reimplement
244
+ it, so a well-formed `:lexbor-contains()` now raises `Makiri::Error`
245
+ ("could not be run") on HTML rather than ever matching (it still PARSES,
246
+ and an actually malformed one still raises `Makiri::CSS::SyntaxError`, as
247
+ before).
119
248
  * Untyped `:*-of-type` (`:first-of-type`, `:nth-of-type(an+b)`, ... with no type
120
249
  selector) is supported and correct on both HTML and XML - the "type" is the
121
250
  element's own expanded name.
@@ -131,12 +260,35 @@ what browsers do - rather than libxml2. Detailed, test-backed notes live in
131
260
  prefix IS resolved against the bindings, when the namespace matters.
132
261
  * `Makiri::XML` resolves CSS prefixes properly - it lowers the selector to the
133
262
  XPath engine, which registers the bindings.
263
+ * The same holds for attribute selectors: HTML `[|href]` (no namespace) and
264
+ plain `[href]` also find an SVG `xlink:href`, which `Nokogiri::HTML5`
265
+ does not.
266
+ `Makiri::XML` reads `[|a]` as the no-namespace attribute.
267
+ * A selector under a node matches the way `Element#querySelectorAll` does in a
268
+ browser, not scoped to that node, on HTML: `at_css("#c").css("div p")` finds a
269
+ `p` inside `#c` when `#c` is itself a `div`, since the selector is matched
270
+ against the whole document and only the results are kept to descendants.
271
+ `Nokogiri::HTML5` and `Nokogiri::XML` scope the selector to the node (`#c`
272
+ cannot be the `div`), and so does `Makiri::XML`, which lowers the selector to
273
+ an XPath from the node. Lexbor has no `:scope`; for a scoped match on HTML,
274
+ use XPath from the node (`xpath(".//div//p")`).
275
+ * The attribute case modifiers (`[a="x" i]`, `[a="x" s]`): HTML supports both,
276
+ through Lexbor's matcher. `Makiri::XML` accepts `s` (case-sensitive, which XML
277
+ values are anyway) and refuses `i` with `Makiri::CSS::SyntaxError`. Nokogiri
278
+ refuses both on either representation.
134
279
  * `#matches?` answers for a DETACHED node (`document.create_element("p")
135
280
  .matches?("p")` is true, on both representations). Nokogiri raises
136
281
  `NoMethodError` there - it implements `#matches?` as a search from
137
282
  `ancestors.last`, which a detached node does not have.
138
- * * Type selectors are ASCII case-insensitive (CSS-correct for HTML; `LI` matches `<li>`)
139
- * `Nokogiri::HTML5` is case-sensitive there.
283
+ * Type selectors in an HTML document follow the HTML Standard's
284
+ case-sensitivity rule, as browsers do: lower-cased for an HTML element (`LI`
285
+ matches `<li>`), as written for any other (`feGaussianBlur` matches the SVG
286
+ element, `fegaussianblur` does not). An HTML element named in upper case
287
+ (`create_element_ns(XHTML, "MY-EL")`, which keeps its name as the DOM does)
288
+ therefore matches no type selector.
289
+ * `Nokogiri::HTML5` is case-sensitive on HTML elements too, so `LI` does not
290
+ match `<li>` there. `Makiri::XML`'s `#css` is case-sensitive, as XML names
291
+ are.
140
292
 
141
293
  ## Serialization
142
294
 
@@ -163,7 +315,10 @@ what browsers do - rather than libxml2. Detailed, test-backed notes live in
163
315
  and attribute values (`[]=`, `set_attribute_ns`) - and stored/read back
164
316
  verbatim, matching the WHATWG DOM / browsers (`document.createTextNode("\0")`).
165
317
  It is still rejected in names, tag names, namespaces, PI target/data, CSS
166
- selectors, and XPath expressions/variable names (a NUL there raises).
318
+ selectors, and XPath expressions/variable names. A NUL in a name given to a
319
+ factory or setter (element, attribute, doctype and PI target names) raises
320
+ `ArgumentError`, as any other refused name does; anywhere else it raises
321
+ `Makiri::Error`.
167
322
  * On re-parse, the HTML tokenizer replaces a U+0000 in text/attributes with
168
323
  U+FFFD (WHATWG), so a serialized-then-reparsed round-trip is not byte-identical.
169
324
  * `Makiri::XML` rejects NUL everywhere: XML 1.0 has no legal U+0000 character,
data/README.md CHANGED
@@ -23,11 +23,13 @@ XPath 1.0 evaluation in its own native engine, with no libxml2 dependency.
23
23
  * Native XML 1.0 parser
24
24
  * A strict, non-validating, fail-closed parser with its own node arena (not
25
25
  Lexbor's HTML DOM), queried through the same native XPath engine, with
26
- in-place tree edits (attributes, content, rename, remove).
26
+ in-place tree edits (attributes, content, remove).
27
27
  * Conformance is held by the W3C XML Conformance Test Suite, an XPath
28
28
  differential, and property-based testing vs Nokogiri (see below).
29
29
  * Bounded, fail-closed execution
30
30
  * XPath evaluation is bounded by per-evaluation limits on work, memory, and recursion.
31
+ * HTML parsing bounds the tree depth (`max_tree_depth:`, default 400) and the
32
+ `<option>`s per `<select>` (10,000).
31
33
  * Ownership and borrowing are kept explicit across layers, with owned/borrowed
32
34
  string types and verified text at engine boundaries.
33
35
  * Programmatic invalid input, limit violations, allocation failures, and unsupported constructs
@@ -65,6 +67,10 @@ link.parent.name # => "div"
65
67
  # Source location (reconstructed from the tokenizer, no Lexbor patches)
66
68
  doc.at_css("p").line # => 3
67
69
 
70
+ # Nesting deeper than 400 elements raises Makiri::Error (Nokogiri's default);
71
+ # max_tree_depth: raises the limit, and a negative value disables it
72
+ Makiri::HTML(deep_html, max_tree_depth: 2000)
73
+
68
74
  # Serialization
69
75
  doc.at_css("#main").to_html # => "<div id=\"main\" ...>...</div>"
70
76
  doc.at_css("#main").inner_html # => "\n <p class=\"lead\">Hello</p>\n..."
@@ -132,11 +138,10 @@ e = doc.at_xpath("//entry")
132
138
  e["id"] = "9" # add or replace an attribute (value escaped on output)
133
139
  e["dc:k"] = "v" # a prefixed name resolves against the in-scope xmlns
134
140
  e.content = "Bye" # replace an element's children with text
135
- e.name = "post" # rename in place (identity + namespace re-resolved)
136
141
  e.delete("id") # remove an attribute
137
142
  doc.at_xpath("//draft").remove
138
143
 
139
- doc.root.to_xml # => "<feed xmlns:dc=\"urn:dc\"><post dc:k=\"v\">Bye</post></feed>"
144
+ doc.root.to_xml # => "<feed xmlns:dc=\"urn:dc\"><entry dc:k=\"v\">Bye</entry></feed>"
140
145
  ```
141
146
 
142
147
  XML subtrees can be built using `Document#create_element` and other node factory methods,