makiri 0.11.0 → 0.12.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (50) hide show
  1. checksums.yaml +4 -4
  2. data/.github/workflows/css-match.yml +180 -0
  3. data/.github/workflows/libfuzzer.yml +3 -1
  4. data/CHANGELOG.md +68 -0
  5. data/NOKOGIRI_DIFFERENCES.md +34 -15
  6. data/Rakefile +5 -2
  7. data/ext/makiri/rust/Cargo.toml +8 -2
  8. data/ext/makiri/rust/build.rs +28 -1
  9. data/ext/makiri/rust/fuzz/Cargo.toml +10 -1
  10. data/ext/makiri/rust/fuzz/css.dict +77 -0
  11. data/ext/makiri/rust/fuzz/fuzz_targets/html_css.rs +13 -10
  12. data/ext/makiri/rust/fuzz/fuzz_targets/html_css_diff.rs +282 -0
  13. data/ext/makiri/rust/src/bridge/fragment.rs +46 -10
  14. data/ext/makiri/rust/src/bridge/html.rs +0 -6
  15. data/ext/makiri/rust/src/bridge/string.rs +24 -0
  16. data/ext/makiri/rust/src/bridge/xml.rs +22 -3
  17. data/ext/makiri/rust/src/falloc/inject.rs +10 -8
  18. data/ext/makiri/rust/src/glue/html_doc.rs +4 -2
  19. data/ext/makiri/rust/src/glue/html_node/mutate.rs +0 -3
  20. data/ext/makiri/rust/src/glue/xml_node/mutate.rs +10 -10
  21. data/ext/makiri/rust/src/glue/xml_node/serialize.rs +7 -3
  22. data/ext/makiri/rust/src/lexbor/abi.rs +21 -2
  23. data/ext/makiri/rust/src/lexbor/adapter/cross_import.rs +65 -127
  24. data/ext/makiri/rust/src/lexbor/adapter/dom_index.rs +19 -2
  25. data/ext/makiri/rust/src/lexbor/adapter/html/build.rs +159 -32
  26. data/ext/makiri/rust/src/lexbor/adapter/html/mod.rs +75 -9
  27. data/ext/makiri/rust/src/lexbor/adapter/html/serialize.rs +435 -0
  28. data/ext/makiri/rust/src/lexbor/adapter/tree_guard.rs +89 -15
  29. data/ext/makiri/rust/src/lexbor/css_match/mod.rs +14 -2
  30. data/ext/makiri/rust/src/lexbor/fragment.rs +122 -25
  31. data/ext/makiri/rust/src/lexbor/mod.rs +3 -2
  32. data/ext/makiri/rust/src/lexbor/selectors.rs +2 -1
  33. data/ext/makiri/rust/src/lexbor/serialize.rs +10 -31
  34. data/ext/makiri/rust/src/lexbor/tests.rs +279 -1
  35. data/ext/makiri/rust/src/lexbor/xpath.rs +3 -2
  36. data/ext/makiri/rust/src/xml/model.rs +10 -7
  37. data/ext/makiri/rust/src/xml/mutate/attr.rs +45 -18
  38. data/ext/makiri/rust/src/xml/mutate/edit.rs +21 -21
  39. data/ext/makiri/rust/src/xml/mutate/factory.rs +19 -11
  40. data/ext/makiri/rust/src/xml/mutate/mod.rs +2 -1
  41. data/ext/makiri/rust/src/xml/selftest.rs +26 -25
  42. data/ext/makiri/rust/src/xml/serialize/c14n.rs +71 -37
  43. data/ext/makiri/rust/src/xml/serialize/mod.rs +11 -6
  44. data/ext/makiri/rust/src/xml/serialize/out.rs +109 -0
  45. data/ext/makiri/rust/src/xml/serialize/xml.rs +3 -6
  46. data/lib/makiri/element.rb +16 -0
  47. data/lib/makiri/node_path.rb +8 -1
  48. data/lib/makiri/version.rb +1 -1
  49. data/script/check_unsafe_boundaries.rb +8 -7
  50. metadata +5 -1
checksums.yaml CHANGED
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  SHA256:
3
- metadata.gz: 9c12838a29d87a6bd9acb72b97e1e25dbdfbdaff37e38609cce6fd562e0f76d7
4
- data.tar.gz: f4531547107db81c7db834d6c991adcf68b8a0d2fd714e158d69b164140a18bc
3
+ metadata.gz: 6432135ca3963c17f432c152ede49162c3f1be48e96179b914c971f0b2da1b76
4
+ data.tar.gz: 2b0478171ee224ea5c29ae19cd0a16a08c3d786e017444241d4bb8f620ecf975
5
5
  SHA512:
6
- metadata.gz: fcdb5e1656c3ab6781c1bae97919b5ce65c5cbaa2bc3a0429a9088483f45407715e060afb2b6478d5ecb8a2d7495c319e3db54acd17f419c728d181a2b9becb4
7
- data.tar.gz: 784ef5eaac9439e44ea99903544bcefac25869f18dc6312f0629669465f92a8f306f72b1bb62d5671babccfca2374f19c026dbb0da2a4f859f917c9822093b9b
6
+ metadata.gz: 8447a816ba507fb6a74ed8e6a5a92deaff6da2453ef29ba6cfe24652ec160b440409ec94b58b4aa89d3d28ab05643086f4dd1bba150339eeff27819d2e3d86d7
7
+ data.tar.gz: 2a9505ac1b1762cbdbc738645542acbda4bb029e06e5b4befe76f63e39e6e55a9f36a34106dd0fe24d3103d17f020939498373d9c872cb2042d71b23bb0d6513
@@ -0,0 +1,180 @@
1
+ name: CSS matcher
2
+
3
+ # The HTML CSS matcher (`lexbor::css_match`) on its own: it answers every
4
+ # `Node#css` / `#at_css` / `#matches?`, and its failure mode is not a crash
5
+ # but a wrong answer, which the general suites only catch by example. Two
6
+ # checks, both against the OLD engine - Lexbor's own `lxb_selectors`, kept as
7
+ # `lexbor::selectors` for exactly this:
8
+ #
9
+ # - the randomized differential tests in `lexbor::tests::css_match`, run long
10
+ # with a fresh seed (MAKIRI_CSS_DIFF_SEED; a failure prints it);
11
+ # - the cargo-fuzz targets `html_css` (memory safety, and the three entry
12
+ # points agreeing) and `html_css_diff` (agreement with Lexbor's engine),
13
+ # with the selector dictionary.
14
+ #
15
+ # A PR that touches the matcher, the Lexbor layer or the pin gets a short run;
16
+ # the nightly one is long. A disagreement on a shape where css_match departs
17
+ # from Lexbor on purpose is not a finding: the departures are listed in
18
+ # `css_match`'s module doc, and both checks leave them out.
19
+
20
+ on:
21
+ push:
22
+ branches: [main, master]
23
+ paths:
24
+ - "ext/makiri/rust/src/lexbor/**"
25
+ - "ext/makiri/rust/fuzz/**"
26
+ - "ext/makiri/rust/Cargo.toml"
27
+ - "vendor/lexbor"
28
+ - ".github/workflows/css-match.yml"
29
+ pull_request:
30
+ paths:
31
+ - "ext/makiri/rust/src/lexbor/**"
32
+ - "ext/makiri/rust/fuzz/**"
33
+ - "ext/makiri/rust/Cargo.toml"
34
+ - "vendor/lexbor"
35
+ - ".github/workflows/css-match.yml"
36
+ schedule:
37
+ - cron: "41 4 * * *"
38
+ workflow_dispatch:
39
+
40
+ env:
41
+ # Short on a push or PR, long on the nightly schedule.
42
+ LONG: ${{ github.event_name == 'schedule' || github.event_name == 'workflow_dispatch' }}
43
+
44
+ jobs:
45
+ differential:
46
+ name: Randomized differential (vs Lexbor's engine)
47
+ runs-on: ubuntu-latest
48
+ timeout-minutes: 60
49
+ steps:
50
+ - name: Checkout (with vendored Lexbor submodule)
51
+ uses: actions/checkout@v6
52
+ with:
53
+ submodules: recursive
54
+
55
+ - name: Ensure cmake is available
56
+ uses: lukka/get-cmake@latest
57
+
58
+ - uses: dtolnay/rust-toolchain@stable
59
+
60
+ - name: Install libclang (bindgen, via rb-sys)
61
+ run: sudo apt-get update && sudo apt-get install -y libclang-dev
62
+
63
+ - name: Set up Ruby
64
+ uses: ruby/setup-ruby@v1
65
+ with:
66
+ ruby-version: "3.4"
67
+ bundler-cache: true
68
+
69
+ # build.rs links the vendored Lexbor archive and reads its headers.
70
+ - name: Compile the extension (builds Lexbor)
71
+ run: bundle exec rake compile
72
+
73
+ # Release: the sweep is the point, and debug is ~10x slower. The seed is
74
+ # fresh per run, so a long series covers new ground; it is printed here
75
+ # and in any failure, so a finding reproduces.
76
+ - name: Run the css_match tests with a fresh seed
77
+ working-directory: ext/makiri/rust
78
+ run: |
79
+ export MAKIRI_CSS_DIFF_SEED=$(printf '0x%x' $(( (GITHUB_RUN_ID * 2654435761 + GITHUB_RUN_ATTEMPT) & 0xffffffffffff )))
80
+ if [ "$LONG" = true ]; then
81
+ export MAKIRI_CSS_DIFF_ITERATIONS=1000000
82
+ else
83
+ export MAKIRI_CSS_DIFF_ITERATIONS=20000
84
+ fi
85
+ echo "MAKIRI_CSS_DIFF_SEED=$MAKIRI_CSS_DIFF_SEED MAKIRI_CSS_DIFF_ITERATIONS=$MAKIRI_CSS_DIFF_ITERATIONS"
86
+ cargo test --release --no-default-features --features lexbor -- \
87
+ lexbor::tests::css_match lexbor::tests::selector_cache
88
+
89
+ fuzz:
90
+ name: cargo-fuzz ${{ matrix.target }}
91
+ runs-on: ubuntu-latest
92
+ timeout-minutes: 120
93
+ strategy:
94
+ fail-fast: false
95
+ matrix:
96
+ target: [html_css, html_css_diff]
97
+ steps:
98
+ - name: Checkout (with vendored Lexbor submodule)
99
+ uses: actions/checkout@v6
100
+ with:
101
+ submodules: recursive
102
+
103
+ - name: Ensure cmake is available
104
+ uses: lukka/get-cmake@latest
105
+
106
+ # -Zsanitizer is unstable, and cargo-fuzz needs it.
107
+ - uses: dtolnay/rust-toolchain@nightly
108
+
109
+ - name: Install libclang (bindgen, via rb-sys)
110
+ run: sudo apt-get update && sudo apt-get install -y libclang-dev
111
+
112
+ - name: Set up Ruby
113
+ uses: ruby/setup-ruby@v1
114
+ with:
115
+ ruby-version: "3.4"
116
+ bundler-cache: true
117
+
118
+ - name: Compile the extension (builds Lexbor)
119
+ run: bundle exec rake compile
120
+
121
+ - name: Install cargo-fuzz
122
+ run: cargo install cargo-fuzz --locked
123
+
124
+ - name: Build the fuzz target
125
+ working-directory: ext/makiri/rust/fuzz
126
+ run: cargo fuzz build ${{ matrix.target }}
127
+
128
+ # As in libfuzzer.yml: nm -u reads 0 on Linux whether or not the binary
129
+ # is instrumented, so assert on the full symbol table.
130
+ - name: Assert the target is instrumented
131
+ working-directory: ext/makiri/rust/fuzz
132
+ run: |
133
+ bin=$(ls target/*/release/${{ matrix.target }})
134
+ reports=$(nm "$bin" | grep -c '__asan_report')
135
+ echo "$bin: __asan_report symbols = $reports"
136
+ test "$reports" -gt 0 || { echo "not ASan-instrumented"; exit 1; }
137
+
138
+ # Both targets read the same `selector NUL document` input, so each
139
+ # also starts from the other's corpus (read-only: new units go to the
140
+ # first directory). The cache prefix is libfuzzer.yml's, so the corpus
141
+ # accumulated there carries over.
142
+ - name: Restore the corpus
143
+ uses: actions/cache@v6
144
+ with:
145
+ path: ext/makiri/rust/fuzz/corpus/${{ matrix.target }}
146
+ key: cargo-fuzz-corpus-${{ matrix.target }}-${{ github.run_id }}
147
+ restore-keys: |
148
+ cargo-fuzz-corpus-${{ matrix.target }}-
149
+
150
+ - name: Restore the sibling corpus
151
+ uses: actions/cache/restore@v6
152
+ with:
153
+ path: ext/makiri/rust/fuzz/corpus/${{ matrix.target == 'html_css' && 'html_css_diff' || 'html_css' }}
154
+ key: cargo-fuzz-corpus-${{ matrix.target == 'html_css' && 'html_css_diff' || 'html_css' }}-${{ github.run_id }}
155
+ restore-keys: |
156
+ cargo-fuzz-corpus-${{ matrix.target == 'html_css' && 'html_css_diff' || 'html_css' }}-
157
+
158
+ # -timeout: html_css_diff runs Lexbor's exhaustively backtracking engine
159
+ # too; the harness skips inputs it cannot afford, and this is the stop
160
+ # for one it misjudged.
161
+ - name: Run cargo-fuzz ${{ matrix.target }}
162
+ working-directory: ext/makiri/rust/fuzz
163
+ run: |
164
+ sibling=${{ matrix.target == 'html_css' && 'html_css_diff' || 'html_css' }}
165
+ mkdir -p corpus/${{ matrix.target }} corpus/$sibling
166
+ if [ "$LONG" = true ]; then seconds=3600; else seconds=180; fi
167
+ cargo fuzz run ${{ matrix.target }} corpus/${{ matrix.target }} corpus/$sibling -- \
168
+ -max_total_time=$seconds \
169
+ -max_len=4096 \
170
+ -timeout=30 \
171
+ -dict=css.dict \
172
+ -print_final_stats=1
173
+
174
+ - name: Upload reproducers
175
+ if: failure()
176
+ uses: actions/upload-artifact@v4
177
+ with:
178
+ name: cargo-fuzz-artifacts-${{ matrix.target }}
179
+ path: ext/makiri/rust/fuzz/artifacts/${{ matrix.target }}
180
+ retention-days: 30
@@ -25,7 +25,9 @@ jobs:
25
25
  strategy:
26
26
  fail-fast: false
27
27
  matrix:
28
- target: [xml, xpath, xml_xpath, css, html, html_xpath, html_css]
28
+ # html_css and html_css_diff run in css-match.yml, with the selector
29
+ # dictionary and a longer budget.
30
+ target: [xml, xpath, xml_xpath, css, html, html_xpath]
29
31
 
30
32
  steps:
31
33
  - name: Checkout (with vendored Lexbor submodule)
data/CHANGELOG.md CHANGED
@@ -1,5 +1,73 @@
1
1
  # Changelog
2
2
 
3
+ ## [0.12.1] - 2026-10-02
4
+
5
+ ### Fixed
6
+
7
+ * Passing a string literal from a file without
8
+ `# frozen_string_literal: true` no longer warns "literal string will be
9
+ frozen in the future" on Ruby 3.4+ (with deprecation warnings enabled).
10
+ * A fragment parsed with a context element from another document
11
+ (`fragment(context:)`, `DocumentFragment.parse(context:)`) no longer keeps
12
+ a reference into that document, which could be read after that document was
13
+ freed.
14
+ * The 10,000-option limit per `<select>` now also counts the options a
15
+ fragment parsed in a select context receives (`inner_html=` and
16
+ `outer_html=` on or inside a select, `fragment(context: "select")`).
17
+ Before, they went uncounted, and inserting them took time quadratic in
18
+ their number.
19
+ * A String `context:` naming a node that is not an element (`"#text"`,
20
+ `"!--"`, ...) raises `ArgumentError`.
21
+
22
+ ### Security
23
+
24
+ * Hardening against Lexbor edge cases in fragment parsing, node copying,
25
+ element creation and serialization. Serialization is now always Makiri's
26
+ own walk, which uses no native recursion, so `to_html` / `inner_html` are
27
+ about 3% slower.
28
+
29
+ ## [0.12.0] - 2026-10-01
30
+
31
+ Most changes accept what the DOM allows where 0.11.0 raised. The XML
32
+ serializers (`to_xml`, `canonicalize`) raise instead when the tree holds
33
+ something XML cannot write.
34
+
35
+ ### Changed
36
+
37
+ HTML:
38
+
39
+ * `to_html` / `inner_html` write a `<template>`'s contents, not its own
40
+ children (added with `add_child`), as browsers do.
41
+ * `create_element_ns` in the HTML namespace accepts an upper-case name
42
+ (`BR`, `DIV`), as the DOM does. It makes an unknown element: not void, and
43
+ not matched by type selectors. Importing such an XHTML element from XML
44
+ behaves the same way.
45
+
46
+ XML:
47
+
48
+ * Text, attribute values and comments accept the characters the DOM allows:
49
+ `"\f"`, U+0001, NUL, and `--` in a comment.
50
+ * `set_attribute_ns` accepts names XML cannot write, such as `p:a}b`.
51
+ * `to_xml` and `canonicalize` raise while any of the above is in the tree.
52
+ * `create_cdata` with `]]>` and `create_processing_instruction` with `?>`
53
+ now raise `ArgumentError`.
54
+ * `canonicalize` adds the namespace declarations a name needs instead of
55
+ raising. It still raises where it would have to invent a prefix.
56
+
57
+ HTML to XML (`import_node`):
58
+
59
+ * The copy is in its namespace at once. Before, it was in none until
60
+ inserted.
61
+ * The copy gets no `xmlns` attribute.
62
+ * An attribute XML cannot write (`x-on:click`, `@click`) is copied as it is
63
+ instead of raising.
64
+ * A `<template>` keeps its own children, after its contents.
65
+
66
+ Other:
67
+
68
+ * `Node#path` round-trips when a sibling with a different prefix or case
69
+ would also match the same XPath step.
70
+
3
71
  ## [0.11.0] - 2026-09-30
4
72
 
5
73
  No code changes since 0.11.0.rc2. Coming from 0.10.x, read the rc2 and rc1
@@ -184,11 +184,11 @@ what browsers do - rather than libxml2. Detailed, test-backed notes live in
184
184
  that does not fit the name (`set_attribute_ns`, `create_element_ns`)
185
185
  raises `Makiri::Error`. Invalid UTF-8 raises `Makiri::Error` for every
186
186
  argument, names included.
187
- * `create_element_ns` refuses an HTML-namespace name in upper case that
188
- lower-cases to an element Lexbor knows (`BR`, `DIV`), where the DOM makes
189
- an unknown element: Lexbor would make that element (`BR` void, its
190
- children never written). Other names keep their case (`MY-EL`). Nokogiri
191
- has no `create_element_ns`.
187
+ * `create_element_ns` keeps the case of an HTML-namespace name, as the DOM
188
+ does: `create_element_ns(XHTML, "BR")` is an unknown element named `BR`,
189
+ not a void `br`, and type selectors do not match it (XPath name tests do,
190
+ folding case on HTML elements as browsers do). Nokogiri has no
191
+ `create_element_ns`.
192
192
  * An HTML `<template>` follows the WHATWG content model, which Nokogiri does
193
193
  not: its parsed contents live in the separate fragment `Element#content_fragment`
194
194
  returns, `template.children` is empty, and `inner_html` / `inner_html=`
@@ -198,6 +198,11 @@ what browsers do - rather than libxml2. Detailed, test-backed notes live in
198
198
  ordinary element, with the parsed nodes as its children, so
199
199
  `template.inner_html` and `template.children` answer the other way round and
200
200
  there is no `content_fragment`.
201
+ * `Makiri::XML` has no template contents: an XHTML `<template>`'s children
202
+ are its children. Crossing into HTML they become its contents, and back
203
+ they become children - what a browser's XML parser, which puts them in the
204
+ contents, and its importNode give for the same document. An XML
205
+ `<template>` whose children should stay children has no way to say so.
201
206
  * An HTML document has one root element and no text child, as the DOM requires;
202
207
  `doc << element` beside an existing root raises.
203
208
  * An insertion the DOM refuses - a child under a text, comment, PI, doctype
@@ -205,13 +210,20 @@ what browsers do - rather than libxml2. Detailed, test-backed notes live in
205
210
  `Makiri::Error` in both representations. Nokogiri refuses the same ones with
206
211
  `ArgumentError` (or `RuntimeError` for a second XML root).
207
212
  * Moving HTML into an XML document (`xml_doc.import_node(html_node)`, or
208
- inserting one) keeps every name's namespace, and refuses what XML cannot
209
- write that way. An attribute in no namespace whose name has a prefix other
210
- than `xml` - `v-on:click`, `fb:like`, an `xlink:href` on an HTML (not SVG)
211
- element - raises `Makiri::Error`: as XML it would be a prefix bound to
212
- nothing. Nokogiri copies it and writes `v-on:click="..."` into output that is
213
- not namespace-well-formed. An element named with a colon (`<fb:like>`)
214
- crosses as a DOM-loose name, which `to_xml` refuses.
213
+ inserting one) copies it as the DOM's clone does: every element in its
214
+ namespace from the start (an imported `<p>` is XHTML before it is
215
+ inserted), every attribute named as it is, and no `xmlns` attribute added -
216
+ `to_xml` and `canonicalize` write the declarations the output needs.
217
+ Nokogiri's copy of an HTML5 `<div>` is in no namespace, and libxml2
218
+ declares `xmlns:svg` on it for an SVG child (in `namespace_definitions`),
219
+ writing the child as `<svg:svg>`. One XML cannot write that way -
220
+ in no namespace with a colon or no XML name, `v-on:click`, `:href`,
221
+ `@click`, `fb:like` - crosses DOM-loose, and `to_xml` refuses the tree
222
+ while it is there; so does an element named with a colon (`<fb:like>`).
223
+ Nokogiri copies them and writes output that is not namespace-well-formed.
224
+ * One exception to the DOM: an HTML attribute named `xml:lang` (in no
225
+ namespace) becomes the XML namespace's `xml:lang`, as the XML reader reads
226
+ that name, so XHTML-style HTML still writes as XML.
215
227
  * A known gap, in Lexbor's tag table: an HTML document that already holds a
216
228
  parsed element named with a colon (`<x:y>`, one local name) and then
217
229
  receives, by `import_node` from another document, a prefixed element
@@ -284,7 +296,7 @@ what browsers do - rather than libxml2. Detailed, test-backed notes live in
284
296
  case-sensitivity rule, as browsers do: lower-cased for an HTML element (`LI`
285
297
  matches `<li>`), as written for any other (`feGaussianBlur` matches the SVG
286
298
  element, `fegaussianblur` does not). An HTML element named in upper case
287
- (`create_element_ns(XHTML, "MY-EL")`, which keeps its name as the DOM does)
299
+ (`create_element_ns(XHTML, "DIV")`, which keeps its name as the DOM does)
288
300
  therefore matches no type selector.
289
301
  * `Nokogiri::HTML5` is case-sensitive on HTML elements too, so `LI` does not
290
302
  match `<li>` there. `Makiri::XML`'s `#css` is case-sensitive, as XML names
@@ -321,5 +333,12 @@ what browsers do - rather than libxml2. Detailed, test-backed notes live in
321
333
  `Makiri::Error`.
322
334
  * On re-parse, the HTML tokenizer replaces a U+0000 in text/attributes with
323
335
  U+FFFD (WHATWG), so a serialized-then-reparsed round-trip is not byte-identical.
324
- * `Makiri::XML` rejects NUL everywhere: XML 1.0 has no legal U+0000 character,
325
- so admitting it would produce non-well-formed XML.
336
+ * `Makiri::XML` holds the character data the DOM holds, NUL included:
337
+ text or an attribute value with a character XML has no `Char` for (`\f`,
338
+ U+0001, U+0000), a comment with `--`. Nokogiri takes the same, but NUL
339
+ (`ArgumentError`, a Ruby C-string limit). Where they part is the output. `to_xml` / `canonicalize` raise for such a tree; Nokogiri writes
340
+ `\f` in text as U+FFFD (the data changes) and a comment's `--` or an
341
+ attribute value's `\f` as it stands (the output does not parse).
342
+ `create_cdata` refuses `]]>` and `create_processing_instruction` `?>`,
343
+ raising `ArgumentError` as the DOM's factories do; Nokogiri takes the PI
344
+ and writes `<?t a?>b?>`, which re-reads as another tree.
data/Rakefile CHANGED
@@ -857,7 +857,9 @@ namespace :fuzz do
857
857
  # library and headers (the crate's build.rs links the archive and generates
858
858
  # the layout from the headers), which `rake compile` produces - hence the
859
859
  # dependency, which is about Lexbor rather than about the bundle.
860
- FUZZ_TARGETS = %w[xml xpath xml_xpath css html html_xpath html_css].freeze
860
+ FUZZ_TARGETS = %w[xml xpath xml_xpath css html html_xpath html_css html_css_diff].freeze
861
+ # The targets whose input starts with a CSS selector take its dictionary.
862
+ FUZZ_CSS_DICT = %w[html_css html_css_diff].freeze
861
863
 
862
864
  # The local mode: TARGETS=css,html_xpath narrows the run to what you touched,
863
865
  # and the time is FUZZ_TIME seconds per target (default 60) or FUZZ_BUDGET
@@ -916,8 +918,9 @@ namespace :fuzz do
916
918
  env, sanitizer = libfuzzer_invocation
917
919
  Dir.chdir("ext/makiri/rust/fuzz") do
918
920
  targets.each do |target|
921
+ dict = FUZZ_CSS_DICT.include?(target) ? ["-dict=css.dict"] : []
919
922
  sh env, "cargo", "fuzz", "run", *sanitizer, target, "--",
920
- "-max_total_time=#{time}", "-max_len=4096"
923
+ "-max_total_time=#{time}", "-max_len=4096", *dict
921
924
  end
922
925
  end
923
926
  end
@@ -44,8 +44,9 @@ expect_used = "warn"
44
44
  panic = "warn"
45
45
 
46
46
  [features]
47
- # Three features, and each names a real optional dependency rather than a step
48
- # of the C port: a Ruby runtime, a vendored C library, and a test hook.
47
+ # Four features. Three name a real optional dependency rather than a step of
48
+ # the C port - a Ruby runtime, a vendored C library, and a test hook - and the
49
+ # fourth, `css-reference`, a test reference (below).
49
50
  #
50
51
  # The port's thirty-seven per-file flags are gone. They existed so one C file at
51
52
  # a time could be replaced behind an identical ABI; there is no C left to
@@ -68,6 +69,11 @@ lexbor = []
68
69
  # `falloc::should_fail` is a const false and the branch disappears.
69
70
  alloc-inject = []
70
71
 
72
+ # The OLD `lxb_selectors` engine (`lexbor::selectors`) outside `cfg(test)`, as
73
+ # the reference the `html_css_diff` fuzz harness holds `css_match` to. Never
74
+ # part of a build that ships.
75
+ css-reference = ["lexbor"]
76
+
71
77
  [dependencies]
72
78
  # Only the `ruby` feature needs these; the engine builds without them, which is
73
79
  # what lets Kani and cargo-fuzz compile it. `rb-sys` is magnus's own binding
@@ -262,6 +262,14 @@ fn main() {
262
262
  .allowlist_function("lxb_dom_attr_set_name")
263
263
  .allowlist_function("lxb_dom_document_create_element")
264
264
  .allowlist_function("lxb_dom_element_create")
265
+ // An HTML element named as written (`DIV`, not `div`): its case-kept
266
+ // tag is Lexbor's `lxb_tag_append` (UNDECLARED_EXPORTS), and the rest
267
+ // is `lxb_dom_element_create`'s own steps with that tag.
268
+ .allowlist_function("lxb_dom_document_create_interface_noi")
269
+ .allowlist_function("lxb_dom_document_destroy_interface_noi")
270
+ .allowlist_function("lxb_ns_prefix_append")
271
+ .allowlist_type("lxb_tag_data_t")
272
+ .allowlist_type("lxb_ns_prefix_data_t")
265
273
  .allowlist_function("lxb_dom_document_create_text_node")
266
274
  .allowlist_function("lxb_dom_document_create_comment")
267
275
  .allowlist_function("lxb_dom_document_create_processing_instruction")
@@ -279,7 +287,14 @@ fn main() {
279
287
  .allowlist_function("lxb_css_property_serialize")
280
288
  .allowlist_function("lxb_css_property_serialize_name")
281
289
  .allowlist_function("lxb_css_selector_serialize_chain")
282
- // The HTML serializers `lexbor::serialize` drives through its sink.
290
+ // One node's own markup, plain and pretty: Makiri drives the walk
291
+ // itself (`adapter::html::serialize`), so a <template> writes its
292
+ // contents and not its own children, as the HTML Standard says, and
293
+ // no tree depth grows the native stack.
294
+ .allowlist_function("lxb_html_serialize_cb")
295
+ .allowlist_function("lxb_html_serialize_pretty_cb")
296
+ // Lexbor's own walks, which those replaced: the reference the
297
+ // `serialize_walk` tests compare against, and called nowhere else.
283
298
  .allowlist_function("lxb_html_serialize_tree_cb")
284
299
  .allowlist_function("lxb_html_serialize_deep_cb")
285
300
  .allowlist_function("lxb_html_serialize_pretty_tree_cb")
@@ -382,6 +397,12 @@ const UNDECLARED_EXPORTS: &[(&str, &str, &str, &str)] = &[
382
397
  "lxb_dom_attr_qualified_name_append",
383
398
  "lexbor_hash_t *hash, const lxb_char_t *name, size_t length",
384
399
  ),
400
+ (
401
+ "lexbor/tag/tag.c",
402
+ "LXB_API const lxb_tag_data_t *",
403
+ "lxb_tag_append",
404
+ "lexbor_hash_t *hash, lxb_tag_id_t tag_id, const lxb_char_t *name, size_t length",
405
+ ),
385
406
  (
386
407
  "lexbor/dom/interfaces/element.c",
387
408
  "LXB_API lxb_status_t",
@@ -389,6 +410,12 @@ const UNDECLARED_EXPORTS: &[(&str, &str, &str, &str)] = &[
389
410
  "lxb_dom_element_t *element, const lxb_char_t *prefix, size_t prefix_len, \
390
411
  const lxb_char_t *lname, size_t lname_len",
391
412
  ),
413
+ (
414
+ "lexbor/ns/ns.c",
415
+ "LXB_API const lxb_ns_data_t *",
416
+ "lxb_ns_append",
417
+ "lexbor_hash_t *hash, const lxb_char_t *link, size_t length",
418
+ ),
392
419
  ];
393
420
 
394
421
  fn check_undeclared_exports(include: &std::path::Path) {
@@ -18,7 +18,9 @@ libfuzzer-sys = "0.4"
18
18
  [dependencies.makiri_rs]
19
19
  path = ".."
20
20
  default-features = false
21
- features = ["lexbor"]
21
+ # `css-reference` adds the OLD CSS engine, the reference `html_css_diff`
22
+ # compares against; it ships in no build of the extension.
23
+ features = ["lexbor", "css-reference"]
22
24
 
23
25
  [[bin]]
24
26
  name = "xml"
@@ -68,3 +70,10 @@ path = "fuzz_targets/html_css.rs"
68
70
  test = false
69
71
  doc = false
70
72
  bench = false
73
+
74
+ [[bin]]
75
+ name = "html_css_diff"
76
+ path = "fuzz_targets/html_css_diff.rs"
77
+ test = false
78
+ doc = false
79
+ bench = false
@@ -0,0 +1,77 @@
1
+ # Selector tokens for the HTML CSS targets (html_css, html_css_diff): the
2
+ # input is `selector NUL document`, and byte-level mutation alone rarely keeps
3
+ # a selector parseable long enough to reach the matcher's deep paths.
4
+ sep="\x00"
5
+ c_desc=" "
6
+ c_child=" > "
7
+ c_next=" + "
8
+ c_sub=" ~ "
9
+ c_col=" || "
10
+ comma=", "
11
+ star="*"
12
+ id="#a"
13
+ cls=".a"
14
+ cls2=".b"
15
+ t_div="div"
16
+ t_p="p"
17
+ t_li="li"
18
+ t_ul="ul"
19
+ t_a="a"
20
+ t_span="span"
21
+ t_input="input"
22
+ t_svg="svg"
23
+ t_ns="svg|circle"
24
+ a_has="[a]"
25
+ a_eq="[a=x]"
26
+ a_inc="[a~=x]"
27
+ a_dash="[a|=x]"
28
+ a_pre="[a^=x]"
29
+ a_suf="[a$=x]"
30
+ a_sub="[a*=x]"
31
+ a_i=" i]"
32
+ a_s=" s]"
33
+ p_first=":first-child"
34
+ p_last=":last-child"
35
+ p_only=":only-child"
36
+ p_fot=":first-of-type"
37
+ p_lot=":last-of-type"
38
+ p_oot=":only-of-type"
39
+ p_empty=":empty"
40
+ p_root=":root"
41
+ p_link=":link"
42
+ p_anylink=":any-link"
43
+ p_req=":required"
44
+ p_opt=":optional"
45
+ p_ro=":read-only"
46
+ p_rw=":read-write"
47
+ p_ph=":placeholder-shown"
48
+ p_hover=":hover"
49
+ p_focus=":focus"
50
+ p_checked=":checked"
51
+ p_disabled=":disabled"
52
+ p_enabled=":enabled"
53
+ f_is=":is("
54
+ f_where=":where("
55
+ f_not=":not("
56
+ f_has=":has("
57
+ f_has_child=":has(> "
58
+ f_has_next=":has(+ "
59
+ f_has_sub=":has(~ "
60
+ f_nth=":nth-child("
61
+ f_nthl=":nth-last-child("
62
+ f_nthot=":nth-of-type("
63
+ f_nthlot=":nth-last-of-type("
64
+ anb_odd="odd"
65
+ anb_even="even"
66
+ anb_1="2n+1"
67
+ anb_2="-n+3"
68
+ anb_of=" of "
69
+ close=")"
70
+ h_div="<div class=a id=a>"
71
+ h_p="<p a=x>"
72
+ h_li="<ul><li>"
73
+ h_a="<a href=x>"
74
+ h_input="<input required placeholder=x>"
75
+ h_svg="<svg><circle r=1></svg>"
76
+ h_close="</div>"
77
+ h_quirks="<!doctype html>"
@@ -62,23 +62,25 @@ fuzz_target!(|data: &[u8]| {
62
62
  // SAFETY: `p` owns the document and outlives every handle made here.
63
63
  let root = unsafe { p.raw_doc().as_doc() }.as_node();
64
64
  for _ in 0..2 {
65
- let checked = selector_cache::with_compiled(&gvl, text.as_bytes(), |groups, scratch| {
65
+ let answered = selector_cache::with_compiled(&gvl, text.as_bytes(), |groups, scratch| {
66
66
  check(root, groups, scratch)
67
67
  });
68
- if checked.is_err() {
68
+ // A refusal is the budget doing its job, and each refused query spends
69
+ // the whole of it: asking again (or the same query a second way) only
70
+ // multiplies the run, which under ASan went past libFuzzer's timeout.
71
+ if !matches!(answered, Ok(true)) {
69
72
  return;
70
73
  }
71
74
  }
72
75
  });
73
76
 
74
- /// The walking query and the one-candidate query agree (module doc).
75
- fn check(root: HtmlNode<'_>, groups: Lists<'_>, scratch: &mut Scratch) {
76
- let all = select_all(scratch, root, groups);
77
- let first = select_first(scratch, root, groups);
78
- let Ok(all) = all else {
79
- return;
77
+ /// The walking query and the one-candidate query agree (module doc). False
78
+ /// when a query refused, so the caller stops.
79
+ fn check(root: HtmlNode<'_>, groups: Lists<'_>, scratch: &mut Scratch) -> bool {
80
+ let Ok(all) = select_all(scratch, root, groups) else {
81
+ return false;
80
82
  };
81
- if let Ok(first) = first {
83
+ if let Ok(first) = select_first(scratch, root, groups) {
82
84
  assert!(
83
85
  first == all.first().copied(),
84
86
  "select_first is not the first of select_all"
@@ -90,7 +92,7 @@ fn check(root: HtmlNode<'_>, groups: Lists<'_>, scratch: &mut Scratch) {
90
92
  .take(ONE_BY_ONE)
91
93
  {
92
94
  let Ok(matched) = matches_any(scratch, groups, el) else {
93
- return;
95
+ return false;
94
96
  };
95
97
  assert_eq!(
96
98
  matched,
@@ -98,4 +100,5 @@ fn check(root: HtmlNode<'_>, groups: Lists<'_>, scratch: &mut Scratch) {
98
100
  "matches? and css disagree on an element"
99
101
  );
100
102
  }
103
+ true
101
104
  }