makiri 0.11.0.rc2-aarch64-linux → 0.12.1-aarch64-linux
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/.github/workflows/css-match.yml +180 -0
- data/.github/workflows/libfuzzer.yml +3 -1
- data/CHANGELOG.md +73 -0
- data/NOKOGIRI_DIFFERENCES.md +34 -15
- data/Rakefile +5 -2
- data/lib/makiri/3.2/makiri.so +0 -0
- data/lib/makiri/3.3/makiri.so +0 -0
- data/lib/makiri/3.4/makiri.so +0 -0
- data/lib/makiri/4.0/makiri.so +0 -0
- data/lib/makiri/element.rb +16 -0
- data/lib/makiri/node_path.rb +8 -1
- data/lib/makiri/version.rb +1 -1
- data/script/check_unsafe_boundaries.rb +8 -7
- metadata +2 -1
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
SHA256:
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
3
|
+
metadata.gz: dad06ebaa7e4fa5fec68080e1714a5be6b72dc89172ff5f1f7840d6e73f4cffe
|
|
4
|
+
data.tar.gz: fe535f36f6d2f28acd4b3bd6c40f2d2f5c1a288affe5dc94ddd22d520c4e8ae7
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: 444bfeb89cb88fe1e3bed19008401e0c8f09043e803722c4dac63c88063a53eaa8a9e097b08f5b11c25f83a76e4d5f7630d99846c4f1c9131feafb72351a17aa
|
|
7
|
+
data.tar.gz: c296890d6e35f332fb975542d2eebe5dd7b4283cbc3012715c43a37917aa82d40492136fe045ab44362e26572b09b3e1f9dc52b9f059d40d1fa93f3f25d5e11c
|
|
@@ -0,0 +1,180 @@
|
|
|
1
|
+
name: CSS matcher
|
|
2
|
+
|
|
3
|
+
# The HTML CSS matcher (`lexbor::css_match`) on its own: it answers every
|
|
4
|
+
# `Node#css` / `#at_css` / `#matches?`, and its failure mode is not a crash
|
|
5
|
+
# but a wrong answer, which the general suites only catch by example. Two
|
|
6
|
+
# checks, both against the OLD engine - Lexbor's own `lxb_selectors`, kept as
|
|
7
|
+
# `lexbor::selectors` for exactly this:
|
|
8
|
+
#
|
|
9
|
+
# - the randomized differential tests in `lexbor::tests::css_match`, run long
|
|
10
|
+
# with a fresh seed (MAKIRI_CSS_DIFF_SEED; a failure prints it);
|
|
11
|
+
# - the cargo-fuzz targets `html_css` (memory safety, and the three entry
|
|
12
|
+
# points agreeing) and `html_css_diff` (agreement with Lexbor's engine),
|
|
13
|
+
# with the selector dictionary.
|
|
14
|
+
#
|
|
15
|
+
# A PR that touches the matcher, the Lexbor layer or the pin gets a short run;
|
|
16
|
+
# the nightly one is long. A disagreement on a shape where css_match departs
|
|
17
|
+
# from Lexbor on purpose is not a finding: the departures are listed in
|
|
18
|
+
# `css_match`'s module doc, and both checks leave them out.
|
|
19
|
+
|
|
20
|
+
on:
|
|
21
|
+
push:
|
|
22
|
+
branches: [main, master]
|
|
23
|
+
paths:
|
|
24
|
+
- "ext/makiri/rust/src/lexbor/**"
|
|
25
|
+
- "ext/makiri/rust/fuzz/**"
|
|
26
|
+
- "ext/makiri/rust/Cargo.toml"
|
|
27
|
+
- "vendor/lexbor"
|
|
28
|
+
- ".github/workflows/css-match.yml"
|
|
29
|
+
pull_request:
|
|
30
|
+
paths:
|
|
31
|
+
- "ext/makiri/rust/src/lexbor/**"
|
|
32
|
+
- "ext/makiri/rust/fuzz/**"
|
|
33
|
+
- "ext/makiri/rust/Cargo.toml"
|
|
34
|
+
- "vendor/lexbor"
|
|
35
|
+
- ".github/workflows/css-match.yml"
|
|
36
|
+
schedule:
|
|
37
|
+
- cron: "41 4 * * *"
|
|
38
|
+
workflow_dispatch:
|
|
39
|
+
|
|
40
|
+
env:
|
|
41
|
+
# Short on a push or PR, long on the nightly schedule.
|
|
42
|
+
LONG: ${{ github.event_name == 'schedule' || github.event_name == 'workflow_dispatch' }}
|
|
43
|
+
|
|
44
|
+
jobs:
|
|
45
|
+
differential:
|
|
46
|
+
name: Randomized differential (vs Lexbor's engine)
|
|
47
|
+
runs-on: ubuntu-latest
|
|
48
|
+
timeout-minutes: 60
|
|
49
|
+
steps:
|
|
50
|
+
- name: Checkout (with vendored Lexbor submodule)
|
|
51
|
+
uses: actions/checkout@v6
|
|
52
|
+
with:
|
|
53
|
+
submodules: recursive
|
|
54
|
+
|
|
55
|
+
- name: Ensure cmake is available
|
|
56
|
+
uses: lukka/get-cmake@latest
|
|
57
|
+
|
|
58
|
+
- uses: dtolnay/rust-toolchain@stable
|
|
59
|
+
|
|
60
|
+
- name: Install libclang (bindgen, via rb-sys)
|
|
61
|
+
run: sudo apt-get update && sudo apt-get install -y libclang-dev
|
|
62
|
+
|
|
63
|
+
- name: Set up Ruby
|
|
64
|
+
uses: ruby/setup-ruby@v1
|
|
65
|
+
with:
|
|
66
|
+
ruby-version: "3.4"
|
|
67
|
+
bundler-cache: true
|
|
68
|
+
|
|
69
|
+
# build.rs links the vendored Lexbor archive and reads its headers.
|
|
70
|
+
- name: Compile the extension (builds Lexbor)
|
|
71
|
+
run: bundle exec rake compile
|
|
72
|
+
|
|
73
|
+
# Release: the sweep is the point, and debug is ~10x slower. The seed is
|
|
74
|
+
# fresh per run, so a long series covers new ground; it is printed here
|
|
75
|
+
# and in any failure, so a finding reproduces.
|
|
76
|
+
- name: Run the css_match tests with a fresh seed
|
|
77
|
+
working-directory: ext/makiri/rust
|
|
78
|
+
run: |
|
|
79
|
+
export MAKIRI_CSS_DIFF_SEED=$(printf '0x%x' $(( (GITHUB_RUN_ID * 2654435761 + GITHUB_RUN_ATTEMPT) & 0xffffffffffff )))
|
|
80
|
+
if [ "$LONG" = true ]; then
|
|
81
|
+
export MAKIRI_CSS_DIFF_ITERATIONS=1000000
|
|
82
|
+
else
|
|
83
|
+
export MAKIRI_CSS_DIFF_ITERATIONS=20000
|
|
84
|
+
fi
|
|
85
|
+
echo "MAKIRI_CSS_DIFF_SEED=$MAKIRI_CSS_DIFF_SEED MAKIRI_CSS_DIFF_ITERATIONS=$MAKIRI_CSS_DIFF_ITERATIONS"
|
|
86
|
+
cargo test --release --no-default-features --features lexbor -- \
|
|
87
|
+
lexbor::tests::css_match lexbor::tests::selector_cache
|
|
88
|
+
|
|
89
|
+
fuzz:
|
|
90
|
+
name: cargo-fuzz ${{ matrix.target }}
|
|
91
|
+
runs-on: ubuntu-latest
|
|
92
|
+
timeout-minutes: 120
|
|
93
|
+
strategy:
|
|
94
|
+
fail-fast: false
|
|
95
|
+
matrix:
|
|
96
|
+
target: [html_css, html_css_diff]
|
|
97
|
+
steps:
|
|
98
|
+
- name: Checkout (with vendored Lexbor submodule)
|
|
99
|
+
uses: actions/checkout@v6
|
|
100
|
+
with:
|
|
101
|
+
submodules: recursive
|
|
102
|
+
|
|
103
|
+
- name: Ensure cmake is available
|
|
104
|
+
uses: lukka/get-cmake@latest
|
|
105
|
+
|
|
106
|
+
# -Zsanitizer is unstable, and cargo-fuzz needs it.
|
|
107
|
+
- uses: dtolnay/rust-toolchain@nightly
|
|
108
|
+
|
|
109
|
+
- name: Install libclang (bindgen, via rb-sys)
|
|
110
|
+
run: sudo apt-get update && sudo apt-get install -y libclang-dev
|
|
111
|
+
|
|
112
|
+
- name: Set up Ruby
|
|
113
|
+
uses: ruby/setup-ruby@v1
|
|
114
|
+
with:
|
|
115
|
+
ruby-version: "3.4"
|
|
116
|
+
bundler-cache: true
|
|
117
|
+
|
|
118
|
+
- name: Compile the extension (builds Lexbor)
|
|
119
|
+
run: bundle exec rake compile
|
|
120
|
+
|
|
121
|
+
- name: Install cargo-fuzz
|
|
122
|
+
run: cargo install cargo-fuzz --locked
|
|
123
|
+
|
|
124
|
+
- name: Build the fuzz target
|
|
125
|
+
working-directory: ext/makiri/rust/fuzz
|
|
126
|
+
run: cargo fuzz build ${{ matrix.target }}
|
|
127
|
+
|
|
128
|
+
# As in libfuzzer.yml: nm -u reads 0 on Linux whether or not the binary
|
|
129
|
+
# is instrumented, so assert on the full symbol table.
|
|
130
|
+
- name: Assert the target is instrumented
|
|
131
|
+
working-directory: ext/makiri/rust/fuzz
|
|
132
|
+
run: |
|
|
133
|
+
bin=$(ls target/*/release/${{ matrix.target }})
|
|
134
|
+
reports=$(nm "$bin" | grep -c '__asan_report')
|
|
135
|
+
echo "$bin: __asan_report symbols = $reports"
|
|
136
|
+
test "$reports" -gt 0 || { echo "not ASan-instrumented"; exit 1; }
|
|
137
|
+
|
|
138
|
+
# Both targets read the same `selector NUL document` input, so each
|
|
139
|
+
# also starts from the other's corpus (read-only: new units go to the
|
|
140
|
+
# first directory). The cache prefix is libfuzzer.yml's, so the corpus
|
|
141
|
+
# accumulated there carries over.
|
|
142
|
+
- name: Restore the corpus
|
|
143
|
+
uses: actions/cache@v6
|
|
144
|
+
with:
|
|
145
|
+
path: ext/makiri/rust/fuzz/corpus/${{ matrix.target }}
|
|
146
|
+
key: cargo-fuzz-corpus-${{ matrix.target }}-${{ github.run_id }}
|
|
147
|
+
restore-keys: |
|
|
148
|
+
cargo-fuzz-corpus-${{ matrix.target }}-
|
|
149
|
+
|
|
150
|
+
- name: Restore the sibling corpus
|
|
151
|
+
uses: actions/cache/restore@v6
|
|
152
|
+
with:
|
|
153
|
+
path: ext/makiri/rust/fuzz/corpus/${{ matrix.target == 'html_css' && 'html_css_diff' || 'html_css' }}
|
|
154
|
+
key: cargo-fuzz-corpus-${{ matrix.target == 'html_css' && 'html_css_diff' || 'html_css' }}-${{ github.run_id }}
|
|
155
|
+
restore-keys: |
|
|
156
|
+
cargo-fuzz-corpus-${{ matrix.target == 'html_css' && 'html_css_diff' || 'html_css' }}-
|
|
157
|
+
|
|
158
|
+
# -timeout: html_css_diff runs Lexbor's exhaustively backtracking engine
|
|
159
|
+
# too; the harness skips inputs it cannot afford, and this is the stop
|
|
160
|
+
# for one it misjudged.
|
|
161
|
+
- name: Run cargo-fuzz ${{ matrix.target }}
|
|
162
|
+
working-directory: ext/makiri/rust/fuzz
|
|
163
|
+
run: |
|
|
164
|
+
sibling=${{ matrix.target == 'html_css' && 'html_css_diff' || 'html_css' }}
|
|
165
|
+
mkdir -p corpus/${{ matrix.target }} corpus/$sibling
|
|
166
|
+
if [ "$LONG" = true ]; then seconds=3600; else seconds=180; fi
|
|
167
|
+
cargo fuzz run ${{ matrix.target }} corpus/${{ matrix.target }} corpus/$sibling -- \
|
|
168
|
+
-max_total_time=$seconds \
|
|
169
|
+
-max_len=4096 \
|
|
170
|
+
-timeout=30 \
|
|
171
|
+
-dict=css.dict \
|
|
172
|
+
-print_final_stats=1
|
|
173
|
+
|
|
174
|
+
- name: Upload reproducers
|
|
175
|
+
if: failure()
|
|
176
|
+
uses: actions/upload-artifact@v4
|
|
177
|
+
with:
|
|
178
|
+
name: cargo-fuzz-artifacts-${{ matrix.target }}
|
|
179
|
+
path: ext/makiri/rust/fuzz/artifacts/${{ matrix.target }}
|
|
180
|
+
retention-days: 30
|
|
@@ -25,7 +25,9 @@ jobs:
|
|
|
25
25
|
strategy:
|
|
26
26
|
fail-fast: false
|
|
27
27
|
matrix:
|
|
28
|
-
|
|
28
|
+
# html_css and html_css_diff run in css-match.yml, with the selector
|
|
29
|
+
# dictionary and a longer budget.
|
|
30
|
+
target: [xml, xpath, xml_xpath, css, html, html_xpath]
|
|
29
31
|
|
|
30
32
|
steps:
|
|
31
33
|
- name: Checkout (with vendored Lexbor submodule)
|
data/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,78 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## [0.12.1] - 2026-10-02
|
|
4
|
+
|
|
5
|
+
### Fixed
|
|
6
|
+
|
|
7
|
+
* Passing a string literal from a file without
|
|
8
|
+
`# frozen_string_literal: true` no longer warns "literal string will be
|
|
9
|
+
frozen in the future" on Ruby 3.4+ (with deprecation warnings enabled).
|
|
10
|
+
* A fragment parsed with a context element from another document
|
|
11
|
+
(`fragment(context:)`, `DocumentFragment.parse(context:)`) no longer keeps
|
|
12
|
+
a reference into that document, which could be read after that document was
|
|
13
|
+
freed.
|
|
14
|
+
* The 10,000-option limit per `<select>` now also counts the options a
|
|
15
|
+
fragment parsed in a select context receives (`inner_html=` and
|
|
16
|
+
`outer_html=` on or inside a select, `fragment(context: "select")`).
|
|
17
|
+
Before, they went uncounted, and inserting them took time quadratic in
|
|
18
|
+
their number.
|
|
19
|
+
* A String `context:` naming a node that is not an element (`"#text"`,
|
|
20
|
+
`"!--"`, ...) raises `ArgumentError`.
|
|
21
|
+
|
|
22
|
+
### Security
|
|
23
|
+
|
|
24
|
+
* Hardening against Lexbor edge cases in fragment parsing, node copying,
|
|
25
|
+
element creation and serialization. Serialization is now always Makiri's
|
|
26
|
+
own walk, which uses no native recursion, so `to_html` / `inner_html` are
|
|
27
|
+
about 3% slower.
|
|
28
|
+
|
|
29
|
+
## [0.12.0] - 2026-10-01
|
|
30
|
+
|
|
31
|
+
Most changes accept what the DOM allows where 0.11.0 raised. The XML
|
|
32
|
+
serializers (`to_xml`, `canonicalize`) raise instead when the tree holds
|
|
33
|
+
something XML cannot write.
|
|
34
|
+
|
|
35
|
+
### Changed
|
|
36
|
+
|
|
37
|
+
HTML:
|
|
38
|
+
|
|
39
|
+
* `to_html` / `inner_html` write a `<template>`'s contents, not its own
|
|
40
|
+
children (added with `add_child`), as browsers do.
|
|
41
|
+
* `create_element_ns` in the HTML namespace accepts an upper-case name
|
|
42
|
+
(`BR`, `DIV`), as the DOM does. It makes an unknown element: not void, and
|
|
43
|
+
not matched by type selectors. Importing such an XHTML element from XML
|
|
44
|
+
behaves the same way.
|
|
45
|
+
|
|
46
|
+
XML:
|
|
47
|
+
|
|
48
|
+
* Text, attribute values and comments accept the characters the DOM allows:
|
|
49
|
+
`"\f"`, U+0001, NUL, and `--` in a comment.
|
|
50
|
+
* `set_attribute_ns` accepts names XML cannot write, such as `p:a}b`.
|
|
51
|
+
* `to_xml` and `canonicalize` raise while any of the above is in the tree.
|
|
52
|
+
* `create_cdata` with `]]>` and `create_processing_instruction` with `?>`
|
|
53
|
+
now raise `ArgumentError`.
|
|
54
|
+
* `canonicalize` adds the namespace declarations a name needs instead of
|
|
55
|
+
raising. It still raises where it would have to invent a prefix.
|
|
56
|
+
|
|
57
|
+
HTML to XML (`import_node`):
|
|
58
|
+
|
|
59
|
+
* The copy is in its namespace at once. Before, it was in none until
|
|
60
|
+
inserted.
|
|
61
|
+
* The copy gets no `xmlns` attribute.
|
|
62
|
+
* An attribute XML cannot write (`x-on:click`, `@click`) is copied as it is
|
|
63
|
+
instead of raising.
|
|
64
|
+
* A `<template>` keeps its own children, after its contents.
|
|
65
|
+
|
|
66
|
+
Other:
|
|
67
|
+
|
|
68
|
+
* `Node#path` round-trips when a sibling with a different prefix or case
|
|
69
|
+
would also match the same XPath step.
|
|
70
|
+
|
|
71
|
+
## [0.11.0] - 2026-09-30
|
|
72
|
+
|
|
73
|
+
No code changes since 0.11.0.rc2. Coming from 0.10.x, read the rc2 and rc1
|
|
74
|
+
notes below as well.
|
|
75
|
+
|
|
3
76
|
## [0.11.0.rc2] - 2026-09-30
|
|
4
77
|
|
|
5
78
|
### Added
|
data/NOKOGIRI_DIFFERENCES.md
CHANGED
|
@@ -184,11 +184,11 @@ what browsers do - rather than libxml2. Detailed, test-backed notes live in
|
|
|
184
184
|
that does not fit the name (`set_attribute_ns`, `create_element_ns`)
|
|
185
185
|
raises `Makiri::Error`. Invalid UTF-8 raises `Makiri::Error` for every
|
|
186
186
|
argument, names included.
|
|
187
|
-
* `create_element_ns`
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
187
|
+
* `create_element_ns` keeps the case of an HTML-namespace name, as the DOM
|
|
188
|
+
does: `create_element_ns(XHTML, "BR")` is an unknown element named `BR`,
|
|
189
|
+
not a void `br`, and type selectors do not match it (XPath name tests do,
|
|
190
|
+
folding case on HTML elements as browsers do). Nokogiri has no
|
|
191
|
+
`create_element_ns`.
|
|
192
192
|
* An HTML `<template>` follows the WHATWG content model, which Nokogiri does
|
|
193
193
|
not: its parsed contents live in the separate fragment `Element#content_fragment`
|
|
194
194
|
returns, `template.children` is empty, and `inner_html` / `inner_html=`
|
|
@@ -198,6 +198,11 @@ what browsers do - rather than libxml2. Detailed, test-backed notes live in
|
|
|
198
198
|
ordinary element, with the parsed nodes as its children, so
|
|
199
199
|
`template.inner_html` and `template.children` answer the other way round and
|
|
200
200
|
there is no `content_fragment`.
|
|
201
|
+
* `Makiri::XML` has no template contents: an XHTML `<template>`'s children
|
|
202
|
+
are its children. Crossing into HTML they become its contents, and back
|
|
203
|
+
they become children - what a browser's XML parser, which puts them in the
|
|
204
|
+
contents, and its importNode give for the same document. An XML
|
|
205
|
+
`<template>` whose children should stay children has no way to say so.
|
|
201
206
|
* An HTML document has one root element and no text child, as the DOM requires;
|
|
202
207
|
`doc << element` beside an existing root raises.
|
|
203
208
|
* An insertion the DOM refuses - a child under a text, comment, PI, doctype
|
|
@@ -205,13 +210,20 @@ what browsers do - rather than libxml2. Detailed, test-backed notes live in
|
|
|
205
210
|
`Makiri::Error` in both representations. Nokogiri refuses the same ones with
|
|
206
211
|
`ArgumentError` (or `RuntimeError` for a second XML root).
|
|
207
212
|
* Moving HTML into an XML document (`xml_doc.import_node(html_node)`, or
|
|
208
|
-
inserting one)
|
|
209
|
-
|
|
210
|
-
|
|
211
|
-
|
|
212
|
-
|
|
213
|
-
|
|
214
|
-
|
|
213
|
+
inserting one) copies it as the DOM's clone does: every element in its
|
|
214
|
+
namespace from the start (an imported `<p>` is XHTML before it is
|
|
215
|
+
inserted), every attribute named as it is, and no `xmlns` attribute added -
|
|
216
|
+
`to_xml` and `canonicalize` write the declarations the output needs.
|
|
217
|
+
Nokogiri's copy of an HTML5 `<div>` is in no namespace, and libxml2
|
|
218
|
+
declares `xmlns:svg` on it for an SVG child (in `namespace_definitions`),
|
|
219
|
+
writing the child as `<svg:svg>`. One XML cannot write that way -
|
|
220
|
+
in no namespace with a colon or no XML name, `v-on:click`, `:href`,
|
|
221
|
+
`@click`, `fb:like` - crosses DOM-loose, and `to_xml` refuses the tree
|
|
222
|
+
while it is there; so does an element named with a colon (`<fb:like>`).
|
|
223
|
+
Nokogiri copies them and writes output that is not namespace-well-formed.
|
|
224
|
+
* One exception to the DOM: an HTML attribute named `xml:lang` (in no
|
|
225
|
+
namespace) becomes the XML namespace's `xml:lang`, as the XML reader reads
|
|
226
|
+
that name, so XHTML-style HTML still writes as XML.
|
|
215
227
|
* A known gap, in Lexbor's tag table: an HTML document that already holds a
|
|
216
228
|
parsed element named with a colon (`<x:y>`, one local name) and then
|
|
217
229
|
receives, by `import_node` from another document, a prefixed element
|
|
@@ -284,7 +296,7 @@ what browsers do - rather than libxml2. Detailed, test-backed notes live in
|
|
|
284
296
|
case-sensitivity rule, as browsers do: lower-cased for an HTML element (`LI`
|
|
285
297
|
matches `<li>`), as written for any other (`feGaussianBlur` matches the SVG
|
|
286
298
|
element, `fegaussianblur` does not). An HTML element named in upper case
|
|
287
|
-
(`create_element_ns(XHTML, "
|
|
299
|
+
(`create_element_ns(XHTML, "DIV")`, which keeps its name as the DOM does)
|
|
288
300
|
therefore matches no type selector.
|
|
289
301
|
* `Nokogiri::HTML5` is case-sensitive on HTML elements too, so `LI` does not
|
|
290
302
|
match `<li>` there. `Makiri::XML`'s `#css` is case-sensitive, as XML names
|
|
@@ -321,5 +333,12 @@ what browsers do - rather than libxml2. Detailed, test-backed notes live in
|
|
|
321
333
|
`Makiri::Error`.
|
|
322
334
|
* On re-parse, the HTML tokenizer replaces a U+0000 in text/attributes with
|
|
323
335
|
U+FFFD (WHATWG), so a serialized-then-reparsed round-trip is not byte-identical.
|
|
324
|
-
* `Makiri::XML`
|
|
325
|
-
|
|
336
|
+
* `Makiri::XML` holds the character data the DOM holds, NUL included:
|
|
337
|
+
text or an attribute value with a character XML has no `Char` for (`\f`,
|
|
338
|
+
U+0001, U+0000), a comment with `--`. Nokogiri takes the same, but NUL
|
|
339
|
+
(`ArgumentError`, a Ruby C-string limit). Where they part is the output. `to_xml` / `canonicalize` raise for such a tree; Nokogiri writes
|
|
340
|
+
`\f` in text as U+FFFD (the data changes) and a comment's `--` or an
|
|
341
|
+
attribute value's `\f` as it stands (the output does not parse).
|
|
342
|
+
`create_cdata` refuses `]]>` and `create_processing_instruction` `?>`,
|
|
343
|
+
raising `ArgumentError` as the DOM's factories do; Nokogiri takes the PI
|
|
344
|
+
and writes `<?t a?>b?>`, which re-reads as another tree.
|
data/Rakefile
CHANGED
|
@@ -857,7 +857,9 @@ namespace :fuzz do
|
|
|
857
857
|
# library and headers (the crate's build.rs links the archive and generates
|
|
858
858
|
# the layout from the headers), which `rake compile` produces - hence the
|
|
859
859
|
# dependency, which is about Lexbor rather than about the bundle.
|
|
860
|
-
FUZZ_TARGETS = %w[xml xpath xml_xpath css html html_xpath html_css].freeze
|
|
860
|
+
FUZZ_TARGETS = %w[xml xpath xml_xpath css html html_xpath html_css html_css_diff].freeze
|
|
861
|
+
# The targets whose input starts with a CSS selector take its dictionary.
|
|
862
|
+
FUZZ_CSS_DICT = %w[html_css html_css_diff].freeze
|
|
861
863
|
|
|
862
864
|
# The local mode: TARGETS=css,html_xpath narrows the run to what you touched,
|
|
863
865
|
# and the time is FUZZ_TIME seconds per target (default 60) or FUZZ_BUDGET
|
|
@@ -916,8 +918,9 @@ namespace :fuzz do
|
|
|
916
918
|
env, sanitizer = libfuzzer_invocation
|
|
917
919
|
Dir.chdir("ext/makiri/rust/fuzz") do
|
|
918
920
|
targets.each do |target|
|
|
921
|
+
dict = FUZZ_CSS_DICT.include?(target) ? ["-dict=css.dict"] : []
|
|
919
922
|
sh env, "cargo", "fuzz", "run", *sanitizer, target, "--",
|
|
920
|
-
"-max_total_time=#{time}", "-max_len=4096"
|
|
923
|
+
"-max_total_time=#{time}", "-max_len=4096", *dict
|
|
921
924
|
end
|
|
922
925
|
end
|
|
923
926
|
end
|
data/lib/makiri/3.2/makiri.so
CHANGED
|
Binary file
|
data/lib/makiri/3.3/makiri.so
CHANGED
|
Binary file
|
data/lib/makiri/3.4/makiri.so
CHANGED
|
Binary file
|
data/lib/makiri/4.0/makiri.so
CHANGED
|
Binary file
|
data/lib/makiri/element.rb
CHANGED
|
@@ -22,5 +22,21 @@ module Makiri
|
|
|
22
22
|
def path_step
|
|
23
23
|
name_step("", unprefixed_element_namespace)
|
|
24
24
|
end
|
|
25
|
+
|
|
26
|
+
# See {NodePath#path}. An element step selects by expanded name, not by the
|
|
27
|
+
# string of the test: `*[local-name()='BR' and ...]`, written for `h:BR`,
|
|
28
|
+
# also selects the unprefixed `BR` beside it. A bare name test on an HTML
|
|
29
|
+
# element folds ASCII case, as the engine (and browsers) match it, so `div`
|
|
30
|
+
# also selects the `DIV` createElementNS makes. Compared as strings, each
|
|
31
|
+
# pair was counted apart, and the path found the other one first.
|
|
32
|
+
def path_selects?(other, test)
|
|
33
|
+
return false unless other.element? && other.namespace_uri.to_s == namespace_uri.to_s
|
|
34
|
+
|
|
35
|
+
mine = local_name
|
|
36
|
+
theirs = other.local_name
|
|
37
|
+
return mine == theirs unless test == name && unprefixed_element_namespace
|
|
38
|
+
|
|
39
|
+
mine.downcase(:ascii) == theirs.downcase(:ascii)
|
|
40
|
+
end
|
|
25
41
|
end
|
|
26
42
|
end
|
data/lib/makiri/node_path.rb
CHANGED
|
@@ -48,7 +48,7 @@ module Makiri
|
|
|
48
48
|
parent_node = parent
|
|
49
49
|
return test unless parent_node # a detached top: #path answers "?" anyway
|
|
50
50
|
|
|
51
|
-
siblings = parent_node.children.select { |c| c
|
|
51
|
+
siblings = parent_node.children.select { |c| path_selects?(c, test) }
|
|
52
52
|
return test if siblings.length <= 1
|
|
53
53
|
|
|
54
54
|
"#{test}[#{siblings.index(self) + 1}]"
|
|
@@ -73,6 +73,13 @@ module Makiri
|
|
|
73
73
|
nil
|
|
74
74
|
end
|
|
75
75
|
|
|
76
|
+
# Whether +test+, this node's own step, selects +other+, a sibling: the
|
|
77
|
+
# position counts exactly those. By default the tests are compared, which
|
|
78
|
+
# is the engine's answer for every kind but an element (see Element).
|
|
79
|
+
def path_selects?(other, test)
|
|
80
|
+
other.path_node_test == test
|
|
81
|
+
end
|
|
82
|
+
|
|
76
83
|
# The name test for an element (+axis+ "") or attribute (+axis+ "@"): the
|
|
77
84
|
# bare name when an unprefixed test reaches this node - it is in
|
|
78
85
|
# +plain_namespace+ and its name is a plain NCName - else by expanded name.
|
data/lib/makiri/version.rb
CHANGED
|
@@ -31,14 +31,14 @@ RUST = File.join(ROOT, "ext/makiri/rust/src")
|
|
|
31
31
|
UNSAFE_ISLANDS = {
|
|
32
32
|
"bridge/alloc.rs" => 4,
|
|
33
33
|
"bridge/doc.rs" => 3,
|
|
34
|
-
"bridge/fragment.rs" =>
|
|
34
|
+
"bridge/fragment.rs" => 6,
|
|
35
35
|
"bridge/gvl.rs" => 5,
|
|
36
36
|
"bridge/html.rs" => 17,
|
|
37
37
|
"bridge/node_set.rs" => 8,
|
|
38
38
|
"bridge/node_wrap.rs" => 1,
|
|
39
39
|
"bridge/ruby.rs" => 18,
|
|
40
40
|
"bridge/stack.rs" => 1,
|
|
41
|
-
"bridge/string.rs" =>
|
|
41
|
+
"bridge/string.rs" => 29,
|
|
42
42
|
"bridge/typed.rs" => 21,
|
|
43
43
|
"bridge/wrapper.rs" => 19,
|
|
44
44
|
"bridge/xml.rs" => 11,
|
|
@@ -56,13 +56,14 @@ UNSAFE_ISLANDS = {
|
|
|
56
56
|
"lexbor/adapter/arena_bytes.rs" => 14,
|
|
57
57
|
"lexbor/adapter/cross_import.rs" => 2,
|
|
58
58
|
"lexbor/adapter/html/attrs.rs" => 20,
|
|
59
|
-
"lexbor/adapter/html/build.rs" =>
|
|
60
|
-
"lexbor/adapter/html/mod.rs" =>
|
|
59
|
+
"lexbor/adapter/html/build.rs" => 24,
|
|
60
|
+
"lexbor/adapter/html/mod.rs" => 62,
|
|
61
61
|
"lexbor/adapter/html/mutate.rs" => 9,
|
|
62
|
+
"lexbor/adapter/html/serialize.rs" => 18,
|
|
62
63
|
"lexbor/adapter/post_parse.rs" => 13,
|
|
63
64
|
"lexbor/adapter/source_loc.rs" => 2,
|
|
64
65
|
"lexbor/adapter/text_index.rs" => 1,
|
|
65
|
-
"lexbor/adapter/tree_guard.rs" =>
|
|
66
|
+
"lexbor/adapter/tree_guard.rs" => 9,
|
|
66
67
|
"lexbor/chunks.rs" => 2,
|
|
67
68
|
"lexbor/css_engine.rs" => 14,
|
|
68
69
|
"lexbor/css_parser.rs" => 21,
|
|
@@ -73,7 +74,7 @@ UNSAFE_ISLANDS = {
|
|
|
73
74
|
"lexbor/selectors.rs" => 9,
|
|
74
75
|
"lexbor/serialize.rs" => 2,
|
|
75
76
|
"lexbor/stylesheet.rs" => 7,
|
|
76
|
-
"lexbor/tests.rs" =>
|
|
77
|
+
"lexbor/tests.rs" => 16,
|
|
77
78
|
"lexbor/xpath.rs" => 9,
|
|
78
79
|
"token.rs" => 1,
|
|
79
80
|
}.freeze
|
|
@@ -461,7 +462,7 @@ end
|
|
|
461
462
|
# is generated, so its signature is Lexbor's. A new hand declaration of a
|
|
462
463
|
# header-declared function belongs in build.rs's allowlist instead; one with no
|
|
463
464
|
# header belongs in build.rs's UNDECLARED_EXPORTS as well as here.
|
|
464
|
-
LEXBOR_HAND_DECLS =
|
|
465
|
+
LEXBOR_HAND_DECLS = 5
|
|
465
466
|
abi_decls = comments_removed(File.binread(File.join(RUST, "lexbor/abi.rs"))).scan(LEXBOR_DECL).length
|
|
466
467
|
if abi_decls != LEXBOR_HAND_DECLS
|
|
467
468
|
errors << "lexbor/abi.rs hand-declares #{abi_decls} Lexbor functions (expected " \
|
metadata
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
--- !ruby/object:Gem::Specification
|
|
2
2
|
name: makiri
|
|
3
3
|
version: !ruby/object:Gem::Version
|
|
4
|
-
version: 0.
|
|
4
|
+
version: 0.12.1
|
|
5
5
|
platform: aarch64-linux
|
|
6
6
|
authors:
|
|
7
7
|
- takahashim
|
|
@@ -64,6 +64,7 @@ extra_rdoc_files: []
|
|
|
64
64
|
files:
|
|
65
65
|
- ".github/workflows/ci.yml"
|
|
66
66
|
- ".github/workflows/conformance.yml"
|
|
67
|
+
- ".github/workflows/css-match.yml"
|
|
67
68
|
- ".github/workflows/libfuzzer.yml"
|
|
68
69
|
- ".github/workflows/release.yml"
|
|
69
70
|
- ".github/workflows/security.yml"
|