testaro 78.0.8 → 78.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.gitattributes +94 -0
- package/.github/workflows/ci.yml +58 -0
- package/.github/workflows/publish.yml +49 -0
- package/.github/workflows/typescript.yml +35 -0
- package/AGENTS.md +2 -2
- package/CLAUDE.md +24 -18
- package/CONTAINERS.md +9 -6
- package/CONTRIBUTING.md +29 -3
- package/Dockerfile +1 -1
- package/README.md +73 -19
- package/UPGRADES.md +4 -0
- package/actSpecs-doc.md +13 -1
- package/actSpecs.js +29 -5
- package/call.js +6 -5
- package/docker-compose.yml +8 -2
- package/docs/checkpoint-scanning.md +199 -0
- package/docs/standard-result-outcome.md +165 -0
- package/env.example +44 -10
- package/eslint.config.mjs +101 -0
- package/netWatch.js +372 -222
- package/package.json +46 -31
- package/pour/README.md +47 -0
- package/pour/pour.min.js +8 -0
- package/procs/actDo.js +808 -0
- package/procs/catalog.d.ts +17 -0
- package/procs/catalog.js +312 -221
- package/procs/catalog.ts +398 -0
- package/procs/checkpoint.js +110 -0
- package/procs/config.d.ts +35 -0
- package/procs/config.js +62 -0
- package/procs/dateTime.js +2 -1
- package/procs/doActs.js +442 -849
- package/procs/doTestAct.js +27 -121
- package/procs/flow.js +221 -0
- package/procs/generateRuleRegistry.js +94 -0
- package/procs/getSource.d.ts +26 -0
- package/procs/job.js +115 -1
- package/procs/launch.d.ts +40 -0
- package/procs/launch.js +239 -108
- package/procs/nu.d.ts +48 -0
- package/procs/scope.js +196 -0
- package/procs/shoot.d.ts +33 -0
- package/procs/standard.d.ts +18 -0
- package/procs/standard.js +83 -0
- package/procs/standard.ts +126 -0
- package/procs/testAct.js +145 -0
- package/procs/testaro.d.ts +20 -0
- package/procs/testaro.js +236 -189
- package/procs/testaro.ts +321 -0
- package/procs/userPath.js +126 -0
- package/procs/xPath.d.ts +4 -0
- package/procs/xPath.js +81 -69
- package/procs/xPath.ts +106 -0
- package/procs/xPathScript.d.ts +6 -0
- package/procs/xPathScript.js +70 -0
- package/run.js +3 -2
- package/surea11y/README.md +44 -0
- package/surea11y/surea11y.browser.js +14 -0
- package/testaro/adbID.d.ts +3 -0
- package/testaro/adbID.js +38 -41
- package/testaro/adbID.ts +66 -0
- package/testaro/allCapStyle.d.ts +3 -0
- package/testaro/allCapStyle.js +31 -34
- package/testaro/allCapStyle.ts +55 -0
- package/testaro/allCaps.d.ts +16 -0
- package/testaro/allCaps.js +179 -151
- package/testaro/allCaps.ts +228 -0
- package/testaro/allHidden.d.ts +13 -0
- package/testaro/allHidden.js +29 -30
- package/testaro/allHidden.ts +50 -0
- package/testaro/allSlanted.d.ts +3 -0
- package/testaro/allSlanted.js +30 -33
- package/testaro/allSlanted.ts +54 -0
- package/testaro/altScheme.d.ts +3 -0
- package/testaro/altScheme.js +27 -30
- package/testaro/altScheme.ts +50 -0
- package/testaro/attVal.d.ts +3 -0
- package/testaro/attVal.js +20 -21
- package/testaro/attVal.ts +52 -0
- package/testaro/autocomplete.d.ts +3 -0
- package/testaro/autocomplete.js +59 -75
- package/testaro/autocomplete.ts +101 -0
- package/testaro/bulk.d.ts +13 -0
- package/testaro/bulk.js +32 -33
- package/testaro/bulk.ts +55 -0
- package/testaro/buttonMenu.d.ts +9 -0
- package/testaro/buttonMenu.js +317 -319
- package/testaro/buttonMenu.ts +391 -0
- package/testaro/captionLoc.d.ts +3 -0
- package/testaro/captionLoc.js +17 -20
- package/testaro/captionLoc.ts +40 -0
- package/testaro/datalistRef.d.ts +3 -0
- package/testaro/datalistRef.js +33 -36
- package/testaro/datalistRef.ts +55 -0
- package/testaro/distortion.d.ts +3 -0
- package/testaro/distortion.js +57 -26
- package/testaro/distortion.ts +81 -0
- package/testaro/docType.d.ts +15 -0
- package/testaro/docType.js +25 -25
- package/testaro/docType.ts +45 -0
- package/testaro/dupAtt.d.ts +16 -0
- package/testaro/dupAtt.js +113 -104
- package/testaro/dupAtt.ts +144 -0
- package/testaro/elements.d.ts +6 -0
- package/testaro/elements.js +153 -153
- package/testaro/elements.ts +215 -0
- package/testaro/embAc.d.ts +3 -0
- package/testaro/embAc.js +19 -20
- package/testaro/embAc.ts +40 -0
- package/testaro/focAll.d.ts +13 -0
- package/testaro/focAll.js +185 -191
- package/testaro/focAll.ts +217 -0
- package/testaro/focAndOp.d.ts +3 -0
- package/testaro/focAndOp.js +100 -104
- package/testaro/focAndOp.ts +128 -0
- package/testaro/focInd.d.ts +3 -0
- package/testaro/focInd.js +64 -66
- package/testaro/focInd.ts +96 -0
- package/testaro/focVis.d.ts +3 -0
- package/testaro/focVis.js +28 -30
- package/testaro/focVis.ts +52 -0
- package/testaro/headEl.d.ts +10 -0
- package/testaro/headEl.js +61 -62
- package/testaro/headEl.ts +82 -0
- package/testaro/headingAmb.d.ts +3 -0
- package/testaro/headingAmb.js +54 -63
- package/testaro/headingAmb.ts +72 -0
- package/testaro/hovInd.d.ts +12 -0
- package/testaro/hovInd.js +163 -130
- package/testaro/hovInd.ts +199 -0
- package/testaro/hover.d.ts +3 -0
- package/testaro/hover.js +155 -126
- package/testaro/hover.ts +154 -0
- package/testaro/hr.d.ts +3 -0
- package/testaro/hr.js +14 -17
- package/testaro/hr.ts +36 -0
- package/testaro/imageLink.d.ts +3 -0
- package/testaro/imageLink.js +17 -20
- package/testaro/imageLink.ts +42 -0
- package/testaro/labClash.d.ts +3 -0
- package/testaro/labClash.js +31 -33
- package/testaro/labClash.ts +54 -0
- package/testaro/legendLoc.d.ts +3 -0
- package/testaro/legendLoc.js +17 -20
- package/testaro/legendLoc.ts +42 -0
- package/testaro/lineHeight.d.ts +3 -0
- package/testaro/lineHeight.js +43 -48
- package/testaro/lineHeight.ts +70 -0
- package/testaro/linkAmb.d.ts +7 -0
- package/testaro/linkAmb.js +79 -80
- package/testaro/linkAmb.ts +105 -0
- package/testaro/linkExt.d.ts +3 -0
- package/testaro/linkExt.js +13 -16
- package/testaro/linkExt.ts +35 -0
- package/testaro/linkOldAtt.d.ts +3 -0
- package/testaro/linkOldAtt.js +25 -28
- package/testaro/linkOldAtt.ts +48 -0
- package/testaro/linkTo.d.ts +3 -0
- package/testaro/linkTo.js +22 -22
- package/testaro/linkTo.ts +43 -0
- package/testaro/linkUl.d.ts +3 -0
- package/testaro/linkUl.js +30 -34
- package/testaro/linkUl.ts +54 -0
- package/testaro/miniText.d.ts +3 -0
- package/testaro/miniText.js +41 -44
- package/testaro/miniText.ts +68 -0
- package/testaro/motion.d.ts +10 -0
- package/testaro/motion.js +92 -96
- package/testaro/motion.ts +125 -0
- package/testaro/nonTable.d.ts +3 -0
- package/testaro/nonTable.js +39 -45
- package/testaro/nonTable.ts +66 -0
- package/testaro/optRoleSel.d.ts +3 -0
- package/testaro/optRoleSel.js +16 -19
- package/testaro/optRoleSel.ts +41 -0
- package/testaro/phOnly.d.ts +3 -0
- package/testaro/phOnly.js +18 -21
- package/testaro/phOnly.ts +43 -0
- package/testaro/pseudoP.d.ts +3 -0
- package/testaro/pseudoP.js +36 -38
- package/testaro/pseudoP.ts +59 -0
- package/testaro/radioSet.d.ts +3 -0
- package/testaro/radioSet.js +57 -59
- package/testaro/radioSet.ts +79 -0
- package/testaro/registry.d.ts +63 -0
- package/testaro/registry.js +67 -0
- package/testaro/registry.ts +141 -0
- package/testaro/role.d.ts +3 -0
- package/testaro/role.js +27 -29
- package/testaro/role.ts +53 -0
- package/testaro/secHeading.d.ts +3 -0
- package/testaro/secHeading.js +30 -33
- package/testaro/secHeading.ts +53 -0
- package/testaro/styleDiff.d.ts +25 -0
- package/testaro/styleDiff.js +248 -252
- package/testaro/styleDiff.ts +303 -0
- package/testaro/tabNav.d.ts +33 -0
- package/testaro/tabNav.js +272 -342
- package/testaro/tabNav.ts +454 -0
- package/testaro/targetsNear.d.ts +9 -0
- package/testaro/targetsNear.js +130 -132
- package/testaro/targetsNear.ts +160 -0
- package/testaro/textNodes.d.ts +6 -0
- package/testaro/textNodes.js +139 -135
- package/testaro/textNodes.ts +185 -0
- package/testaro/textSem.d.ts +3 -0
- package/testaro/textSem.js +25 -28
- package/testaro/textSem.ts +47 -0
- package/testaro/title.d.ts +9 -0
- package/testaro/title.js +16 -13
- package/testaro/title.ts +31 -0
- package/testaro/titledEl.d.ts +3 -0
- package/testaro/titledEl.js +15 -18
- package/testaro/titledEl.ts +38 -0
- package/testaro/zIndex.d.ts +3 -0
- package/testaro/zIndex.js +19 -22
- package/testaro/zIndex.ts +42 -0
- package/tests/alfa.d.ts +45 -0
- package/tests/alfa.js +138 -141
- package/tests/alfa.ts +214 -0
- package/tests/aslint.d.ts +33 -0
- package/tests/aslint.js +272 -249
- package/tests/aslint.ts +301 -0
- package/tests/axe.d.ts +27 -0
- package/tests/axe.js +199 -200
- package/tests/axe.ts +277 -0
- package/tests/ed11y.d.ts +28 -0
- package/tests/ed11y.js +141 -99
- package/tests/ed11y.ts +178 -0
- package/tests/htmlcs.d.ts +21 -0
- package/tests/htmlcs.js +174 -140
- package/tests/htmlcs.ts +181 -0
- package/tests/ibm.d.ts +52 -0
- package/tests/ibm.js +164 -166
- package/tests/ibm.ts +251 -0
- package/tests/nuVal.d.ts +13 -0
- package/tests/nuVal.js +107 -112
- package/tests/nuVal.ts +145 -0
- package/tests/nuVnu.d.ts +15 -0
- package/tests/nuVnu.js +141 -111
- package/tests/nuVnu.ts +144 -0
- package/tests/pour.d.ts +31 -0
- package/tests/pour.js +242 -0
- package/tests/pour.ts +273 -0
- package/tests/qualWeb.d.ts +39 -0
- package/tests/qualWeb.js +302 -272
- package/tests/qualWeb.ts +415 -0
- package/tests/surea11y.d.ts +33 -0
- package/tests/surea11y.js +288 -0
- package/tests/surea11y.ts +334 -0
- package/tests/testaro.d.ts +25 -0
- package/tests/testaro.js +745 -652
- package/tests/testaro.ts +862 -0
- package/tests/wave.d.ts +46 -0
- package/tests/wave.js +166 -177
- package/tests/wave.ts +252 -0
- package/tsconfig.json +17 -0
- package/types.d.ts +243 -0
- package/types.js +10 -0
- package/types.ts +376 -0
- package/validation/act/README.md +46 -0
- package/validation/act/capture.js +424 -0
- package/validation/act/chromium-issue-draft.md +66 -0
- package/validation/act/fp-triage-2026-08-22.md +86 -0
- package/validation/act/isolation-notes.md +159 -0
- package/validation/act/playwright-issue-draft.md +73 -0
- package/validation/act/propose-mappings.js +0 -0
- package/validation/act/repro-cdp-raw.js +105 -0
- package/validation/act/repro-metarefresh.js +65 -0
- package/validation/act/score.js +184 -0
- package/validation/act/stage3a-stress-report.md +69 -0
- package/validation/act/stage3b-mapping-proposals.md +94 -0
- package/validation/act/stage3b-triage-draft.md +138 -0
- package/validation/act/surea11y-track-a-2026-09-01.md +50 -0
- package/validation/executors/netWatch.js +180 -90
- package/validation/executors/test.js +17 -2
- package/validation/executors/tests.js +118 -10
- package/validation/jobs/reports/raw/260901T1000-surea11y-validation.json +964 -0
- package/validation/jobs/todo/240101T1200-simple-example.json +14 -6
- package/validation/jobs/todo/240101T1300-shoot-example.json +2 -1
- package/validation/jobs/todo/260821T1900-pour-validation.json +45 -0
- package/validation/jobs/todo/260901T1000-surea11y-validation.json +45 -0
- package/validation/knownFailures.json +1 -0
- package/validation/tests/jobProperties/adbID.json +27 -2
- package/validation/tests/jobProperties/{focOp.json → allCapStyle.json} +56 -53
- package/validation/tests/jobProperties/allCaps.json +25 -0
- package/validation/tests/jobProperties/allHidden.json +133 -13
- package/validation/tests/jobProperties/allSlanted.json +3 -3
- package/validation/tests/jobProperties/altScheme.json +23 -8
- package/validation/tests/jobProperties/attVal.json +57 -57
- package/validation/tests/jobProperties/autocomplete.json +11 -1
- package/validation/tests/jobProperties/bulk.json +6 -1
- package/validation/tests/jobProperties/buttonMenu.json +63 -42
- package/validation/tests/jobProperties/captionLoc.json +1 -6
- package/validation/tests/jobProperties/checkpoint-browser.json +401 -0
- package/validation/tests/jobProperties/checkpoint-page.json +397 -0
- package/validation/tests/jobProperties/checkpoint.json +395 -0
- package/validation/tests/jobProperties/datalistRef.json +20 -5
- package/validation/tests/jobProperties/distortion.json +28 -3
- package/validation/tests/jobProperties/docType.json +2 -2
- package/validation/tests/jobProperties/dupAtt.json +48 -33
- package/validation/tests/jobProperties/elements.json +28 -28
- package/validation/tests/jobProperties/embAc.json +36 -31
- package/validation/tests/jobProperties/focAndOp.json +284 -0
- package/validation/tests/jobProperties/focInd.json +34 -34
- package/validation/tests/jobProperties/focVis.json +1 -1
- package/validation/tests/jobProperties/hover.json +39 -37
- package/validation/tests/jobProperties/hr.json +13 -3
- package/validation/tests/jobProperties/imageLink.json +0 -5
- package/validation/tests/jobProperties/labClash.json +68 -33
- package/validation/tests/jobProperties/legendLoc.json +1 -6
- package/validation/tests/jobProperties/lineHeight.json +2 -2
- package/validation/tests/jobProperties/linkAmb.json +13 -18
- package/validation/tests/jobProperties/linkExt.json +1 -1
- package/validation/tests/jobProperties/linkOldAtt.json +12 -2
- package/validation/tests/jobProperties/linkTo.json +1 -1
- package/validation/tests/jobProperties/linkUl.json +65 -65
- package/validation/tests/jobProperties/miniText.json +7 -2
- package/validation/tests/jobProperties/motion.json +4 -50
- package/validation/tests/jobProperties/nonTable.json +48 -3
- package/validation/tests/jobProperties/optRoleSel.json +12 -2
- package/validation/tests/jobProperties/phOnly.json +6 -16
- package/validation/tests/jobProperties/pseudoP.json +17 -2
- package/validation/tests/jobProperties/radioSet.json +30 -30
- package/validation/tests/jobProperties/role.json +25 -5
- package/validation/tests/jobProperties/secHeading.json +26 -21
- package/validation/tests/jobProperties/styleDiff.json +35 -5
- package/validation/tests/jobProperties/tabNav.json +3 -1
- package/validation/tests/jobProperties/targetsNear.json +237 -0
- package/validation/tests/jobProperties/textNodes.json +29 -29
- package/validation/tests/jobProperties/textSem.json +26 -1
- package/validation/tests/jobProperties/title.json +12 -2
- package/validation/tests/jobProperties/titledEl.json +39 -9
- package/validation/tests/jobProperties/userPath.json +317 -0
- package/validation/tests/jobProperties/zIndex.json +40 -35
- package/validation/tests/targets/allCapStyle/index.html +26 -0
- package/validation/tests/targets/checkpoint/index.html +38 -0
- package/validation/tests/targets/datalistRef/index.html +1 -1
- package/validation/tests/targets/focAndOp/bad.html +29 -0
- package/validation/tests/targets/{focOp → focAndOp}/good.html +3 -1
- package/validation/tests/targets/focInd/bad.html +2 -1
- package/validation/tests/targets/headEl/index.html +10 -1
- package/validation/tests/targets/{targetSmall → targetsNear}/index.html +15 -1
- package/validation/tests/targets/userPath/index.html +42 -0
- package/validation/validateTest.js +61 -9
- package/.claude/settings.local.json +0 -11
- package/.eslintrc.json +0 -41
- package/htmlcs/.eslintrc.json +0 -67
- package/validation/tests/jobProperties/linkTitle.json +0 -127
- package/validation/tests/jobProperties/opFoc.json +0 -164
- package/validation/tests/jobProperties/targetSmall.json +0 -152
- package/validation/tests/jobProperties/targetTiny.json +0 -142
- package/validation/tests/targets/focOp/bad.html +0 -25
- package/validation/tests/targets/linkTitle/index.html +0 -24
- package/validation/tests/targets/opFoc/bad.html +0 -26
- package/validation/tests/targets/opFoc/good.html +0 -23
|
@@ -0,0 +1,159 @@
|
|
|
1
|
+
# Long-lived-browser injection failure — evidence log (2026-08-22)
|
|
2
|
+
|
|
3
|
+
Motivation beyond this harness: production testaro launches a fresh browser per test act because
|
|
4
|
+
of recurring spurious test-isolation bugs, at a large throughput cost. Anything learned here about
|
|
5
|
+
what long-lived-browser state actually breaks — and what level of recycling (context vs browser)
|
|
6
|
+
resets it — bears on whether that cost can be reduced.
|
|
7
|
+
|
|
8
|
+
## The incident
|
|
9
|
+
|
|
10
|
+
Full ACT capture (1,213 testcases × pour + axe, one Chromium browser + one context for the whole
|
|
11
|
+
run, fresh page per testcase × engine, pages closed after use). From testcase 274 — exactly the
|
|
12
|
+
first fixture of rule `7d6734`, immediately after the last `c487ae` fixture — **every** pour row
|
|
13
|
+
for the remainder of the run (940 rows) failed the same way: script element inserted, no
|
|
14
|
+
exception, `pourEngine` global never defined. Axe, whose adapter executes via `page.evaluate`
|
|
15
|
+
instead of a `<script>` tag, kept detecting instances to the final row. No recovery for ~940
|
|
16
|
+
consecutive pages over ~13 minutes.
|
|
17
|
+
|
|
18
|
+
## Experiments
|
|
19
|
+
|
|
20
|
+
| # | Setup | Result |
|
|
21
|
+
| --- | --- | --- |
|
|
22
|
+
| 1 | Fresh process, the boundary fixtures (`7d6734`, `afw4f7`) | All pass — fixtures innocent in isolation |
|
|
23
|
+
| 2 | 700 cycles, local benign fixtures, one browser, axe-style page interleaved (data-xpath stamping + axe evaluate), no recycling | 0 failures, flat RSS — page count + interleave alone insufficient |
|
|
24
|
+
| 3 | Full-feed pour re-capture, browser recycled every 100 testcases, no axe interleave | Clean through the entire former cliff region (34/34 on `7d6734`/`307n5z`/`a25f45`) |
|
|
25
|
+
| 4 | Cliff-neighborhood replay (`c487ae`→`a25f45`, 69 testcases), pour + axe interleaved, **no recycling**, fresh process | 69/69 clean — the fixture region + interleave is not a deterministic trigger |
|
|
26
|
+
| 5 | Full-feed pour-only recapture, hardened harness, solo machine, recycling every 100 | **Cliff recurred**: 41 consecutive rows dead starting mid-`1a02b0` (video-transcript fixtures) — in a browser only ~36 pages old. Rules out cumulative-pages-per-browser; the concurrent-load theory weakens too (run was solo). |
|
|
27
|
+
| 6 | Meta-refresh(0) fixtures, local repro (`repro-metarefresh.js`): goto → evaluate (interrupted by the refresh navigation) → `page.close()` | **`page.close()` hangs indefinitely** — never resolves, never rejects — on ~half the rounds, on both a single-redirect page and an a→b→a redirect loop. Playwright 1.62.1, Chromium headless. The context stays usable for *new* pages afterward. Independently, both full captures wedged at exactly the first `bc659a` ("Meta element has no refresh delay") fixture — its passed examples are `content="0; URL=…"` immediate redirects — frozen >35 min in an untimeouted `await page.close()`. |
|
|
28
|
+
| 7 | Replay of the recurrence region (`1a02b0`,`2ee8b8`,`59br37`, 61 testcases), no recycling, fresh process | 61/61 clean — the video region is not a deterministic trigger either. |
|
|
29
|
+
| 8 | Wedged-browser aftermath (from the bc659a incident): `newPage` on the wedged browser | Throws `Protocol error (Target.createTarget): Not supported`, then `Target page, context or browser has been closed` once the process dies — a recognizable signature usable as a replace trigger. A deadline-only trigger let 73 rows (pour) / 886 rows (qualWeb) fail before the next scheduled recycle. |
|
|
30
|
+
|
|
31
|
+
## Conclusion: probabilistic wedge events, answered by canary-triggered recycling
|
|
32
|
+
|
|
33
|
+
The injection cliff is **nondeterministic**: it recurred at a different position in a nearly-new
|
|
34
|
+
browser on a quiet machine, and every regional replay of a cliff neighborhood comes back clean.
|
|
35
|
+
Best model: races between fixture-initiated behavior (media load, client-side navigation) and
|
|
36
|
+
harness operations occasionally wedge browser-wide `<script>`-element execution — same family as
|
|
37
|
+
the reproducible close() hang, at roughly 1–2 events per ~1,200 real-page cycles. Chasing a
|
|
38
|
+
deterministic trigger further has diminishing value; the effective countermeasures, all now in
|
|
39
|
+
`capture.js`, are:
|
|
40
|
+
|
|
41
|
+
1. **Deadline-race every browser-touching await** (nothing can hang the loop);
|
|
42
|
+
2. **Canary assertions** (expected tool global present on text/html pages);
|
|
43
|
+
3. **Replace the browser on any wedge signature**: tripped deadline, `Protocol error
|
|
44
|
+
(Target.createTarget)`, "has been closed", or a failed canary.
|
|
45
|
+
|
|
46
|
+
## Upstream status (2026-08-22)
|
|
47
|
+
|
|
48
|
+
- Playwright: filed as <https://github.com/microsoft/playwright/issues/42366> (likely duplicate
|
|
49
|
+
of #42068, which the team milestoned v1.63); **fix PR submitted:**
|
|
50
|
+
<https://github.com/microsoft/playwright/pull/42367> (bounded re-issue of `Target.closeTarget`
|
|
51
|
+
in `CRBrowser._closePage` + stale-fixme removal; their close-related suites pass 41/41).
|
|
52
|
+
Note: Playwright's rolled Chromium 152.0.7977.54 narrows the race window sharply through
|
|
53
|
+
Playwright's call timing (their reload-trigger test passes 6/6 unpatched), but raw CDP still
|
|
54
|
+
reproduces the contract violation on 152 (~2/10) — so channel builds (Chrome/Edge stable,
|
|
55
|
+
151-class, 70–80% hit rate) remain the population the guard protects.
|
|
56
|
+
- testaro: `browserClose` hardened with a 10 s settle-deadline per close (try/catch cannot catch
|
|
57
|
+
non-settlement) — PR <https://github.com/YRA-Tech/testaro/pull/92>. (Discovered en route:
|
|
58
|
+
jrpool/testaro now redirects to YRA-Tech/testaro — the repo was transferred, so #92 IS the
|
|
59
|
+
upstream PR.) The same patch is applied to the working tree used by the harness.
|
|
60
|
+
- Chromium: already tracked as <https://issues.chromium.org/issues/536385539> (filed 2026-07-19
|
|
61
|
+
by a Google engineer; P2/S2; **pending code change 8251379**). Their C++ root cause matches our
|
|
62
|
+
protocol-level one exactly: `WebContentsImpl::ClosePage` arms the close on the current main
|
|
63
|
+
RenderFrameHost; a racing navigation commit swaps the RFH; the pending `ClosePage` callback
|
|
64
|
+
dies with the old RFH and is never re-issued. Our meta-refresh(0) trigger, hit rates, and
|
|
65
|
+
workaround matrix posted there as comment #2; both issues cross-linked. Practical upshot for
|
|
66
|
+
testaro: a fix is in flight upstream, but the deadline+canary+context-recycle defenses remain
|
|
67
|
+
necessary until it ships in a released Chromium — and for the injection cliff, which has no
|
|
68
|
+
upstream issue yet.
|
|
69
|
+
|
|
70
|
+
## Discovery #2 — root-caused (2026-08-22 evening)
|
|
71
|
+
|
|
72
|
+
A `DEBUG=pw:protocol` trace of a hanging round shows the full mechanism: Playwright sends
|
|
73
|
+
`Target.closeTarget`, Chromium replies `{"result":{"success":true}}`, but the target — mid-commit
|
|
74
|
+
in the meta-refresh navigation — never closes; the same session then emits the redirect
|
|
75
|
+
destination's entire load lifecycle after the "successful" close, `Target.targetDestroyed` never
|
|
76
|
+
fires, and `page.close()` waits for it forever. Chromium acknowledges a close it does not
|
|
77
|
+
perform; Playwright trusts the acknowledgment with no timeout.
|
|
78
|
+
|
|
79
|
+
Matrix (10 rounds/cell): **Chromium-only** (Firefox 153 and WebKit: 0/10 everywhere), and
|
|
80
|
+
**default `close()` only** — `close({runBeforeUnload: true})` 0/10, `context.close()` 0/10.
|
|
81
|
+
After a hang, `context.close()` still settles and destroys the page (**8/8 recoveries**).
|
|
82
|
+
|
|
83
|
+
The recovery result upgrades the production recommendation: for this wedge class,
|
|
84
|
+
**context-level teardown is a sufficient and reliable recovery** — no browser relaunch needed
|
|
85
|
+
(~10 ms vs ~300 ms). A pooled-browser testaro could run context-per-act with deadline-raced
|
|
86
|
+
closes, escalating to browser replacement only if the context teardown itself trips its
|
|
87
|
+
deadline. (The injection cliff's recovery level remains untested — it is not reproducible on
|
|
88
|
+
demand — so the escalation path should stay.)
|
|
89
|
+
|
|
90
|
+
## Discovery #2 (original observation): page.close() can hang forever after a meta-refresh navigation
|
|
91
|
+
|
|
92
|
+
This is a second, *reproducible-at-will* failure mode, distinct from the injection cliff but in
|
|
93
|
+
the same family (navigation racing teardown corrupts page state):
|
|
94
|
+
|
|
95
|
+
- Signature: `close()` neither resolves nor rejects, so `await page.close().catch(...)` blocks
|
|
96
|
+
forever — `.catch` guards rejection, not non-settlement. A loop with no deadline around close
|
|
97
|
+
freezes permanently, silently.
|
|
98
|
+
- Nondeterministic (~50% of rounds locally): it is a race with the refresh-triggered navigation,
|
|
99
|
+
which also explains why one run passes a fixture another run wedges on.
|
|
100
|
+
- No matching upstream issue found (2026-08-22 search); nearest family member is
|
|
101
|
+
[playwright#33806](https://github.com/microsoft/playwright/issues/33806) (hangs waiting for
|
|
102
|
+
pending navigations). **Fileable** with `repro-metarefresh.js` (~40 lines, local pages, no
|
|
103
|
+
external deps beyond playwright + any large script to inject — the script is incidental).
|
|
104
|
+
- Harness guard shipped in `capture.js`: every browser-touching await is raced against a
|
|
105
|
+
deadline (`withDeadline`), a tripped deadline presumes a wedged browser and replaces it
|
|
106
|
+
(`replaceBrowser`, itself deadline-guarded), and captures resume by skipping rows already in
|
|
107
|
+
the out file. Verified against the live `bc659a` fixtures: one real close-hang tripped the 5 s
|
|
108
|
+
deadline, the browser was replaced, and all 15 fixtures completed.
|
|
109
|
+
|
|
110
|
+
### Production translation (testaro)
|
|
111
|
+
|
|
112
|
+
Pages that navigate during scanning (meta refresh is common on real sites) can hang `close()`
|
|
113
|
+
forever. In a browser-per-act world the leaked hang is bounded by the act's child process exit;
|
|
114
|
+
in any pooled/long-lived design it is fatal without deadlines. Concretely: (1) every
|
|
115
|
+
`page.close()`/`browserClose()` should be deadline-raced, (2) a tripped deadline should recycle
|
|
116
|
+
the browser, not just the page, and (3) the meta-refresh fixture class belongs in any isolation
|
|
117
|
+
test suite as the canonical wedge-inducer.
|
|
118
|
+
|
|
119
|
+
The overall answer to "could testaro keep a browser across acts?": the observed isolation
|
|
120
|
+
failures here were **rare, probabilistic wedge events with recognizable signatures**, not
|
|
121
|
+
gradual deterministic leakage. That favors a middle path over browser-per-act: pool the browser,
|
|
122
|
+
assert a cheap per-act canary (script-injection round-trip + expected globals), and recycle the
|
|
123
|
+
browser only on a wedge signature or every N acts — the three-countermeasure pattern above,
|
|
124
|
+
already exercised at ~1,200-page scale by this harness. The remaining unknown for production is
|
|
125
|
+
whether *cross-page contamination* (state leaking between scanned pages, distinct from wedges)
|
|
126
|
+
also occurs; this harness's same-tool-different-page design cannot see that class, so a pooled
|
|
127
|
+
design should keep context-per-act (cheap) even if the browser is shared.
|
|
128
|
+
|
|
129
|
+
## Explained failure classes (distinct from the cliff)
|
|
130
|
+
|
|
131
|
+
- `contentType application/xml` fixtures: `document.createElement('script')` in an XML document
|
|
132
|
+
creates a null-namespace element that never executes — expected, benign, now visible via the
|
|
133
|
+
adapter's diagnostic error (content type, script chars, ready state).
|
|
134
|
+
- "Resulting promise was garbage collected": fixtures that navigate (meta refresh) during the
|
|
135
|
+
in-page evaluate. All observed on `inapplicable` testcases.
|
|
136
|
+
|
|
137
|
+
## Current best theory
|
|
138
|
+
|
|
139
|
+
Not page count, not the axe interleave, not any single fixture so far: either (a) cumulative
|
|
140
|
+
state that only the full 273-testcase real-page history builds up, or (b) a nondeterministic
|
|
141
|
+
in-browser event (renderer/helper-process crash or OOM kill — the host was carrying ~31 GB of
|
|
142
|
+
desktop Chrome RSS during the incident) that wedges `<script>`-element execution while leaving
|
|
143
|
+
CDP `evaluate` execution intact. All fixtures being same-origin (`www.w3.org`) means Chromium packs
|
|
144
|
+
them into shared renderer processes, which is consistent with one wedged process poisoning every
|
|
145
|
+
subsequent same-origin page.
|
|
146
|
+
|
|
147
|
+
## Production relevance (testaro browser-per-act)
|
|
148
|
+
|
|
149
|
+
- The failure mode was **silent** — no exception anywhere; only an output-level check (expected
|
|
150
|
+
global absent) caught it. Any move away from browser-per-act needs a per-act canary assertion
|
|
151
|
+
of this kind, not just error handling.
|
|
152
|
+
- Recycling every N acts (experiment 3) is so far sufficient as a guard and is the cheap middle
|
|
153
|
+
ground between browser-per-act and browser-forever; whether *context*-per-act (much cheaper
|
|
154
|
+
than browser launch) also resets the poisoned state is untested — worth adding
|
|
155
|
+
`--recycle-context` bisection if experiment 5 reproduces the cliff.
|
|
156
|
+
- If experiment 5 comes back clean, the honest conclusion is: one observed environmental
|
|
157
|
+
incident in ~5,000 page-cycles, mitigated by periodic recycling — suggestive that testaro's
|
|
158
|
+
isolation bugs may likewise be rare wedge events rather than deterministic leakage, which would
|
|
159
|
+
argue for recycle-on-anomaly (canary-triggered) rather than always-fresh browsers.
|
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
# Playwright bug report — page.close() never settles after meta-refresh navigation
|
|
2
|
+
|
|
3
|
+
**FILED 2026-08-22: <https://github.com/microsoft/playwright/issues/42366>**
|
|
4
|
+
Minimal repro: `repro-metarefresh.js` in this directory (self-contained, local HTTP server).
|
|
5
|
+
Chromium side: already tracked as <https://issues.chromium.org/issues/536385539> (pending code
|
|
6
|
+
change 8251379); cross-linked in both directions 2026-08-22 — see `chromium-issue-draft.md`.
|
|
7
|
+
|
|
8
|
+
---
|
|
9
|
+
|
|
10
|
+
**Title:** [Bug]: page.close() hangs forever — Chromium's Target.closeTarget returns success:true
|
|
11
|
+
without closing a target that is mid-commit in a client-side (meta refresh) navigation
|
|
12
|
+
|
|
13
|
+
## System info
|
|
14
|
+
|
|
15
|
+
- Playwright Version: 1.62.1 (latest at time of report)
|
|
16
|
+
- Operating System: Linux (7.0.0-30-generic)
|
|
17
|
+
- Browser: Chromium (bundled, HeadlessChrome/151.0.7922.34) — **Chromium only**; Firefox and
|
|
18
|
+
WebKit are unaffected (0/10 each in the same harness)
|
|
19
|
+
- Other info: Node 24.14.1
|
|
20
|
+
|
|
21
|
+
## Repro
|
|
22
|
+
|
|
23
|
+
1. Serve a page with `<meta http-equiv="refresh" content="0; URL='/target.html'">` (immediate
|
|
24
|
+
client-side redirect; the W3C ACT-Rules bc659a "passed" fixtures are real-world examples).
|
|
25
|
+
2. `page.goto(url, {waitUntil: 'load'})` — resolves.
|
|
26
|
+
3. `page.evaluate(...)` — rejects with "Execution context was destroyed" (expected; the refresh
|
|
27
|
+
navigation fired mid-evaluate).
|
|
28
|
+
4. `await page.close()` — **hangs on ~70–80% of rounds** (8/10, 7/10 across runs): the promise
|
|
29
|
+
never resolves and never rejects. `.catch()` never fires; a loop awaiting it freezes forever.
|
|
30
|
+
|
|
31
|
+
## Root cause (from a `DEBUG=pw:protocol` trace of a hanging round)
|
|
32
|
+
|
|
33
|
+
- Playwright sends `Target.closeTarget {targetId}`.
|
|
34
|
+
- The browser replies `{"result":{"success":true}}`.
|
|
35
|
+
- The target does not close: the meta-refresh navigation was mid-commit, and the same session
|
|
36
|
+
then emits the full lifecycle of the redirect destination — `Page.frameNavigated`,
|
|
37
|
+
`DOMContentLoaded`, `load`, paint events, `networkIdle` — all *after* the successful-looking
|
|
38
|
+
`closeTarget` response.
|
|
39
|
+
- `Target.targetDestroyed` is never emitted; `page.close()` waits for it indefinitely.
|
|
40
|
+
|
|
41
|
+
So Chromium acknowledges the close without performing it when the close races a navigation
|
|
42
|
+
commit, and Playwright trusts the acknowledgment with no timeout or retry.
|
|
43
|
+
|
|
44
|
+
The browser-side half is confirmed **without Playwright**: a raw-CDP repro (plain WebSocket, no
|
|
45
|
+
client library) shows `Target.closeTarget` → `success:true` with no `Target.targetDestroyed`
|
|
46
|
+
ever emitted and the target still listed in `Target.getTargets`, ~3/10 rounds. A Chromium bug is
|
|
47
|
+
being filed in parallel (link when available). Playwright still has its own half: `page.close()`
|
|
48
|
+
trusts the acknowledgment indefinitely — a deadline/retry (or routing default close through the
|
|
49
|
+
`runBeforeUnload` path, which is immune) would make clients robust to the browser defect.
|
|
50
|
+
|
|
51
|
+
## Matrix (10 rounds per cell, same repro)
|
|
52
|
+
|
|
53
|
+
| | `close()` | `close({runBeforeUnload:true})` | `context.close()` |
|
|
54
|
+
| --- | --- | --- | --- |
|
|
55
|
+
| Chromium | **8/10 hang** | 0/10 | 0/10 |
|
|
56
|
+
| Firefox | 0/10 | 0/10 | 0/10 |
|
|
57
|
+
| WebKit | 0/10 | 0/10 | 0/10 |
|
|
58
|
+
|
|
59
|
+
After a hang, `context.close()` still settles and destroys the page (**8/8 recoveries**) — the
|
|
60
|
+
context teardown path is unaffected.
|
|
61
|
+
|
|
62
|
+
## Expected behavior
|
|
63
|
+
|
|
64
|
+
`page.close()` always settles — resolves once the page is gone, or rejects.
|
|
65
|
+
|
|
66
|
+
**Workarounds** (verified): `page.close({runBeforeUnload: true})`; or race `close()` against a
|
|
67
|
+
deadline and call `context.close()` on expiry (recovers reliably). Without a guard, a wedged
|
|
68
|
+
browser's subsequent `newPage` can then fail with `Protocol error (Target.createTarget): Not
|
|
69
|
+
supported` followed by "Target page, context or browser has been closed".
|
|
70
|
+
|
|
71
|
+
Possibly related: #33806 (hangs while waiting for pending navigations) — different in that no
|
|
72
|
+
dialog is involved and the acknowledged-but-not-performed `Target.closeTarget` is visible in the
|
|
73
|
+
protocol trace.
|
|
Binary file
|
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
// Playwright-free reproduction attempt: raw CDP over WebSocket (Node's
|
|
2
|
+
// native WebSocket client). Question: does Chromium's Target.closeTarget
|
|
3
|
+
// return {success:true} WITHOUT closing (no Target.targetDestroyed, target
|
|
4
|
+
// still listed) when the close races a meta-refresh navigation commit?
|
|
5
|
+
// If yes → browser-side contract violation, independent of Playwright.
|
|
6
|
+
const http = require('http');
|
|
7
|
+
const {spawn} = require('child_process');
|
|
8
|
+
// Any Chromium binary works; defaults to Playwright's bundled one.
|
|
9
|
+
const {chromium} = require('playwright');
|
|
10
|
+
const CHROME = process.env.CHROME_PATH || chromium.executablePath();
|
|
11
|
+
|
|
12
|
+
const pages = {
|
|
13
|
+
'/a.html': `<!DOCTYPE html><html lang="en"><head><title>a</title>
|
|
14
|
+
<meta http-equiv="refresh" content="0; URL='/target.html'"></head>
|
|
15
|
+
<body><p>redirecting</p></body></html>`,
|
|
16
|
+
'/target.html': `<!DOCTYPE html><html lang="en"><head><title>t</title></head>
|
|
17
|
+
<body><p>arrived</p></body></html>`
|
|
18
|
+
};
|
|
19
|
+
|
|
20
|
+
(async () => {
|
|
21
|
+
const server = http.createServer((req, res) => {
|
|
22
|
+
res.setHeader('content-type', 'text/html');
|
|
23
|
+
res.end(pages[req.url] || pages['/target.html']);
|
|
24
|
+
});
|
|
25
|
+
await new Promise(resolve => server.listen(0, resolve));
|
|
26
|
+
const base = `http://127.0.0.1:${server.address().port}`;
|
|
27
|
+
|
|
28
|
+
// Launch the same Chromium binary Playwright uses, but bare.
|
|
29
|
+
const chrome = spawn(CHROME, [
|
|
30
|
+
'--headless=new', '--remote-debugging-port=0', '--no-first-run', '--no-default-browser-check',
|
|
31
|
+
'--user-data-dir=/tmp/cdp-repro-profile'
|
|
32
|
+
]);
|
|
33
|
+
const wsURL = await new Promise((resolve, reject) => {
|
|
34
|
+
let stderr = '';
|
|
35
|
+
chrome.stderr.on('data', chunk => {
|
|
36
|
+
stderr += chunk;
|
|
37
|
+
const match = stderr.match(/DevTools listening on (ws:\/\/\S+)/);
|
|
38
|
+
if (match) resolve(match[1]);
|
|
39
|
+
});
|
|
40
|
+
setTimeout(() => reject(new Error('no DevTools ws URL')), 15000);
|
|
41
|
+
});
|
|
42
|
+
|
|
43
|
+
const ws = new WebSocket(wsURL);
|
|
44
|
+
await new Promise(resolve => ws.onopen = resolve);
|
|
45
|
+
let nextId = 1;
|
|
46
|
+
const pending = new Map();
|
|
47
|
+
const events = [];
|
|
48
|
+
ws.onmessage = messageEvent => {
|
|
49
|
+
const message = JSON.parse(messageEvent.data);
|
|
50
|
+
if (message.id && pending.has(message.id)) {
|
|
51
|
+
pending.get(message.id)(message);
|
|
52
|
+
pending.delete(message.id);
|
|
53
|
+
}
|
|
54
|
+
else if (message.method) {
|
|
55
|
+
events.push(message);
|
|
56
|
+
}
|
|
57
|
+
};
|
|
58
|
+
const send = (method, params = {}, sessionId) => new Promise(resolve => {
|
|
59
|
+
const id = nextId++;
|
|
60
|
+
pending.set(id, resolve);
|
|
61
|
+
ws.send(JSON.stringify({id, method, params, ...(sessionId ? {sessionId} : {})}));
|
|
62
|
+
});
|
|
63
|
+
const wait = ms => new Promise(resolve => setTimeout(resolve, ms));
|
|
64
|
+
|
|
65
|
+
await send('Target.setDiscoverTargets', {discover: true});
|
|
66
|
+
let contractViolations = 0;
|
|
67
|
+
const ROUNDS = 10;
|
|
68
|
+
for (let round = 0; round < ROUNDS; round++) {
|
|
69
|
+
const {result: {targetId}} = await send('Target.createTarget', {url: 'about:blank'});
|
|
70
|
+
const {result: {sessionId}} = await send('Target.attachToTarget', {targetId, flatten: true});
|
|
71
|
+
await send('Page.enable', {}, sessionId);
|
|
72
|
+
await send('Runtime.enable', {}, sessionId);
|
|
73
|
+
await send('Page.navigate', {url: `${base}/a.html`}, sessionId);
|
|
74
|
+
// Leave an evaluate in flight for the refresh navigation to destroy
|
|
75
|
+
// (mirrors the real-world trigger), then close during the commit window.
|
|
76
|
+
send('Runtime.evaluate', {
|
|
77
|
+
expression: 'new Promise(r => setTimeout(r, 100))',
|
|
78
|
+
awaitPromise: true
|
|
79
|
+
}, sessionId);
|
|
80
|
+
await wait(30 + (round % 5) * 15);
|
|
81
|
+
events.length = 0;
|
|
82
|
+
const closeReply = await send('Target.closeTarget', {targetId});
|
|
83
|
+
const success = closeReply.result && closeReply.result.success;
|
|
84
|
+
await wait(3000);
|
|
85
|
+
const destroyed = events.some(
|
|
86
|
+
e => e.method === 'Target.targetDestroyed' && e.params.targetId === targetId
|
|
87
|
+
);
|
|
88
|
+
const {result: {targetInfos}} = await send('Target.getTargets');
|
|
89
|
+
const stillListed = targetInfos.some(t => t.targetId === targetId);
|
|
90
|
+
const verdict = success && ! destroyed && stillListed ? 'CONTRACT VIOLATION'
|
|
91
|
+
: success && destroyed ? 'closed properly'
|
|
92
|
+
: `other (success=${success} destroyed=${destroyed} listed=${stillListed})`;
|
|
93
|
+
console.log(`round ${round}: closeTarget success=${success}, targetDestroyed=${destroyed}, stillListed=${stillListed} → ${verdict}`);
|
|
94
|
+
if (verdict === 'CONTRACT VIOLATION') {
|
|
95
|
+
contractViolations++;
|
|
96
|
+
// Clean up the orphan so rounds stay independent.
|
|
97
|
+
await send('Target.closeTarget', {targetId});
|
|
98
|
+
await wait(500);
|
|
99
|
+
}
|
|
100
|
+
}
|
|
101
|
+
console.log(`${contractViolations}/${ROUNDS} rounds: success:true with target neither destroyed nor removed`);
|
|
102
|
+
chrome.kill();
|
|
103
|
+
server.close();
|
|
104
|
+
process.exit(0);
|
|
105
|
+
})();
|
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
/*
|
|
2
|
+
© 2026 Jeff Witt.
|
|
3
|
+
|
|
4
|
+
Licensed under the MIT License. See LICENSE file at the project root or
|
|
5
|
+
https://opensource.org/license/mit/ for details.
|
|
6
|
+
|
|
7
|
+
SPDX-License-Identifier: MIT
|
|
8
|
+
*/
|
|
9
|
+
|
|
10
|
+
/*
|
|
11
|
+
repro-metarefresh.js
|
|
12
|
+
Minimal reproduction: page.close() intermittently NEVER SETTLES (neither
|
|
13
|
+
resolves nor rejects) when the page performed a meta-refresh(0) navigation
|
|
14
|
+
that interrupted an in-flight evaluate. See playwright-issue-draft.md.
|
|
15
|
+
Observed on playwright 1.62.1, Chromium headless: ~50% of rounds hang.
|
|
16
|
+
Run from the repository root: node validation/act/repro-metarefresh.js
|
|
17
|
+
*/
|
|
18
|
+
|
|
19
|
+
const http = require('http');
|
|
20
|
+
const {chromium} = require('playwright');
|
|
21
|
+
|
|
22
|
+
const pages = {
|
|
23
|
+
'/a.html': `<!DOCTYPE html><html lang="en"><head><title>a</title>
|
|
24
|
+
<meta http-equiv="refresh" content="0; URL='/target.html'"></head>
|
|
25
|
+
<body><p>redirecting</p></body></html>`,
|
|
26
|
+
'/target.html': `<!DOCTYPE html><html lang="en"><head><title>t</title></head>
|
|
27
|
+
<body><p>arrived</p></body></html>`
|
|
28
|
+
};
|
|
29
|
+
|
|
30
|
+
// Reports whether a promise settles within ms.
|
|
31
|
+
const settles = (promise, ms) => Promise.race([
|
|
32
|
+
promise.then(() => 'resolved').catch(() => 'rejected'),
|
|
33
|
+
new Promise(resolve => setTimeout(() => resolve('NEVER SETTLED'), ms))
|
|
34
|
+
]);
|
|
35
|
+
|
|
36
|
+
(async () => {
|
|
37
|
+
const server = http.createServer((req, res) => {
|
|
38
|
+
res.setHeader('content-type', 'text/html');
|
|
39
|
+
res.end(pages[req.url] || pages['/target.html']);
|
|
40
|
+
});
|
|
41
|
+
await new Promise(resolve => server.listen(0, resolve));
|
|
42
|
+
const base = `http://127.0.0.1:${server.address().port}`;
|
|
43
|
+
const browser = await chromium.launch();
|
|
44
|
+
const context = await browser.newContext();
|
|
45
|
+
let hangs = 0;
|
|
46
|
+
const ROUNDS = 10;
|
|
47
|
+
for (let round = 0; round < ROUNDS; round++) {
|
|
48
|
+
const page = await context.newPage();
|
|
49
|
+
await page.goto(`${base}/a.html`, {waitUntil: 'load'});
|
|
50
|
+
// An evaluate for the meta refresh to interrupt (it rejects with
|
|
51
|
+
// "Execution context was destroyed" — expected and irrelevant).
|
|
52
|
+
await page.evaluate(
|
|
53
|
+
() => new Promise(resolve => setTimeout(resolve, 100))
|
|
54
|
+
).catch(() => {});
|
|
55
|
+
const outcome = await settles(page.close(), 10000);
|
|
56
|
+
console.log(`round ${round}: page.close() ${outcome}`);
|
|
57
|
+
if (outcome === 'NEVER SETTLED') {
|
|
58
|
+
hangs++;
|
|
59
|
+
}
|
|
60
|
+
}
|
|
61
|
+
console.log(`${hangs}/${ROUNDS} rounds hung`);
|
|
62
|
+
await Promise.race([browser.close(), new Promise(r => setTimeout(r, 5000))]);
|
|
63
|
+
server.close();
|
|
64
|
+
process.exit(hangs ? 1 : 0);
|
|
65
|
+
})();
|
|
@@ -0,0 +1,184 @@
|
|
|
1
|
+
/*
|
|
2
|
+
© 2026 Jeff Witt.
|
|
3
|
+
|
|
4
|
+
Licensed under the MIT License. See LICENSE file at the project root or
|
|
5
|
+
https://opensource.org/license/mit/ for details.
|
|
6
|
+
|
|
7
|
+
SPDX-License-Identifier: MIT
|
|
8
|
+
*/
|
|
9
|
+
|
|
10
|
+
/*
|
|
11
|
+
score.js
|
|
12
|
+
Track-A scoring: computes a per-engine, per-ACT-rule confusion matrix from a
|
|
13
|
+
capture JSONL (capture.js output).
|
|
14
|
+
|
|
15
|
+
Usage:
|
|
16
|
+
node validation/act/score.js --in results/act-....jsonl [--band asserted|review|both]
|
|
17
|
+
[--json out.json]
|
|
18
|
+
|
|
19
|
+
Scoring policy (criterion-level comparability layer, v1):
|
|
20
|
+
- An ACT rule's positive criteria are its forConformance WCAG 2.x success
|
|
21
|
+
criteria from the testcase feed. Rules with none (technique/ARIA-only
|
|
22
|
+
rules) are reported separately as unscoreable.
|
|
23
|
+
- An engine FLAGS a testcase when it reports ≥1 finding on any of the rule's
|
|
24
|
+
criteria. Band `asserted` (default) counts definite failures only (standard
|
|
25
|
+
instance outcome `failed`); `review` counts engine-flagged uncertainty
|
|
26
|
+
(outcome `cantTell`); `both` counts either.
|
|
27
|
+
- failed testcases are the positive class; passed + inapplicable are the
|
|
28
|
+
negative class. Sensitivity = TP/(TP+FN); specificity = TN/(TN+FP).
|
|
29
|
+
Known generosities/strictnesses, stated wherever results publish: a finding
|
|
30
|
+
on the right criterion from an unrelated check still counts (generous); a
|
|
31
|
+
real finding reported under a different criterion does not (strict).
|
|
32
|
+
*/
|
|
33
|
+
|
|
34
|
+
// IMPORTS
|
|
35
|
+
|
|
36
|
+
const fs = require('fs');
|
|
37
|
+
const path = require('path');
|
|
38
|
+
|
|
39
|
+
// FUNCTIONS
|
|
40
|
+
|
|
41
|
+
const parseArgs = argv => {
|
|
42
|
+
const args = {};
|
|
43
|
+
for (let i = 2; i < argv.length; i += 2) {
|
|
44
|
+
args[argv[i].replace(/^--/, '')] = argv[i + 1];
|
|
45
|
+
}
|
|
46
|
+
return args;
|
|
47
|
+
};
|
|
48
|
+
|
|
49
|
+
const percent = (numerator, denominator) =>
|
|
50
|
+
denominator ? `${(100 * numerator / denominator).toFixed(1)}%` : '—';
|
|
51
|
+
|
|
52
|
+
// OPERATION
|
|
53
|
+
|
|
54
|
+
const args = parseArgs(process.argv);
|
|
55
|
+
if (! args.in) {
|
|
56
|
+
console.log('Usage: node validation/act/score.js --in <capture.jsonl> [--band asserted|review|both]');
|
|
57
|
+
process.exit(1);
|
|
58
|
+
}
|
|
59
|
+
const band = args.band || 'asserted';
|
|
60
|
+
const rows = fs.readFileSync(args.in, 'utf8')
|
|
61
|
+
.split('\n')
|
|
62
|
+
.filter(Boolean)
|
|
63
|
+
.map(line => JSON.parse(line));
|
|
64
|
+
|
|
65
|
+
// ACT-rule criterion sets from the cached feed.
|
|
66
|
+
const feed = JSON.parse(
|
|
67
|
+
fs.readFileSync(path.join(__dirname, 'cache', 'testcases.json'), 'utf8')
|
|
68
|
+
);
|
|
69
|
+
const ruleCriteria = {};
|
|
70
|
+
const ruleNames = {};
|
|
71
|
+
feed.testcases.forEach(testcase => {
|
|
72
|
+
ruleNames[testcase.ruleId] = testcase.ruleName;
|
|
73
|
+
if (! ruleCriteria[testcase.ruleId]) {
|
|
74
|
+
const criteria = Object.entries(testcase.ruleAccessibilityRequirements || {})
|
|
75
|
+
.filter(([key, value]) => /^wcag2\d:/.test(key) && value && value.forConformance)
|
|
76
|
+
.map(([key]) => key.split(':')[1]);
|
|
77
|
+
ruleCriteria[testcase.ruleId] = new Set(criteria);
|
|
78
|
+
}
|
|
79
|
+
});
|
|
80
|
+
|
|
81
|
+
// Whether a capture row flags its testcase under the scoring policy. Rows
|
|
82
|
+
// with exact ACT-rule maps (actAsserted/actReview — qualWeb) are scored on
|
|
83
|
+
// those; rows scored at the exact level bypass the unscoreable-criteria gate
|
|
84
|
+
// since the mapping needs no WCAG SC bridge.
|
|
85
|
+
const flags = row => {
|
|
86
|
+
if (row.actAsserted || row.actReview) {
|
|
87
|
+
const buckets = [];
|
|
88
|
+
if (band === 'asserted' || band === 'both') {
|
|
89
|
+
buckets.push(row.actAsserted || {});
|
|
90
|
+
}
|
|
91
|
+
if (band === 'review' || band === 'both') {
|
|
92
|
+
buckets.push(row.actReview || {});
|
|
93
|
+
}
|
|
94
|
+
return buckets.some(bucket => !! bucket[row.ruleId]);
|
|
95
|
+
}
|
|
96
|
+
const criteria = ruleCriteria[row.ruleId];
|
|
97
|
+
if (! criteria || ! criteria.size) {
|
|
98
|
+
return null;
|
|
99
|
+
}
|
|
100
|
+
const buckets = [];
|
|
101
|
+
if (band === 'asserted' || band === 'both') {
|
|
102
|
+
buckets.push(row.asserted || {});
|
|
103
|
+
}
|
|
104
|
+
if (band === 'review' || band === 'both') {
|
|
105
|
+
buckets.push(row.review || {});
|
|
106
|
+
}
|
|
107
|
+
return buckets.some(
|
|
108
|
+
bucket => Object.entries(bucket).some(([criterion, count]) => count && criteria.has(criterion))
|
|
109
|
+
);
|
|
110
|
+
};
|
|
111
|
+
|
|
112
|
+
// Tally per engine × rule.
|
|
113
|
+
const tallies = {};
|
|
114
|
+
const unscoreable = new Set();
|
|
115
|
+
const errored = {};
|
|
116
|
+
rows.forEach(row => {
|
|
117
|
+
if (row.prevented) {
|
|
118
|
+
errored[row.engine] = (errored[row.engine] || 0) + 1;
|
|
119
|
+
return;
|
|
120
|
+
}
|
|
121
|
+
const flagged = flags(row);
|
|
122
|
+
if (flagged === null) {
|
|
123
|
+
unscoreable.add(row.ruleId);
|
|
124
|
+
return;
|
|
125
|
+
}
|
|
126
|
+
tallies[row.engine] ??= {};
|
|
127
|
+
const tally = tallies[row.engine][row.ruleId] ??= {TP: 0, FP: 0, TN: 0, FN: 0};
|
|
128
|
+
if (row.expected === 'failed') {
|
|
129
|
+
tally[flagged ? 'TP' : 'FN']++;
|
|
130
|
+
}
|
|
131
|
+
else {
|
|
132
|
+
tally[flagged ? 'FP' : 'TN']++;
|
|
133
|
+
}
|
|
134
|
+
});
|
|
135
|
+
|
|
136
|
+
// Report.
|
|
137
|
+
const report = {band, engines: {}};
|
|
138
|
+
Object.entries(tallies).forEach(([engine, ruleTallies]) => {
|
|
139
|
+
console.log(`\n## ${engine} (band: ${band})\n`);
|
|
140
|
+
console.log('| ACT rule | Name | TP | FN | FP | TN | Sens | Spec |');
|
|
141
|
+
console.log('| --- | --- | --- | --- | --- | --- | --- | --- |');
|
|
142
|
+
const totals = {TP: 0, FP: 0, TN: 0, FN: 0};
|
|
143
|
+
const perRule = {};
|
|
144
|
+
Object.entries(ruleTallies)
|
|
145
|
+
.sort(([, a], [, b]) => (b.TP + b.FN) - (a.TP + a.FN))
|
|
146
|
+
.forEach(([ruleId, tally]) => {
|
|
147
|
+
['TP', 'FP', 'TN', 'FN'].forEach(key => {
|
|
148
|
+
totals[key] += tally[key];
|
|
149
|
+
});
|
|
150
|
+
perRule[ruleId] = {
|
|
151
|
+
...tally,
|
|
152
|
+
sensitivity: tally.TP + tally.FN ? tally.TP / (tally.TP + tally.FN) : null,
|
|
153
|
+
specificity: tally.TN + tally.FP ? tally.TN / (tally.TN + tally.FP) : null
|
|
154
|
+
};
|
|
155
|
+
// Rows where the engine saw nothing at all and nothing was expected are
|
|
156
|
+
// uninformative for display; keep them in totals and JSON regardless.
|
|
157
|
+
if (tally.TP + tally.FN === 0 && tally.FP === 0) {
|
|
158
|
+
return;
|
|
159
|
+
}
|
|
160
|
+
console.log(
|
|
161
|
+
`| ${ruleId} | ${ruleNames[ruleId].slice(0, 45)} | ${tally.TP} | ${tally.FN} | ${tally.FP} `
|
|
162
|
+
+ `| ${tally.TN} | ${percent(tally.TP, tally.TP + tally.FN)} `
|
|
163
|
+
+ `| ${percent(tally.TN, tally.TN + tally.FP)} |`
|
|
164
|
+
);
|
|
165
|
+
});
|
|
166
|
+
console.log(
|
|
167
|
+
`| **all** | | ${totals.TP} | ${totals.FN} | ${totals.FP} | ${totals.TN} `
|
|
168
|
+
+ `| ${percent(totals.TP, totals.TP + totals.FN)} `
|
|
169
|
+
+ `| ${percent(totals.TN, totals.TN + totals.FP)} |`
|
|
170
|
+
);
|
|
171
|
+
report.engines[engine] = {totals, perRule, errored: errored[engine] || 0};
|
|
172
|
+
});
|
|
173
|
+
if (unscoreable.size) {
|
|
174
|
+
console.log(
|
|
175
|
+
`\nUnscoreable (no forConformance WCAG 2.x SC): ${[...unscoreable].join(', ')}`
|
|
176
|
+
);
|
|
177
|
+
}
|
|
178
|
+
Object.entries(errored).forEach(([engine, count]) => {
|
|
179
|
+
console.log(`Errored/prevented rows for ${engine}: ${count}`);
|
|
180
|
+
});
|
|
181
|
+
if (args.json) {
|
|
182
|
+
fs.writeFileSync(args.json, JSON.stringify(report, null, 2));
|
|
183
|
+
console.log(`\nJSON → ${args.json}`);
|
|
184
|
+
}
|
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
# Stage-3a stress report — pour on 300 real pages (2026-08-22)
|
|
2
|
+
|
|
3
|
+
Capture: `results/stage3a-2026-08-22.jsonl` — 300 pages (250 seeded-random US CrUX tail + 50
|
|
4
|
+
top-1000 head; `yra-100k/data/stage3a-sample-300.txt`, seed 42) × pour, axe, ed11y, qualWeb, via
|
|
5
|
+
the hardened URL-mode harness (per-instance element paths captured for Stage 3b). The run
|
|
6
|
+
completed unattended: 4 canary-triggered browser replacements, zero stalls.
|
|
7
|
+
|
|
8
|
+
## Gate verdict: PASS
|
|
9
|
+
|
|
10
|
+
| Engine | Raw ran-rate | Effective ran-rate¹ | p50 | p95 | Instances |
|
|
11
|
+
| --- | --- | --- | --- | --- | --- |
|
|
12
|
+
| pour | 96.3% | **99.0%** | 2.0 s | 7.7 s | 50,123 |
|
|
13
|
+
| axe | 97.0% | 99.7% | 1.7 s | 6.5 s | 32,071 |
|
|
14
|
+
| ed11y | 96.7% | 99.3% | 1.0 s | 4.3 s | 6,029 |
|
|
15
|
+
| qualWeb | 94.7% | 97.3% | 4.1 s | 13.4 s | 113,656 |
|
|
16
|
+
|
|
17
|
+
¹ Excluding the 8 pages unreachable for **every** engine (dead/hostile sites — page problems,
|
|
18
|
+
not engine problems). Pour's gate was ≥95%: passed on either denominator. Latency is well inside
|
|
19
|
+
the worker budget. No trigger-happy rule: nothing asserts on >90% of pages (max:
|
|
20
|
+
`target-size-enhanced`, 86% — an AAA rule where near-ubiquity is plausible; flagged for tic
|
|
21
|
+
quality review, not a defect).
|
|
22
|
+
|
|
23
|
+
## Prevention taxonomy (engine-specific residue is small)
|
|
24
|
+
|
|
25
|
+
Navigation failures/timeouts dominate (6–7 per engine — the dead-site class). Residue:
|
|
26
|
+
`navigated-during-eval` 1–3/engine (meta-refresh-class pages; results legitimately unobtainable);
|
|
27
|
+
reporter timeouts on 2 extremely heavy pages (pour 1, ed11y 3); qualWeb "No DOM" ×7 (its own
|
|
28
|
+
HTML ingestion fails on some pages — an incumbent robustness datum, not a candidate problem);
|
|
29
|
+
one axe adapter TypeError on one page (page-specific, pre-existing adapter code). Zero CSP
|
|
30
|
+
preventions — the nonce passthrough works in the wild.
|
|
31
|
+
|
|
32
|
+
## Volume anatomy — the kitchen-sink effect
|
|
33
|
+
|
|
34
|
+
Pour's headline 50k instances is dominated by AAA and best-practice rules (~30k: contrast-
|
|
35
|
+
enhanced 13.1k, target-size-enhanced 9.2k, region 8.1k), a direct consequence of the adapter's
|
|
36
|
+
run-every-rule setting (matching axe's). The A/AA-comparable picture is much closer, and on the
|
|
37
|
+
single biggest A/AA category the two engines nearly agree:
|
|
38
|
+
|
|
39
|
+
- **color-contrast (1.4.3): pour 8,829 instances / 74% of pages vs axe 8,795 / 76%** — near-
|
|
40
|
+
identical at scale.
|
|
41
|
+
- link-name: pour 536/30% vs axe 660/31%. heading-order: 153/30% vs 269/32%.
|
|
42
|
+
- Divergences to watch in Stage 3b: `region` (pour 8.1k vs axe 2.7k — counting granularity) and
|
|
43
|
+
axe's `hidden-content` (11.3k, 87% of pages — its own kitchen-sink review rule).
|
|
44
|
+
|
|
45
|
+
## Assertion-band inversion (the vendor claim, reproduced)
|
|
46
|
+
|
|
47
|
+
Of pour's instances, **72% are asserted violations** (severity 2–3) and 28% review; axe inverts:
|
|
48
|
+
**34% violations, 66% review/incomplete**. This independently reproduces pour.dev's "twice the
|
|
49
|
+
failing elements with fewer check-by-eye verdicts" claim — and it makes Stage-4 triage of pour's
|
|
50
|
+
*asserted* band the critical validation: pour stakes much more on definite assertions than axe
|
|
51
|
+
does.
|
|
52
|
+
|
|
53
|
+
## ACT-identified FP classes at real-world scale
|
|
54
|
+
|
|
55
|
+
- **Contrast** (both rules): 21.9k instances = 44% of pour's volume — the FP mechanisms found on
|
|
56
|
+
ACT fixtures (symbol-only text, letter-as-icon glyphs) live inside this mass. Highest-priority
|
|
57
|
+
triage sample for Stage 4.
|
|
58
|
+
- **Language applicability** (`valid-lang-parts`): fired on **zero** real pages — the ACT FP
|
|
59
|
+
class is real but rare in the wild. `html-lang` asserts on 31 pages (10.7%) — consistent with
|
|
60
|
+
WebAIM's ~13% missing-language prevalence, so likely mostly true positives.
|
|
61
|
+
- `button-name` 111 instances / 32 pages, `link-name` 536 / 88 pages — moderate volumes; the
|
|
62
|
+
known name-computation gap (descendant `aria-labelledby`) warrants a targeted triage slice.
|
|
63
|
+
|
|
64
|
+
## Next (Stage 3b)
|
|
65
|
+
|
|
66
|
+
Per-instance element paths are already captured for all four engines. Run
|
|
67
|
+
`propose-tic-mappings`-style co-occurrence between pour rule IDs and the mapped incumbents'
|
|
68
|
+
issues over this same JSONL; disposition all 70 pour rules that fired (plus the ~16 that
|
|
69
|
+
didn't); then Stage-4 triage sampling with the contrast and link/button-name slices prioritized.
|