blastproof 0.4.0 → 0.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +8 -2
- package/dist/cli.js +40 -10
- package/dist/cli.js.map +1 -1
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -44,8 +44,14 @@ Before `run`, `plan` or `test` do anything, they check what they are about to sp
|
|
|
44
44
|
|
|
45
45
|
**Not supported yet:** `iframe` content (so hosted payment widgets like Stripe Elements are invisible — an embedded checkout cannot be driven end to end), hover, scroll-to, drag and drop, file upload, multiple tabs, native `alert`/`confirm` dialogs. Page snapshots are capped at 200 lines by default, so very dense pages are truncated — raise it with `browser.max_snapshot_lines` if your pages need more; truncation is always marked in the snapshot so the model is never misled into thinking it saw the whole page.
|
|
46
46
|
|
|
47
|
+
**Point it at disposable data.** A step that fails is retried, and the agent decides how to recover — which can mean performing the step's action again. A test that submits a form and then fails its check may submit that form more than once, so a run against real data can leave more records behind than the journey describes. In one verification run a badly written step produced three issues where the test intended one. Use a seeded database, a staging environment you can reset, or a throwaway account; do not gate on a run against production data.
|
|
48
|
+
|
|
47
49
|
`browser.timeout_ms` bounds every wait — resolving a target element from the accessibility tree, and navigation — not only the click or fill performed afterwards. Raise it for an application that is merely slow to hydrate; the trade-off is that a genuinely missing element then takes longer to fail. It never changes how many self-healing retries a step gets — waiting and retrying are deliberately separate.
|
|
48
50
|
|
|
51
|
+
Writing each step so it states its own outcome helps here as well as everywhere else: `submit the form, then verify the confirmation shows the reference number` gives the agent something to check, where `click the submit button` leaves it to invent an expectation — and a poor invented expectation is what turns one submission into three.
|
|
52
|
+
|
|
53
|
+
**This is now load-bearing, not just advisable.** The judge decides whether a step's own outcome holds, using the model's expectation only as the claim offered in support of it — a step that never says what its outcome is gives the judge nothing to anchor on beyond whatever the model happened to check that turn. `verify the confirmation shows the reference number` gives the judge a real question; `verify it worked` does not, and may now fail where a looser judge previously let a true-but-unrelated claim pass it.
|
|
54
|
+
|
|
49
55
|
## Writing tests
|
|
50
56
|
|
|
51
57
|
Tests live in `.blastproof/tests/` as plain-English YAML — no selectors:
|
|
@@ -107,9 +113,9 @@ jobs:
|
|
|
107
113
|
|
|
108
114
|
- run: npm start & # however your app boots
|
|
109
115
|
|
|
110
|
-
- uses: hamc/blastproof@v0.
|
|
116
|
+
- uses: hamc/blastproof@v0.6.0
|
|
111
117
|
with:
|
|
112
|
-
version: '0.
|
|
118
|
+
version: '0.6.0' # pin both when this gates merges
|
|
113
119
|
api-key: ${{ secrets.ANTHROPIC_API_KEY }}
|
|
114
120
|
base: ${{ github.event.pull_request.base.ref }}
|
|
115
121
|
min-score: '80'
|
package/dist/cli.js
CHANGED
|
@@ -256,7 +256,7 @@ function estimateMaxModelCalls(tests, maxIterationsPerStep, maxRetriesPerStep) {
|
|
|
256
256
|
(sum, test) => sum + (test.setup?.length ?? 0) + test.steps.length,
|
|
257
257
|
0
|
|
258
258
|
);
|
|
259
|
-
return totalSteps * (maxIterationsPerStep + maxRetriesPerStep);
|
|
259
|
+
return totalSteps * (maxIterationsPerStep + maxRetriesPerStep + Math.min(maxIterationsPerStep, maxRetriesPerStep));
|
|
260
260
|
}
|
|
261
261
|
|
|
262
262
|
// src/runner/env.ts
|
|
@@ -439,6 +439,19 @@ async function performAction(page, action, ctx) {
|
|
|
439
439
|
|
|
440
440
|
// src/runner/executor.ts
|
|
441
441
|
var DEFAULT_MAX_ITERATIONS_PER_STEP = 15;
|
|
442
|
+
var SETTLE_TIMEOUT_MS = 2e3;
|
|
443
|
+
async function waitForSettled(page) {
|
|
444
|
+
const deadline = Date.now() + SETTLE_TIMEOUT_MS;
|
|
445
|
+
const urlBefore = page.url();
|
|
446
|
+
await page.waitForLoadState("networkidle", { timeout: SETTLE_TIMEOUT_MS }).catch(() => {
|
|
447
|
+
});
|
|
448
|
+
if (page.url() === urlBefore) return;
|
|
449
|
+
const remaining = deadline - Date.now();
|
|
450
|
+
if (remaining > 0) {
|
|
451
|
+
await page.waitForLoadState("networkidle", { timeout: remaining }).catch(() => {
|
|
452
|
+
});
|
|
453
|
+
}
|
|
454
|
+
}
|
|
442
455
|
async function defaultSnapshot(page, maxLines) {
|
|
443
456
|
const { captureSnapshot } = await import("./snapshot-CAIB2OHX.js");
|
|
444
457
|
return captureSnapshot(page, { maxLines });
|
|
@@ -496,6 +509,7 @@ async function executeTest(page, test, options) {
|
|
|
496
509
|
if (iterations >= maxIterationsPerStep) {
|
|
497
510
|
throw new StepFailure(`step exceeded ${maxIterationsPerStep} actions without completing`);
|
|
498
511
|
}
|
|
512
|
+
await waitForSettled(page);
|
|
499
513
|
const snap = await takeSnapshot(page);
|
|
500
514
|
let action;
|
|
501
515
|
try {
|
|
@@ -527,7 +541,12 @@ async function executeTest(page, test, options) {
|
|
|
527
541
|
}
|
|
528
542
|
if (action.action === "assert") {
|
|
529
543
|
const expectation = action.expectation ?? action.reasoning;
|
|
530
|
-
|
|
544
|
+
let judgment = await brain.judge(mask(step), mask(expectation), mask(snap));
|
|
545
|
+
if (!judgment.pass) {
|
|
546
|
+
await waitForSettled(page);
|
|
547
|
+
const freshSnap = await takeSnapshot(page);
|
|
548
|
+
judgment = await brain.judge(mask(step), mask(expectation), mask(freshSnap));
|
|
549
|
+
}
|
|
531
550
|
const result = judgment.pass ? `ok: assertion passed: ${judgment.reason}` : `assertion failed: ${judgment.reason}`;
|
|
532
551
|
emitAction(index, action, result);
|
|
533
552
|
if (judgment.pass) {
|
|
@@ -704,7 +723,7 @@ async function fromSteps(options) {
|
|
|
704
723
|
);
|
|
705
724
|
}
|
|
706
725
|
if (auth.verify) {
|
|
707
|
-
const judgment = await brain.judge(auth.verify, mask.mask(await takeSnapshot(page)));
|
|
726
|
+
const judgment = await brain.judge(auth.verify, auth.verify, mask.mask(await takeSnapshot(page)));
|
|
708
727
|
if (!judgment.pass) {
|
|
709
728
|
throw new AuthError(`Authentication could not be verified: ${mask.mask(judgment.reason)}`);
|
|
710
729
|
}
|
|
@@ -1033,6 +1052,7 @@ Rules:
|
|
|
1033
1052
|
- Return "done" when the current step's outcome holds \u2014 including when it already held before you acted, or was achieved by your previous action. "Already true" is done, never failure. Do not return "done" for work belonging to later steps.
|
|
1034
1053
|
- Return "fail" only when the step's outcome cannot be reached: the element is still absent after retries, the page cannot support the step, or an error blocks progress. Never return "fail" because the work appears to have been done already.
|
|
1035
1054
|
- If your previous action errored, re-read the fresh snapshot and choose an alternative element or approach. Do not repeat the exact same failing action.
|
|
1055
|
+
- \`***\` in a snapshot is a redacted secret \u2014 a password, token or key deliberately withheld from you. Seeing it is expected and is not a problem. A field showing \`***\` after you filled it from an {{env.VAR}} placeholder means the fill worked; treat that as success and move on. Never retry a fill because its value is redacted, and never report failure because a value was withheld.
|
|
1036
1056
|
- Keep reasoning to one short sentence.`;
|
|
1037
1057
|
}
|
|
1038
1058
|
function agentUserPrompt(input) {
|
|
@@ -1053,16 +1073,26 @@ function agentUserPrompt(input) {
|
|
|
1053
1073
|
return parts.join("\n");
|
|
1054
1074
|
}
|
|
1055
1075
|
function assertSystemPrompt() {
|
|
1056
|
-
return `You are a QA judge. You receive a
|
|
1076
|
+
return `You are a QA judge. You receive a test step, the model's expectation for the current page, and a page accessibility snapshot. Decide whether the STEP's own outcome holds \u2014 the expectation is the claim the model is offering in support of that, not a substitute question of its own. A claim can be true of the snapshot and still fail the step, if it does not establish what the step actually describes: only pass when the snapshot itself shows the step's outcome, never merely because the expectation offered happens to be true of something else on the page.
|
|
1077
|
+
|
|
1078
|
+
A value sitting in a control that was just typed into \u2014 an open dialog's textbox, an unsubmitted form field \u2014 is not the same as a committed outcome. When the step describes an outcome (something now appears in a list, is saved, is confirmed, is created), text visible only inside an editable, not-yet-submitted control does not satisfy it; look for the outcome committed outside that control (the dialog closed, the item is listed on its own, a confirmation appeared). This is specifically about that confusion, not a license to fail anything you are merely unsure about \u2014 if the snapshot plainly shows the step's outcome, pass it.
|
|
1079
|
+
|
|
1080
|
+
A step that names an ACTION (submit, click, create, add, ...) is satisfied by evidence the action took effect, not by the action's own control still being on the page. A successful action ordinarily replaces or moves past exactly the form, button or field the step names, so that control's absence is normal evidence of success, not evidence the step is unverifiable \u2014 do not fail such a step only because you can no longer see the thing it names. Fail it instead when the snapshot shows the action did NOT take effect: an error message, a validation warning, or the very same pre-action page still in front of you with nothing changed. A different page, a new state, or the result the action was meant to produce counts as evidence it worked.
|
|
1081
|
+
|
|
1082
|
+
Be strict about what the step asks, not about withholding a pass you can plainly see is earned. Answer with pass=true/false and a one-sentence reason.
|
|
1083
|
+
|
|
1084
|
+
\`***\` marks a secret deliberately withheld from you \u2014 a password, token or key. Seeing it is expected. A field holding \`***\` is filled, not empty, so do not fail a step on the grounds that a value was redacted. This applies only to the redaction itself: everything else the step asks for must still be visibly satisfied by the snapshot, and a step you genuinely cannot check against what you were shown still fails.`;
|
|
1057
1085
|
}
|
|
1058
|
-
function assertUserPrompt(expectation, snapshot) {
|
|
1086
|
+
function assertUserPrompt(step, expectation, snapshot) {
|
|
1059
1087
|
return [
|
|
1060
|
-
`
|
|
1088
|
+
`Step under test: ${step}`,
|
|
1089
|
+
"",
|
|
1090
|
+
`Model's expectation (the claim offered in support of the step, not the question itself): ${expectation}`,
|
|
1061
1091
|
"",
|
|
1062
1092
|
"Page accessibility snapshot:",
|
|
1063
1093
|
snapshot,
|
|
1064
1094
|
"",
|
|
1065
|
-
"Does the snapshot
|
|
1095
|
+
"Does the snapshot establish that the step's own outcome holds?"
|
|
1066
1096
|
].join("\n");
|
|
1067
1097
|
}
|
|
1068
1098
|
function plannerSystemPrompt() {
|
|
@@ -1152,12 +1182,12 @@ function createBrain(model, generate = generateObject, budget) {
|
|
|
1152
1182
|
}
|
|
1153
1183
|
return parsed.data;
|
|
1154
1184
|
},
|
|
1155
|
-
async judge(expectation, snapshot) {
|
|
1185
|
+
async judge(step, expectation, snapshot) {
|
|
1156
1186
|
const result = await countedGenerate(generate, budget, {
|
|
1157
1187
|
model,
|
|
1158
1188
|
schema: assertJudgmentSchema,
|
|
1159
1189
|
system: assertSystemPrompt(),
|
|
1160
|
-
prompt: assertUserPrompt(expectation, snapshot)
|
|
1190
|
+
prompt: assertUserPrompt(step, expectation, snapshot)
|
|
1161
1191
|
});
|
|
1162
1192
|
const parsed = assertJudgmentSchema.safeParse(result.object);
|
|
1163
1193
|
if (!parsed.success) {
|
|
@@ -2534,7 +2564,7 @@ function parsePositiveNumber(flag) {
|
|
|
2534
2564
|
};
|
|
2535
2565
|
}
|
|
2536
2566
|
var program = new Command();
|
|
2537
|
-
program.name("blastproof").description("Open-source AI testing agent: plain-English YAML tests executed agentically on a real browser.").version("0.
|
|
2567
|
+
program.name("blastproof").description("Open-source AI testing agent: plain-English YAML tests executed agentically on a real browser.").version("0.6.0");
|
|
2538
2568
|
program.command("init").description("Scaffold .blastproof/ (config, tests, sample tests) in the current directory").action(async () => {
|
|
2539
2569
|
try {
|
|
2540
2570
|
const result = await initProject(process.cwd());
|