blastproof 0.5.0 → 0.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +8 -2
- package/dist/cli.js +99 -20
- package/dist/cli.js.map +1 -1
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -44,8 +44,14 @@ Before `run`, `plan` or `test` do anything, they check what they are about to sp
|
|
|
44
44
|
|
|
45
45
|
**Not supported yet:** `iframe` content (so hosted payment widgets like Stripe Elements are invisible — an embedded checkout cannot be driven end to end), hover, scroll-to, drag and drop, file upload, multiple tabs, native `alert`/`confirm` dialogs. Page snapshots are capped at 200 lines by default, so very dense pages are truncated — raise it with `browser.max_snapshot_lines` if your pages need more; truncation is always marked in the snapshot so the model is never misled into thinking it saw the whole page.
|
|
46
46
|
|
|
47
|
+
**Point it at disposable data.** Within a step, an action that commits — a click, or pressing Enter — is never performed twice: the runner refuses the repeat and tells the agent it already did that. This closes the case that used to produce duplicate records, where a submit answered by a redirect came back to a reset form and the agent, seeing no evidence of its own work, submitted again. It is not a guarantee of zero duplicate writes: an agent that reaches the same effect by a genuinely different route — another control with the same effect — is not caught. Use a seeded database, a staging environment you can reset, or a throwaway account; do not gate on a run against production data.
|
|
48
|
+
|
|
47
49
|
`browser.timeout_ms` bounds every wait — resolving a target element from the accessibility tree, and navigation — not only the click or fill performed afterwards. Raise it for an application that is merely slow to hydrate; the trade-off is that a genuinely missing element then takes longer to fail. It never changes how many self-healing retries a step gets — waiting and retrying are deliberately separate.
|
|
48
50
|
|
|
51
|
+
Writing each step so it states its own outcome helps here as well as everywhere else: `submit the form, then verify the confirmation shows the reference number` gives the agent something to check, where `click the submit button` leaves it to invent an expectation — and a poor invented expectation is what turns one submission into three.
|
|
52
|
+
|
|
53
|
+
**This is now load-bearing, not just advisable.** The judge decides whether a step's own outcome holds, using the model's expectation only as the claim offered in support of it — a step that never says what its outcome is gives the judge nothing to anchor on beyond whatever the model happened to check that turn. `verify the confirmation shows the reference number` gives the judge a real question; `verify it worked` does not, and may now fail where a looser judge previously let a true-but-unrelated claim pass it.
|
|
54
|
+
|
|
49
55
|
## Writing tests
|
|
50
56
|
|
|
51
57
|
Tests live in `.blastproof/tests/` as plain-English YAML — no selectors:
|
|
@@ -107,9 +113,9 @@ jobs:
|
|
|
107
113
|
|
|
108
114
|
- run: npm start & # however your app boots
|
|
109
115
|
|
|
110
|
-
- uses: hamc/blastproof@v0.
|
|
116
|
+
- uses: hamc/blastproof@v0.7.0
|
|
111
117
|
with:
|
|
112
|
-
version: '0.
|
|
118
|
+
version: '0.6.0' # pin both when this gates merges
|
|
113
119
|
api-key: ${{ secrets.ANTHROPIC_API_KEY }}
|
|
114
120
|
base: ${{ github.event.pull_request.base.ref }}
|
|
115
121
|
min-score: '80'
|
package/dist/cli.js
CHANGED
|
@@ -437,6 +437,59 @@ async function performAction(page, action, ctx) {
|
|
|
437
437
|
}
|
|
438
438
|
}
|
|
439
439
|
|
|
440
|
+
// src/runner/recovery.ts
|
|
441
|
+
var COMMIT_ACTIONS = /* @__PURE__ */ new Set(["click", "press"]);
|
|
442
|
+
var COMMIT_KEYS = /* @__PURE__ */ new Set(["Enter", "NumpadEnter", " ", "Space", "Spacebar"]);
|
|
443
|
+
function describeAction(action) {
|
|
444
|
+
const target = action.target ? ` ${action.target.role ?? ""} "${action.target.name ?? action.target.text ?? ""}"` : "";
|
|
445
|
+
const value = action.value ? ` [${action.value}]` : "";
|
|
446
|
+
return `${action.action}${target}${value}`;
|
|
447
|
+
}
|
|
448
|
+
function identity(action) {
|
|
449
|
+
return JSON.stringify([
|
|
450
|
+
action.action,
|
|
451
|
+
action.target?.role ?? "",
|
|
452
|
+
action.target?.name ?? "",
|
|
453
|
+
action.target?.text ?? "",
|
|
454
|
+
action.value ?? ""
|
|
455
|
+
]);
|
|
456
|
+
}
|
|
457
|
+
var StepRecovery = class {
|
|
458
|
+
performed = /* @__PURE__ */ new Set();
|
|
459
|
+
history = [];
|
|
460
|
+
/** Records an action that was actually performed and succeeded. */
|
|
461
|
+
record(action, description, result) {
|
|
462
|
+
this.performed.add(identity(action));
|
|
463
|
+
this.history.push({ action: description, result });
|
|
464
|
+
}
|
|
465
|
+
/**
|
|
466
|
+
* The reason to refuse `action`, or `undefined` when it may be performed.
|
|
467
|
+
*
|
|
468
|
+
* Refusing rather than failing keeps the model in the loop with an
|
|
469
|
+
* explanation instead of ending the step outright, which would trade
|
|
470
|
+
* duplicate writes for false failures.
|
|
471
|
+
*
|
|
472
|
+
* A genuine retry — a commit that landed but had no effect, which the model
|
|
473
|
+
* repeats for good reason — is refused too, and that is the deliberate cost.
|
|
474
|
+
* The two cases are indistinguishable from the accessibility tree: "the
|
|
475
|
+
* click did nothing" and "the click worked and the redirect erased the
|
|
476
|
+
* proof" produce the same snapshot. The asymmetry decides it. A refused
|
|
477
|
+
* legitimate retry costs a visible failed step that someone investigates; an
|
|
478
|
+
* allowed duplicate commit costs a silent extra row in someone's database,
|
|
479
|
+
* and #28 has now produced one on three applications.
|
|
480
|
+
*/
|
|
481
|
+
refusalFor(action) {
|
|
482
|
+
if (!COMMIT_ACTIONS.has(action.action)) return void 0;
|
|
483
|
+
if (action.action === "press" && !COMMIT_KEYS.has(action.value ?? "")) return void 0;
|
|
484
|
+
if (!this.performed.has(identity(action))) return void 0;
|
|
485
|
+
return `refused: this exact action already succeeded earlier in this step, so it was NOT performed again. Repeating something that commits repeats whatever it changed in the application. If the page no longer shows that it worked, that is normal for a submit answered by a redirect \u2014 check the record of what you have already done. Verify the step's outcome another way, or fail the step.`;
|
|
486
|
+
}
|
|
487
|
+
/** The step's history so far, oldest first, for the model's prompt. */
|
|
488
|
+
stepHistory() {
|
|
489
|
+
return this.history;
|
|
490
|
+
}
|
|
491
|
+
};
|
|
492
|
+
|
|
440
493
|
// src/runner/executor.ts
|
|
441
494
|
var DEFAULT_MAX_ITERATIONS_PER_STEP = 15;
|
|
442
495
|
var SETTLE_TIMEOUT_MS = 2e3;
|
|
@@ -504,6 +557,7 @@ async function executeTest(page, test, options) {
|
|
|
504
557
|
let failedAttempts = 0;
|
|
505
558
|
let lastResult;
|
|
506
559
|
let stepFailedReason;
|
|
560
|
+
const recovery = new StepRecovery();
|
|
507
561
|
try {
|
|
508
562
|
while (true) {
|
|
509
563
|
if (iterations >= maxIterationsPerStep) {
|
|
@@ -518,6 +572,9 @@ async function executeTest(page, test, options) {
|
|
|
518
572
|
isSetup: setup,
|
|
519
573
|
snapshot: mask(snap),
|
|
520
574
|
lastResult: lastResult === void 0 ? void 0 : mask(lastResult),
|
|
575
|
+
// Already masked when recorded, on the same boundary as everything
|
|
576
|
+
// else crossing into a prompt (design contained-recovery, D2).
|
|
577
|
+
stepHistory: recovery.stepHistory(),
|
|
521
578
|
retriesLeft: maxRetries - failedAttempts,
|
|
522
579
|
iterationsLeft: maxIterationsPerStep - iterations
|
|
523
580
|
});
|
|
@@ -541,11 +598,11 @@ async function executeTest(page, test, options) {
|
|
|
541
598
|
}
|
|
542
599
|
if (action.action === "assert") {
|
|
543
600
|
const expectation = action.expectation ?? action.reasoning;
|
|
544
|
-
let judgment = await brain.judge(mask(expectation), mask(snap));
|
|
601
|
+
let judgment = await brain.judge(mask(step), mask(expectation), mask(snap));
|
|
545
602
|
if (!judgment.pass) {
|
|
546
603
|
await waitForSettled(page);
|
|
547
604
|
const freshSnap = await takeSnapshot(page);
|
|
548
|
-
judgment = await brain.judge(mask(expectation), mask(freshSnap));
|
|
605
|
+
judgment = await brain.judge(mask(step), mask(expectation), mask(freshSnap));
|
|
549
606
|
}
|
|
550
607
|
const result = judgment.pass ? `ok: assertion passed: ${judgment.reason}` : `assertion failed: ${judgment.reason}`;
|
|
551
608
|
emitAction(index, action, result);
|
|
@@ -560,6 +617,16 @@ async function executeTest(page, test, options) {
|
|
|
560
617
|
}
|
|
561
618
|
continue;
|
|
562
619
|
}
|
|
620
|
+
const refusal = recovery.refusalFor(action);
|
|
621
|
+
if (refusal) {
|
|
622
|
+
failedAttempts++;
|
|
623
|
+
lastResult = refusal;
|
|
624
|
+
emitAction(index, action, refusal);
|
|
625
|
+
if (failedAttempts >= maxRetries) {
|
|
626
|
+
throw new StepFailure(refusal);
|
|
627
|
+
}
|
|
628
|
+
continue;
|
|
629
|
+
}
|
|
563
630
|
try {
|
|
564
631
|
const result = await performAction(page, action, {
|
|
565
632
|
baseUrl,
|
|
@@ -568,6 +635,7 @@ async function executeTest(page, test, options) {
|
|
|
568
635
|
resolveTimeoutMs: timeoutMs
|
|
569
636
|
});
|
|
570
637
|
lastResult = result;
|
|
638
|
+
recovery.record(action, mask(describeAction(action)), mask(result));
|
|
571
639
|
emitAction(index, action, result);
|
|
572
640
|
} catch (error) {
|
|
573
641
|
failedAttempts++;
|
|
@@ -723,7 +791,7 @@ async function fromSteps(options) {
|
|
|
723
791
|
);
|
|
724
792
|
}
|
|
725
793
|
if (auth.verify) {
|
|
726
|
-
const judgment = await brain.judge(auth.verify, mask.mask(await takeSnapshot(page)));
|
|
794
|
+
const judgment = await brain.judge(auth.verify, auth.verify, mask.mask(await takeSnapshot(page)));
|
|
727
795
|
if (!judgment.pass) {
|
|
728
796
|
throw new AuthError(`Authentication could not be verified: ${mask.mask(judgment.reason)}`);
|
|
729
797
|
}
|
|
@@ -1052,16 +1120,21 @@ Rules:
|
|
|
1052
1120
|
- Return "done" when the current step's outcome holds \u2014 including when it already held before you acted, or was achieved by your previous action. "Already true" is done, never failure. Do not return "done" for work belonging to later steps.
|
|
1053
1121
|
- Return "fail" only when the step's outcome cannot be reached: the element is still absent after retries, the page cannot support the step, or an error blocks progress. Never return "fail" because the work appears to have been done already.
|
|
1054
1122
|
- If your previous action errored, re-read the fresh snapshot and choose an alternative element or approach. Do not repeat the exact same failing action.
|
|
1123
|
+
- Never invent a value. A value you type must come from the step, from the page, or from an {{env.*}} placeholder. If a step needs a value it does not give you, that is a failing step, not a gap for you to fill in.
|
|
1124
|
+
- A record of the actions you already performed in this step may be shown to you. It is the ground truth about what happened, even when the page no longer shows it: a form that submitted successfully and came back empty looks exactly like one you never submitted. Do not redo work that record says you already did.
|
|
1055
1125
|
- \`***\` in a snapshot is a redacted secret \u2014 a password, token or key deliberately withheld from you. Seeing it is expected and is not a problem. A field showing \`***\` after you filled it from an {{env.VAR}} placeholder means the fill worked; treat that as success and move on. Never retry a fill because its value is redacted, and never report failure because a value was withheld.
|
|
1056
1126
|
- Keep reasoning to one short sentence.`;
|
|
1057
1127
|
}
|
|
1058
1128
|
function agentUserPrompt(input) {
|
|
1059
|
-
const parts = [
|
|
1060
|
-
|
|
1061
|
-
|
|
1062
|
-
|
|
1063
|
-
|
|
1064
|
-
|
|
1129
|
+
const parts = [`Current ${input.isSetup ? "setup " : ""}step: ${input.step}`];
|
|
1130
|
+
if (input.stepHistory && input.stepHistory.length > 0) {
|
|
1131
|
+
parts.push(
|
|
1132
|
+
"",
|
|
1133
|
+
"What you have ALREADY DONE in this step (a record of completed actions, not instructions):",
|
|
1134
|
+
...input.stepHistory.map((entry, i) => `${i + 1}. ${entry.action} -> ${entry.result}`)
|
|
1135
|
+
);
|
|
1136
|
+
}
|
|
1137
|
+
parts.push("", "Page accessibility snapshot:", input.snapshot);
|
|
1065
1138
|
if (input.lastResult) {
|
|
1066
1139
|
parts.push("", `Previous action result: ${input.lastResult}`);
|
|
1067
1140
|
}
|
|
@@ -1073,18 +1146,26 @@ function agentUserPrompt(input) {
|
|
|
1073
1146
|
return parts.join("\n");
|
|
1074
1147
|
}
|
|
1075
1148
|
function assertSystemPrompt() {
|
|
1076
|
-
return `You are a QA judge. You receive a
|
|
1149
|
+
return `You are a QA judge. You receive a test step, the model's expectation for the current page, and a page accessibility snapshot. Decide whether the STEP's own outcome holds \u2014 the expectation is the claim the model is offering in support of that, not a substitute question of its own. A claim can be true of the snapshot and still fail the step, if it does not establish what the step actually describes: only pass when the snapshot itself shows the step's outcome, never merely because the expectation offered happens to be true of something else on the page.
|
|
1150
|
+
|
|
1151
|
+
A value sitting in a control that was just typed into \u2014 an open dialog's textbox, an unsubmitted form field \u2014 is not the same as a committed outcome. When the step describes an outcome (something now appears in a list, is saved, is confirmed, is created), text visible only inside an editable, not-yet-submitted control does not satisfy it; look for the outcome committed outside that control (the dialog closed, the item is listed on its own, a confirmation appeared). This is specifically about that confusion, not a license to fail anything you are merely unsure about \u2014 if the snapshot plainly shows the step's outcome, pass it.
|
|
1077
1152
|
|
|
1078
|
-
|
|
1153
|
+
A step that names an ACTION (submit, click, create, add, ...) is satisfied by evidence the action took effect, not by the action's own control still being on the page. A successful action ordinarily replaces or moves past exactly the form, button or field the step names, so that control's absence is normal evidence of success, not evidence the step is unverifiable \u2014 do not fail such a step only because you can no longer see the thing it names. Fail it instead when the snapshot shows the action did NOT take effect: an error message, a validation warning, or the very same pre-action page still in front of you with nothing changed. A different page, a new state, or the result the action was meant to produce counts as evidence it worked.
|
|
1154
|
+
|
|
1155
|
+
Be strict about what the step asks, not about withholding a pass you can plainly see is earned. Answer with pass=true/false and a one-sentence reason.
|
|
1156
|
+
|
|
1157
|
+
\`***\` marks a secret deliberately withheld from you \u2014 a password, token or key. Seeing it is expected. A field holding \`***\` is filled, not empty, so do not fail a step on the grounds that a value was redacted. This applies only to the redaction itself: everything else the step asks for must still be visibly satisfied by the snapshot, and a step you genuinely cannot check against what you were shown still fails.`;
|
|
1079
1158
|
}
|
|
1080
|
-
function assertUserPrompt(expectation, snapshot) {
|
|
1159
|
+
function assertUserPrompt(step, expectation, snapshot) {
|
|
1081
1160
|
return [
|
|
1082
|
-
`
|
|
1161
|
+
`Step under test: ${step}`,
|
|
1162
|
+
"",
|
|
1163
|
+
`Model's expectation (the claim offered in support of the step, not the question itself): ${expectation}`,
|
|
1083
1164
|
"",
|
|
1084
1165
|
"Page accessibility snapshot:",
|
|
1085
1166
|
snapshot,
|
|
1086
1167
|
"",
|
|
1087
|
-
"Does the snapshot
|
|
1168
|
+
"Does the snapshot establish that the step's own outcome holds?"
|
|
1088
1169
|
].join("\n");
|
|
1089
1170
|
}
|
|
1090
1171
|
function plannerSystemPrompt() {
|
|
@@ -1174,12 +1255,12 @@ function createBrain(model, generate = generateObject, budget) {
|
|
|
1174
1255
|
}
|
|
1175
1256
|
return parsed.data;
|
|
1176
1257
|
},
|
|
1177
|
-
async judge(expectation, snapshot) {
|
|
1258
|
+
async judge(step, expectation, snapshot) {
|
|
1178
1259
|
const result = await countedGenerate(generate, budget, {
|
|
1179
1260
|
model,
|
|
1180
1261
|
schema: assertJudgmentSchema,
|
|
1181
1262
|
system: assertSystemPrompt(),
|
|
1182
|
-
prompt: assertUserPrompt(expectation, snapshot)
|
|
1263
|
+
prompt: assertUserPrompt(step, expectation, snapshot)
|
|
1183
1264
|
});
|
|
1184
1265
|
const parsed = assertJudgmentSchema.safeParse(result.object);
|
|
1185
1266
|
if (!parsed.success) {
|
|
@@ -1884,9 +1965,7 @@ function printEvent(event) {
|
|
|
1884
1965
|
break;
|
|
1885
1966
|
case "action": {
|
|
1886
1967
|
const { action, result } = event;
|
|
1887
|
-
|
|
1888
|
-
const value = action.value ? ` [${action.value}]` : "";
|
|
1889
|
-
console.log(` -> ${action.action}${target}${value} :: ${result}`);
|
|
1968
|
+
console.log(` -> ${describeAction(action)} :: ${result}`);
|
|
1890
1969
|
break;
|
|
1891
1970
|
}
|
|
1892
1971
|
case "step-end":
|
|
@@ -2556,7 +2635,7 @@ function parsePositiveNumber(flag) {
|
|
|
2556
2635
|
};
|
|
2557
2636
|
}
|
|
2558
2637
|
var program = new Command();
|
|
2559
|
-
program.name("blastproof").description("Open-source AI testing agent: plain-English YAML tests executed agentically on a real browser.").version("0.
|
|
2638
|
+
program.name("blastproof").description("Open-source AI testing agent: plain-English YAML tests executed agentically on a real browser.").version("0.7.0");
|
|
2560
2639
|
program.command("init").description("Scaffold .blastproof/ (config, tests, sample tests) in the current directory").action(async () => {
|
|
2561
2640
|
try {
|
|
2562
2641
|
const result = await initProject(process.cwd());
|