blastproof 0.5.0 → 0.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -44,8 +44,14 @@ Before `run`, `plan` or `test` do anything, they check what they are about to sp
44
44
 
45
45
  **Not supported yet:** `iframe` content (so hosted payment widgets like Stripe Elements are invisible — an embedded checkout cannot be driven end to end), hover, scroll-to, drag and drop, file upload, multiple tabs, native `alert`/`confirm` dialogs. Page snapshots are capped at 200 lines by default, so very dense pages are truncated — raise it with `browser.max_snapshot_lines` if your pages need more; truncation is always marked in the snapshot so the model is never misled into thinking it saw the whole page.
46
46
 
47
+ **Point it at disposable data.** A step that fails is retried, and the agent decides how to recover — which can mean performing the step's action again. A test that submits a form and then fails its check may submit that form more than once, so a run against real data can leave more records behind than the journey describes. In one verification run a badly written step produced three issues where the test intended one. Use a seeded database, a staging environment you can reset, or a throwaway account; do not gate on a run against production data.
48
+
47
49
  `browser.timeout_ms` bounds every wait — resolving a target element from the accessibility tree, and navigation — not only the click or fill performed afterwards. Raise it for an application that is merely slow to hydrate; the trade-off is that a genuinely missing element then takes longer to fail. It never changes how many self-healing retries a step gets — waiting and retrying are deliberately separate.
48
50
 
51
+ Writing each step so it states its own outcome helps here as well as everywhere else: `submit the form, then verify the confirmation shows the reference number` gives the agent something to check, where `click the submit button` leaves it to invent an expectation — and a poor invented expectation is what turns one submission into three.
52
+
53
+ **This is now load-bearing, not just advisable.** The judge decides whether a step's own outcome holds, using the model's expectation only as the claim offered in support of it — a step that never says what its outcome is gives the judge nothing to anchor on beyond whatever the model happened to check that turn. `verify the confirmation shows the reference number` gives the judge a real question; `verify it worked` does not, and may now fail where a looser judge previously let a true-but-unrelated claim pass it.
54
+
49
55
  ## Writing tests
50
56
 
51
57
  Tests live in `.blastproof/tests/` as plain-English YAML — no selectors:
@@ -107,9 +113,9 @@ jobs:
107
113
 
108
114
  - run: npm start & # however your app boots
109
115
 
110
- - uses: hamc/blastproof@v0.5.0
116
+ - uses: hamc/blastproof@v0.6.0
111
117
  with:
112
- version: '0.5.0' # pin both when this gates merges
118
+ version: '0.6.0' # pin both when this gates merges
113
119
  api-key: ${{ secrets.ANTHROPIC_API_KEY }}
114
120
  base: ${{ github.event.pull_request.base.ref }}
115
121
  min-score: '80'
package/dist/cli.js CHANGED
@@ -541,11 +541,11 @@ async function executeTest(page, test, options) {
541
541
  }
542
542
  if (action.action === "assert") {
543
543
  const expectation = action.expectation ?? action.reasoning;
544
- let judgment = await brain.judge(mask(expectation), mask(snap));
544
+ let judgment = await brain.judge(mask(step), mask(expectation), mask(snap));
545
545
  if (!judgment.pass) {
546
546
  await waitForSettled(page);
547
547
  const freshSnap = await takeSnapshot(page);
548
- judgment = await brain.judge(mask(expectation), mask(freshSnap));
548
+ judgment = await brain.judge(mask(step), mask(expectation), mask(freshSnap));
549
549
  }
550
550
  const result = judgment.pass ? `ok: assertion passed: ${judgment.reason}` : `assertion failed: ${judgment.reason}`;
551
551
  emitAction(index, action, result);
@@ -723,7 +723,7 @@ async function fromSteps(options) {
723
723
  );
724
724
  }
725
725
  if (auth.verify) {
726
- const judgment = await brain.judge(auth.verify, mask.mask(await takeSnapshot(page)));
726
+ const judgment = await brain.judge(auth.verify, auth.verify, mask.mask(await takeSnapshot(page)));
727
727
  if (!judgment.pass) {
728
728
  throw new AuthError(`Authentication could not be verified: ${mask.mask(judgment.reason)}`);
729
729
  }
@@ -1073,18 +1073,26 @@ function agentUserPrompt(input) {
1073
1073
  return parts.join("\n");
1074
1074
  }
1075
1075
  function assertSystemPrompt() {
1076
- return `You are a QA judge. You receive a page accessibility snapshot and an expectation. Decide whether the snapshot satisfies the expectation. Be strict: only pass when the expectation is clearly met by the snapshot content. Answer with pass=true/false and a one-sentence reason.
1076
+ return `You are a QA judge. You receive a test step, the model's expectation for the current page, and a page accessibility snapshot. Decide whether the STEP's own outcome holds \u2014 the expectation is the claim the model is offering in support of that, not a substitute question of its own. A claim can be true of the snapshot and still fail the step, if it does not establish what the step actually describes: only pass when the snapshot itself shows the step's outcome, never merely because the expectation offered happens to be true of something else on the page.
1077
1077
 
1078
- \`***\` marks a secret deliberately withheld from you \u2014 a password, token or key. Seeing it is expected. A field holding \`***\` is filled, not empty, so do not fail an expectation on the grounds that a value was redacted. This applies only to the redaction itself: everything else the expectation asks for must still be visibly satisfied by the snapshot, and an expectation you genuinely cannot check against what you were shown still fails.`;
1078
+ A value sitting in a control that was just typed into \u2014 an open dialog's textbox, an unsubmitted form field \u2014 is not the same as a committed outcome. When the step describes an outcome (something now appears in a list, is saved, is confirmed, is created), text visible only inside an editable, not-yet-submitted control does not satisfy it; look for the outcome committed outside that control (the dialog closed, the item is listed on its own, a confirmation appeared). This is specifically about that confusion, not a license to fail anything you are merely unsure about \u2014 if the snapshot plainly shows the step's outcome, pass it.
1079
+
1080
+ A step that names an ACTION (submit, click, create, add, ...) is satisfied by evidence the action took effect, not by the action's own control still being on the page. A successful action ordinarily replaces or moves past exactly the form, button or field the step names, so that control's absence is normal evidence of success, not evidence the step is unverifiable \u2014 do not fail such a step only because you can no longer see the thing it names. Fail it instead when the snapshot shows the action did NOT take effect: an error message, a validation warning, or the very same pre-action page still in front of you with nothing changed. A different page, a new state, or the result the action was meant to produce counts as evidence it worked.
1081
+
1082
+ Be strict about what the step asks, not about withholding a pass you can plainly see is earned. Answer with pass=true/false and a one-sentence reason.
1083
+
1084
+ \`***\` marks a secret deliberately withheld from you \u2014 a password, token or key. Seeing it is expected. A field holding \`***\` is filled, not empty, so do not fail a step on the grounds that a value was redacted. This applies only to the redaction itself: everything else the step asks for must still be visibly satisfied by the snapshot, and a step you genuinely cannot check against what you were shown still fails.`;
1079
1085
  }
1080
- function assertUserPrompt(expectation, snapshot) {
1086
+ function assertUserPrompt(step, expectation, snapshot) {
1081
1087
  return [
1082
- `Expectation: ${expectation}`,
1088
+ `Step under test: ${step}`,
1089
+ "",
1090
+ `Model's expectation (the claim offered in support of the step, not the question itself): ${expectation}`,
1083
1091
  "",
1084
1092
  "Page accessibility snapshot:",
1085
1093
  snapshot,
1086
1094
  "",
1087
- "Does the snapshot satisfy the expectation?"
1095
+ "Does the snapshot establish that the step's own outcome holds?"
1088
1096
  ].join("\n");
1089
1097
  }
1090
1098
  function plannerSystemPrompt() {
@@ -1174,12 +1182,12 @@ function createBrain(model, generate = generateObject, budget) {
1174
1182
  }
1175
1183
  return parsed.data;
1176
1184
  },
1177
- async judge(expectation, snapshot) {
1185
+ async judge(step, expectation, snapshot) {
1178
1186
  const result = await countedGenerate(generate, budget, {
1179
1187
  model,
1180
1188
  schema: assertJudgmentSchema,
1181
1189
  system: assertSystemPrompt(),
1182
- prompt: assertUserPrompt(expectation, snapshot)
1190
+ prompt: assertUserPrompt(step, expectation, snapshot)
1183
1191
  });
1184
1192
  const parsed = assertJudgmentSchema.safeParse(result.object);
1185
1193
  if (!parsed.success) {
@@ -2556,7 +2564,7 @@ function parsePositiveNumber(flag) {
2556
2564
  };
2557
2565
  }
2558
2566
  var program = new Command();
2559
- program.name("blastproof").description("Open-source AI testing agent: plain-English YAML tests executed agentically on a real browser.").version("0.5.0");
2567
+ program.name("blastproof").description("Open-source AI testing agent: plain-English YAML tests executed agentically on a real browser.").version("0.6.0");
2560
2568
  program.command("init").description("Scaffold .blastproof/ (config, tests, sample tests) in the current directory").action(async () => {
2561
2569
  try {
2562
2570
  const result = await initProject(process.cwd());