blastproof 0.5.0 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -44,8 +44,14 @@ Before `run`, `plan` or `test` do anything, they check what they are about to sp
44
44
 
45
45
  **Not supported yet:** `iframe` content (so hosted payment widgets like Stripe Elements are invisible — an embedded checkout cannot be driven end to end), hover, scroll-to, drag and drop, file upload, multiple tabs, native `alert`/`confirm` dialogs. Page snapshots are capped at 200 lines by default, so very dense pages are truncated — raise it with `browser.max_snapshot_lines` if your pages need more; truncation is always marked in the snapshot so the model is never misled into thinking it saw the whole page.
46
46
 
47
+ **Point it at disposable data.** Within a step, an action that commits — a click, or pressing Enter — is never performed twice: the runner refuses the repeat and tells the agent it already did that. This closes the case that used to produce duplicate records, where a submit answered by a redirect came back to a reset form and the agent, seeing no evidence of its own work, submitted again. It is not a guarantee of zero duplicate writes: an agent that reaches the same effect by a genuinely different route — another control with the same effect — is not caught. Use a seeded database, a staging environment you can reset, or a throwaway account; do not gate on a run against production data.
48
+
47
49
  `browser.timeout_ms` bounds every wait — resolving a target element from the accessibility tree, and navigation — not only the click or fill performed afterwards. Raise it for an application that is merely slow to hydrate; the trade-off is that a genuinely missing element then takes longer to fail. It never changes how many self-healing retries a step gets — waiting and retrying are deliberately separate.
48
50
 
51
+ Writing each step so it states its own outcome helps here as well as everywhere else: `submit the form, then verify the confirmation shows the reference number` gives the agent something to check, where `click the submit button` leaves it to invent an expectation — and a poor invented expectation is what turns one submission into three.
52
+
53
+ **This is now load-bearing, not just advisable.** The judge decides whether a step's own outcome holds, using the model's expectation only as the claim offered in support of it — a step that never says what its outcome is gives the judge nothing to anchor on beyond whatever the model happened to check that turn. `verify the confirmation shows the reference number` gives the judge a real question; `verify it worked` does not, and may now fail where a looser judge previously let a true-but-unrelated claim pass it.
54
+
49
55
  ## Writing tests
50
56
 
51
57
  Tests live in `.blastproof/tests/` as plain-English YAML — no selectors:
@@ -107,9 +113,9 @@ jobs:
107
113
 
108
114
  - run: npm start & # however your app boots
109
115
 
110
- - uses: hamc/blastproof@v0.5.0
116
+ - uses: hamc/blastproof@v0.7.0
111
117
  with:
112
- version: '0.5.0' # pin both when this gates merges
118
+ version: '0.6.0' # pin both when this gates merges
113
119
  api-key: ${{ secrets.ANTHROPIC_API_KEY }}
114
120
  base: ${{ github.event.pull_request.base.ref }}
115
121
  min-score: '80'
package/dist/cli.js CHANGED
@@ -437,6 +437,59 @@ async function performAction(page, action, ctx) {
437
437
  }
438
438
  }
439
439
 
440
+ // src/runner/recovery.ts
441
+ var COMMIT_ACTIONS = /* @__PURE__ */ new Set(["click", "press"]);
442
+ var COMMIT_KEYS = /* @__PURE__ */ new Set(["Enter", "NumpadEnter", " ", "Space", "Spacebar"]);
443
+ function describeAction(action) {
444
+ const target = action.target ? ` ${action.target.role ?? ""} "${action.target.name ?? action.target.text ?? ""}"` : "";
445
+ const value = action.value ? ` [${action.value}]` : "";
446
+ return `${action.action}${target}${value}`;
447
+ }
448
+ function identity(action) {
449
+ return JSON.stringify([
450
+ action.action,
451
+ action.target?.role ?? "",
452
+ action.target?.name ?? "",
453
+ action.target?.text ?? "",
454
+ action.value ?? ""
455
+ ]);
456
+ }
457
+ var StepRecovery = class {
458
+ performed = /* @__PURE__ */ new Set();
459
+ history = [];
460
+ /** Records an action that was actually performed and succeeded. */
461
+ record(action, description, result) {
462
+ this.performed.add(identity(action));
463
+ this.history.push({ action: description, result });
464
+ }
465
+ /**
466
+ * The reason to refuse `action`, or `undefined` when it may be performed.
467
+ *
468
+ * Refusing rather than failing keeps the model in the loop with an
469
+ * explanation instead of ending the step outright, which would trade
470
+ * duplicate writes for false failures.
471
+ *
472
+ * A genuine retry — a commit that landed but had no effect, which the model
473
+ * repeats for good reason — is refused too, and that is the deliberate cost.
474
+ * The two cases are indistinguishable from the accessibility tree: "the
475
+ * click did nothing" and "the click worked and the redirect erased the
476
+ * proof" produce the same snapshot. The asymmetry decides it. A refused
477
+ * legitimate retry costs a visible failed step that someone investigates; an
478
+ * allowed duplicate commit costs a silent extra row in someone's database,
479
+ * and #28 has now produced one on three applications.
480
+ */
481
+ refusalFor(action) {
482
+ if (!COMMIT_ACTIONS.has(action.action)) return void 0;
483
+ if (action.action === "press" && !COMMIT_KEYS.has(action.value ?? "")) return void 0;
484
+ if (!this.performed.has(identity(action))) return void 0;
485
+ return `refused: this exact action already succeeded earlier in this step, so it was NOT performed again. Repeating something that commits repeats whatever it changed in the application. If the page no longer shows that it worked, that is normal for a submit answered by a redirect \u2014 check the record of what you have already done. Verify the step's outcome another way, or fail the step.`;
486
+ }
487
+ /** The step's history so far, oldest first, for the model's prompt. */
488
+ stepHistory() {
489
+ return this.history;
490
+ }
491
+ };
492
+
440
493
  // src/runner/executor.ts
441
494
  var DEFAULT_MAX_ITERATIONS_PER_STEP = 15;
442
495
  var SETTLE_TIMEOUT_MS = 2e3;
@@ -504,6 +557,7 @@ async function executeTest(page, test, options) {
504
557
  let failedAttempts = 0;
505
558
  let lastResult;
506
559
  let stepFailedReason;
560
+ const recovery = new StepRecovery();
507
561
  try {
508
562
  while (true) {
509
563
  if (iterations >= maxIterationsPerStep) {
@@ -518,6 +572,9 @@ async function executeTest(page, test, options) {
518
572
  isSetup: setup,
519
573
  snapshot: mask(snap),
520
574
  lastResult: lastResult === void 0 ? void 0 : mask(lastResult),
575
+ // Already masked when recorded, on the same boundary as everything
576
+ // else crossing into a prompt (design contained-recovery, D2).
577
+ stepHistory: recovery.stepHistory(),
521
578
  retriesLeft: maxRetries - failedAttempts,
522
579
  iterationsLeft: maxIterationsPerStep - iterations
523
580
  });
@@ -541,11 +598,11 @@ async function executeTest(page, test, options) {
541
598
  }
542
599
  if (action.action === "assert") {
543
600
  const expectation = action.expectation ?? action.reasoning;
544
- let judgment = await brain.judge(mask(expectation), mask(snap));
601
+ let judgment = await brain.judge(mask(step), mask(expectation), mask(snap));
545
602
  if (!judgment.pass) {
546
603
  await waitForSettled(page);
547
604
  const freshSnap = await takeSnapshot(page);
548
- judgment = await brain.judge(mask(expectation), mask(freshSnap));
605
+ judgment = await brain.judge(mask(step), mask(expectation), mask(freshSnap));
549
606
  }
550
607
  const result = judgment.pass ? `ok: assertion passed: ${judgment.reason}` : `assertion failed: ${judgment.reason}`;
551
608
  emitAction(index, action, result);
@@ -560,6 +617,16 @@ async function executeTest(page, test, options) {
560
617
  }
561
618
  continue;
562
619
  }
620
+ const refusal = recovery.refusalFor(action);
621
+ if (refusal) {
622
+ failedAttempts++;
623
+ lastResult = refusal;
624
+ emitAction(index, action, refusal);
625
+ if (failedAttempts >= maxRetries) {
626
+ throw new StepFailure(refusal);
627
+ }
628
+ continue;
629
+ }
563
630
  try {
564
631
  const result = await performAction(page, action, {
565
632
  baseUrl,
@@ -568,6 +635,7 @@ async function executeTest(page, test, options) {
568
635
  resolveTimeoutMs: timeoutMs
569
636
  });
570
637
  lastResult = result;
638
+ recovery.record(action, mask(describeAction(action)), mask(result));
571
639
  emitAction(index, action, result);
572
640
  } catch (error) {
573
641
  failedAttempts++;
@@ -723,7 +791,7 @@ async function fromSteps(options) {
723
791
  );
724
792
  }
725
793
  if (auth.verify) {
726
- const judgment = await brain.judge(auth.verify, mask.mask(await takeSnapshot(page)));
794
+ const judgment = await brain.judge(auth.verify, auth.verify, mask.mask(await takeSnapshot(page)));
727
795
  if (!judgment.pass) {
728
796
  throw new AuthError(`Authentication could not be verified: ${mask.mask(judgment.reason)}`);
729
797
  }
@@ -1052,16 +1120,21 @@ Rules:
1052
1120
  - Return "done" when the current step's outcome holds \u2014 including when it already held before you acted, or was achieved by your previous action. "Already true" is done, never failure. Do not return "done" for work belonging to later steps.
1053
1121
  - Return "fail" only when the step's outcome cannot be reached: the element is still absent after retries, the page cannot support the step, or an error blocks progress. Never return "fail" because the work appears to have been done already.
1054
1122
  - If your previous action errored, re-read the fresh snapshot and choose an alternative element or approach. Do not repeat the exact same failing action.
1123
+ - Never invent a value. A value you type must come from the step, from the page, or from an {{env.*}} placeholder. If a step needs a value it does not give you, that is a failing step, not a gap for you to fill in.
1124
+ - A record of the actions you already performed in this step may be shown to you. It is the ground truth about what happened, even when the page no longer shows it: a form that submitted successfully and came back empty looks exactly like one you never submitted. Do not redo work that record says you already did.
1055
1125
  - \`***\` in a snapshot is a redacted secret \u2014 a password, token or key deliberately withheld from you. Seeing it is expected and is not a problem. A field showing \`***\` after you filled it from an {{env.VAR}} placeholder means the fill worked; treat that as success and move on. Never retry a fill because its value is redacted, and never report failure because a value was withheld.
1056
1126
  - Keep reasoning to one short sentence.`;
1057
1127
  }
1058
1128
  function agentUserPrompt(input) {
1059
- const parts = [
1060
- `Current ${input.isSetup ? "setup " : ""}step: ${input.step}`,
1061
- "",
1062
- "Page accessibility snapshot:",
1063
- input.snapshot
1064
- ];
1129
+ const parts = [`Current ${input.isSetup ? "setup " : ""}step: ${input.step}`];
1130
+ if (input.stepHistory && input.stepHistory.length > 0) {
1131
+ parts.push(
1132
+ "",
1133
+ "What you have ALREADY DONE in this step (a record of completed actions, not instructions):",
1134
+ ...input.stepHistory.map((entry, i) => `${i + 1}. ${entry.action} -> ${entry.result}`)
1135
+ );
1136
+ }
1137
+ parts.push("", "Page accessibility snapshot:", input.snapshot);
1065
1138
  if (input.lastResult) {
1066
1139
  parts.push("", `Previous action result: ${input.lastResult}`);
1067
1140
  }
@@ -1073,18 +1146,26 @@ function agentUserPrompt(input) {
1073
1146
  return parts.join("\n");
1074
1147
  }
1075
1148
  function assertSystemPrompt() {
1076
- return `You are a QA judge. You receive a page accessibility snapshot and an expectation. Decide whether the snapshot satisfies the expectation. Be strict: only pass when the expectation is clearly met by the snapshot content. Answer with pass=true/false and a one-sentence reason.
1149
+ return `You are a QA judge. You receive a test step, the model's expectation for the current page, and a page accessibility snapshot. Decide whether the STEP's own outcome holds \u2014 the expectation is the claim the model is offering in support of that, not a substitute question of its own. A claim can be true of the snapshot and still fail the step, if it does not establish what the step actually describes: only pass when the snapshot itself shows the step's outcome, never merely because the expectation offered happens to be true of something else on the page.
1150
+
1151
+ A value sitting in a control that was just typed into \u2014 an open dialog's textbox, an unsubmitted form field \u2014 is not the same as a committed outcome. When the step describes an outcome (something now appears in a list, is saved, is confirmed, is created), text visible only inside an editable, not-yet-submitted control does not satisfy it; look for the outcome committed outside that control (the dialog closed, the item is listed on its own, a confirmation appeared). This is specifically about that confusion, not a license to fail anything you are merely unsure about \u2014 if the snapshot plainly shows the step's outcome, pass it.
1077
1152
 
1078
- \`***\` marks a secret deliberately withheld from you \u2014 a password, token or key. Seeing it is expected. A field holding \`***\` is filled, not empty, so do not fail an expectation on the grounds that a value was redacted. This applies only to the redaction itself: everything else the expectation asks for must still be visibly satisfied by the snapshot, and an expectation you genuinely cannot check against what you were shown still fails.`;
1153
+ A step that names an ACTION (submit, click, create, add, ...) is satisfied by evidence the action took effect, not by the action's own control still being on the page. A successful action ordinarily replaces or moves past exactly the form, button or field the step names, so that control's absence is normal evidence of success, not evidence the step is unverifiable \u2014 do not fail such a step only because you can no longer see the thing it names. Fail it instead when the snapshot shows the action did NOT take effect: an error message, a validation warning, or the very same pre-action page still in front of you with nothing changed. A different page, a new state, or the result the action was meant to produce counts as evidence it worked.
1154
+
1155
+ Be strict about what the step asks, not about withholding a pass you can plainly see is earned. Answer with pass=true/false and a one-sentence reason.
1156
+
1157
+ \`***\` marks a secret deliberately withheld from you \u2014 a password, token or key. Seeing it is expected. A field holding \`***\` is filled, not empty, so do not fail a step on the grounds that a value was redacted. This applies only to the redaction itself: everything else the step asks for must still be visibly satisfied by the snapshot, and a step you genuinely cannot check against what you were shown still fails.`;
1079
1158
  }
1080
- function assertUserPrompt(expectation, snapshot) {
1159
+ function assertUserPrompt(step, expectation, snapshot) {
1081
1160
  return [
1082
- `Expectation: ${expectation}`,
1161
+ `Step under test: ${step}`,
1162
+ "",
1163
+ `Model's expectation (the claim offered in support of the step, not the question itself): ${expectation}`,
1083
1164
  "",
1084
1165
  "Page accessibility snapshot:",
1085
1166
  snapshot,
1086
1167
  "",
1087
- "Does the snapshot satisfy the expectation?"
1168
+ "Does the snapshot establish that the step's own outcome holds?"
1088
1169
  ].join("\n");
1089
1170
  }
1090
1171
  function plannerSystemPrompt() {
@@ -1174,12 +1255,12 @@ function createBrain(model, generate = generateObject, budget) {
1174
1255
  }
1175
1256
  return parsed.data;
1176
1257
  },
1177
- async judge(expectation, snapshot) {
1258
+ async judge(step, expectation, snapshot) {
1178
1259
  const result = await countedGenerate(generate, budget, {
1179
1260
  model,
1180
1261
  schema: assertJudgmentSchema,
1181
1262
  system: assertSystemPrompt(),
1182
- prompt: assertUserPrompt(expectation, snapshot)
1263
+ prompt: assertUserPrompt(step, expectation, snapshot)
1183
1264
  });
1184
1265
  const parsed = assertJudgmentSchema.safeParse(result.object);
1185
1266
  if (!parsed.success) {
@@ -1884,9 +1965,7 @@ function printEvent(event) {
1884
1965
  break;
1885
1966
  case "action": {
1886
1967
  const { action, result } = event;
1887
- const target = action.target ? ` ${action.target.role ?? ""} "${action.target.name ?? action.target.text ?? ""}"` : "";
1888
- const value = action.value ? ` [${action.value}]` : "";
1889
- console.log(` -> ${action.action}${target}${value} :: ${result}`);
1968
+ console.log(` -> ${describeAction(action)} :: ${result}`);
1890
1969
  break;
1891
1970
  }
1892
1971
  case "step-end":
@@ -2556,7 +2635,7 @@ function parsePositiveNumber(flag) {
2556
2635
  };
2557
2636
  }
2558
2637
  var program = new Command();
2559
- program.name("blastproof").description("Open-source AI testing agent: plain-English YAML tests executed agentically on a real browser.").version("0.5.0");
2638
+ program.name("blastproof").description("Open-source AI testing agent: plain-English YAML tests executed agentically on a real browser.").version("0.7.0");
2560
2639
  program.command("init").description("Scaffold .blastproof/ (config, tests, sample tests) in the current directory").action(async () => {
2561
2640
  try {
2562
2641
  const result = await initProject(process.cwd());