blastproof 0.22.0 → 0.23.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -210,7 +210,7 @@ Common flags — `blastproof <command> --help` has the full list:
210
210
  | `--write` | `plan` only — persist drafts instead of previewing |
211
211
  | `--max-llm-calls` · `--max-tokens` · `--max-duration` | [Bound what a run may spend](./docs/configuration.md#budget--bounding-what-a-run-spends) |
212
212
 
213
- Exit codes: **0** pass, **1** the gate failed, **2** usage or config error.
213
+ Exit codes: **0** pass, **1** the gate failed or the run stopped (its budget, or the model provider refusing a call: `Run incomplete:`), **2** usage or config error.
214
214
 
215
215
  **Generated drafts are never executed and never affect the score.** An unreviewed model-written test in the merge path fails in two directions: a hallucinated expectation blocks a correct PR, and a credulous one waves a broken change through while looking like coverage. `plan` makes the gap visible with a draft to review; it does not make an uncovered route safe.
216
216
 
package/dist/cli.js CHANGED
@@ -219,7 +219,9 @@ function describeLimit(limit, observed, configured) {
219
219
  return `deadline exceeded: reached the configured maximum of ${(configured / 1e3).toFixed(0)}s (elapsed ${(observed / 1e3).toFixed(1)}s)`;
220
220
  }
221
221
  }
222
- var BudgetExhaustedError = class extends Error {
222
+ var RunStoppedError = class extends Error {
223
+ };
224
+ var BudgetExhaustedError = class extends RunStoppedError {
223
225
  limit;
224
226
  observed;
225
227
  configured;
@@ -231,6 +233,30 @@ var BudgetExhaustedError = class extends Error {
231
233
  this.configured = configured;
232
234
  }
233
235
  };
236
+ function refusalRemedy(statusCode) {
237
+ if (statusCode === void 0) {
238
+ return "The provider could not be reached after 3 attempts: check llm.base_url and the network, then run again.";
239
+ }
240
+ if (statusCode === 401 || statusCode === 403) {
241
+ return "Check the API key in the variable named by llm.api_key_env, then run again.";
242
+ }
243
+ if (statusCode === 402) return "Add credit to the provider account, then run again.";
244
+ if (statusCode === 408 || statusCode === 409 || statusCode === 429 || statusCode >= 500) {
245
+ return "The provider was unavailable after 3 attempts: run again later.";
246
+ }
247
+ return void 0;
248
+ }
249
+ var ProviderRefusedError = class extends RunStoppedError {
250
+ statusCode;
251
+ constructor(statusCode, detail) {
252
+ const status = statusCode === void 0 ? "no response" : `HTTP ${statusCode}`;
253
+ const remedy = refusalRemedy(statusCode);
254
+ const separator = /[.!?]$/.test(detail.trim()) ? " " : ". ";
255
+ super(`model provider refused the request (${status}): ${detail}${remedy ? `${separator}${remedy}` : ""}`);
256
+ this.name = "ProviderRefusedError";
257
+ this.statusCode = statusCode;
258
+ }
259
+ };
234
260
  var RunBudget = class {
235
261
  maxCalls;
236
262
  maxTokens;
@@ -600,11 +626,13 @@ function describeAction(action) {
600
626
  return `${action.action}${target}${value}`;
601
627
  }
602
628
  function identity(action) {
629
+ const role = action.target?.role ?? "";
630
+ const name = action.target?.name ?? "";
603
631
  return JSON.stringify([
604
632
  action.action,
605
- action.target?.role ?? "",
606
- action.target?.name ?? "",
607
- action.target?.text ?? "",
633
+ role,
634
+ normalise(name),
635
+ role || name ? "" : action.target?.text ?? "",
608
636
  action.value ?? ""
609
637
  ]);
610
638
  }
@@ -854,7 +882,7 @@ async function executeTest(page, test, options) {
854
882
  iterationsLeft: maxIterationsPerStep - iterations
855
883
  });
856
884
  } catch (error) {
857
- if (error instanceof BudgetExhaustedError) throw error;
885
+ if (error instanceof RunStoppedError) throw error;
858
886
  failedAttempts++;
859
887
  lastResult = `error: ${error instanceof Error ? error.message : String(error)}`;
860
888
  if (failedAttempts >= maxRetries) {
@@ -882,23 +910,35 @@ async function executeTest(page, test, options) {
882
910
  }
883
911
  if (action.action === "assert") {
884
912
  const expectation = action.expectation ?? action.reasoning;
885
- let judgment = await brain.judge(
886
- mask(step),
887
- mask(expectation),
888
- maskedSnap,
889
- recovery.stepHistory()
890
- );
891
- if (!judgment.pass) {
892
- await waitForSettled(page);
893
- const freshSnap = await takeSnapshot(page);
894
- const maskedFresh = mask(freshSnap);
895
- recovery.observe(maskedFresh);
913
+ let judgment;
914
+ try {
896
915
  judgment = await brain.judge(
897
916
  mask(step),
898
917
  mask(expectation),
899
- maskedFresh,
918
+ maskedSnap,
900
919
  recovery.stepHistory()
901
920
  );
921
+ if (!judgment.pass) {
922
+ await waitForSettled(page);
923
+ const freshSnap = await takeSnapshot(page);
924
+ const maskedFresh = mask(freshSnap);
925
+ recovery.observe(maskedFresh);
926
+ judgment = await brain.judge(
927
+ mask(step),
928
+ mask(expectation),
929
+ maskedFresh,
930
+ recovery.stepHistory()
931
+ );
932
+ }
933
+ } catch (error) {
934
+ if (error instanceof RunStoppedError) throw error;
935
+ failedAttempts++;
936
+ lastResult = `error: ${error instanceof Error ? error.message : String(error)}`;
937
+ emitAction(index, action, lastResult);
938
+ if (failedAttempts >= maxRetries) {
939
+ throw new StepFailure(lastResult);
940
+ }
941
+ continue;
902
942
  }
903
943
  const result = judgment.pass ? `ok: assertion passed: ${judgment.reason}` : `assertion failed: ${judgment.reason}`;
904
944
  emitAction(index, action, result);
@@ -943,7 +983,7 @@ async function executeTest(page, test, options) {
943
983
  }
944
984
  }
945
985
  } catch (error) {
946
- if (error instanceof BudgetExhaustedError) throw error;
986
+ if (error instanceof RunStoppedError) throw error;
947
987
  stepFailedReason = mask(error instanceof Error ? error.message : String(error));
948
988
  }
949
989
  const status = stepFailedReason ? "failed" : "passed";
@@ -1044,12 +1084,24 @@ async function runJourney(...args) {
1044
1084
  try {
1045
1085
  return await executeTest(...args);
1046
1086
  } catch (error) {
1047
- if (error instanceof BudgetExhaustedError) throw error;
1087
+ if (error instanceof RunStoppedError) throw error;
1048
1088
  throw new AuthError(
1049
1089
  `Authentication could not run: ${error instanceof Error ? error.message : String(error)}`
1050
1090
  );
1051
1091
  }
1052
1092
  }
1093
+ async function verifyJudgment(brain, verify, snapshot, maskText, maxRetries = 3) {
1094
+ let lastError = "";
1095
+ for (let attempt = 0; attempt < maxRetries; attempt++) {
1096
+ try {
1097
+ return await brain.judge(verify, verify, await snapshot());
1098
+ } catch (error) {
1099
+ if (error instanceof RunStoppedError) throw error;
1100
+ lastError = error instanceof Error ? error.message : String(error);
1101
+ }
1102
+ }
1103
+ throw new AuthError(`Authentication could not be verified: ${maskText(lastError)}`);
1104
+ }
1053
1105
  async function fromSteps(options) {
1054
1106
  const { auth, baseUrl, browser, brain, maxRetries, snapshot, timeoutMs, maxSnapshotLines, onEvent, mask } = options;
1055
1107
  const steps = auth.steps;
@@ -1087,7 +1139,13 @@ async function fromSteps(options) {
1087
1139
  );
1088
1140
  }
1089
1141
  if (auth.verify) {
1090
- const judgment = await brain.judge(auth.verify, auth.verify, mask.mask(await takeSnapshot(page)));
1142
+ const judgment = await verifyJudgment(
1143
+ brain,
1144
+ auth.verify,
1145
+ async () => mask.mask(await takeSnapshot(page)),
1146
+ (text) => mask.mask(text),
1147
+ maxRetries
1148
+ );
1091
1149
  if (!judgment.pass) {
1092
1150
  throw new AuthError(`Authentication could not be verified: ${mask.mask(judgment.reason)}`);
1093
1151
  }
@@ -1459,7 +1517,7 @@ warning: ${findings.length} file(s) changed in the working tree are not in the d
1459
1517
  }
1460
1518
 
1461
1519
  // src/llm/brain.ts
1462
- import { generateObject } from "ai";
1520
+ import { APICallError, generateObject, RetryError } from "ai";
1463
1521
 
1464
1522
  // src/llm/prompts.ts
1465
1523
  function agentSystemPrompt() {
@@ -1511,6 +1569,8 @@ A value sitting in a control that was just typed into \u2014 an open dialog's te
1511
1569
 
1512
1570
  A step that names an ACTION (submit, click, create, add, ...) is satisfied by evidence the action took effect, not by the action's own control still being on the page. A successful action ordinarily replaces or moves past exactly the form, button or field the step names, so that control's absence is normal evidence of success, not evidence the step is unverifiable \u2014 do not fail such a step only because you can no longer see the thing it names. Fail it instead when the snapshot shows the action did NOT take effect: an error message, a validation warning, or the very same pre-action page still in front of you with nothing changed. A different page, a new state, or the result the action was meant to produce counts as evidence it worked.
1513
1571
 
1572
+ An outcome may already hold before this step acts: an earlier step, or the application itself, got there first. A step asking for something to be gone, closed or dismissed is satisfied by that thing being absent from the snapshot. It does not also require the thing to have been there, or this step to have removed it, and a different element that is present (another dialog, another banner) is not the one the step names. Whether the step's action ran is not what you decide: judge the state the step asks for.
1573
+
1514
1574
  You may also be shown the actions already performed in this step, with their results. That record tells you what was ATTEMPTED and what it produced \u2014 for instance that a navigation was performed and which URL the server ultimately served, or that a form was submitted. Use it to avoid concluding that something never happened when the page simply cannot show it any more: a navigation the server redirected does not leave the browser at the path that was requested, and that is what success looks like, not failure.
1515
1575
 
1516
1576
  The record is not evidence that the step's outcome holds. An action reported as \`ok\` establishes that it ran and what it returned; whether the thing the step describes is now TRUE is still decided by the snapshot alone. Never pass a step because the record shows an action succeeded while the snapshot does not show the outcome.
@@ -1616,9 +1676,12 @@ function parseAgentAction(value) {
1616
1676
  return parsedAgentActionSchema.safeParse(value);
1617
1677
  }
1618
1678
  var assertJudgmentSchema = z2.object({
1619
- reason: z2.string().describe("One sentence: what the STEP requires to be true now, and whether the snapshot shows it."),
1679
+ outcome: z2.string().describe(
1680
+ 'The state the STEP asks for, rewritten as a sentence about the page with its action removed: "click Save and verify the note is listed" becomes "the note is listed"; "open the Account menu and verify it shows the email" becomes "the Account menu shows the email".'
1681
+ ),
1682
+ reason: z2.string().describe("One sentence: whether the snapshot shows that outcome."),
1620
1683
  pass: z2.boolean().describe(
1621
- "Whether the snapshot shows the STEP's own outcome. The expectation is only a claim offered in support; it never replaces the step. False if any part of the outcome the step asks for is not shown, or cannot be assessed from this snapshot."
1684
+ "Whether the snapshot shows that outcome. The expectation is only a claim offered in support; it never replaces the step. False if any part of the outcome the step asks for is not shown, or cannot be assessed from this snapshot. An action the step names (click, dismiss, submit) is how its outcome is reached, not part of it: an outcome that holds passes whether or not that action was needed. An outcome that is an absence (gone, closed, dismissed, removed) is shown by the thing being absent."
1622
1685
  )
1623
1686
  });
1624
1687
  var generatedTestSchema = z2.object({
@@ -1647,13 +1710,20 @@ function withProviderDetail(error) {
1647
1710
  error.message = `${error.message} \u2014 provider said: ${detail}`;
1648
1711
  return error;
1649
1712
  }
1713
+ function asProviderRefusal(error) {
1714
+ const cause = RetryError.isInstance(error) ? error.lastError : error;
1715
+ if (!APICallError.isInstance(cause)) return error;
1716
+ if (cause.statusCode !== void 0 && cause.statusCode < 400) return error;
1717
+ withProviderDetail(cause);
1718
+ return new ProviderRefusedError(cause.statusCode, cause.message);
1719
+ }
1650
1720
  async function countedGenerate(generate, budget, options) {
1651
1721
  budget.check();
1652
1722
  let result;
1653
1723
  try {
1654
1724
  result = await generate(options);
1655
1725
  } catch (error) {
1656
- throw withProviderDetail(error);
1726
+ throw asProviderRefusal(withProviderDetail(error));
1657
1727
  }
1658
1728
  budget.record(result.usage);
1659
1729
  return result;
@@ -1712,6 +1782,7 @@ ${e.result}`).join("\n"));
1712
1782
  if (onPage.includes(name) || inRecord.includes(name) || others.length === 0) continue;
1713
1783
  const shown = others.map((other) => redactionLabel(other)).join(", ");
1714
1784
  return {
1785
+ ...judgment,
1715
1786
  pass: false,
1716
1787
  reason: `The step names ${redactionLabel(name)}, which appears neither on the page nor in this step's actions, while the page shows ${shown}: a different secret cannot satisfy it. (The judge had said: ${judgment.reason})`
1717
1788
  };
@@ -2577,7 +2648,7 @@ X ${r.summary} (${r.file})`);
2577
2648
  }
2578
2649
  if (notRun.length > 0) {
2579
2650
  console.log(`
2580
- ${notRun.length} test(s) not run (run stopped by its budget or deadline):`);
2651
+ ${notRun.length} test(s) not run (run stopped: ${notRun[0].reason}):`);
2581
2652
  for (const r of notRun) console.log(` - ${r.summary} (${r.file})`);
2582
2653
  }
2583
2654
  }
@@ -2902,7 +2973,7 @@ ${results.length} test file(s) could not be parsed:`);
2902
2973
  onEvent: printEvent
2903
2974
  });
2904
2975
  } catch (error) {
2905
- if (error instanceof BudgetExhaustedError) {
2976
+ if (error instanceof RunStoppedError) {
2906
2977
  incomplete = error;
2907
2978
  } else if (error instanceof AuthError) {
2908
2979
  console.error(`error: ${error.message}`);
@@ -2932,7 +3003,7 @@ ${results.length} test file(s) could not be parsed:`);
2932
3003
  if (!streaming) for (const line of lines) console.log(line);
2933
3004
  return { index, result };
2934
3005
  } catch (error) {
2935
- if (error instanceof BudgetExhaustedError) {
3006
+ if (error instanceof RunStoppedError) {
2936
3007
  stoppedBy ??= error;
2937
3008
  if (!streaming) for (const line of lines) console.log(line);
2938
3009
  return { index, stopped: error };
@@ -3116,7 +3187,7 @@ async function planCommand(options) {
3116
3187
  onEvent: printAuthEvent
3117
3188
  });
3118
3189
  } catch (error) {
3119
- if (error instanceof BudgetExhaustedError) {
3190
+ if (error instanceof RunStoppedError) {
3120
3191
  incomplete = error;
3121
3192
  } else if (error instanceof AuthError) {
3122
3193
  console.error(`error: ${error.message}`);
@@ -3144,7 +3215,7 @@ async function planCommand(options) {
3144
3215
  timeoutMs: config.browser.timeout_ms
3145
3216
  });
3146
3217
  } catch (error) {
3147
- if (error instanceof BudgetExhaustedError) {
3218
+ if (error instanceof RunStoppedError) {
3148
3219
  incomplete = error;
3149
3220
  break;
3150
3221
  }
@@ -3198,7 +3269,7 @@ ${renderTestYaml(draft, { route, base })}`);
3198
3269
  if (incomplete) {
3199
3270
  console.log(`Stopped: ${incomplete.message}`);
3200
3271
  if (notAttempted.length > 0) {
3201
- console.log("Not attempted (run out of budget):");
3272
+ console.log("Not attempted (run stopped):");
3202
3273
  for (const route of notAttempted) console.log(` ${route}`);
3203
3274
  }
3204
3275
  }
@@ -3292,7 +3363,7 @@ function parsePositiveNumber(flag) {
3292
3363
  };
3293
3364
  }
3294
3365
  var program = new Command();
3295
- program.name("blastproof").description("Open-source AI testing agent: plain-English YAML tests executed agentically on a real browser.").version("0.22.0");
3366
+ program.name("blastproof").description("Open-source AI testing agent: plain-English YAML tests executed agentically on a real browser.").version("0.23.0");
3296
3367
  program.command("init").description("Scaffold .blastproof/ (config, tests, sample tests) in the current directory").action(async () => {
3297
3368
  try {
3298
3369
  const result = await initProject(process.cwd());