@skyramp/mcp 0.4.2-rc.1 → 0.4.2-rc.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (104) hide show
  1. package/build/commands/localDevTestChangesCommand.js +1 -1
  2. package/build/commands/recommendTestsAndExecuteCommand.js +10 -1
  3. package/build/commands/testThisEndpointCommand.js +19 -2
  4. package/build/execution/wrapperConfig.d.ts +56 -0
  5. package/build/execution/wrapperConfig.js +155 -0
  6. package/build/index.js +6 -6
  7. package/build/playwright/registerPlaywrightTools.js +14 -0
  8. package/build/playwright/traceExportStore.d.ts +22 -0
  9. package/build/playwright/traceExportStore.js +81 -0
  10. package/build/playwright/traceRecordingPrompt.js +2 -1
  11. package/build/prompts/code-reuse.js +24 -21
  12. package/build/prompts/local-dev/local-dev-plan.js +6 -23
  13. package/build/prompts/local-dev/local-dev-prompts.js +1 -1
  14. package/build/prompts/shared-helper-policy.d.ts +36 -0
  15. package/build/prompts/shared-helper-policy.js +33 -1
  16. package/build/prompts/startTraceCollectionPrompts.js +1 -1
  17. package/build/prompts/sut-setup/modes/adaptWorkflowPrompt.js +6 -7
  18. package/build/prompts/sut-setup/shared.d.ts +1 -1
  19. package/build/prompts/sut-setup/shared.js +5 -3
  20. package/build/prompts/test-maintenance/drift-analysis-prompt.d.ts +16 -8
  21. package/build/prompts/test-maintenance/drift-analysis-prompt.js +90 -36
  22. package/build/prompts/test-maintenance/uiDriftAnalysisSections.js +1 -1
  23. package/build/prompts/test-recommendation/recommendationShared.d.ts +1 -1
  24. package/build/prompts/test-recommendation/recommendationShared.js +0 -1
  25. package/build/prompts/testbot/testbot-prompts.js +11 -9
  26. package/build/services/TestDiscoveryService.js +32 -4
  27. package/build/skills/runTestSkill.d.ts +6 -0
  28. package/build/skills/runTestSkill.js +17 -0
  29. package/build/tool-phases.js +0 -1
  30. package/build/tools/budgetExcuse.d.ts +15 -0
  31. package/build/tools/budgetExcuse.js +113 -0
  32. package/build/tools/code-refactor/utils-verify-gates.js +17 -3
  33. package/build/tools/executeSkyrampTestTool.d.ts +97 -48
  34. package/build/tools/executeSkyrampTestTool.js +775 -449
  35. package/build/tools/submitReportTool.js +128 -0
  36. package/build/tools/test-management/actionsTool.js +31 -0
  37. package/build/tools/test-management/analyzeChangesTool.d.ts +4 -4
  38. package/build/tools/test-management/analyzeTestHealthTool.d.ts +0 -11
  39. package/build/tools/test-management/analyzeTestHealthTool.js +7 -63
  40. package/build/tools/test-management/testsOwedBeforeRun.d.ts +28 -0
  41. package/build/tools/test-management/testsOwedBeforeRun.js +53 -0
  42. package/build/tools/trace/stopTraceCollectionTool.js +1 -1
  43. package/build/types/RepositoryAnalysis.d.ts +32 -32
  44. package/build/types/ReuseOutcome.d.ts +4 -3
  45. package/build/types/TestExecution.d.ts +2 -2
  46. package/build/types/TestTypes.d.ts +3 -0
  47. package/build/types/TestTypes.js +6 -0
  48. package/build/utils/AnalysisStateManager.d.ts +0 -7
  49. package/build/utils/AnalysisStateManager.js +1 -1
  50. package/build/utils/connectionErrors.d.ts +10 -0
  51. package/build/utils/connectionErrors.js +10 -0
  52. package/build/utils/language-helper.js +24 -3
  53. package/build/utils/progress.d.ts +1 -1
  54. package/build/utils/progress.js +1 -1
  55. package/build/utils/rebaselineSnapshots.d.ts +1 -1
  56. package/build/utils/rebaselineSnapshots.js +6 -16
  57. package/build/utils/reuseRouting.d.ts +10 -0
  58. package/build/utils/reuseRouting.js +15 -0
  59. package/build/utils/runContextGauge.d.ts +27 -0
  60. package/build/utils/runContextGauge.js +181 -0
  61. package/build/utils/skyrampMdContent.d.ts +1 -1
  62. package/build/utils/skyrampMdContent.js +1 -1
  63. package/build/utils/skyrampSdkVersion.d.ts +9 -0
  64. package/build/utils/skyrampSdkVersion.js +16 -0
  65. package/build/utils/testDependencyPolicy.js +21 -0
  66. package/build/utils/testExecutionRecord.d.ts +5 -1
  67. package/build/utils/testExecutionRecord.js +3 -1
  68. package/build/utils/testFileClassification.d.ts +8 -0
  69. package/build/utils/testFileClassification.js +36 -3
  70. package/build/utils/utils-verify/action-key.d.ts +42 -0
  71. package/build/utils/utils-verify/action-key.js +118 -36
  72. package/build/utils/utils-verify/action-sites.d.ts +32 -0
  73. package/build/utils/utils-verify/action-sites.js +202 -0
  74. package/build/utils/utils-verify/body-reach.js +2 -4
  75. package/build/utils/utils-verify/call-sites.d.ts +25 -6
  76. package/build/utils/utils-verify/call-sites.js +8 -5
  77. package/build/utils/utils-verify/index.d.ts +1 -0
  78. package/build/utils/utils-verify/index.js +1 -0
  79. package/build/utils/utils-verify/language-spec.d.ts +25 -0
  80. package/build/utils/utils-verify/language-spec.js +16 -2
  81. package/build/utils/utils-verify/parse.d.ts +10 -1
  82. package/build/utils/utils-verify/parse.js +19 -2
  83. package/build/utils/utils-verify/verify.d.ts +3 -2
  84. package/build/utils/utils-verify/verify.js +16 -18
  85. package/build/workspace/workspace.d.ts +72 -52
  86. package/build/workspace/workspace.js +12 -8
  87. package/package.json +1 -1
  88. package/plugin/prompts/testbot-task1.md +0 -2
  89. package/plugin/skills/fix-test-import-errors/SKILL.md +2 -1
  90. package/plugin/skills/run-test/SKILL.md +16 -0
  91. package/build/adapters/jestAdapter.d.ts +0 -14
  92. package/build/adapters/jestAdapter.js +0 -131
  93. package/build/adapters/mochaAdapter.d.ts +0 -13
  94. package/build/adapters/mochaAdapter.js +0 -93
  95. package/build/adapters/playwrightAdapter.d.ts +0 -17
  96. package/build/adapters/playwrightAdapter.js +0 -184
  97. package/build/adapters/pytestAdapter.d.ts +0 -15
  98. package/build/adapters/pytestAdapter.js +0 -119
  99. package/build/tools/runExistingTestsTool.d.ts +0 -138
  100. package/build/tools/runExistingTestsTool.js +0 -666
  101. package/build/types/ExternalTestExecution.d.ts +0 -67
  102. package/build/types/ExternalTestExecution.js +0 -8
  103. package/build/workspace/testSuites.d.ts +0 -20
  104. package/build/workspace/testSuites.js +0 -17
@@ -1,44 +1,46 @@
1
1
  import { z } from "zod";
2
- import { pendingReuseDebt } from "./code-refactor/reuse-state.js";
3
- import { assertionFeedbackForExecution, recordAssertionExecution, } from "./code-refactor/assertion-state.js";
4
- import { stageAndRecordRetrofits } from "./code-refactor/retrofit-state.js";
2
+ import { TestExecutionService } from "../services/TestExecutionService.js";
3
+ import { getWorkspaceBaseUrl } from "../utils/workspaceAuth.js";
4
+ import { spawn } from "child_process";
5
+ import crypto from "crypto";
6
+ import * as fs from "fs";
5
7
  import path from "path";
6
8
  import { stripVTControlCharacters } from "util";
7
- import { TestExecutionService } from "../services/TestExecutionService.js";
8
- import { AnalyticsService } from "../services/AnalyticsService.js";
9
9
  import { makeProgressReporter } from "../utils/progress.js";
10
+ import { pendingReuseDebt } from "./code-refactor/reuse-state.js";
11
+ import { assertionFeedbackForExecution, canonicalTestPath, recordAssertionExecution, } from "./code-refactor/assertion-state.js";
12
+ import { stageAndRecordRetrofits } from "./code-refactor/retrofit-state.js";
13
+ import { AnalyticsService } from "../services/AnalyticsService.js";
10
14
  import { TestExecutionStatus, } from "../types/TestExecution.js";
11
- import { getWorkspaceBaseUrl } from "../utils/workspaceAuth.js";
12
15
  import { ProgrammingLanguage, TestType } from "../types/TestTypes.js";
13
- import { StateManager, currentRunStateFile, getTestsRepoDir, resolveOwnRunStatePath, resolveRunStatePath, } from "../utils/AnalysisStateManager.js";
14
- import { DriftAction, TestSource, } from "../types/TestAnalysis.js";
16
+ import { StateManager, currentRunStateFile, getPrimaryRepository, getTestsRepoDir, resolveOwnRunStatePath, resolveRunStatePath, } from "../utils/AnalysisStateManager.js";
17
+ import { DriftAction, } from "../types/TestAnalysis.js";
15
18
  import { logger } from "../utils/logger.js";
16
19
  import { toolError } from "../utils/utils.js";
17
20
  import { recordExecutionVideo } from "./execution-video-state.js";
18
21
  import { stageGeneratedPaths } from "../utils/gitStaging.js";
22
+ import { walkDir } from "../utils/fileWalk.js";
23
+ import { findRepoPlaywrightConfig, writeWrapperConfig, commandPassesBrowserFlag, } from "../execution/wrapperConfig.js";
19
24
  import { canonicalStateFilePath, persistTestExecutionResult, readPinnedMaxFixAttempts, readRepoSectionOrThrow, reserveTestExecutionAttempt, stateFileKey, } from "../utils/testExecutionRecord.js";
20
- import { canonicalTestPath } from "./code-refactor/assertion-state.js";
21
25
  import { getMaxFixAttempts } from "../utils/fixAttempts.js";
22
26
  import { runSerialized } from "../utils/runSerialized.js";
23
- import * as fs from "fs";
24
27
  import { sha256Of } from "../utils/assertion-verify/index.js";
25
28
  import { baselineFileMatchesStem, baselineStem, rebaselineSnapshotNameSchema, snapshotDirFor, } from "../utils/rebaselineSnapshots.js";
26
- import { EXECUTOR_DOCKER_IMAGE } from "../utils/versions.js";
27
- const TOOL_NAME = "skyramp_execute_test";
28
- export const CONTRACT_EXECUTION_MODES = ["provider", "consumer"];
29
+ export const TOOL_NAME = "skyramp_execute_test";
30
+ const DEFAULT_TIMEOUT_MS = 300_000;
31
+ const MAX_TIMEOUT_MS = 3_600_000;
32
+ /** Output kept per run, from its end: it is held in memory, saved to the state file, and returned to the agent. */
33
+ export const MAX_OUTPUT_CHARS = 200_000;
29
34
  /**
30
- * Resolve the effective auth token for test execution.
31
- * `unauthenticated: true` forces no token (empty string) so the container
32
- * env omits SKYRAMP_TEST_TOKEN entirely — unauthenticated endpoints won't
33
- * receive an empty Authorization header that triggers encoding errors (E7).
35
+ * `unauthenticated: true` forces no token, so the child env carries no
36
+ * SKYRAMP_TEST_TOKEN at all — an empty Authorization header triggers encoding
37
+ * errors on unauthenticated endpoints (E7). An empty `token` means "use the
38
+ * server's environment", as the local-dev prompt passes it.
34
39
  */
35
40
  export function resolveEffectiveToken(unauthenticated, paramToken, envToken) {
36
41
  if (unauthenticated)
37
42
  return "";
38
- return paramToken ?? envToken ?? "";
39
- }
40
- export function shouldInjectSkyrampBaseUrl(testType, contractMode) {
41
- return testType !== TestType.CONTRACT || contractMode !== "consumer";
43
+ return paramToken || envToken || "";
42
44
  }
43
45
  /**
44
46
  * Append the recorded video path to execution output.
@@ -51,6 +53,242 @@ export function shouldInjectSkyrampBaseUrl(testType, contractMode) {
51
53
  export function withVideoInfo(output, videoPath) {
52
54
  return videoPath ? `${output}\n\nVideo recording: ${videoPath}` : output;
53
55
  }
56
+ function isBrowserTest(testType) {
57
+ return testType === TestType.UI || testType === TestType.E2E;
58
+ }
59
+ /**
60
+ * One directory per run: a unique name leaves nothing to clean up between runs, and
61
+ * the random suffix keeps two runs in the same millisecond apart.
62
+ */
63
+ export function videoSubdirName(testFile) {
64
+ const basename = path
65
+ .basename(testFile, path.extname(testFile))
66
+ .replace(/[^A-Za-z0-9._-]/g, "-");
67
+ const hash = crypto
68
+ .createHash("sha256")
69
+ .update(testFile)
70
+ .digest("hex")
71
+ .slice(0, 8);
72
+ return `${basename}-${hash}-${Date.now()}-${crypto.randomBytes(3).toString("hex")}`;
73
+ }
74
+ /** The first `video.webm` under a run's video directory, if one was recorded. */
75
+ export function collectVideoPath(videoDir) {
76
+ try {
77
+ for (const [entry, fullPath] of walkDir(videoDir)) {
78
+ if (entry.name === "video.webm")
79
+ return fullPath;
80
+ }
81
+ }
82
+ catch (err) {
83
+ logger.warning(`Could not scan ${videoDir} for a video`, {
84
+ error: String(err),
85
+ });
86
+ }
87
+ return undefined;
88
+ }
89
+ /**
90
+ * The directory whose `.skyramp/workspace.yml` describes this run: the nearest
91
+ * ancestor of `cwd` that has one, else the run's primary checkout (a tests repo
92
+ * delivered apart from the SUT has no workspace.yml of its own), else `cwd`.
93
+ */
94
+ export function resolveWorkspaceRoot(cwd) {
95
+ let dir = path.resolve(cwd);
96
+ for (;;) {
97
+ if (fs.existsSync(path.join(dir, ".skyramp", "workspace.yml")))
98
+ return dir;
99
+ const parent = path.dirname(dir);
100
+ if (parent === dir)
101
+ break;
102
+ dir = parent;
103
+ }
104
+ return getPrimaryRepository()?.repositoryPath ?? path.resolve(cwd);
105
+ }
106
+ /** pytest splits PYTEST_ADDOPTS with shlex, so a path with whitespace needs quotes. */
107
+ export function appendPytestVideoOpts(existing, videoDir) {
108
+ const dir = /\s/.test(videoDir) ? JSON.stringify(videoDir) : videoDir;
109
+ return [existing?.trim(), `--video on --output ${dir}`]
110
+ .filter(Boolean)
111
+ .join(" ");
112
+ }
113
+ /**
114
+ * The names by which a command can select `testFile`: its basename, and for Java the
115
+ * class name, because Maven selects a test with `-Dtest=FooTest`.
116
+ */
117
+ export function testFileNames(testFile) {
118
+ const basename = path.basename(testFile);
119
+ return path.extname(basename) === ".java"
120
+ ? [basename, path.basename(basename, ".java")]
121
+ : [basename];
122
+ }
123
+ /**
124
+ * Backslashes are ignored: Playwright and Jest read the path as a regular
125
+ * expression, so the agent escapes it (`a\.spec\.ts`).
126
+ *
127
+ * A trailing shell comment is removed first. `npm test # a_test.py` names the
128
+ * file only in a comment and would run the whole suite, and the check exists
129
+ * to stop exactly that. This is a name check, not a parse: a command that
130
+ * mentions the file in some other way it does not run still passes.
131
+ */
132
+ /** The config an explicit `--config <path>` names, resolved against `cwd`, or
133
+ * undefined when the command carries none. `$SKYRAMP_PLAYWRIGHT_CONFIG` is
134
+ * ours and is not a repository config. */
135
+ export function explicitPlaywrightConfig(command, cwd) {
136
+ const m = /--config[= ]\s*("[^"]+"|'[^']+'|[^\s]+)/.exec(command);
137
+ if (!m)
138
+ return undefined;
139
+ const raw = m[1].replace(/^["']|["']$/g, "");
140
+ if (raw.includes("SKYRAMP_PLAYWRIGHT_CONFIG"))
141
+ return undefined;
142
+ const resolved = path.isAbsolute(raw) ? raw : path.join(cwd, raw);
143
+ return fs.existsSync(resolved) ? resolved : undefined;
144
+ }
145
+ /** The command with its `--config <repo config>` pointed at the wrapper. The
146
+ * wrapper imports that same config, so the run keeps the repository's testDir
147
+ * and projects and gains the video overlay. Without this swap a command that
148
+ * names a config runs the config directly and records nothing. */
149
+ export function pointConfigAtWrapper(command, wrapperPath) {
150
+ return command.replace(/(--config[= ]\s*)("[^"]+"|'[^']+'|[^\s]+)/, (_m, flag) => `${flag}${JSON.stringify(wrapperPath)}`);
151
+ }
152
+ export function commandNamesTestFile(command, testFile) {
153
+ const withoutComment = command.replace(/(^|\s)#.*$/, "$1");
154
+ const unescaped = withoutComment.replace(/\\/g, "");
155
+ return testFileNames(testFile).some((name) => unescaped.includes(name));
156
+ }
157
+ /** pytest exits 4 on a usage error; without pytest-playwright `--video` is one. */
158
+ export function pytestRejectedVideoOptions(run) {
159
+ return (run.exitCode === 4 &&
160
+ /unrecognized arguments:[^\n]*--video/.test(run.output));
161
+ }
162
+ /** What the output says when the command ended before any test ran. Each entry
163
+ * is the line the runner prints instead of a result: a missing package, or a
164
+ * file the runner never matched. Node, Playwright and Jest all exit 1 for
165
+ * these, so without this the run is recorded as a failing test. */
166
+ const NO_RUN_PATTERNS = [
167
+ /^.*\bError: Cannot find module\b.*$/m,
168
+ /^Cannot find module\b.*$/m,
169
+ /^.*\bERR_MODULE_NOT_FOUND\b.*$/m,
170
+ /^(?:Error: )?No tests found\b.*$/im,
171
+ // npx --no-install refuses to fetch a runner the repository does not have.
172
+ /^.*\bnpx canceled due to missing packages\b.*$/m,
173
+ // Jest matched the file and loaded it, and it declared no test.
174
+ /^.*\bYour test suite must contain at least one test\b.*$/m,
175
+ ];
176
+ /** The line proving no test ran, or undefined when the output does not say so. */
177
+ export function noTestRanReason(output) {
178
+ for (const pattern of NO_RUN_PATTERNS) {
179
+ const hit = pattern.exec(output);
180
+ if (hit)
181
+ return hit[0].trim();
182
+ }
183
+ return undefined;
184
+ }
185
+ export function verdictForExit(exitCode, timedOut, output = "") {
186
+ if (timedOut)
187
+ return TestExecutionStatus.Error;
188
+ if (exitCode === 0)
189
+ return TestExecutionStatus.Pass;
190
+ // Exit 1 is the failing-test code, but it is also what a runner returns when
191
+ // it never got as far as a test. Error keeps those out of the failing count.
192
+ if (exitCode === 1)
193
+ return noTestRanReason(output)
194
+ ? TestExecutionStatus.Error
195
+ : TestExecutionStatus.Fail;
196
+ return TestExecutionStatus.Error;
197
+ }
198
+ /**
199
+ * How long output may keep arriving after the shell exits. A background process the
200
+ * command started holds the pipes open, so waiting for them to close would wait for it.
201
+ */
202
+ const EXIT_OUTPUT_GRACE_MS = 2_000;
203
+ /**
204
+ * Runs `command` through the shell in its own process group, so a timeout kills
205
+ * the runner's children too. stdout and stderr share one buffer in arrival order.
206
+ */
207
+ export function spawnTestCommand(opts) {
208
+ const startedAt = Date.now();
209
+ return new Promise((resolve) => {
210
+ let output = "";
211
+ let timedOut = false;
212
+ let spawnError;
213
+ let exitCode = null;
214
+ let settled = false;
215
+ let grace;
216
+ const child = spawn(opts.command, {
217
+ cwd: opts.cwd,
218
+ env: opts.env,
219
+ shell: true,
220
+ detached: process.platform !== "win32",
221
+ });
222
+ const killGroup = () => {
223
+ try {
224
+ if (process.platform !== "win32" && child.pid) {
225
+ process.kill(-child.pid, "SIGKILL");
226
+ }
227
+ else if (child.pid) {
228
+ // Windows has no process group: child.kill reaches the shell and
229
+ // leaves the runner beneath it alive, still writing to the checkout.
230
+ // /T takes the tree, /F forces it.
231
+ spawn("taskkill", ["/pid", String(child.pid), "/T", "/F"], {
232
+ stdio: "ignore",
233
+ });
234
+ }
235
+ }
236
+ catch {
237
+ // already gone
238
+ }
239
+ };
240
+ const finish = () => {
241
+ if (settled)
242
+ return;
243
+ settled = true;
244
+ clearTimeout(timer);
245
+ clearTimeout(grace);
246
+ if (output.length > MAX_OUTPUT_CHARS) {
247
+ output = output.slice(-MAX_OUTPUT_CHARS);
248
+ truncated = true;
249
+ }
250
+ resolve({
251
+ exitCode,
252
+ output: (truncated
253
+ ? `[output truncated to its last ${MAX_OUTPUT_CHARS} characters]\n`
254
+ : "") + stripVTControlCharacters(output),
255
+ timedOut,
256
+ spawnError,
257
+ duration: Date.now() - startedAt,
258
+ });
259
+ };
260
+ const timer = setTimeout(() => {
261
+ timedOut = true;
262
+ killGroup();
263
+ }, opts.timeoutMs);
264
+ let truncated = false;
265
+ const append = (d) => {
266
+ output += String(d);
267
+ if (output.length > 2 * MAX_OUTPUT_CHARS) {
268
+ output = output.slice(-MAX_OUTPUT_CHARS);
269
+ truncated = true;
270
+ }
271
+ };
272
+ child.stdout?.on("data", append);
273
+ child.stderr?.on("data", append);
274
+ child.on("error", (err) => {
275
+ spawnError = String(err);
276
+ finish();
277
+ });
278
+ child.on("exit", (code) => {
279
+ exitCode = code;
280
+ clearTimeout(timer);
281
+ grace = setTimeout(() => {
282
+ killGroup();
283
+ finish();
284
+ }, EXIT_OUTPUT_GRACE_MS);
285
+ });
286
+ child.on("close", (code) => {
287
+ exitCode = code ?? exitCode;
288
+ finish();
289
+ });
290
+ });
291
+ }
54
292
  /**
55
293
  * Where a real HTTP 401 shows up in runner output. Every entry is a shape taken from
56
294
  * actual skyramp_execute_test output in the eval logs, not from guesswork:
@@ -81,7 +319,6 @@ const HTTP_401_SHAPES = [
81
319
  // there, while a test TITLE ("should return 401 Unauthorized for an expired
82
320
  // token") always has words in front of it and is not evidence of a response.
83
321
  /^\s*401\s+unauthori[sz]ed\b/im,
84
- // Status line, as curl -i and Go's httputil print it.
85
322
  /\bHTTP\/[\d.]+\s+401\b/i,
86
323
  // A status FIELD set to 401 — the value the response carried. Only `:` is
87
324
  // accepted: `== 401` and `= 401` are an assertion or echoed test source, which
@@ -90,8 +327,6 @@ const HTTP_401_SHAPES = [
90
327
  // (`Expected: {"status": 401}`) alongside `Received: {"status": 500}` is not a
91
328
  // 401 the app sent. Reject the line rather than the value.
92
329
  /^(?!.*\bexpected\b).*\b(?:status|status[_-]?code|statuscode|code|errorcode)\\?"?\s*:\s*401\b/im,
93
- // Playwright prints both compared values. `Received` is what the app sent;
94
- // `Expected` is what the test wanted, so it is not evidence of a 401.
95
330
  /^\s*Received:\s*401\b/im,
96
331
  // pytest assertion rewriting. A Skyramp-generated test reads
97
332
  // `assert response.status_code == N`, so the OBSERVED value is on the left and
@@ -124,14 +359,10 @@ export function resolveRebaselineSnapshots(requested, phase) {
124
359
  return { snapshots };
125
360
  }
126
361
  /**
127
- * Snapshot of the requested baselines under `<spec>-snapshots/` (SKYR-4298): for
128
- * each requested name, every PNG whose name matches the stem (a spec may hold both
129
- * `<stem>-linux.png` and `<stem>-chromium-linux.png`; latching onto one of them would
130
- * misreport the other) with its size and mtime. Taken before and after the run so
131
- * the tool can tell the agent which baselines were actually rewritten — SmartPlaywright
132
- * is the only party that knows the exact filename, and an executor image that lacks
133
- * SKYRAMP_UPDATE_SNAPSHOTS (or a name matching no toHaveScreenshot call) leaves
134
- * every file untouched.
362
+ * For each requested name, every PNG under `<spec>-snapshots/` whose name matches the
363
+ * stem, with size and mtime. A spec may hold both `<stem>-linux.png` and
364
+ * `<stem>-chromium-linux.png`; latching onto one would misreport the other. Taken
365
+ * before and after the run to tell which baselines were actually rewritten.
135
366
  */
136
367
  export function readBaselineState(specFile, requested) {
137
368
  const dir = snapshotDirFor(specFile);
@@ -159,10 +390,6 @@ export function readBaselineState(specFile, requested) {
159
390
  }
160
391
  return state;
161
392
  }
162
- /**
163
- * Which requested baselines changed on disk between two readBaselineState calls, and
164
- * which files carried the change (the ones to stage).
165
- */
166
393
  export function diffBaselineState(before, after) {
167
394
  const refreshed = [];
168
395
  const notRefreshed = [];
@@ -236,16 +463,13 @@ export function authorizeRebaseline(stateData, testFile, requested) {
236
463
  return {};
237
464
  }
238
465
  /**
239
- * Reconcile the persisted verdict with what the executor actually did (SKYR-4298).
240
- * An executor image whose @skyramp/skyramp predates SKYRAMP_UPDATE_SNAPSHOTS rewrites
241
- * nothing; left alone, the verdict would still promise a refresh, the report gate
242
- * would refuse the report, and nothing in the prompt makes the agent's way out
243
- * deterministic. So: names that were not refreshed are dropped from the verdict; a
244
- * rebaseline-only UPDATE with nothing left becomes VERIFY (the test stays red, the
245
- * rationale says why), and an UPDATE that also carried edits keeps UPDATE and is
246
- * held to its edit. The report then reflects what happened, not what was asked.
466
+ * Reconcile the persisted verdict with what the run actually did (SKYR-4298). A
467
+ * @skyramp/skyramp that predates SKYRAMP_UPDATE_SNAPSHOTS rewrites nothing; left
468
+ * alone, the verdict would still promise a refresh and the report gate would refuse
469
+ * the report. Names not refreshed are dropped; a rebaseline-only UPDATE with nothing
470
+ * left becomes VERIFY, and an UPDATE that also carried edits is held to its edit.
247
471
  */
248
- export function applyRefreshOutcomeToVerdicts(verdicts, testFile, outcome, executorImage) {
472
+ export function applyRefreshOutcomeToVerdicts(verdicts, testFile, outcome) {
249
473
  if (outcome.notRefreshed.length === 0)
250
474
  return { verdicts };
251
475
  let note;
@@ -255,7 +479,7 @@ export function applyRefreshOutcomeToVerdicts(verdicts, testFile, outcome, execu
255
479
  v.action !== DriftAction.Update)
256
480
  return v;
257
481
  const remaining = (v.rebaselineSnapshots ?? []).filter((n) => !outcome.notRefreshed.includes(n));
258
- const reason = `baseline refresh of ${outcome.notRefreshed.join(", ")} was not applied by ${executorImage} (its @skyramp/skyramp lacks SKYRAMP_UPDATE_SNAPSHOTS, or the name matches no toHaveScreenshot call)`;
482
+ const reason = `baseline refresh of ${outcome.notRefreshed.join(", ")} was not applied by the test run (its @skyramp/skyramp lacks SKYRAMP_UPDATE_SNAPSHOTS, or the name matches no toHaveScreenshot call)`;
259
483
  if (remaining.length === 0 && v.rebaselineOnly) {
260
484
  note = `Verdict for ${path.basename(testFile)} downgraded UPDATE → VERIFY: ${reason}. The test stays as it is; report it honestly.`;
261
485
  const { rebaselineSnapshots: _dropped, rebaselineOnly: _only, ...rest } = v;
@@ -290,10 +514,11 @@ export function describeRefreshOutcome(outcome) {
290
514
  parts.push(`Visual baselines refreshed: ${outcome.refreshed.join(", ")}.`);
291
515
  }
292
516
  if (outcome.notRefreshed.length > 0) {
293
- parts.push(`Visual baselines NOT refreshed: ${outcome.notRefreshed.join(", ")} — the executor image may lack SKYRAMP_UPDATE_SNAPSHOTS support (needs @skyramp/skyramp with SKYR-4298), or the name matches no toHaveScreenshot() call in this spec. Do not report these as refreshed.`);
517
+ parts.push(`Visual baselines NOT refreshed: ${outcome.notRefreshed.join(", ")} — the test's @skyramp/skyramp may lack SKYRAMP_UPDATE_SNAPSHOTS support, or the name matches no toHaveScreenshot() call in this spec. Do not report these as refreshed.`);
294
518
  }
295
519
  return parts.join(" ");
296
520
  }
521
+ /**
297
522
  /**
298
523
  * The fix-and-rerun attempt cap (SKYR-4460), enforced where the prompt's prose
299
524
  * cannot be: a file that has already received `cap` runs in the final phase
@@ -339,13 +564,6 @@ export function setTransientRetryDelayForTests(ms) {
339
564
  * error itself: two executions, one attempt. The prompt withholds the agent's
340
565
  * own unchanged re-run when it sees this, so one attempt is never three runs. */
341
566
  const TRANSIENT_RETRY_NOTE = `\n\nNote: the executor hit a transient connection error and re-ran this file once unchanged before answering; the two runs count as one attempt. Do not re-run it unchanged again — fix the cause the output names, or report it.`;
342
- /** The warning both result paths append when the run's state could not take
343
- * the result: the attempt was counted at reservation, the record was not. */
344
- function persistFailureWarning(stateFile, reason) {
345
- return (`\n\nWarning: this run's result could not be recorded in stateFile ` +
346
- `${stateFile} (${reason}). The attempt was counted; the report will read ` +
347
- `this file's result as Unknown unless the state file is repaired.`);
348
- }
349
567
  export function describeAttempt(priorAfterRuns, cap, opts = {}) {
350
568
  const attempt = priorAfterRuns + 1;
351
569
  const remaining = Math.max(0, cap - attempt);
@@ -384,16 +602,12 @@ export function buildExecutionFailureText(result, opts = {}) {
384
602
  sections.push(output);
385
603
  }
386
604
  else if (errors.length > 0) {
387
- // The executor reported a cause of its own (e.g. "Docker image setup
388
- // failed"). Saying the cause cannot be determined would contradict the
389
- // Errors line below and throw away the only thing known about the failure.
390
- sections.push("The executor captured no test output, so there are no per-test diagnostics. " +
391
- "It did report the error below, which is the cause on record — use it. Do " +
392
- "NOT infer anything the Errors line does not say, and leave the generated " +
393
- "test unchanged.");
605
+ sections.push("The command printed no output, so there are no per-test diagnostics. " +
606
+ "The error below is the cause on record — use it. Do NOT infer anything " +
607
+ "the Errors line does not say, and leave the generated test unchanged.");
394
608
  }
395
609
  else {
396
- sections.push("The executor captured no output, so this failure carries no diagnostics of " +
610
+ sections.push("The command printed no output, so this failure carries no diagnostics of " +
397
611
  "its own and the cause cannot be determined from it. The test runner, the " +
398
612
  "test process, or the application could each have died silently. Do NOT " +
399
613
  "report the app as unreachable or misconfigured on this basis — nothing " +
@@ -422,69 +636,415 @@ export function buildExecutionFailureText(result, opts = {}) {
422
636
  }
423
637
  return sections.join("\n\n");
424
638
  }
639
+ export const inputSchema = {
640
+ commandOverride: z
641
+ .string()
642
+ .trim()
643
+ .min(1)
644
+ .optional()
645
+ .describe("The shell command that runs this one test file with the repository's own test runner, run on this host. Prefer it: the test then runs the way the repository's own CI runs it. Take the command from the suite's testRunCommand in .skyramp/workspace.yml, a Makefile target, a package script, or the command the repository's CI runs, and limit it to this one file. When workspace.yml records more than one suite, take it from the suite whose pathGlobs match the file: another suite's runner either fails to load it or reports a result for a file it never ran. If the repository has none, use the framework's default: `npx --no-install <runner> <file>` for playwright, jest, vitest or mocha, so a runner the repository does not have fails the run instead of downloading a different version; `python -m pytest <file> -q` with the interpreter the repository's own tests use; `mvn -q test -Dtest=<class> -DfailIfNoTests=false` for junit. It must name the test file; a command that does not is refused without running. For a TypeScript or JavaScript ui or e2e test, the server sets SKYRAMP_PLAYWRIGHT_CONFIG to a config that records video. Pass it even when you expect the runner or a package to be missing: the failure output is what names what is missing, and a check you make before the run reaches no repair. Omit it to run the test in the Skyramp executor container instead, which needs workspacePath — that is the fallback for a repository this host cannot run, not the first choice."),
646
+ cwd: z
647
+ .string()
648
+ .optional()
649
+ .describe("Absolute path of the directory commandOverride runs in. Required with commandOverride."),
650
+ workspacePath: z
651
+ .string()
652
+ .optional()
653
+ .describe("Absolute path of the Skyramp workspace. Required when commandOverride is omitted, because the executor resolves the service and its base URL from it."),
654
+ testFile: z.string().describe("Absolute path to the test file to execute."),
655
+ language: z
656
+ .nativeEnum(ProgrammingLanguage)
657
+ .describe("Programming language of the test file to execute (e.g., python, javascript, typescript, java)"),
658
+ testType: z
659
+ .nativeEnum(TestType)
660
+ .describe("Type of the test to execute. Note: 'mock' is NOT a valid test type — mock files are deployed via their apply_mock() function, not executed as tests."),
661
+ phase: z
662
+ .enum(["before", "after"])
663
+ .default("after")
664
+ .describe("Execution phase for maintained tests: 'before' captures pre-edit baseline; 'after' (default) records post-edit result."),
665
+ stateFile: z
666
+ .string()
667
+ .optional()
668
+ .describe("Path to state file from skyramp_analyze_changes. Always pass when available — results are written back so skyramp_submit_report can override before/afterStatus with ground-truth pass/fail."),
669
+ repository: z
670
+ .string()
671
+ .trim()
672
+ .min(1)
673
+ .describe("The owner/repo whose analysis section to write execution results into (e.g. 'letsramp/api-insight'). Set it on every call — the primary's owner/repo for a primary-repo test, or a related repo's owner/repo for that repo's section of the run-scoped stateFile."),
674
+ token: z
675
+ .string()
676
+ .optional()
677
+ .describe("Explicit authentication token, set as SKYRAMP_TEST_TOKEN. Omit it, or pass an empty string, to use SKYRAMP_TEST_TOKEN from the server's environment. Use `unauthenticated: true` for no auth."),
678
+ unauthenticated: z
679
+ .boolean()
680
+ .optional()
681
+ .describe("Set true to force this test execution to carry NO auth token, even if SKYRAMP_TEST_TOKEN is set in the environment or a token was passed. Use for tests that must assert 401/403 unauthenticated behavior."),
682
+ rebaselineSnapshots: z
683
+ .array(rebaselineSnapshotNameSchema)
684
+ .optional()
685
+ .describe('UI tests only. toHaveScreenshot() baseline filenames (e.g. ["page-001.png"]) this run REPLACES instead of comparing against, because the PR intentionally changed how the captured page/element/region looks (SKYR-4298). ' +
686
+ "Pass exactly the list skyramp_actions returned as rebaseline_snapshots for this spec — the tool checks it against the persisted UPDATE verdict in stateFile (so stateFile is required) and refuses names the verdict did not authorize, a spec with no such verdict, or a spec whose phase: 'before' run has not been recorded yet. Use the name as the test passes it (page-001.png), not the on-disk file (page-001-chromium-linux.png). " +
687
+ "The refreshed PNGs land beside the spec and are delivered with the Testbot PR as image diffs; the result names which baselines were refreshed and which were not. Never use this to silence a screenshot mismatch the diff does not explain."),
688
+ timeout: z
689
+ .number()
690
+ .int()
691
+ .positive()
692
+ .max(MAX_TIMEOUT_MS, `timeout must be at most ${MAX_TIMEOUT_MS} ms`)
693
+ .optional()
694
+ .describe(`Milliseconds before the command is killed and the run recorded as Error. Default ${DEFAULT_TIMEOUT_MS}.`),
695
+ };
696
+ /** Returns a warning when the result, or its pre-edit baseline, was not saved. */
697
+ async function writeExecutionToState(params, result, reserved) {
698
+ if (!params.stateFile)
699
+ return undefined;
700
+ const where = `testFile ${params.testFile} in repository ${params.repository}`;
701
+ try {
702
+ const written = await persistTestExecutionResult(params.stateFile, params.repository, params.testType, params.phase, result,
703
+ // A reserved attempt already advanced the counters before the run.
704
+ { reserved });
705
+ if (!written.saved) {
706
+ return `This result was not saved: the stateFile has no section for repository ${params.repository}.`;
707
+ }
708
+ if (params.phase === "before" && !written.matchedExistingTest) {
709
+ return `The phase: "before" result was not saved as a baseline: the stateFile lists no existing test for ${where}.`;
710
+ }
711
+ return undefined;
712
+ }
713
+ catch (err) {
714
+ // The attempt was counted at reservation, so the count is right. The RESULT
715
+ // is what the report reads for afterStatus, so an unrecorded one has to
716
+ // reach the agent, not only the log.
717
+ return (`This result for ${where} was not saved to the stateFile: ${err.message}. ` +
718
+ `The attempt was counted; the report will read this file's result as Unknown ` +
719
+ `unless the state file is repaired.`);
720
+ }
721
+ }
722
+ /** Stages rewritten baselines and reconciles the verdict; returns the text to append. */
723
+ async function settleRefresh(params, snapshots, before) {
724
+ const outcome = diffBaselineState(before, readBaselineState(params.testFile, snapshots));
725
+ let text = describeRefreshOutcome(outcome);
726
+ // The rewritten files themselves, never the directory: `git add` on the directory
727
+ // would ship anything else sitting there under one authorized baseline's authority.
728
+ for (const name of outcome.refreshed) {
729
+ for (const file of outcome.refreshedFiles[name] ?? []) {
730
+ try {
731
+ await stageGeneratedPaths(path.join(snapshotDirFor(params.testFile), file));
732
+ }
733
+ catch (err) {
734
+ logger.warning(`Could not stage refreshed visual baseline ${file}: ${err.message}`);
735
+ }
736
+ }
737
+ }
738
+ if (outcome.notRefreshed.length > 0 && params.stateFile) {
739
+ try {
740
+ const stateManager = StateManager.fromStatePath(params.stateFile);
741
+ const repo = params.repository;
742
+ const stateData = await stateManager.readRepoData(repo);
743
+ if (repo && stateData?.maintenanceVerdicts) {
744
+ const reconciled = applyRefreshOutcomeToVerdicts(stateData.maintenanceVerdicts, params.testFile, outcome);
745
+ await stateManager.updateRepoData({ ...stateData, maintenanceVerdicts: reconciled.verdicts }, { repo });
746
+ if (reconciled.note)
747
+ text += ` ${reconciled.note}`;
748
+ }
749
+ }
750
+ catch (err) {
751
+ logger.warning(`Could not reconcile the maintenance verdict with the refresh outcome: ${err.message}`);
752
+ }
753
+ }
754
+ return text;
755
+ }
756
+ /** The fallback: no commandOverride, so the Skyramp executor container runs the
757
+ * test. Kept while host execution is proven — the container resolves the
758
+ * service and its base URL from the workspace, which the host path leaves to
759
+ * the agent's own command.
760
+ *
761
+ * It records the same result the host path records, through the same writer,
762
+ * so the report cannot tell which path produced a run. */
763
+ async function runInExecutor(params, snapshots, budget, onProgress, sendProgress) {
764
+ // runTest validated this for the executor mode before anything wrote state.
765
+ const workspacePath = params.workspacePath;
766
+ const token = resolveEffectiveToken(params.unauthenticated, params.token, process.env.SKYRAMP_TEST_TOKEN);
767
+ // Delivered tests and the state file sit outside workspacePath in a
768
+ // cross-repo run, so the executor needs the tests-repo root both to mount
769
+ // them and to match the service that owns the file.
770
+ const testRepoPath = getTestsRepoDir();
771
+ const { baseUrl, candidates, dockerNetwork } = await getWorkspaceBaseUrl(workspacePath, params.testFile, params.language, testRepoPath);
772
+ if (!baseUrl && candidates.length > 0) {
773
+ return toolError([
774
+ "Cannot determine SKYRAMP_TEST_BASE_URL — the test file matches more than one service:",
775
+ ...candidates.map((c) => ` \u2022 ${c.serviceName}: ${c.baseUrl}`),
776
+ "",
777
+ "Set SKYRAMP_TEST_BASE_URL to the right service URL, or give each service its own testDirectory in .skyramp/workspace.yml.",
778
+ ].join("\n"));
779
+ }
780
+ let injectedBaseUrl = false;
781
+ if (baseUrl && !process.env.SKYRAMP_TEST_BASE_URL) {
782
+ process.env.SKYRAMP_TEST_BASE_URL = baseUrl;
783
+ injectedBaseUrl = true;
784
+ }
785
+ const execOptions = {
786
+ testFile: params.testFile,
787
+ workspacePath,
788
+ testRepoPath,
789
+ language: params.language,
790
+ testType: params.testType,
791
+ token,
792
+ dockerNetwork,
793
+ useHostNetwork: false,
794
+ ...(snapshots.length > 0 ? { rebaselineSnapshots: snapshots } : {}),
795
+ ...(params.timeout !== undefined ? { timeout: params.timeout } : {}),
796
+ };
797
+ let result;
798
+ // The executor's own retry of a transient connection error: two executions
799
+ // billed as ONE attempt, on purpose — the SUT's hiccup is not the agent's fix
800
+ // to make. An EOF or a refused connection means the SUT is not ready yet, not
801
+ // that the test failed. Only a throw from the executor is retried here; a
802
+ // connection error inside the test's own output reaches the agent as a normal
803
+ // failure.
804
+ let transientRetried = false;
805
+ try {
806
+ const service = new TestExecutionService();
807
+ try {
808
+ result = await service.executeTest(execOptions, onProgress);
809
+ }
810
+ catch (firstErr) {
811
+ const errMsg = firstErr instanceof Error ? firstErr.message : String(firstErr);
812
+ if (!/\bEOF\b|connection refused|ECONNREFUSED|ECONNRESET/i.test(errMsg)) {
813
+ throw firstErr;
814
+ }
815
+ logger.info(`Test execution hit transient connection error, retrying after ${transientRetryDelayMs}ms...`, { error: errMsg });
816
+ await sendProgress(50, 100, "SUT connection error — retrying in 10s...");
817
+ await new Promise((r) => setTimeout(r, transientRetryDelayMs));
818
+ transientRetried = true;
819
+ try {
820
+ result = await service.executeTest(execOptions, onProgress);
821
+ }
822
+ catch (secondErr) {
823
+ // Both runs threw: the caller's catch owes the agent the retry note, or
824
+ // the prompt lets it spend an unchanged re-run the tool already used.
825
+ throw Object.assign(secondErr, { transientRetried: true });
826
+ }
827
+ }
828
+ }
829
+ finally {
830
+ // Unset only what this call set; a caller's own value must survive.
831
+ if (injectedBaseUrl)
832
+ delete process.env.SKYRAMP_TEST_BASE_URL;
833
+ }
834
+ const warnings = [...result.warnings];
835
+ if (transientRetried) {
836
+ warnings.push("The executor hit a transient connection error and re-ran this file once unchanged before answering. This is the second run's result, counted as one attempt. Do not re-run it unchanged again — fix the cause the output names, or report it.");
837
+ }
838
+ const stateWarning = await writeExecutionToState(params, result, budget.reserved);
839
+ if (stateWarning)
840
+ warnings.push(stateWarning);
841
+ await recordExecutionVideo(result, params.stateFile);
842
+ const compose = (text) => [
843
+ withVideoInfo(text, result.videoPath),
844
+ ...warnings.map((w) => `Warning: ${w}`),
845
+ ]
846
+ .filter(Boolean)
847
+ .join("\n\n") + budget.noRunNote;
848
+ if (result.status !== TestExecutionStatus.Pass) {
849
+ return toolError(compose(buildExecutionFailureText(result, {
850
+ unchangedRerun: budget.unchangedRerun,
851
+ }) + budget.attemptLine));
852
+ }
853
+ return {
854
+ content: [
855
+ {
856
+ type: "text",
857
+ text: compose(`Test execution passed in the executor (duration=${result.duration}ms).\n\n${result.output ?? ""}`),
858
+ },
859
+ ],
860
+ };
861
+ }
862
+ /** A throw after the attempt was reserved has spent it (a Docker image setup,
863
+ * a workspace check, a spawn that died). Land an Error result so the record is
864
+ * not left as the reservation placeholder, and tell the agent the attempt went. */
865
+ async function accountForThrow(params, err, budget) {
866
+ let persistNote = "";
867
+ if (budget.reserved) {
868
+ const warning = await writeExecutionToState(params, {
869
+ testFile: params.testFile,
870
+ status: TestExecutionStatus.Error,
871
+ executedAt: new Date().toISOString(),
872
+ duration: 0,
873
+ errors: [err.message],
874
+ warnings: [],
875
+ output: "",
876
+ }, true);
877
+ if (warning)
878
+ persistNote = `\n\nWarning: ${warning}`;
879
+ }
880
+ const retryNote = err.transientRetried
881
+ ? TRANSIENT_RETRY_NOTE
882
+ : "";
883
+ return toolError(`Test execution failed: ${err.message}${retryNote}${budget.attemptLine}${persistNote}${budget.noRunNote}`);
884
+ }
885
+ /** The preferred path: the agent supplied the command, so the server runs it on
886
+ * this host in `cwd` and reads the verdict off the exit code. */
887
+ async function runOnHost(params, command, hostCwd, snapshots, budget) {
888
+ const warnings = [];
889
+ const invEnv = {};
890
+ const workspaceRoot = resolveWorkspaceRoot(hostCwd);
891
+ const token = resolveEffectiveToken(params.unauthenticated, params.token, process.env.SKYRAMP_TEST_TOKEN);
892
+ if (token)
893
+ invEnv.SKYRAMP_TEST_TOKEN = token;
894
+ else if (!params.unauthenticated) {
895
+ warnings.push("No auth token available — authenticated endpoints will likely return 401. Set SKYRAMP_TEST_TOKEN or pass token/unauthenticated.");
896
+ }
897
+ if (snapshots.length > 0) {
898
+ invEnv.SKYRAMP_UPDATE_SNAPSHOTS = snapshots.join(",");
899
+ }
900
+ let videoDir;
901
+ let wrapper;
902
+ /** The config the server picked because the command named none. */
903
+ let autoFoundConfig;
904
+ if (isBrowserTest(params.testType)) {
905
+ videoDir = path.join(workspaceRoot, ".skyramp", "videos", videoSubdirName(params.testFile));
906
+ fs.mkdirSync(videoDir, { recursive: true });
907
+ if (params.language === ProgrammingLanguage.PYTHON) {
908
+ invEnv.PYTEST_ADDOPTS = appendPytestVideoOpts(process.env.PYTEST_ADDOPTS, videoDir);
909
+ }
910
+ else if (params.language === ProgrammingLanguage.TYPESCRIPT ||
911
+ params.language === ProgrammingLanguage.JAVASCRIPT) {
912
+ const namedConfig = explicitPlaywrightConfig(command, hostCwd);
913
+ // A repository with one Playwright config per suite gets whichever sits
914
+ // nearest the test file, and its testDir may not collect that file. The
915
+ // agent cannot see the choice, so name it when the run finds no test.
916
+ autoFoundConfig = namedConfig
917
+ ? undefined
918
+ : findRepoPlaywrightConfig(params.testFile, hostCwd);
919
+ wrapper = writeWrapperConfig({
920
+ // The wrapper replaces whatever --config the command carried, so an
921
+ // explicit one must be imported or the repository loses its projects,
922
+ // testDir and CI settings -- the thing the wrapper exists to keep.
923
+ repoConfigPath: namedConfig ?? autoFoundConfig,
924
+ fallbackDir: hostCwd,
925
+ outputDir: videoDir,
926
+ testFile: path.resolve(params.testFile),
927
+ commandPassesBrowserFlag: commandPassesBrowserFlag(command),
928
+ });
929
+ invEnv.SKYRAMP_PLAYWRIGHT_CONFIG = wrapper.path;
930
+ // A command that names its own config would otherwise load it directly and
931
+ // record no video, which is how a passing run loses its evidence.
932
+ if (namedConfig)
933
+ command = pointConfigAtWrapper(command, wrapper.path);
934
+ }
935
+ }
936
+ // testbot exports NODE_PATH=<mcp>/node_modules job-wide. Inheriting it lets the test
937
+ // load our @skyramp/skyramp and @playwright/test instead of the repository's own.
938
+ const env = { ...process.env, ...invEnv };
939
+ delete env.NODE_PATH;
940
+ if (!token)
941
+ delete env.SKYRAMP_TEST_TOKEN;
942
+ const baselinesBefore = snapshots.length > 0 ? readBaselineState(params.testFile, snapshots) : {};
943
+ const executedAt = new Date().toISOString();
944
+ const timeoutMs = params.timeout ?? DEFAULT_TIMEOUT_MS;
945
+ let run;
946
+ try {
947
+ run = await spawnTestCommand({
948
+ command,
949
+ cwd: hostCwd,
950
+ env,
951
+ timeoutMs,
952
+ });
953
+ }
954
+ finally {
955
+ // The wrapper sits in the customer's repo; a broad `git add` would ship it.
956
+ wrapper?.cleanup();
957
+ }
958
+ if (invEnv.PYTEST_ADDOPTS && pytestRejectedVideoOptions(run)) {
959
+ const plainEnv = { ...env };
960
+ if (process.env.PYTEST_ADDOPTS === undefined)
961
+ delete plainEnv.PYTEST_ADDOPTS;
962
+ else
963
+ plainEnv.PYTEST_ADDOPTS = process.env.PYTEST_ADDOPTS;
964
+ run = await spawnTestCommand({
965
+ command,
966
+ cwd: hostCwd,
967
+ env: plainEnv,
968
+ timeoutMs,
969
+ });
970
+ warnings.push("pytest did not recognize the added --video and --output options, so the test ran again without them. No video was recorded.");
971
+ videoDir = undefined;
972
+ }
973
+ const errors = [];
974
+ if (run.timedOut)
975
+ errors.push(`the command timed out after ${timeoutMs} ms`);
976
+ if (run.spawnError)
977
+ errors.push(run.spawnError);
978
+ const videoPath = videoDir ? collectVideoPath(videoDir) : undefined;
979
+ if (videoDir && !videoPath) {
980
+ warnings.push(`No video.webm was recorded under ${videoDir}. The verdict stands; the report has no video for this test.`);
981
+ }
982
+ const noRunReason = run.exitCode === 1 ? noTestRanReason(run.output) : undefined;
983
+ if (noRunReason) {
984
+ warnings.push(`No test ran here: the output reports "${noRunReason}". Exit 1 is the runner's, not a test's.`);
985
+ if (autoFoundConfig) {
986
+ warnings.push(`SKYRAMP_PLAYWRIGHT_CONFIG wraps ${path.relative(hostCwd, autoFoundConfig) || autoFoundConfig}, the nearest Playwright config above the test file, because the command named none. Its testDir decides which files the run collects — name the config the file belongs to if this is the wrong one.`);
987
+ }
988
+ }
989
+ const result = {
990
+ testFile: params.testFile,
991
+ status: verdictForExit(run.exitCode, run.timedOut || !!run.spawnError, run.output),
992
+ executedAt,
993
+ duration: run.duration,
994
+ errors,
995
+ warnings,
996
+ output: run.output,
997
+ ...(run.exitCode !== null ? { exitCode: run.exitCode } : {}),
998
+ ...(videoPath ? { videoPath } : {}),
999
+ };
1000
+ const stateWarning = await writeExecutionToState(params, result, budget.reserved);
1001
+ if (stateWarning)
1002
+ warnings.push(stateWarning);
1003
+ await recordExecutionVideo(result, params.stateFile);
1004
+ const refreshText = snapshots.length > 0
1005
+ ? await settleRefresh(params, snapshots, baselinesBefore)
1006
+ : "";
1007
+ const compose = (text) => [
1008
+ withVideoInfo(text, videoPath),
1009
+ refreshText,
1010
+ ...warnings.map((w) => `Warning: ${w}`),
1011
+ ]
1012
+ .filter(Boolean)
1013
+ .join("\n\n") + budget.noRunNote;
1014
+ if (result.status !== TestExecutionStatus.Pass) {
1015
+ return toolError(compose(buildExecutionFailureText(result, {
1016
+ unchangedRerun: budget.unchangedRerun,
1017
+ }) + budget.attemptLine));
1018
+ }
1019
+ return {
1020
+ content: [
1021
+ {
1022
+ type: "text",
1023
+ text: compose(`Test execution passed (exitCode=0, duration=${result.duration}ms).\n\n${result.output ?? ""}`),
1024
+ },
1025
+ ],
1026
+ };
1027
+ }
425
1028
  export function registerExecuteSkyrampTestTool(server) {
426
1029
  server.registerTool(TOOL_NAME, {
427
- description: `Execute a Skyramp-generated test in isolated containerized environments for reliable, deterministic testing. Call this once a test file exists on disk (from a skyramp_*_test_generation tool). First-time execution may take longer while Docker images download — this is expected, not a failure.`,
1030
+ description: "Run one test file and record the verdict from the exit code: 0 is Pass, 1 is Fail, any other exit or a timeout is Error. With `commandOverride` the server runs that command on this machine in `cwd`, with the server's environment. Without it the Skyramp executor container runs the test, which needs `workspacePath`. Either way the output is kept from its end, up to 200,000 characters.",
428
1031
  annotations: {
429
1032
  readOnlyHint: false,
430
- destructiveHint: false,
1033
+ destructiveHint: true,
431
1034
  idempotentHint: false,
432
1035
  openWorldHint: true,
433
1036
  },
434
- inputSchema: {
435
- workspacePath: z
436
- .string()
437
- .describe("The path to the workspace directory where the test file is located"),
438
- language: z
439
- .nativeEnum(ProgrammingLanguage)
440
- .describe("Programming language of the test file to execute (e.g., python, javascript, typescript, java)"),
441
- testType: z
442
- .nativeEnum(TestType)
443
- .describe("Type of the test to execute. Note: 'mock' is NOT a valid test type — mock files are deployed via their apply_mock() function, not executed as tests."),
444
- testFile: z
445
- .string()
446
- .describe("Absolute path to the test file to execute."),
447
- contractMode: z
448
- .enum(CONTRACT_EXECUTION_MODES)
449
- .optional()
450
- .describe("Only applies when testType is 'contract'. Use 'provider' for provider contract tests that hit the real service under test and need SKYRAMP_TEST_BASE_URL. Use 'consumer' only for consumer contract tests with inline mocks that do not hit the real service. Defaults to provider behavior when omitted."),
451
- token: z
452
- .string()
453
- .optional()
454
- .describe("Explicit Skyramp authentication token for test execution. Omit this parameter to use SKYRAMP_TEST_TOKEN from the environment. An empty string is passed through as a literal token value, not a 'no auth' signal — use `unauthenticated: true` for tests that must run without credentials."),
455
- unauthenticated: z
456
- .boolean()
457
- .optional()
458
- .describe("Set true to force this test execution to carry NO auth token, even if SKYRAMP_TEST_TOKEN is set in the environment or a token was passed. Use for tests that must assert 401/403 unauthenticated behavior."),
459
- playwrightSaveStoragePath: z
460
- .string()
461
- .optional()
462
- .describe("Path to save Playwright session storage after test execution for authentication purposes. Can be a relative path to the workspace (e.g., 'auth-session.json') or an absolute path. The session will be saved after the test completes."),
463
- stateFile: z
464
- .string()
465
- .trim()
466
- .optional()
467
- .describe("Path to state file from skyramp_analyze_changes. Always pass when available — results are written back so skyramp_submit_report can override before/afterStatus with ground-truth pass/fail."),
468
- phase: z
469
- .enum(["before", "after"])
470
- .default("after")
471
- .describe("Execution phase for maintained tests: 'before' captures pre-edit baseline; 'after' (default) records post-edit result."),
472
- rebaselineSnapshots: z
473
- .array(rebaselineSnapshotNameSchema)
474
- .optional()
475
- .describe('UI tests only. toHaveScreenshot() baseline filenames (e.g. ["page-001.png"]) this run REPLACES instead of comparing against, because the PR intentionally changed how the captured page/element/region looks (SKYR-4298). ' +
476
- "Pass exactly the list skyramp_actions returned as rebaseline_snapshots for this spec — the tool checks it against the persisted UPDATE verdict in stateFile (so stateFile is required) and refuses names the verdict did not authorize, a spec with no such verdict, or a spec whose phase: 'before' run has not been recorded yet. Use the name as the test passes it (page-001.png), not the on-disk file (page-001-chromium-linux.png). " +
477
- "The refreshed PNGs land beside the spec and are delivered with the Testbot PR as image diffs; the result names which baselines were refreshed and which were not. Never use this to silence a screenshot mismatch the diff does not explain."),
478
- repository: z
479
- .string()
480
- .trim()
481
- .min(1)
482
- .describe("The owner/repo whose analysis section to write execution results into (e.g. 'letsramp/api-insight'). Set it on every call — the primary's owner/repo for a primary-repo test, or a related repo's owner/repo for that repo's section of the run-scoped stateFile."),
483
- },
1037
+ inputSchema,
484
1038
  _meta: {
485
1039
  keywords: ["run test", "execute test"],
486
1040
  },
487
1041
  }, async (params, extra) => {
1042
+ const sendProgress = makeProgressReporter(extra);
1043
+ // Progress callback adapter for TestExecutionService.
1044
+ const onExecutionProgress = async (progress) => {
1045
+ await sendProgress(progress.percent, 100, progress.message);
1046
+ };
1047
+ await sendProgress(0, 100, "Starting execution...");
488
1048
  // `stateFile` may be omitted by the agent; the run's own state file (the
489
1049
  // one skyramp_analyze_changes wrote for this run) is the fallback, so a
490
1050
  // dropped argument cannot dodge the attempt cap or lose the result the
@@ -524,9 +1084,6 @@ export function registerExecuteSkyrampTestTool(server) {
524
1084
  resolvedRunStatePath: resolveRunStatePath(),
525
1085
  });
526
1086
  }
527
- const noRunNote = stateFilePath
528
- ? ""
529
- : `\n\nNote: no run state file was found (none passed, and no active run), so this execution was not counted against an attempt cap and its result is not recorded for the report.`;
530
1087
  // One execution at a time per run: everything below that touches the
531
1088
  // state file — the cap read, the reservation, the retrofit staging, the
532
1089
  // assertion proof, the result, the video record, the baseline verdict —
@@ -542,49 +1099,50 @@ export function registerExecuteSkyrampTestTool(server) {
542
1099
  ? stateFileKey(stateFilePath)
543
1100
  : canonicalTestPath(params.testFile), async () => {
544
1101
  let errorResult;
545
- // Attempt-cap bookkeeping (SKYR-4460), visible to the outer catch so a
546
- // throw after the reservation can still account for the attempt.
547
- let maxFixAttempts = getMaxFixAttempts();
548
- let priorAfterRuns = 0;
549
- let reserved = false;
550
- // Set when the tool re-ran a thrown transient connection error itself
551
- // (see the executor call): both the result path and the outer catch
552
- // must say so, or the agent is told it may re-run unchanged once more.
553
- let transientRetried = false;
554
- // The record the cap check read, and whether this run re-executes the
555
- // same file contents it ran last time (SKYR-4460: the one unchanged
556
- // re-run a file gets is then spent, and the texts say so).
557
- let unchangedRerun = false;
558
- // `repository` names the state section; `stateFilePath` (the argument, else
559
- // this run's own file) says whether there is any state to write. A standalone
560
- // execution — the local-dev workflow, an IDE call, no run — names its
561
- // repository but writes nothing.
562
- const stateSection = stateFilePath
563
- ? { file: stateFilePath, repo: params.repository }
564
- : undefined;
565
- // Helper to send progress notifications to the MCP client.
566
- const sendProgress = makeProgressReporter(extra);
567
- // Send immediate acknowledgment
568
- await sendProgress(0, 100, "Starting execution...");
569
- // Progress callback adapter for TestExecutionService
570
- const onExecutionProgress = async (progress) => {
571
- await sendProgress(progress.percent, 100, progress.message);
572
- };
573
- const previousBaseUrl = process.env.SKYRAMP_TEST_BASE_URL;
574
- let didSetSkyrampBaseUrl = false;
575
- let dockerNetwork;
576
1102
  try {
577
1103
  if (!path.isAbsolute(params.testFile)) {
578
1104
  errorResult = toolError(`testFile must be an absolute path, got: ${params.testFile}`);
579
1105
  return errorResult;
580
1106
  }
581
- // A missing test file is an argument error: refuse it before any
582
- // side effect (retrofit staging, the assertion proof record, the
583
- // attempt reservation) has recorded an execution that never ran.
1107
+ // A missing test file is an argument error: refuse it before any side effect
1108
+ // (retrofit staging, the assertion proof record, the attempt reservation) has
1109
+ // recorded an execution that never ran.
584
1110
  if (!fs.existsSync(params.testFile)) {
585
1111
  errorResult = toolError(`Test file does not exist: ${params.testFile}`);
586
1112
  return errorResult;
587
1113
  }
1114
+ // Validate the inputs of whichever mode this call is in, before anything
1115
+ // below writes state: a call that cannot run must not record that it did.
1116
+ const command = params.commandOverride;
1117
+ const cwd = params.cwd;
1118
+ if (command === undefined) {
1119
+ if (params.workspacePath === undefined) {
1120
+ errorResult = toolError("workspacePath is required when commandOverride is omitted: the executor resolves the service and its base URL from the workspace. Pass commandOverride to run the test on this host instead.");
1121
+ return errorResult;
1122
+ }
1123
+ if (!path.isAbsolute(params.workspacePath)) {
1124
+ errorResult = toolError(`workspacePath must be an absolute path, got: ${params.workspacePath}`);
1125
+ return errorResult;
1126
+ }
1127
+ }
1128
+ else {
1129
+ if (cwd === undefined) {
1130
+ errorResult = toolError("cwd is required with commandOverride.");
1131
+ return errorResult;
1132
+ }
1133
+ if (!path.isAbsolute(cwd)) {
1134
+ errorResult = toolError(`cwd must be an absolute path, got: ${cwd}`);
1135
+ return errorResult;
1136
+ }
1137
+ if (!fs.statSync(cwd, { throwIfNoEntry: false })?.isDirectory()) {
1138
+ errorResult = toolError(`cwd ${cwd} is not an existing directory. The command was not run.`);
1139
+ return errorResult;
1140
+ }
1141
+ if (!commandNamesTestFile(command, params.testFile)) {
1142
+ errorResult = toolError(`commandOverride does not contain ${testFileNames(params.testFile).join(" or ")}, the name of testFile. The command was not run.`);
1143
+ return errorResult;
1144
+ }
1145
+ }
588
1146
  // Hashed once, before any state write: recorded with the reservation
589
1147
  // so the NEXT run can tell an unchanged re-run from a fixed file.
590
1148
  const fileHash = sha256Of(fs.readFileSync(params.testFile, "utf8"));
@@ -593,63 +1151,49 @@ export function registerExecuteSkyrampTestTool(server) {
593
1151
  errorResult = toolError(rebaseline.error);
594
1152
  return errorResult;
595
1153
  }
596
- const rebaselineSnapshots = rebaseline.snapshots;
597
- if (rebaselineSnapshots.length > 0) {
598
- if (!stateSection) {
1154
+ const snapshots = rebaseline.snapshots;
1155
+ if (snapshots.length > 0) {
1156
+ if (!stateFilePath) {
599
1157
  errorResult = toolError("rebaselineSnapshots requires stateFile: the refresh is authorized against the UPDATE verdict skyramp_actions persisted there.");
600
1158
  return errorResult;
601
1159
  }
602
- const authState = await StateManager.fromStatePath(stateSection.file).readRepoData(stateSection.repo);
603
- const auth = authorizeRebaseline(authState, params.testFile, rebaselineSnapshots);
1160
+ const authState = await StateManager.fromStatePath(stateFilePath).readRepoData(params.repository);
1161
+ const auth = authorizeRebaseline(authState, params.testFile, snapshots);
604
1162
  if (auth.error) {
605
1163
  errorResult = toolError(auth.error);
606
1164
  return errorResult;
607
1165
  }
608
1166
  }
609
- // Deterministic external-test guard (SKYR-3924): this tool runs Skyramp-generated
610
- // tests in the executor and cannot run a repo's native (user-written) suite, so a
611
- // run on an external test only errors (e.g. pytest import/collection failure).
612
- // If the stateFile records this test as external, skip execution — the prompt asks
613
- // the agent to exclude them, and this enforces it. Best-effort and conservative:
614
- // only skip on a positive external match; any state-read issue falls through to
615
- // normal execution, and a not-found test (e.g. a newly generated one) is not skipped.
616
- if (stateSection) {
617
- try {
618
- const stateManager = StateManager.fromStatePath(stateSection.file);
619
- const stateData = await stateManager.readRepoData(stateSection.repo);
620
- const entry = stateData?.existingTests?.find((t) => canonicalTestPath(t.testFile) ===
621
- canonicalTestPath(params.testFile));
622
- if (entry?.source === TestSource.External) {
623
- logger.info(`Skipping execution of external test ${params.testFile} — native suites are not run by ${TOOL_NAME}`);
624
- return {
625
- content: [
626
- {
627
- type: "text",
628
- text: `Skipped execution: ${params.testFile} is marked external in the state file. ${TOOL_NAME} runs Skyramp-generated tests only; external (native) suites are not executed here.`,
629
- },
630
- ],
631
- };
632
- }
633
- }
634
- catch (err) {
635
- logger.warning(`External-test guard could not read stateFile (${err.message}); proceeding with execution`);
636
- }
637
- }
1167
+ // `stateFile` says whether there is any run state to count against and write
1168
+ // into; `repository` names the section inside it. A standalone execution —
1169
+ // the local-dev workflow, an IDE call, no run — names its repository but
1170
+ // writes nothing, and the tool says so in its answer.
1171
+ const stateSection = stateFilePath
1172
+ ? { file: stateFilePath, repo: params.repository }
1173
+ : undefined;
1174
+ const noRunNote = stateSection
1175
+ ? ""
1176
+ : `\n\nNote: no run state file was found (none passed, and no active run), so this execution was not counted against an attempt cap and its result is not recorded for the report.`;
638
1177
  // Fix-and-rerun attempt cap (SKYR-4460). With a stateFile the cap is the one
639
- // pinned in the run's state (the first call pins the prompt's value), and
640
- // the check fails CLOSED: this is the enforcement boundary, so a state that
641
- // cannot be read refuses the run instead of counting as zero prior
642
- // attempts. A call without any run state resolves from the environment and
643
- // the default and enforces nothing — there is no run to count against.
644
- maxFixAttempts = getMaxFixAttempts();
1178
+ // pinned in the run's state (the first call pins the prompt's value), and the
1179
+ // check fails CLOSED: this is the enforcement boundary, so a state that cannot
1180
+ // be read refuses the run instead of counting as zero prior attempts. A call
1181
+ // without any run state resolves from the environment and the default and
1182
+ // enforces nothing — there is no run to count against.
1183
+ let maxFixAttempts = getMaxFixAttempts();
1184
+ // Whether this run re-executes the same file contents as the last
1185
+ // one (SKYR-4460): the single unchanged re-run a file gets is then
1186
+ // spent, and the failure texts say so instead of offering another.
1187
+ let unchangedRerun = false;
1188
+ let priorAfterRuns = 0;
645
1189
  if (stateSection) {
646
1190
  let record;
647
1191
  try {
648
1192
  const stateData = await readRepoSectionOrThrow(stateSection.file, stateSection.repo);
649
- // The pin is run-wide and lives at the ROOT, whichever section
650
- // this file's record is in. A pinned value is re-validated: a
651
- // hand-edited or half-written state must not turn into "cap 0,
652
- // nothing ever runs" or "no cap".
1193
+ // The pin is run-wide and lives at the ROOT, whichever section this
1194
+ // file's record is in. A pinned value is re-validated: a hand-edited or
1195
+ // half-written state must not turn into "cap 0, nothing ever runs" or
1196
+ // "no cap".
653
1197
  const pinned = await readPinnedMaxFixAttempts(stateSection.file);
654
1198
  const current = getMaxFixAttempts();
655
1199
  maxFixAttempts = pinned ?? current;
@@ -677,112 +1221,26 @@ export function registerExecuteSkyrampTestTool(server) {
677
1221
  return errorResult;
678
1222
  }
679
1223
  }
680
- // SKYR-4220: stage the shared utils file (see stageUtilsArtifacts) — execution is
681
- // the last step before delivery, and a utils file left unstaged here ships a
682
- // test importing a module the PR lacks. After the external-test skip: a native
683
- // suite this tool will not run gets no scan and no staging.
1224
+ // SKYR-4220: a utils file left unstaged ships a test importing a module the PR lacks.
684
1225
  await stageAndRecordRetrofits(params.testFile, undefined, stateFilePath);
685
- // SKYR-4115 backstop: enhance_assertions carries the same check, but nothing
686
- // guarantees the agent calls it, and execution is the last step that still
687
- // precedes reporting. See pendingReuseVerification for why enforcing the verify
688
- // CALL needs no retry budget.
689
- //
690
- // Pass the run's state path through: without it the state path resolves
691
- // from the CI/Testbot anchor alone and the check returns early — failing
692
- // OPEN — for a caller that supplies a valid stateFile outside CI. It is the
693
- // canonical path (explicit argument → this run's own file → undefined),
694
- // the same one the external-test guard above and every write below use,
695
- // so a symlinked argument cannot fork the state into a second file.
696
- const owedReuseVerification = await pendingReuseDebt(params.testFile, stateFilePath, params.testType);
697
- if (owedReuseVerification) {
698
- errorResult = toolError(owedReuseVerification);
1226
+ const owedReuse = await pendingReuseDebt(params.testFile, stateFilePath, params.testType);
1227
+ if (owedReuse) {
1228
+ errorResult = toolError(owedReuse);
699
1229
  return errorResult;
700
1230
  }
701
- // Deterministic assertion-enhancement check: the server verifies the file
702
- // itself here (never relying on the agent to call `verify: true` — prose
703
- // can be ignored, this cannot). Insufficient assertions return feedback
704
- // instead of executing; a fixed file passes on the next execute call.
705
- // Sits BELOW the external-test guard: an external test is skipped, not
706
- // executed, so deferring it for assertion work would demand fixes to a
707
- // file this tool will never run.
708
1231
  const assertionFeedback = await assertionFeedbackForExecution(params.testFile, stateFilePath);
709
1232
  if (assertionFeedback) {
710
1233
  errorResult = toolError(assertionFeedback);
711
1234
  return errorResult;
712
1235
  }
713
- // Proof-of-work substrate (SKYR-4262 follow-up): count this execution
714
- // server-side so the report can cross-check generated vs executed —
715
- // the narrative is LLM-authored, this number is not. Best-effort.
716
1236
  await recordAssertionExecution(params.testFile, params.testType, stateFilePath);
717
- // Send initial progress
718
- await sendProgress(5, 100, "Starting test execution...");
719
- // Cross-repo run (SKYR-3819): tests delivered under the run's testsRepoDir
720
- // (set by skyramp_analyze_changes, run-scoped) live outside workspacePath.
721
- // Thread it through so the executor mounts the test repo and service
722
- // matching still resolves the SUT baseUrl/dockerNetwork.
723
- const testRepoPath = getTestsRepoDir();
724
- // Resolve workspace config for base URL injection and Docker network
725
- // attachment. A compose dockerNetwork is not host networking: on macOS
726
- // the executor still needs localhost rewritten to host.docker.internal.
727
- if (params.workspacePath) {
728
- const workspaceConfig = await getWorkspaceBaseUrl(params.workspacePath, params.testFile, params.language, testRepoPath);
729
- const { baseUrl, candidates } = workspaceConfig;
730
- dockerNetwork = workspaceConfig.dockerNetwork;
731
- const shouldInjectBaseUrl = shouldInjectSkyrampBaseUrl(params.testType, params.contractMode);
732
- if (shouldInjectBaseUrl && !process.env.SKYRAMP_TEST_BASE_URL) {
733
- if (baseUrl) {
734
- process.env.SKYRAMP_TEST_BASE_URL = baseUrl;
735
- didSetSkyrampBaseUrl = true;
736
- }
737
- else if (candidates.length > 0) {
738
- errorResult = toolError([
739
- `Cannot determine SKYRAMP_TEST_BASE_URL — test file matches multiple services:`,
740
- ...candidates.map((c) => ` • ${c.serviceName}: ${c.baseUrl}`),
741
- ``,
742
- `Re-invoke with SKYRAMP_TEST_BASE_URL set to the correct service URL, or make each service's testDirectory unique in .skyramp/workspace.yml.`,
743
- ].join("\n"));
744
- return errorResult;
745
- }
746
- }
747
- }
748
- const executionService = new TestExecutionService();
749
- const effectiveToken = resolveEffectiveToken(params.unauthenticated, params.token, process.env.SKYRAMP_TEST_TOKEN);
750
- if (!effectiveToken &&
751
- !params.unauthenticated &&
752
- params.token === undefined) {
753
- logger.warning("No auth token available — authenticated endpoints will likely return 401. Set SKYRAMP_TEST_TOKEN or pass token/unauthenticated.");
754
- }
755
- // Execute test with progress callback - reports Docker cache/pull status.
756
- // Retry once on transient connection errors (EOF, connection refused) —
757
- // these indicate the SUT isn't fully ready yet, not a test failure.
758
- // Retrying inside the tool call saves agent turns vs failing and
759
- // requiring the agent to re-invoke. This covers a connection error the
760
- // executor THROWS, under the one attempt already reserved; a connection
761
- // error that surfaces inside the test's own output reaches the agent as
762
- // a normal failure, and the prompt's unchanged-re-run allowance is for
763
- // that one.
764
- const execOptions = {
765
- testFile: params.testFile,
766
- workspacePath: params.workspacePath,
767
- testRepoPath,
768
- language: params.language,
769
- testType: params.testType,
770
- token: effectiveToken,
771
- playwrightSaveStoragePath: params.playwrightSaveStoragePath,
772
- dockerNetwork,
773
- useHostNetwork: false,
774
- ...(rebaselineSnapshots.length > 0
775
- ? { rebaselineSnapshots }
776
- : {}),
777
- };
778
- // Identity of the requested baselines before the run, to report afterwards
779
- // which ones SmartPlaywright actually rewrote (SKYR-4298).
780
- const baselinesBefore = rebaselineSnapshots.length > 0
781
- ? readBaselineState(params.testFile, rebaselineSnapshots)
782
- : {};
783
- // Reserve the attempt before the run (SKYR-4460): the counters advance now
784
- // and the cap is pinned, so a state write that fails after the run cannot
785
- // hand this file a free attempt. Fails closed, like the check above.
1237
+ // Reserve the attempt before the run (SKYR-4460). It sits ABOVE the mode
1238
+ // dispatch on purpose: the host path spawns the command itself and never
1239
+ // reaches TestExecutionService, so a reservation taken around the executor
1240
+ // call would count nothing on the path this tool now prefers. The counters
1241
+ // advance now and the cap is pinned, so a state write that fails after the
1242
+ // run cannot hand this file a free attempt. Fails closed, like the check above.
1243
+ let reserved = false;
786
1244
  if (stateSection) {
787
1245
  try {
788
1246
  const reservation = await reserveTestExecutionAttempt(stateSection.file, stateSection.repo, params.testType, params.phase, params.testFile, maxFixAttempts, fileHash);
@@ -795,173 +1253,41 @@ export function registerExecuteSkyrampTestTool(server) {
795
1253
  return errorResult;
796
1254
  }
797
1255
  }
798
- let result;
799
- // The tool's own retry of a transient connection error the executor
800
- // THREW (a resolved Error/Fail result whose output mentions
801
- // ECONNREFUSED is not retried here — TestExecutionService converts
802
- // execution-time exceptions into resolved results): two executions
803
- // billed as ONE attempt, on purpose — the SUT's hiccup is not the
804
- // agent's fix to make. The response says when it happened, and the
805
- // prompt withholds the agent's own unchanged re-run only then, so
806
- // one attempt is never more than two runs.
1256
+ const budget = {
1257
+ // The budget line only means something when the run counts attempts, i.e.
1258
+ // when a stateFile records them and this is a final-phase run.
1259
+ attemptLine: stateSection && params.phase === "after"
1260
+ ? `\n\n${describeAttempt(priorAfterRuns, maxFixAttempts, { unchangedRerun })}`
1261
+ : "",
1262
+ reserved,
1263
+ noRunNote,
1264
+ unchangedRerun,
1265
+ };
1266
+ await sendProgress(5, 100, "Starting test execution...");
1267
+ // Both modes owe the checks above. Only now does the path diverge.
1268
+ // Both take the RESOLVED state path, not the argument: everything they
1269
+ // hand it to — the result write, the video record, the verdict
1270
+ // reconciliation — must name the same file every check above read.
1271
+ const runParams = {
1272
+ ...params,
1273
+ stateFile: stateFilePath,
1274
+ };
807
1275
  try {
808
- result = await executionService.executeTest(execOptions, onExecutionProgress);
809
- }
810
- catch (firstErr) {
811
- const errMsg = firstErr instanceof Error ? firstErr.message : String(firstErr);
812
- if (/\bEOF\b|connection refused|ECONNREFUSED|ECONNRESET/i.test(errMsg)) {
813
- logger.info(`Test execution hit transient connection error, retrying after ${transientRetryDelayMs}ms...`, { error: errMsg });
814
- await sendProgress(50, 100, "SUT connection error — retrying in 10s...");
815
- await new Promise((r) => setTimeout(r, transientRetryDelayMs));
816
- transientRetried = true;
817
- result = await executionService.executeTest(execOptions, onExecutionProgress);
818
- }
819
- else {
820
- throw firstErr;
821
- }
1276
+ return command === undefined
1277
+ ? await runInExecutor(runParams, snapshots, budget, onExecutionProgress, sendProgress)
1278
+ : await runOnHost(runParams, command, cwd, snapshots, budget);
822
1279
  }
823
- const transientRetryNote = transientRetried
824
- ? TRANSIENT_RETRY_NOTE
825
- : "";
826
- // Update stateFile with execution results if provided. Multi-repo: write
827
- // into the section for `repository` of the run-scoped file.
828
- let persistWarning = noRunNote;
829
- if (stateSection) {
830
- try {
831
- await persistTestExecutionResult(stateSection.file, stateSection.repo, params.testType, params.phase, result, { reserved: true });
832
- logger.info(`Updated stateFile with execution results for ${params.testFile}`);
833
- }
834
- catch (err) {
835
- // The attempt was reserved before the run, so the count is right.
836
- // The RESULT is what the report reads for afterStatus, though, so
837
- // an unrecorded one must reach the agent, not only the log.
838
- logger.error(`Failed to update stateFile: ${err.message}`);
839
- persistWarning = persistFailureWarning(stateSection.file, err.message);
840
- }
841
- }
842
- // Record the recording for the report before returning, so it happens on the
843
- // failure path too (SKYR-4156). skyramp_submit_report reads these records to
844
- // populate testResults[].videoPath — testbot uploads only the video
845
- // directories the report references, so an unrecorded video is never seen.
846
- await recordExecutionVideo(result, stateFilePath);
847
- // Which requested baselines were actually rewritten (SKYR-4298). Reported on
848
- // pass and fail alike; a refreshed PNG is a deliverable like a generated
849
- // spec, so stage the spec's snapshot directory whenever one was rewritten —
850
- // even on a failing run, since the report gate checks the PNG, not the
851
- // status — so the eval harness commit and the artifact collector see it
852
- // (production delivery adds the whole test directory anyway). No-op outside
853
- // a testbot run, like every other stageGeneratedPaths call; never fails the
854
- // execution.
855
- let refreshOutcomeText = "";
856
- if (rebaselineSnapshots.length > 0) {
857
- const after = readBaselineState(params.testFile, rebaselineSnapshots);
858
- const outcome = diffBaselineState(baselinesBefore, after);
859
- refreshOutcomeText = describeRefreshOutcome(outcome);
860
- // Stage the rewritten files themselves, never the directory: `git add` on
861
- // the directory would ship anything else sitting there under one
862
- // authorized baseline's authority.
863
- for (const name of outcome.refreshed) {
864
- for (const file of outcome.refreshedFiles[name] ?? []) {
865
- try {
866
- await stageGeneratedPaths(path.join(snapshotDirFor(params.testFile), file));
867
- }
868
- catch (err) {
869
- logger.warning(`Could not stage refreshed visual baseline ${file} for ${params.testFile}: ${err.message}`);
870
- }
871
- }
872
- }
873
- if (outcome.notRefreshed.length > 0 && stateSection) {
874
- try {
875
- const stateManager = StateManager.fromStatePath(stateSection.file);
876
- const stateData = await stateManager.readRepoData(stateSection.repo);
877
- if (stateData?.maintenanceVerdicts) {
878
- const reconciled = applyRefreshOutcomeToVerdicts(stateData.maintenanceVerdicts, params.testFile, outcome, EXECUTOR_DOCKER_IMAGE);
879
- await stateManager.updateRepoData({
880
- ...stateData,
881
- maintenanceVerdicts: reconciled.verdicts,
882
- }, { repo: stateSection.repo });
883
- if (reconciled.note)
884
- refreshOutcomeText += ` ${reconciled.note}`;
885
- }
886
- }
887
- catch (err) {
888
- logger.warning(`Could not reconcile the maintenance verdict with the refresh outcome: ${err.message}`);
889
- }
890
- }
1280
+ catch (err) {
1281
+ return accountForThrow(runParams, err, budget);
891
1282
  }
892
- const withRefreshOutcome = (text) => (refreshOutcomeText ? `${text}\n\n${refreshOutcomeText}` : text) +
893
- transientRetryNote +
894
- persistWarning;
895
- // Progress is already reported by TestExecutionService
896
- // Only report final status if not already at 100%
897
- if (result.status !== TestExecutionStatus.Pass) {
898
- // The budget line only means something when the run counts attempts,
899
- // i.e. when a stateFile records them and this is a final-phase run.
900
- const attemptLine = stateSection && params.phase === "after"
901
- ? `\n\n${describeAttempt(priorAfterRuns, maxFixAttempts, { unchangedRerun })}`
902
- : "";
903
- errorResult = toolError(withRefreshOutcome(withVideoInfo(buildExecutionFailureText(result, { unchangedRerun }) +
904
- attemptLine, result.videoPath)));
905
- return errorResult;
906
- }
907
- // Success - progress already reported by TestExecutionService
908
- return {
909
- content: [
910
- {
911
- type: "text",
912
- text: withRefreshOutcome(withVideoInfo(`Test execution result: ${stripVTControlCharacters(result.output || "")}`, result.videoPath)),
913
- },
914
- ],
915
- };
916
1283
  }
917
1284
  catch (err) {
918
- const message = err.message;
919
- let attemptNote = "";
920
- let persistNote = "";
921
- // Both executions threw: the retry happened and the agent must still
922
- // hear it, or the prompt lets it spend a third unchanged run.
923
- const retryNote = transientRetried ? TRANSIENT_RETRY_NOTE : "";
924
- // A throw after the reservation (Docker image setup, a workspace check,
925
- // the rethrow from the transient-retry path) has spent an attempt; say
926
- // so, and land an Error result so the record is not left as the
927
- // reservation placeholder.
928
- if (reserved && stateSection) {
929
- try {
930
- await persistTestExecutionResult(stateSection.file, stateSection.repo, params.testType, params.phase, {
931
- testFile: params.testFile,
932
- status: TestExecutionStatus.Error,
933
- executedAt: new Date().toISOString(),
934
- duration: 0,
935
- errors: [message],
936
- warnings: [],
937
- output: "",
938
- }, { reserved: true });
939
- }
940
- catch (persistErr) {
941
- // Same as the main path: the count is right (reserved), the
942
- // record is not, and the agent must hear it, not only the log.
943
- logger.error(`Failed to record the execution error in stateFile: ${persistErr.message}`);
944
- persistNote = persistFailureWarning(stateSection.file, persistErr.message);
945
- }
946
- if (params.phase === "after") {
947
- attemptNote = `\n\n${describeAttempt(priorAfterRuns, maxFixAttempts, { unchangedRerun })}`;
948
- }
949
- }
950
- errorResult = toolError(`Test execution failed: ${message}${retryNote}${attemptNote}${persistNote}${noRunNote}`);
1285
+ errorResult = toolError(`Test execution failed: ${err.message}`);
951
1286
  return errorResult;
952
1287
  }
953
1288
  finally {
954
- if (didSetSkyrampBaseUrl) {
955
- if (previousBaseUrl === undefined) {
956
- delete process.env.SKYRAMP_TEST_BASE_URL;
957
- }
958
- else {
959
- process.env.SKYRAMP_TEST_BASE_URL = previousBaseUrl;
960
- }
961
- }
962
1289
  AnalyticsService.pushMCPToolEvent(TOOL_NAME, errorResult, {
963
1290
  testFile: params.testFile,
964
- workspacePath: params.workspacePath,
965
1291
  language: params.language,
966
1292
  testType: params.testType,
967
1293
  }).catch((err) => {