@skyramp/mcp 0.4.2-rc.1 → 0.4.2-rc.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/build/commands/localDevTestChangesCommand.js +1 -1
- package/build/commands/recommendTestsAndExecuteCommand.js +10 -1
- package/build/commands/testThisEndpointCommand.js +19 -2
- package/build/execution/wrapperConfig.d.ts +56 -0
- package/build/execution/wrapperConfig.js +155 -0
- package/build/index.js +6 -6
- package/build/playwright/registerPlaywrightTools.js +14 -0
- package/build/playwright/traceExportStore.d.ts +22 -0
- package/build/playwright/traceExportStore.js +81 -0
- package/build/playwright/traceRecordingPrompt.js +2 -1
- package/build/prompts/code-reuse.js +24 -21
- package/build/prompts/local-dev/local-dev-plan.js +6 -23
- package/build/prompts/local-dev/local-dev-prompts.js +1 -1
- package/build/prompts/shared-helper-policy.d.ts +36 -0
- package/build/prompts/shared-helper-policy.js +33 -1
- package/build/prompts/startTraceCollectionPrompts.js +1 -1
- package/build/prompts/sut-setup/modes/adaptWorkflowPrompt.js +6 -7
- package/build/prompts/sut-setup/shared.d.ts +1 -1
- package/build/prompts/sut-setup/shared.js +5 -3
- package/build/prompts/test-maintenance/drift-analysis-prompt.d.ts +16 -8
- package/build/prompts/test-maintenance/drift-analysis-prompt.js +90 -36
- package/build/prompts/test-maintenance/uiDriftAnalysisSections.js +1 -1
- package/build/prompts/test-recommendation/recommendationShared.d.ts +1 -1
- package/build/prompts/test-recommendation/recommendationShared.js +0 -1
- package/build/prompts/testbot/testbot-prompts.js +11 -9
- package/build/services/TestDiscoveryService.js +32 -4
- package/build/skills/runTestSkill.d.ts +6 -0
- package/build/skills/runTestSkill.js +17 -0
- package/build/tool-phases.js +0 -1
- package/build/tools/budgetExcuse.d.ts +15 -0
- package/build/tools/budgetExcuse.js +113 -0
- package/build/tools/code-refactor/utils-verify-gates.js +17 -3
- package/build/tools/executeSkyrampTestTool.d.ts +97 -48
- package/build/tools/executeSkyrampTestTool.js +775 -449
- package/build/tools/generate-tests/generateE2ERestTool.d.ts +0 -1
- package/build/tools/generate-tests/generateE2ERestTool.js +1 -9
- package/build/tools/generate-tests/generateUIRestTool.d.ts +0 -2
- package/build/tools/generate-tests/generateUIRestTool.js +1 -9
- package/build/tools/submitReportTool.js +128 -0
- package/build/tools/test-management/actionsTool.js +31 -0
- package/build/tools/test-management/analyzeChangesTool.d.ts +4 -4
- package/build/tools/test-management/analyzeTestHealthTool.d.ts +0 -11
- package/build/tools/test-management/analyzeTestHealthTool.js +7 -63
- package/build/tools/test-management/testsOwedBeforeRun.d.ts +28 -0
- package/build/tools/test-management/testsOwedBeforeRun.js +53 -0
- package/build/tools/trace/stopTraceCollectionTool.js +1 -1
- package/build/types/RepositoryAnalysis.d.ts +32 -32
- package/build/types/ReuseOutcome.d.ts +4 -3
- package/build/types/TestExecution.d.ts +2 -2
- package/build/types/TestTypes.d.ts +3 -7
- package/build/types/TestTypes.js +6 -20
- package/build/utils/AnalysisStateManager.d.ts +0 -7
- package/build/utils/AnalysisStateManager.js +1 -1
- package/build/utils/connectionErrors.d.ts +10 -0
- package/build/utils/connectionErrors.js +10 -0
- package/build/utils/language-helper.js +24 -3
- package/build/utils/progress.d.ts +1 -1
- package/build/utils/progress.js +1 -1
- package/build/utils/rebaselineSnapshots.d.ts +1 -1
- package/build/utils/rebaselineSnapshots.js +6 -16
- package/build/utils/reuseRouting.d.ts +10 -0
- package/build/utils/reuseRouting.js +15 -0
- package/build/utils/runContextGauge.d.ts +27 -0
- package/build/utils/runContextGauge.js +181 -0
- package/build/utils/skyrampMdContent.d.ts +1 -1
- package/build/utils/skyrampMdContent.js +1 -1
- package/build/utils/skyrampSdkVersion.d.ts +9 -0
- package/build/utils/skyrampSdkVersion.js +16 -0
- package/build/utils/testDependencyPolicy.js +21 -0
- package/build/utils/testExecutionRecord.d.ts +5 -1
- package/build/utils/testExecutionRecord.js +3 -1
- package/build/utils/testFileClassification.d.ts +8 -0
- package/build/utils/testFileClassification.js +36 -3
- package/build/utils/utils-verify/action-key.d.ts +42 -0
- package/build/utils/utils-verify/action-key.js +118 -36
- package/build/utils/utils-verify/action-sites.d.ts +32 -0
- package/build/utils/utils-verify/action-sites.js +202 -0
- package/build/utils/utils-verify/body-reach.js +2 -4
- package/build/utils/utils-verify/call-sites.d.ts +25 -6
- package/build/utils/utils-verify/call-sites.js +8 -5
- package/build/utils/utils-verify/index.d.ts +1 -0
- package/build/utils/utils-verify/index.js +1 -0
- package/build/utils/utils-verify/language-spec.d.ts +25 -0
- package/build/utils/utils-verify/language-spec.js +16 -2
- package/build/utils/utils-verify/parse.d.ts +10 -1
- package/build/utils/utils-verify/parse.js +19 -2
- package/build/utils/utils-verify/verify.d.ts +3 -2
- package/build/utils/utils-verify/verify.js +16 -18
- package/build/workspace/workspace.d.ts +72 -52
- package/build/workspace/workspace.js +12 -8
- package/package.json +1 -1
- package/plugin/prompts/testbot-task1.md +0 -2
- package/plugin/skills/enhance-assertions/reference/shared-rules.md +1 -1
- package/plugin/skills/enhance-assertions/reference/ui.md +1 -1
- package/plugin/skills/fix-test-import-errors/SKILL.md +2 -1
- package/plugin/skills/run-test/SKILL.md +16 -0
- package/build/adapters/jestAdapter.d.ts +0 -14
- package/build/adapters/jestAdapter.js +0 -131
- package/build/adapters/mochaAdapter.d.ts +0 -13
- package/build/adapters/mochaAdapter.js +0 -93
- package/build/adapters/playwrightAdapter.d.ts +0 -17
- package/build/adapters/playwrightAdapter.js +0 -184
- package/build/adapters/pytestAdapter.d.ts +0 -15
- package/build/adapters/pytestAdapter.js +0 -119
- package/build/tools/runExistingTestsTool.d.ts +0 -138
- package/build/tools/runExistingTestsTool.js +0 -666
- package/build/types/ExternalTestExecution.d.ts +0 -67
- package/build/types/ExternalTestExecution.js +0 -8
- package/build/workspace/testSuites.d.ts +0 -20
- package/build/workspace/testSuites.js +0 -17
|
@@ -1,44 +1,46 @@
|
|
|
1
1
|
import { z } from "zod";
|
|
2
|
-
import {
|
|
3
|
-
import {
|
|
4
|
-
import {
|
|
2
|
+
import { TestExecutionService } from "../services/TestExecutionService.js";
|
|
3
|
+
import { getWorkspaceBaseUrl } from "../utils/workspaceAuth.js";
|
|
4
|
+
import { spawn } from "child_process";
|
|
5
|
+
import crypto from "crypto";
|
|
6
|
+
import * as fs from "fs";
|
|
5
7
|
import path from "path";
|
|
6
8
|
import { stripVTControlCharacters } from "util";
|
|
7
|
-
import { TestExecutionService } from "../services/TestExecutionService.js";
|
|
8
|
-
import { AnalyticsService } from "../services/AnalyticsService.js";
|
|
9
9
|
import { makeProgressReporter } from "../utils/progress.js";
|
|
10
|
+
import { pendingReuseDebt } from "./code-refactor/reuse-state.js";
|
|
11
|
+
import { assertionFeedbackForExecution, canonicalTestPath, recordAssertionExecution, } from "./code-refactor/assertion-state.js";
|
|
12
|
+
import { stageAndRecordRetrofits } from "./code-refactor/retrofit-state.js";
|
|
13
|
+
import { AnalyticsService } from "../services/AnalyticsService.js";
|
|
10
14
|
import { TestExecutionStatus, } from "../types/TestExecution.js";
|
|
11
|
-
import { getWorkspaceBaseUrl } from "../utils/workspaceAuth.js";
|
|
12
15
|
import { ProgrammingLanguage, TestType } from "../types/TestTypes.js";
|
|
13
|
-
import { StateManager, currentRunStateFile, getTestsRepoDir, resolveOwnRunStatePath, resolveRunStatePath, } from "../utils/AnalysisStateManager.js";
|
|
14
|
-
import { DriftAction,
|
|
16
|
+
import { StateManager, currentRunStateFile, getPrimaryRepository, getTestsRepoDir, resolveOwnRunStatePath, resolveRunStatePath, } from "../utils/AnalysisStateManager.js";
|
|
17
|
+
import { DriftAction, } from "../types/TestAnalysis.js";
|
|
15
18
|
import { logger } from "../utils/logger.js";
|
|
16
19
|
import { toolError } from "../utils/utils.js";
|
|
17
20
|
import { recordExecutionVideo } from "./execution-video-state.js";
|
|
18
21
|
import { stageGeneratedPaths } from "../utils/gitStaging.js";
|
|
22
|
+
import { walkDir } from "../utils/fileWalk.js";
|
|
23
|
+
import { findRepoPlaywrightConfig, writeWrapperConfig, commandPassesBrowserFlag, } from "../execution/wrapperConfig.js";
|
|
19
24
|
import { canonicalStateFilePath, persistTestExecutionResult, readPinnedMaxFixAttempts, readRepoSectionOrThrow, reserveTestExecutionAttempt, stateFileKey, } from "../utils/testExecutionRecord.js";
|
|
20
|
-
import { canonicalTestPath } from "./code-refactor/assertion-state.js";
|
|
21
25
|
import { getMaxFixAttempts } from "../utils/fixAttempts.js";
|
|
22
26
|
import { runSerialized } from "../utils/runSerialized.js";
|
|
23
|
-
import * as fs from "fs";
|
|
24
27
|
import { sha256Of } from "../utils/assertion-verify/index.js";
|
|
25
28
|
import { baselineFileMatchesStem, baselineStem, rebaselineSnapshotNameSchema, snapshotDirFor, } from "../utils/rebaselineSnapshots.js";
|
|
26
|
-
|
|
27
|
-
const
|
|
28
|
-
|
|
29
|
+
export const TOOL_NAME = "skyramp_execute_test";
|
|
30
|
+
const DEFAULT_TIMEOUT_MS = 300_000;
|
|
31
|
+
const MAX_TIMEOUT_MS = 3_600_000;
|
|
32
|
+
/** Output kept per run, from its end: it is held in memory, saved to the state file, and returned to the agent. */
|
|
33
|
+
export const MAX_OUTPUT_CHARS = 200_000;
|
|
29
34
|
/**
|
|
30
|
-
*
|
|
31
|
-
*
|
|
32
|
-
*
|
|
33
|
-
*
|
|
35
|
+
* `unauthenticated: true` forces no token, so the child env carries no
|
|
36
|
+
* SKYRAMP_TEST_TOKEN at all — an empty Authorization header triggers encoding
|
|
37
|
+
* errors on unauthenticated endpoints (E7). An empty `token` means "use the
|
|
38
|
+
* server's environment", as the local-dev prompt passes it.
|
|
34
39
|
*/
|
|
35
40
|
export function resolveEffectiveToken(unauthenticated, paramToken, envToken) {
|
|
36
41
|
if (unauthenticated)
|
|
37
42
|
return "";
|
|
38
|
-
return paramToken
|
|
39
|
-
}
|
|
40
|
-
export function shouldInjectSkyrampBaseUrl(testType, contractMode) {
|
|
41
|
-
return testType !== TestType.CONTRACT || contractMode !== "consumer";
|
|
43
|
+
return paramToken || envToken || "";
|
|
42
44
|
}
|
|
43
45
|
/**
|
|
44
46
|
* Append the recorded video path to execution output.
|
|
@@ -51,6 +53,242 @@ export function shouldInjectSkyrampBaseUrl(testType, contractMode) {
|
|
|
51
53
|
export function withVideoInfo(output, videoPath) {
|
|
52
54
|
return videoPath ? `${output}\n\nVideo recording: ${videoPath}` : output;
|
|
53
55
|
}
|
|
56
|
+
function isBrowserTest(testType) {
|
|
57
|
+
return testType === TestType.UI || testType === TestType.E2E;
|
|
58
|
+
}
|
|
59
|
+
/**
|
|
60
|
+
* One directory per run: a unique name leaves nothing to clean up between runs, and
|
|
61
|
+
* the random suffix keeps two runs in the same millisecond apart.
|
|
62
|
+
*/
|
|
63
|
+
export function videoSubdirName(testFile) {
|
|
64
|
+
const basename = path
|
|
65
|
+
.basename(testFile, path.extname(testFile))
|
|
66
|
+
.replace(/[^A-Za-z0-9._-]/g, "-");
|
|
67
|
+
const hash = crypto
|
|
68
|
+
.createHash("sha256")
|
|
69
|
+
.update(testFile)
|
|
70
|
+
.digest("hex")
|
|
71
|
+
.slice(0, 8);
|
|
72
|
+
return `${basename}-${hash}-${Date.now()}-${crypto.randomBytes(3).toString("hex")}`;
|
|
73
|
+
}
|
|
74
|
+
/** The first `video.webm` under a run's video directory, if one was recorded. */
|
|
75
|
+
export function collectVideoPath(videoDir) {
|
|
76
|
+
try {
|
|
77
|
+
for (const [entry, fullPath] of walkDir(videoDir)) {
|
|
78
|
+
if (entry.name === "video.webm")
|
|
79
|
+
return fullPath;
|
|
80
|
+
}
|
|
81
|
+
}
|
|
82
|
+
catch (err) {
|
|
83
|
+
logger.warning(`Could not scan ${videoDir} for a video`, {
|
|
84
|
+
error: String(err),
|
|
85
|
+
});
|
|
86
|
+
}
|
|
87
|
+
return undefined;
|
|
88
|
+
}
|
|
89
|
+
/**
|
|
90
|
+
* The directory whose `.skyramp/workspace.yml` describes this run: the nearest
|
|
91
|
+
* ancestor of `cwd` that has one, else the run's primary checkout (a tests repo
|
|
92
|
+
* delivered apart from the SUT has no workspace.yml of its own), else `cwd`.
|
|
93
|
+
*/
|
|
94
|
+
export function resolveWorkspaceRoot(cwd) {
|
|
95
|
+
let dir = path.resolve(cwd);
|
|
96
|
+
for (;;) {
|
|
97
|
+
if (fs.existsSync(path.join(dir, ".skyramp", "workspace.yml")))
|
|
98
|
+
return dir;
|
|
99
|
+
const parent = path.dirname(dir);
|
|
100
|
+
if (parent === dir)
|
|
101
|
+
break;
|
|
102
|
+
dir = parent;
|
|
103
|
+
}
|
|
104
|
+
return getPrimaryRepository()?.repositoryPath ?? path.resolve(cwd);
|
|
105
|
+
}
|
|
106
|
+
/** pytest splits PYTEST_ADDOPTS with shlex, so a path with whitespace needs quotes. */
|
|
107
|
+
export function appendPytestVideoOpts(existing, videoDir) {
|
|
108
|
+
const dir = /\s/.test(videoDir) ? JSON.stringify(videoDir) : videoDir;
|
|
109
|
+
return [existing?.trim(), `--video on --output ${dir}`]
|
|
110
|
+
.filter(Boolean)
|
|
111
|
+
.join(" ");
|
|
112
|
+
}
|
|
113
|
+
/**
|
|
114
|
+
* The names by which a command can select `testFile`: its basename, and for Java the
|
|
115
|
+
* class name, because Maven selects a test with `-Dtest=FooTest`.
|
|
116
|
+
*/
|
|
117
|
+
export function testFileNames(testFile) {
|
|
118
|
+
const basename = path.basename(testFile);
|
|
119
|
+
return path.extname(basename) === ".java"
|
|
120
|
+
? [basename, path.basename(basename, ".java")]
|
|
121
|
+
: [basename];
|
|
122
|
+
}
|
|
123
|
+
/**
|
|
124
|
+
* Backslashes are ignored: Playwright and Jest read the path as a regular
|
|
125
|
+
* expression, so the agent escapes it (`a\.spec\.ts`).
|
|
126
|
+
*
|
|
127
|
+
* A trailing shell comment is removed first. `npm test # a_test.py` names the
|
|
128
|
+
* file only in a comment and would run the whole suite, and the check exists
|
|
129
|
+
* to stop exactly that. This is a name check, not a parse: a command that
|
|
130
|
+
* mentions the file in some other way it does not run still passes.
|
|
131
|
+
*/
|
|
132
|
+
/** The config an explicit `--config <path>` names, resolved against `cwd`, or
|
|
133
|
+
* undefined when the command carries none. `$SKYRAMP_PLAYWRIGHT_CONFIG` is
|
|
134
|
+
* ours and is not a repository config. */
|
|
135
|
+
export function explicitPlaywrightConfig(command, cwd) {
|
|
136
|
+
const m = /--config[= ]\s*("[^"]+"|'[^']+'|[^\s]+)/.exec(command);
|
|
137
|
+
if (!m)
|
|
138
|
+
return undefined;
|
|
139
|
+
const raw = m[1].replace(/^["']|["']$/g, "");
|
|
140
|
+
if (raw.includes("SKYRAMP_PLAYWRIGHT_CONFIG"))
|
|
141
|
+
return undefined;
|
|
142
|
+
const resolved = path.isAbsolute(raw) ? raw : path.join(cwd, raw);
|
|
143
|
+
return fs.existsSync(resolved) ? resolved : undefined;
|
|
144
|
+
}
|
|
145
|
+
/** The command with its `--config <repo config>` pointed at the wrapper. The
|
|
146
|
+
* wrapper imports that same config, so the run keeps the repository's testDir
|
|
147
|
+
* and projects and gains the video overlay. Without this swap a command that
|
|
148
|
+
* names a config runs the config directly and records nothing. */
|
|
149
|
+
export function pointConfigAtWrapper(command, wrapperPath) {
|
|
150
|
+
return command.replace(/(--config[= ]\s*)("[^"]+"|'[^']+'|[^\s]+)/, (_m, flag) => `${flag}${JSON.stringify(wrapperPath)}`);
|
|
151
|
+
}
|
|
152
|
+
export function commandNamesTestFile(command, testFile) {
|
|
153
|
+
const withoutComment = command.replace(/(^|\s)#.*$/, "$1");
|
|
154
|
+
const unescaped = withoutComment.replace(/\\/g, "");
|
|
155
|
+
return testFileNames(testFile).some((name) => unescaped.includes(name));
|
|
156
|
+
}
|
|
157
|
+
/** pytest exits 4 on a usage error; without pytest-playwright `--video` is one. */
|
|
158
|
+
export function pytestRejectedVideoOptions(run) {
|
|
159
|
+
return (run.exitCode === 4 &&
|
|
160
|
+
/unrecognized arguments:[^\n]*--video/.test(run.output));
|
|
161
|
+
}
|
|
162
|
+
/** What the output says when the command ended before any test ran. Each entry
|
|
163
|
+
* is the line the runner prints instead of a result: a missing package, or a
|
|
164
|
+
* file the runner never matched. Node, Playwright and Jest all exit 1 for
|
|
165
|
+
* these, so without this the run is recorded as a failing test. */
|
|
166
|
+
const NO_RUN_PATTERNS = [
|
|
167
|
+
/^.*\bError: Cannot find module\b.*$/m,
|
|
168
|
+
/^Cannot find module\b.*$/m,
|
|
169
|
+
/^.*\bERR_MODULE_NOT_FOUND\b.*$/m,
|
|
170
|
+
/^(?:Error: )?No tests found\b.*$/im,
|
|
171
|
+
// npx --no-install refuses to fetch a runner the repository does not have.
|
|
172
|
+
/^.*\bnpx canceled due to missing packages\b.*$/m,
|
|
173
|
+
// Jest matched the file and loaded it, and it declared no test.
|
|
174
|
+
/^.*\bYour test suite must contain at least one test\b.*$/m,
|
|
175
|
+
];
|
|
176
|
+
/** The line proving no test ran, or undefined when the output does not say so. */
|
|
177
|
+
export function noTestRanReason(output) {
|
|
178
|
+
for (const pattern of NO_RUN_PATTERNS) {
|
|
179
|
+
const hit = pattern.exec(output);
|
|
180
|
+
if (hit)
|
|
181
|
+
return hit[0].trim();
|
|
182
|
+
}
|
|
183
|
+
return undefined;
|
|
184
|
+
}
|
|
185
|
+
export function verdictForExit(exitCode, timedOut, output = "") {
|
|
186
|
+
if (timedOut)
|
|
187
|
+
return TestExecutionStatus.Error;
|
|
188
|
+
if (exitCode === 0)
|
|
189
|
+
return TestExecutionStatus.Pass;
|
|
190
|
+
// Exit 1 is the failing-test code, but it is also what a runner returns when
|
|
191
|
+
// it never got as far as a test. Error keeps those out of the failing count.
|
|
192
|
+
if (exitCode === 1)
|
|
193
|
+
return noTestRanReason(output)
|
|
194
|
+
? TestExecutionStatus.Error
|
|
195
|
+
: TestExecutionStatus.Fail;
|
|
196
|
+
return TestExecutionStatus.Error;
|
|
197
|
+
}
|
|
198
|
+
/**
|
|
199
|
+
* How long output may keep arriving after the shell exits. A background process the
|
|
200
|
+
* command started holds the pipes open, so waiting for them to close would wait for it.
|
|
201
|
+
*/
|
|
202
|
+
const EXIT_OUTPUT_GRACE_MS = 2_000;
|
|
203
|
+
/**
|
|
204
|
+
* Runs `command` through the shell in its own process group, so a timeout kills
|
|
205
|
+
* the runner's children too. stdout and stderr share one buffer in arrival order.
|
|
206
|
+
*/
|
|
207
|
+
export function spawnTestCommand(opts) {
|
|
208
|
+
const startedAt = Date.now();
|
|
209
|
+
return new Promise((resolve) => {
|
|
210
|
+
let output = "";
|
|
211
|
+
let timedOut = false;
|
|
212
|
+
let spawnError;
|
|
213
|
+
let exitCode = null;
|
|
214
|
+
let settled = false;
|
|
215
|
+
let grace;
|
|
216
|
+
const child = spawn(opts.command, {
|
|
217
|
+
cwd: opts.cwd,
|
|
218
|
+
env: opts.env,
|
|
219
|
+
shell: true,
|
|
220
|
+
detached: process.platform !== "win32",
|
|
221
|
+
});
|
|
222
|
+
const killGroup = () => {
|
|
223
|
+
try {
|
|
224
|
+
if (process.platform !== "win32" && child.pid) {
|
|
225
|
+
process.kill(-child.pid, "SIGKILL");
|
|
226
|
+
}
|
|
227
|
+
else if (child.pid) {
|
|
228
|
+
// Windows has no process group: child.kill reaches the shell and
|
|
229
|
+
// leaves the runner beneath it alive, still writing to the checkout.
|
|
230
|
+
// /T takes the tree, /F forces it.
|
|
231
|
+
spawn("taskkill", ["/pid", String(child.pid), "/T", "/F"], {
|
|
232
|
+
stdio: "ignore",
|
|
233
|
+
});
|
|
234
|
+
}
|
|
235
|
+
}
|
|
236
|
+
catch {
|
|
237
|
+
// already gone
|
|
238
|
+
}
|
|
239
|
+
};
|
|
240
|
+
const finish = () => {
|
|
241
|
+
if (settled)
|
|
242
|
+
return;
|
|
243
|
+
settled = true;
|
|
244
|
+
clearTimeout(timer);
|
|
245
|
+
clearTimeout(grace);
|
|
246
|
+
if (output.length > MAX_OUTPUT_CHARS) {
|
|
247
|
+
output = output.slice(-MAX_OUTPUT_CHARS);
|
|
248
|
+
truncated = true;
|
|
249
|
+
}
|
|
250
|
+
resolve({
|
|
251
|
+
exitCode,
|
|
252
|
+
output: (truncated
|
|
253
|
+
? `[output truncated to its last ${MAX_OUTPUT_CHARS} characters]\n`
|
|
254
|
+
: "") + stripVTControlCharacters(output),
|
|
255
|
+
timedOut,
|
|
256
|
+
spawnError,
|
|
257
|
+
duration: Date.now() - startedAt,
|
|
258
|
+
});
|
|
259
|
+
};
|
|
260
|
+
const timer = setTimeout(() => {
|
|
261
|
+
timedOut = true;
|
|
262
|
+
killGroup();
|
|
263
|
+
}, opts.timeoutMs);
|
|
264
|
+
let truncated = false;
|
|
265
|
+
const append = (d) => {
|
|
266
|
+
output += String(d);
|
|
267
|
+
if (output.length > 2 * MAX_OUTPUT_CHARS) {
|
|
268
|
+
output = output.slice(-MAX_OUTPUT_CHARS);
|
|
269
|
+
truncated = true;
|
|
270
|
+
}
|
|
271
|
+
};
|
|
272
|
+
child.stdout?.on("data", append);
|
|
273
|
+
child.stderr?.on("data", append);
|
|
274
|
+
child.on("error", (err) => {
|
|
275
|
+
spawnError = String(err);
|
|
276
|
+
finish();
|
|
277
|
+
});
|
|
278
|
+
child.on("exit", (code) => {
|
|
279
|
+
exitCode = code;
|
|
280
|
+
clearTimeout(timer);
|
|
281
|
+
grace = setTimeout(() => {
|
|
282
|
+
killGroup();
|
|
283
|
+
finish();
|
|
284
|
+
}, EXIT_OUTPUT_GRACE_MS);
|
|
285
|
+
});
|
|
286
|
+
child.on("close", (code) => {
|
|
287
|
+
exitCode = code ?? exitCode;
|
|
288
|
+
finish();
|
|
289
|
+
});
|
|
290
|
+
});
|
|
291
|
+
}
|
|
54
292
|
/**
|
|
55
293
|
* Where a real HTTP 401 shows up in runner output. Every entry is a shape taken from
|
|
56
294
|
* actual skyramp_execute_test output in the eval logs, not from guesswork:
|
|
@@ -81,7 +319,6 @@ const HTTP_401_SHAPES = [
|
|
|
81
319
|
// there, while a test TITLE ("should return 401 Unauthorized for an expired
|
|
82
320
|
// token") always has words in front of it and is not evidence of a response.
|
|
83
321
|
/^\s*401\s+unauthori[sz]ed\b/im,
|
|
84
|
-
// Status line, as curl -i and Go's httputil print it.
|
|
85
322
|
/\bHTTP\/[\d.]+\s+401\b/i,
|
|
86
323
|
// A status FIELD set to 401 — the value the response carried. Only `:` is
|
|
87
324
|
// accepted: `== 401` and `= 401` are an assertion or echoed test source, which
|
|
@@ -90,8 +327,6 @@ const HTTP_401_SHAPES = [
|
|
|
90
327
|
// (`Expected: {"status": 401}`) alongside `Received: {"status": 500}` is not a
|
|
91
328
|
// 401 the app sent. Reject the line rather than the value.
|
|
92
329
|
/^(?!.*\bexpected\b).*\b(?:status|status[_-]?code|statuscode|code|errorcode)\\?"?\s*:\s*401\b/im,
|
|
93
|
-
// Playwright prints both compared values. `Received` is what the app sent;
|
|
94
|
-
// `Expected` is what the test wanted, so it is not evidence of a 401.
|
|
95
330
|
/^\s*Received:\s*401\b/im,
|
|
96
331
|
// pytest assertion rewriting. A Skyramp-generated test reads
|
|
97
332
|
// `assert response.status_code == N`, so the OBSERVED value is on the left and
|
|
@@ -124,14 +359,10 @@ export function resolveRebaselineSnapshots(requested, phase) {
|
|
|
124
359
|
return { snapshots };
|
|
125
360
|
}
|
|
126
361
|
/**
|
|
127
|
-
*
|
|
128
|
-
*
|
|
129
|
-
* `<stem>-
|
|
130
|
-
*
|
|
131
|
-
* the tool can tell the agent which baselines were actually rewritten — SmartPlaywright
|
|
132
|
-
* is the only party that knows the exact filename, and an executor image that lacks
|
|
133
|
-
* SKYRAMP_UPDATE_SNAPSHOTS (or a name matching no toHaveScreenshot call) leaves
|
|
134
|
-
* every file untouched.
|
|
362
|
+
* For each requested name, every PNG under `<spec>-snapshots/` whose name matches the
|
|
363
|
+
* stem, with size and mtime. A spec may hold both `<stem>-linux.png` and
|
|
364
|
+
* `<stem>-chromium-linux.png`; latching onto one would misreport the other. Taken
|
|
365
|
+
* before and after the run to tell which baselines were actually rewritten.
|
|
135
366
|
*/
|
|
136
367
|
export function readBaselineState(specFile, requested) {
|
|
137
368
|
const dir = snapshotDirFor(specFile);
|
|
@@ -159,10 +390,6 @@ export function readBaselineState(specFile, requested) {
|
|
|
159
390
|
}
|
|
160
391
|
return state;
|
|
161
392
|
}
|
|
162
|
-
/**
|
|
163
|
-
* Which requested baselines changed on disk between two readBaselineState calls, and
|
|
164
|
-
* which files carried the change (the ones to stage).
|
|
165
|
-
*/
|
|
166
393
|
export function diffBaselineState(before, after) {
|
|
167
394
|
const refreshed = [];
|
|
168
395
|
const notRefreshed = [];
|
|
@@ -236,16 +463,13 @@ export function authorizeRebaseline(stateData, testFile, requested) {
|
|
|
236
463
|
return {};
|
|
237
464
|
}
|
|
238
465
|
/**
|
|
239
|
-
* Reconcile the persisted verdict with what the
|
|
240
|
-
*
|
|
241
|
-
*
|
|
242
|
-
*
|
|
243
|
-
*
|
|
244
|
-
* rebaseline-only UPDATE with nothing left becomes VERIFY (the test stays red, the
|
|
245
|
-
* rationale says why), and an UPDATE that also carried edits keeps UPDATE and is
|
|
246
|
-
* held to its edit. The report then reflects what happened, not what was asked.
|
|
466
|
+
* Reconcile the persisted verdict with what the run actually did (SKYR-4298). A
|
|
467
|
+
* @skyramp/skyramp that predates SKYRAMP_UPDATE_SNAPSHOTS rewrites nothing; left
|
|
468
|
+
* alone, the verdict would still promise a refresh and the report gate would refuse
|
|
469
|
+
* the report. Names not refreshed are dropped; a rebaseline-only UPDATE with nothing
|
|
470
|
+
* left becomes VERIFY, and an UPDATE that also carried edits is held to its edit.
|
|
247
471
|
*/
|
|
248
|
-
export function applyRefreshOutcomeToVerdicts(verdicts, testFile, outcome
|
|
472
|
+
export function applyRefreshOutcomeToVerdicts(verdicts, testFile, outcome) {
|
|
249
473
|
if (outcome.notRefreshed.length === 0)
|
|
250
474
|
return { verdicts };
|
|
251
475
|
let note;
|
|
@@ -255,7 +479,7 @@ export function applyRefreshOutcomeToVerdicts(verdicts, testFile, outcome, execu
|
|
|
255
479
|
v.action !== DriftAction.Update)
|
|
256
480
|
return v;
|
|
257
481
|
const remaining = (v.rebaselineSnapshots ?? []).filter((n) => !outcome.notRefreshed.includes(n));
|
|
258
|
-
const reason = `baseline refresh of ${outcome.notRefreshed.join(", ")} was not applied by
|
|
482
|
+
const reason = `baseline refresh of ${outcome.notRefreshed.join(", ")} was not applied by the test run (its @skyramp/skyramp lacks SKYRAMP_UPDATE_SNAPSHOTS, or the name matches no toHaveScreenshot call)`;
|
|
259
483
|
if (remaining.length === 0 && v.rebaselineOnly) {
|
|
260
484
|
note = `Verdict for ${path.basename(testFile)} downgraded UPDATE → VERIFY: ${reason}. The test stays as it is; report it honestly.`;
|
|
261
485
|
const { rebaselineSnapshots: _dropped, rebaselineOnly: _only, ...rest } = v;
|
|
@@ -290,10 +514,11 @@ export function describeRefreshOutcome(outcome) {
|
|
|
290
514
|
parts.push(`Visual baselines refreshed: ${outcome.refreshed.join(", ")}.`);
|
|
291
515
|
}
|
|
292
516
|
if (outcome.notRefreshed.length > 0) {
|
|
293
|
-
parts.push(`Visual baselines NOT refreshed: ${outcome.notRefreshed.join(", ")} — the
|
|
517
|
+
parts.push(`Visual baselines NOT refreshed: ${outcome.notRefreshed.join(", ")} — the test's @skyramp/skyramp may lack SKYRAMP_UPDATE_SNAPSHOTS support, or the name matches no toHaveScreenshot() call in this spec. Do not report these as refreshed.`);
|
|
294
518
|
}
|
|
295
519
|
return parts.join(" ");
|
|
296
520
|
}
|
|
521
|
+
/**
|
|
297
522
|
/**
|
|
298
523
|
* The fix-and-rerun attempt cap (SKYR-4460), enforced where the prompt's prose
|
|
299
524
|
* cannot be: a file that has already received `cap` runs in the final phase
|
|
@@ -339,13 +564,6 @@ export function setTransientRetryDelayForTests(ms) {
|
|
|
339
564
|
* error itself: two executions, one attempt. The prompt withholds the agent's
|
|
340
565
|
* own unchanged re-run when it sees this, so one attempt is never three runs. */
|
|
341
566
|
const TRANSIENT_RETRY_NOTE = `\n\nNote: the executor hit a transient connection error and re-ran this file once unchanged before answering; the two runs count as one attempt. Do not re-run it unchanged again — fix the cause the output names, or report it.`;
|
|
342
|
-
/** The warning both result paths append when the run's state could not take
|
|
343
|
-
* the result: the attempt was counted at reservation, the record was not. */
|
|
344
|
-
function persistFailureWarning(stateFile, reason) {
|
|
345
|
-
return (`\n\nWarning: this run's result could not be recorded in stateFile ` +
|
|
346
|
-
`${stateFile} (${reason}). The attempt was counted; the report will read ` +
|
|
347
|
-
`this file's result as Unknown unless the state file is repaired.`);
|
|
348
|
-
}
|
|
349
567
|
export function describeAttempt(priorAfterRuns, cap, opts = {}) {
|
|
350
568
|
const attempt = priorAfterRuns + 1;
|
|
351
569
|
const remaining = Math.max(0, cap - attempt);
|
|
@@ -384,16 +602,12 @@ export function buildExecutionFailureText(result, opts = {}) {
|
|
|
384
602
|
sections.push(output);
|
|
385
603
|
}
|
|
386
604
|
else if (errors.length > 0) {
|
|
387
|
-
|
|
388
|
-
|
|
389
|
-
|
|
390
|
-
sections.push("The executor captured no test output, so there are no per-test diagnostics. " +
|
|
391
|
-
"It did report the error below, which is the cause on record — use it. Do " +
|
|
392
|
-
"NOT infer anything the Errors line does not say, and leave the generated " +
|
|
393
|
-
"test unchanged.");
|
|
605
|
+
sections.push("The command printed no output, so there are no per-test diagnostics. " +
|
|
606
|
+
"The error below is the cause on record — use it. Do NOT infer anything " +
|
|
607
|
+
"the Errors line does not say, and leave the generated test unchanged.");
|
|
394
608
|
}
|
|
395
609
|
else {
|
|
396
|
-
sections.push("The
|
|
610
|
+
sections.push("The command printed no output, so this failure carries no diagnostics of " +
|
|
397
611
|
"its own and the cause cannot be determined from it. The test runner, the " +
|
|
398
612
|
"test process, or the application could each have died silently. Do NOT " +
|
|
399
613
|
"report the app as unreachable or misconfigured on this basis — nothing " +
|
|
@@ -422,69 +636,415 @@ export function buildExecutionFailureText(result, opts = {}) {
|
|
|
422
636
|
}
|
|
423
637
|
return sections.join("\n\n");
|
|
424
638
|
}
|
|
639
|
+
export const inputSchema = {
|
|
640
|
+
commandOverride: z
|
|
641
|
+
.string()
|
|
642
|
+
.trim()
|
|
643
|
+
.min(1)
|
|
644
|
+
.optional()
|
|
645
|
+
.describe("The shell command that runs this one test file with the repository's own test runner, run on this host. Prefer it: the test then runs the way the repository's own CI runs it. Take the command from the suite's testRunCommand in .skyramp/workspace.yml, a Makefile target, a package script, or the command the repository's CI runs, and limit it to this one file. When workspace.yml records more than one suite, take it from the suite whose pathGlobs match the file: another suite's runner either fails to load it or reports a result for a file it never ran. If the repository has none, use the framework's default: `npx --no-install <runner> <file>` for playwright, jest, vitest or mocha, so a runner the repository does not have fails the run instead of downloading a different version; `python -m pytest <file> -q` with the interpreter the repository's own tests use; `mvn -q test -Dtest=<class> -DfailIfNoTests=false` for junit. It must name the test file; a command that does not is refused without running. For a TypeScript or JavaScript ui or e2e test, the server sets SKYRAMP_PLAYWRIGHT_CONFIG to a config that records video. Pass it even when you expect the runner or a package to be missing: the failure output is what names what is missing, and a check you make before the run reaches no repair. Omit it to run the test in the Skyramp executor container instead, which needs workspacePath — that is the fallback for a repository this host cannot run, not the first choice."),
|
|
646
|
+
cwd: z
|
|
647
|
+
.string()
|
|
648
|
+
.optional()
|
|
649
|
+
.describe("Absolute path of the directory commandOverride runs in. Required with commandOverride."),
|
|
650
|
+
workspacePath: z
|
|
651
|
+
.string()
|
|
652
|
+
.optional()
|
|
653
|
+
.describe("Absolute path of the Skyramp workspace. Required when commandOverride is omitted, because the executor resolves the service and its base URL from it."),
|
|
654
|
+
testFile: z.string().describe("Absolute path to the test file to execute."),
|
|
655
|
+
language: z
|
|
656
|
+
.nativeEnum(ProgrammingLanguage)
|
|
657
|
+
.describe("Programming language of the test file to execute (e.g., python, javascript, typescript, java)"),
|
|
658
|
+
testType: z
|
|
659
|
+
.nativeEnum(TestType)
|
|
660
|
+
.describe("Type of the test to execute. Note: 'mock' is NOT a valid test type — mock files are deployed via their apply_mock() function, not executed as tests."),
|
|
661
|
+
phase: z
|
|
662
|
+
.enum(["before", "after"])
|
|
663
|
+
.default("after")
|
|
664
|
+
.describe("Execution phase for maintained tests: 'before' captures pre-edit baseline; 'after' (default) records post-edit result."),
|
|
665
|
+
stateFile: z
|
|
666
|
+
.string()
|
|
667
|
+
.optional()
|
|
668
|
+
.describe("Path to state file from skyramp_analyze_changes. Always pass when available — results are written back so skyramp_submit_report can override before/afterStatus with ground-truth pass/fail."),
|
|
669
|
+
repository: z
|
|
670
|
+
.string()
|
|
671
|
+
.trim()
|
|
672
|
+
.min(1)
|
|
673
|
+
.describe("The owner/repo whose analysis section to write execution results into (e.g. 'letsramp/api-insight'). Set it on every call — the primary's owner/repo for a primary-repo test, or a related repo's owner/repo for that repo's section of the run-scoped stateFile."),
|
|
674
|
+
token: z
|
|
675
|
+
.string()
|
|
676
|
+
.optional()
|
|
677
|
+
.describe("Explicit authentication token, set as SKYRAMP_TEST_TOKEN. Omit it, or pass an empty string, to use SKYRAMP_TEST_TOKEN from the server's environment. Use `unauthenticated: true` for no auth."),
|
|
678
|
+
unauthenticated: z
|
|
679
|
+
.boolean()
|
|
680
|
+
.optional()
|
|
681
|
+
.describe("Set true to force this test execution to carry NO auth token, even if SKYRAMP_TEST_TOKEN is set in the environment or a token was passed. Use for tests that must assert 401/403 unauthenticated behavior."),
|
|
682
|
+
rebaselineSnapshots: z
|
|
683
|
+
.array(rebaselineSnapshotNameSchema)
|
|
684
|
+
.optional()
|
|
685
|
+
.describe('UI tests only. toHaveScreenshot() baseline filenames (e.g. ["page-001.png"]) this run REPLACES instead of comparing against, because the PR intentionally changed how the captured page/element/region looks (SKYR-4298). ' +
|
|
686
|
+
"Pass exactly the list skyramp_actions returned as rebaseline_snapshots for this spec — the tool checks it against the persisted UPDATE verdict in stateFile (so stateFile is required) and refuses names the verdict did not authorize, a spec with no such verdict, or a spec whose phase: 'before' run has not been recorded yet. Use the name as the test passes it (page-001.png), not the on-disk file (page-001-chromium-linux.png). " +
|
|
687
|
+
"The refreshed PNGs land beside the spec and are delivered with the Testbot PR as image diffs; the result names which baselines were refreshed and which were not. Never use this to silence a screenshot mismatch the diff does not explain."),
|
|
688
|
+
timeout: z
|
|
689
|
+
.number()
|
|
690
|
+
.int()
|
|
691
|
+
.positive()
|
|
692
|
+
.max(MAX_TIMEOUT_MS, `timeout must be at most ${MAX_TIMEOUT_MS} ms`)
|
|
693
|
+
.optional()
|
|
694
|
+
.describe(`Milliseconds before the command is killed and the run recorded as Error. Default ${DEFAULT_TIMEOUT_MS}.`),
|
|
695
|
+
};
|
|
696
|
+
/** Returns a warning when the result, or its pre-edit baseline, was not saved. */
|
|
697
|
+
async function writeExecutionToState(params, result, reserved) {
|
|
698
|
+
if (!params.stateFile)
|
|
699
|
+
return undefined;
|
|
700
|
+
const where = `testFile ${params.testFile} in repository ${params.repository}`;
|
|
701
|
+
try {
|
|
702
|
+
const written = await persistTestExecutionResult(params.stateFile, params.repository, params.testType, params.phase, result,
|
|
703
|
+
// A reserved attempt already advanced the counters before the run.
|
|
704
|
+
{ reserved });
|
|
705
|
+
if (!written.saved) {
|
|
706
|
+
return `This result was not saved: the stateFile has no section for repository ${params.repository}.`;
|
|
707
|
+
}
|
|
708
|
+
if (params.phase === "before" && !written.matchedExistingTest) {
|
|
709
|
+
return `The phase: "before" result was not saved as a baseline: the stateFile lists no existing test for ${where}.`;
|
|
710
|
+
}
|
|
711
|
+
return undefined;
|
|
712
|
+
}
|
|
713
|
+
catch (err) {
|
|
714
|
+
// The attempt was counted at reservation, so the count is right. The RESULT
|
|
715
|
+
// is what the report reads for afterStatus, so an unrecorded one has to
|
|
716
|
+
// reach the agent, not only the log.
|
|
717
|
+
return (`This result for ${where} was not saved to the stateFile: ${err.message}. ` +
|
|
718
|
+
`The attempt was counted; the report will read this file's result as Unknown ` +
|
|
719
|
+
`unless the state file is repaired.`);
|
|
720
|
+
}
|
|
721
|
+
}
|
|
722
|
+
/** Stages rewritten baselines and reconciles the verdict; returns the text to append. */
|
|
723
|
+
async function settleRefresh(params, snapshots, before) {
|
|
724
|
+
const outcome = diffBaselineState(before, readBaselineState(params.testFile, snapshots));
|
|
725
|
+
let text = describeRefreshOutcome(outcome);
|
|
726
|
+
// The rewritten files themselves, never the directory: `git add` on the directory
|
|
727
|
+
// would ship anything else sitting there under one authorized baseline's authority.
|
|
728
|
+
for (const name of outcome.refreshed) {
|
|
729
|
+
for (const file of outcome.refreshedFiles[name] ?? []) {
|
|
730
|
+
try {
|
|
731
|
+
await stageGeneratedPaths(path.join(snapshotDirFor(params.testFile), file));
|
|
732
|
+
}
|
|
733
|
+
catch (err) {
|
|
734
|
+
logger.warning(`Could not stage refreshed visual baseline ${file}: ${err.message}`);
|
|
735
|
+
}
|
|
736
|
+
}
|
|
737
|
+
}
|
|
738
|
+
if (outcome.notRefreshed.length > 0 && params.stateFile) {
|
|
739
|
+
try {
|
|
740
|
+
const stateManager = StateManager.fromStatePath(params.stateFile);
|
|
741
|
+
const repo = params.repository;
|
|
742
|
+
const stateData = await stateManager.readRepoData(repo);
|
|
743
|
+
if (repo && stateData?.maintenanceVerdicts) {
|
|
744
|
+
const reconciled = applyRefreshOutcomeToVerdicts(stateData.maintenanceVerdicts, params.testFile, outcome);
|
|
745
|
+
await stateManager.updateRepoData({ ...stateData, maintenanceVerdicts: reconciled.verdicts }, { repo });
|
|
746
|
+
if (reconciled.note)
|
|
747
|
+
text += ` ${reconciled.note}`;
|
|
748
|
+
}
|
|
749
|
+
}
|
|
750
|
+
catch (err) {
|
|
751
|
+
logger.warning(`Could not reconcile the maintenance verdict with the refresh outcome: ${err.message}`);
|
|
752
|
+
}
|
|
753
|
+
}
|
|
754
|
+
return text;
|
|
755
|
+
}
|
|
756
|
+
/** The fallback: no commandOverride, so the Skyramp executor container runs the
|
|
757
|
+
* test. Kept while host execution is proven — the container resolves the
|
|
758
|
+
* service and its base URL from the workspace, which the host path leaves to
|
|
759
|
+
* the agent's own command.
|
|
760
|
+
*
|
|
761
|
+
* It records the same result the host path records, through the same writer,
|
|
762
|
+
* so the report cannot tell which path produced a run. */
|
|
763
|
+
async function runInExecutor(params, snapshots, budget, onProgress, sendProgress) {
|
|
764
|
+
// runTest validated this for the executor mode before anything wrote state.
|
|
765
|
+
const workspacePath = params.workspacePath;
|
|
766
|
+
const token = resolveEffectiveToken(params.unauthenticated, params.token, process.env.SKYRAMP_TEST_TOKEN);
|
|
767
|
+
// Delivered tests and the state file sit outside workspacePath in a
|
|
768
|
+
// cross-repo run, so the executor needs the tests-repo root both to mount
|
|
769
|
+
// them and to match the service that owns the file.
|
|
770
|
+
const testRepoPath = getTestsRepoDir();
|
|
771
|
+
const { baseUrl, candidates, dockerNetwork } = await getWorkspaceBaseUrl(workspacePath, params.testFile, params.language, testRepoPath);
|
|
772
|
+
if (!baseUrl && candidates.length > 0) {
|
|
773
|
+
return toolError([
|
|
774
|
+
"Cannot determine SKYRAMP_TEST_BASE_URL — the test file matches more than one service:",
|
|
775
|
+
...candidates.map((c) => ` \u2022 ${c.serviceName}: ${c.baseUrl}`),
|
|
776
|
+
"",
|
|
777
|
+
"Set SKYRAMP_TEST_BASE_URL to the right service URL, or give each service its own testDirectory in .skyramp/workspace.yml.",
|
|
778
|
+
].join("\n"));
|
|
779
|
+
}
|
|
780
|
+
let injectedBaseUrl = false;
|
|
781
|
+
if (baseUrl && !process.env.SKYRAMP_TEST_BASE_URL) {
|
|
782
|
+
process.env.SKYRAMP_TEST_BASE_URL = baseUrl;
|
|
783
|
+
injectedBaseUrl = true;
|
|
784
|
+
}
|
|
785
|
+
const execOptions = {
|
|
786
|
+
testFile: params.testFile,
|
|
787
|
+
workspacePath,
|
|
788
|
+
testRepoPath,
|
|
789
|
+
language: params.language,
|
|
790
|
+
testType: params.testType,
|
|
791
|
+
token,
|
|
792
|
+
dockerNetwork,
|
|
793
|
+
useHostNetwork: false,
|
|
794
|
+
...(snapshots.length > 0 ? { rebaselineSnapshots: snapshots } : {}),
|
|
795
|
+
...(params.timeout !== undefined ? { timeout: params.timeout } : {}),
|
|
796
|
+
};
|
|
797
|
+
let result;
|
|
798
|
+
// The executor's own retry of a transient connection error: two executions
|
|
799
|
+
// billed as ONE attempt, on purpose — the SUT's hiccup is not the agent's fix
|
|
800
|
+
// to make. An EOF or a refused connection means the SUT is not ready yet, not
|
|
801
|
+
// that the test failed. Only a throw from the executor is retried here; a
|
|
802
|
+
// connection error inside the test's own output reaches the agent as a normal
|
|
803
|
+
// failure.
|
|
804
|
+
let transientRetried = false;
|
|
805
|
+
try {
|
|
806
|
+
const service = new TestExecutionService();
|
|
807
|
+
try {
|
|
808
|
+
result = await service.executeTest(execOptions, onProgress);
|
|
809
|
+
}
|
|
810
|
+
catch (firstErr) {
|
|
811
|
+
const errMsg = firstErr instanceof Error ? firstErr.message : String(firstErr);
|
|
812
|
+
if (!/\bEOF\b|connection refused|ECONNREFUSED|ECONNRESET/i.test(errMsg)) {
|
|
813
|
+
throw firstErr;
|
|
814
|
+
}
|
|
815
|
+
logger.info(`Test execution hit transient connection error, retrying after ${transientRetryDelayMs}ms...`, { error: errMsg });
|
|
816
|
+
await sendProgress(50, 100, "SUT connection error — retrying in 10s...");
|
|
817
|
+
await new Promise((r) => setTimeout(r, transientRetryDelayMs));
|
|
818
|
+
transientRetried = true;
|
|
819
|
+
try {
|
|
820
|
+
result = await service.executeTest(execOptions, onProgress);
|
|
821
|
+
}
|
|
822
|
+
catch (secondErr) {
|
|
823
|
+
// Both runs threw: the caller's catch owes the agent the retry note, or
|
|
824
|
+
// the prompt lets it spend an unchanged re-run the tool already used.
|
|
825
|
+
throw Object.assign(secondErr, { transientRetried: true });
|
|
826
|
+
}
|
|
827
|
+
}
|
|
828
|
+
}
|
|
829
|
+
finally {
|
|
830
|
+
// Unset only what this call set; a caller's own value must survive.
|
|
831
|
+
if (injectedBaseUrl)
|
|
832
|
+
delete process.env.SKYRAMP_TEST_BASE_URL;
|
|
833
|
+
}
|
|
834
|
+
const warnings = [...result.warnings];
|
|
835
|
+
if (transientRetried) {
|
|
836
|
+
warnings.push("The executor hit a transient connection error and re-ran this file once unchanged before answering. This is the second run's result, counted as one attempt. Do not re-run it unchanged again — fix the cause the output names, or report it.");
|
|
837
|
+
}
|
|
838
|
+
const stateWarning = await writeExecutionToState(params, result, budget.reserved);
|
|
839
|
+
if (stateWarning)
|
|
840
|
+
warnings.push(stateWarning);
|
|
841
|
+
await recordExecutionVideo(result, params.stateFile);
|
|
842
|
+
const compose = (text) => [
|
|
843
|
+
withVideoInfo(text, result.videoPath),
|
|
844
|
+
...warnings.map((w) => `Warning: ${w}`),
|
|
845
|
+
]
|
|
846
|
+
.filter(Boolean)
|
|
847
|
+
.join("\n\n") + budget.noRunNote;
|
|
848
|
+
if (result.status !== TestExecutionStatus.Pass) {
|
|
849
|
+
return toolError(compose(buildExecutionFailureText(result, {
|
|
850
|
+
unchangedRerun: budget.unchangedRerun,
|
|
851
|
+
}) + budget.attemptLine));
|
|
852
|
+
}
|
|
853
|
+
return {
|
|
854
|
+
content: [
|
|
855
|
+
{
|
|
856
|
+
type: "text",
|
|
857
|
+
text: compose(`Test execution passed in the executor (duration=${result.duration}ms).\n\n${result.output ?? ""}`),
|
|
858
|
+
},
|
|
859
|
+
],
|
|
860
|
+
};
|
|
861
|
+
}
|
|
862
|
+
/** A throw after the attempt was reserved has spent it (a Docker image setup,
|
|
863
|
+
* a workspace check, a spawn that died). Land an Error result so the record is
|
|
864
|
+
* not left as the reservation placeholder, and tell the agent the attempt went. */
|
|
865
|
+
async function accountForThrow(params, err, budget) {
|
|
866
|
+
let persistNote = "";
|
|
867
|
+
if (budget.reserved) {
|
|
868
|
+
const warning = await writeExecutionToState(params, {
|
|
869
|
+
testFile: params.testFile,
|
|
870
|
+
status: TestExecutionStatus.Error,
|
|
871
|
+
executedAt: new Date().toISOString(),
|
|
872
|
+
duration: 0,
|
|
873
|
+
errors: [err.message],
|
|
874
|
+
warnings: [],
|
|
875
|
+
output: "",
|
|
876
|
+
}, true);
|
|
877
|
+
if (warning)
|
|
878
|
+
persistNote = `\n\nWarning: ${warning}`;
|
|
879
|
+
}
|
|
880
|
+
const retryNote = err.transientRetried
|
|
881
|
+
? TRANSIENT_RETRY_NOTE
|
|
882
|
+
: "";
|
|
883
|
+
return toolError(`Test execution failed: ${err.message}${retryNote}${budget.attemptLine}${persistNote}${budget.noRunNote}`);
|
|
884
|
+
}
|
|
885
|
+
/** The preferred path: the agent supplied the command, so the server runs it on
|
|
886
|
+
* this host in `cwd` and reads the verdict off the exit code. */
|
|
887
|
+
async function runOnHost(params, command, hostCwd, snapshots, budget) {
|
|
888
|
+
const warnings = [];
|
|
889
|
+
const invEnv = {};
|
|
890
|
+
const workspaceRoot = resolveWorkspaceRoot(hostCwd);
|
|
891
|
+
const token = resolveEffectiveToken(params.unauthenticated, params.token, process.env.SKYRAMP_TEST_TOKEN);
|
|
892
|
+
if (token)
|
|
893
|
+
invEnv.SKYRAMP_TEST_TOKEN = token;
|
|
894
|
+
else if (!params.unauthenticated) {
|
|
895
|
+
warnings.push("No auth token available — authenticated endpoints will likely return 401. Set SKYRAMP_TEST_TOKEN or pass token/unauthenticated.");
|
|
896
|
+
}
|
|
897
|
+
if (snapshots.length > 0) {
|
|
898
|
+
invEnv.SKYRAMP_UPDATE_SNAPSHOTS = snapshots.join(",");
|
|
899
|
+
}
|
|
900
|
+
let videoDir;
|
|
901
|
+
let wrapper;
|
|
902
|
+
/** The config the server picked because the command named none. */
|
|
903
|
+
let autoFoundConfig;
|
|
904
|
+
if (isBrowserTest(params.testType)) {
|
|
905
|
+
videoDir = path.join(workspaceRoot, ".skyramp", "videos", videoSubdirName(params.testFile));
|
|
906
|
+
fs.mkdirSync(videoDir, { recursive: true });
|
|
907
|
+
if (params.language === ProgrammingLanguage.PYTHON) {
|
|
908
|
+
invEnv.PYTEST_ADDOPTS = appendPytestVideoOpts(process.env.PYTEST_ADDOPTS, videoDir);
|
|
909
|
+
}
|
|
910
|
+
else if (params.language === ProgrammingLanguage.TYPESCRIPT ||
|
|
911
|
+
params.language === ProgrammingLanguage.JAVASCRIPT) {
|
|
912
|
+
const namedConfig = explicitPlaywrightConfig(command, hostCwd);
|
|
913
|
+
// A repository with one Playwright config per suite gets whichever sits
|
|
914
|
+
// nearest the test file, and its testDir may not collect that file. The
|
|
915
|
+
// agent cannot see the choice, so name it when the run finds no test.
|
|
916
|
+
autoFoundConfig = namedConfig
|
|
917
|
+
? undefined
|
|
918
|
+
: findRepoPlaywrightConfig(params.testFile, hostCwd);
|
|
919
|
+
wrapper = writeWrapperConfig({
|
|
920
|
+
// The wrapper replaces whatever --config the command carried, so an
|
|
921
|
+
// explicit one must be imported or the repository loses its projects,
|
|
922
|
+
// testDir and CI settings -- the thing the wrapper exists to keep.
|
|
923
|
+
repoConfigPath: namedConfig ?? autoFoundConfig,
|
|
924
|
+
fallbackDir: hostCwd,
|
|
925
|
+
outputDir: videoDir,
|
|
926
|
+
testFile: path.resolve(params.testFile),
|
|
927
|
+
commandPassesBrowserFlag: commandPassesBrowserFlag(command),
|
|
928
|
+
});
|
|
929
|
+
invEnv.SKYRAMP_PLAYWRIGHT_CONFIG = wrapper.path;
|
|
930
|
+
// A command that names its own config would otherwise load it directly and
|
|
931
|
+
// record no video, which is how a passing run loses its evidence.
|
|
932
|
+
if (namedConfig)
|
|
933
|
+
command = pointConfigAtWrapper(command, wrapper.path);
|
|
934
|
+
}
|
|
935
|
+
}
|
|
936
|
+
// testbot exports NODE_PATH=<mcp>/node_modules job-wide. Inheriting it lets the test
|
|
937
|
+
// load our @skyramp/skyramp and @playwright/test instead of the repository's own.
|
|
938
|
+
const env = { ...process.env, ...invEnv };
|
|
939
|
+
delete env.NODE_PATH;
|
|
940
|
+
if (!token)
|
|
941
|
+
delete env.SKYRAMP_TEST_TOKEN;
|
|
942
|
+
const baselinesBefore = snapshots.length > 0 ? readBaselineState(params.testFile, snapshots) : {};
|
|
943
|
+
const executedAt = new Date().toISOString();
|
|
944
|
+
const timeoutMs = params.timeout ?? DEFAULT_TIMEOUT_MS;
|
|
945
|
+
let run;
|
|
946
|
+
try {
|
|
947
|
+
run = await spawnTestCommand({
|
|
948
|
+
command,
|
|
949
|
+
cwd: hostCwd,
|
|
950
|
+
env,
|
|
951
|
+
timeoutMs,
|
|
952
|
+
});
|
|
953
|
+
}
|
|
954
|
+
finally {
|
|
955
|
+
// The wrapper sits in the customer's repo; a broad `git add` would ship it.
|
|
956
|
+
wrapper?.cleanup();
|
|
957
|
+
}
|
|
958
|
+
if (invEnv.PYTEST_ADDOPTS && pytestRejectedVideoOptions(run)) {
|
|
959
|
+
const plainEnv = { ...env };
|
|
960
|
+
if (process.env.PYTEST_ADDOPTS === undefined)
|
|
961
|
+
delete plainEnv.PYTEST_ADDOPTS;
|
|
962
|
+
else
|
|
963
|
+
plainEnv.PYTEST_ADDOPTS = process.env.PYTEST_ADDOPTS;
|
|
964
|
+
run = await spawnTestCommand({
|
|
965
|
+
command,
|
|
966
|
+
cwd: hostCwd,
|
|
967
|
+
env: plainEnv,
|
|
968
|
+
timeoutMs,
|
|
969
|
+
});
|
|
970
|
+
warnings.push("pytest did not recognize the added --video and --output options, so the test ran again without them. No video was recorded.");
|
|
971
|
+
videoDir = undefined;
|
|
972
|
+
}
|
|
973
|
+
const errors = [];
|
|
974
|
+
if (run.timedOut)
|
|
975
|
+
errors.push(`the command timed out after ${timeoutMs} ms`);
|
|
976
|
+
if (run.spawnError)
|
|
977
|
+
errors.push(run.spawnError);
|
|
978
|
+
const videoPath = videoDir ? collectVideoPath(videoDir) : undefined;
|
|
979
|
+
if (videoDir && !videoPath) {
|
|
980
|
+
warnings.push(`No video.webm was recorded under ${videoDir}. The verdict stands; the report has no video for this test.`);
|
|
981
|
+
}
|
|
982
|
+
const noRunReason = run.exitCode === 1 ? noTestRanReason(run.output) : undefined;
|
|
983
|
+
if (noRunReason) {
|
|
984
|
+
warnings.push(`No test ran here: the output reports "${noRunReason}". Exit 1 is the runner's, not a test's.`);
|
|
985
|
+
if (autoFoundConfig) {
|
|
986
|
+
warnings.push(`SKYRAMP_PLAYWRIGHT_CONFIG wraps ${path.relative(hostCwd, autoFoundConfig) || autoFoundConfig}, the nearest Playwright config above the test file, because the command named none. Its testDir decides which files the run collects — name the config the file belongs to if this is the wrong one.`);
|
|
987
|
+
}
|
|
988
|
+
}
|
|
989
|
+
const result = {
|
|
990
|
+
testFile: params.testFile,
|
|
991
|
+
status: verdictForExit(run.exitCode, run.timedOut || !!run.spawnError, run.output),
|
|
992
|
+
executedAt,
|
|
993
|
+
duration: run.duration,
|
|
994
|
+
errors,
|
|
995
|
+
warnings,
|
|
996
|
+
output: run.output,
|
|
997
|
+
...(run.exitCode !== null ? { exitCode: run.exitCode } : {}),
|
|
998
|
+
...(videoPath ? { videoPath } : {}),
|
|
999
|
+
};
|
|
1000
|
+
const stateWarning = await writeExecutionToState(params, result, budget.reserved);
|
|
1001
|
+
if (stateWarning)
|
|
1002
|
+
warnings.push(stateWarning);
|
|
1003
|
+
await recordExecutionVideo(result, params.stateFile);
|
|
1004
|
+
const refreshText = snapshots.length > 0
|
|
1005
|
+
? await settleRefresh(params, snapshots, baselinesBefore)
|
|
1006
|
+
: "";
|
|
1007
|
+
const compose = (text) => [
|
|
1008
|
+
withVideoInfo(text, videoPath),
|
|
1009
|
+
refreshText,
|
|
1010
|
+
...warnings.map((w) => `Warning: ${w}`),
|
|
1011
|
+
]
|
|
1012
|
+
.filter(Boolean)
|
|
1013
|
+
.join("\n\n") + budget.noRunNote;
|
|
1014
|
+
if (result.status !== TestExecutionStatus.Pass) {
|
|
1015
|
+
return toolError(compose(buildExecutionFailureText(result, {
|
|
1016
|
+
unchangedRerun: budget.unchangedRerun,
|
|
1017
|
+
}) + budget.attemptLine));
|
|
1018
|
+
}
|
|
1019
|
+
return {
|
|
1020
|
+
content: [
|
|
1021
|
+
{
|
|
1022
|
+
type: "text",
|
|
1023
|
+
text: compose(`Test execution passed (exitCode=0, duration=${result.duration}ms).\n\n${result.output ?? ""}`),
|
|
1024
|
+
},
|
|
1025
|
+
],
|
|
1026
|
+
};
|
|
1027
|
+
}
|
|
425
1028
|
export function registerExecuteSkyrampTestTool(server) {
|
|
426
1029
|
server.registerTool(TOOL_NAME, {
|
|
427
|
-
description:
|
|
1030
|
+
description: "Run one test file and record the verdict from the exit code: 0 is Pass, 1 is Fail, any other exit or a timeout is Error. With `commandOverride` the server runs that command on this machine in `cwd`, with the server's environment. Without it the Skyramp executor container runs the test, which needs `workspacePath`. Either way the output is kept from its end, up to 200,000 characters.",
|
|
428
1031
|
annotations: {
|
|
429
1032
|
readOnlyHint: false,
|
|
430
|
-
destructiveHint:
|
|
1033
|
+
destructiveHint: true,
|
|
431
1034
|
idempotentHint: false,
|
|
432
1035
|
openWorldHint: true,
|
|
433
1036
|
},
|
|
434
|
-
inputSchema
|
|
435
|
-
workspacePath: z
|
|
436
|
-
.string()
|
|
437
|
-
.describe("The path to the workspace directory where the test file is located"),
|
|
438
|
-
language: z
|
|
439
|
-
.nativeEnum(ProgrammingLanguage)
|
|
440
|
-
.describe("Programming language of the test file to execute (e.g., python, javascript, typescript, java)"),
|
|
441
|
-
testType: z
|
|
442
|
-
.nativeEnum(TestType)
|
|
443
|
-
.describe("Type of the test to execute. Note: 'mock' is NOT a valid test type — mock files are deployed via their apply_mock() function, not executed as tests."),
|
|
444
|
-
testFile: z
|
|
445
|
-
.string()
|
|
446
|
-
.describe("Absolute path to the test file to execute."),
|
|
447
|
-
contractMode: z
|
|
448
|
-
.enum(CONTRACT_EXECUTION_MODES)
|
|
449
|
-
.optional()
|
|
450
|
-
.describe("Only applies when testType is 'contract'. Use 'provider' for provider contract tests that hit the real service under test and need SKYRAMP_TEST_BASE_URL. Use 'consumer' only for consumer contract tests with inline mocks that do not hit the real service. Defaults to provider behavior when omitted."),
|
|
451
|
-
token: z
|
|
452
|
-
.string()
|
|
453
|
-
.optional()
|
|
454
|
-
.describe("Explicit Skyramp authentication token for test execution. Omit this parameter to use SKYRAMP_TEST_TOKEN from the environment. An empty string is passed through as a literal token value, not a 'no auth' signal — use `unauthenticated: true` for tests that must run without credentials."),
|
|
455
|
-
unauthenticated: z
|
|
456
|
-
.boolean()
|
|
457
|
-
.optional()
|
|
458
|
-
.describe("Set true to force this test execution to carry NO auth token, even if SKYRAMP_TEST_TOKEN is set in the environment or a token was passed. Use for tests that must assert 401/403 unauthenticated behavior."),
|
|
459
|
-
playwrightSaveStoragePath: z
|
|
460
|
-
.string()
|
|
461
|
-
.optional()
|
|
462
|
-
.describe("Path to save Playwright session storage after test execution for authentication purposes. Can be a relative path to the workspace (e.g., 'auth-session.json') or an absolute path. The session will be saved after the test completes."),
|
|
463
|
-
stateFile: z
|
|
464
|
-
.string()
|
|
465
|
-
.trim()
|
|
466
|
-
.optional()
|
|
467
|
-
.describe("Path to state file from skyramp_analyze_changes. Always pass when available — results are written back so skyramp_submit_report can override before/afterStatus with ground-truth pass/fail."),
|
|
468
|
-
phase: z
|
|
469
|
-
.enum(["before", "after"])
|
|
470
|
-
.default("after")
|
|
471
|
-
.describe("Execution phase for maintained tests: 'before' captures pre-edit baseline; 'after' (default) records post-edit result."),
|
|
472
|
-
rebaselineSnapshots: z
|
|
473
|
-
.array(rebaselineSnapshotNameSchema)
|
|
474
|
-
.optional()
|
|
475
|
-
.describe('UI tests only. toHaveScreenshot() baseline filenames (e.g. ["page-001.png"]) this run REPLACES instead of comparing against, because the PR intentionally changed how the captured page/element/region looks (SKYR-4298). ' +
|
|
476
|
-
"Pass exactly the list skyramp_actions returned as rebaseline_snapshots for this spec — the tool checks it against the persisted UPDATE verdict in stateFile (so stateFile is required) and refuses names the verdict did not authorize, a spec with no such verdict, or a spec whose phase: 'before' run has not been recorded yet. Use the name as the test passes it (page-001.png), not the on-disk file (page-001-chromium-linux.png). " +
|
|
477
|
-
"The refreshed PNGs land beside the spec and are delivered with the Testbot PR as image diffs; the result names which baselines were refreshed and which were not. Never use this to silence a screenshot mismatch the diff does not explain."),
|
|
478
|
-
repository: z
|
|
479
|
-
.string()
|
|
480
|
-
.trim()
|
|
481
|
-
.min(1)
|
|
482
|
-
.describe("The owner/repo whose analysis section to write execution results into (e.g. 'letsramp/api-insight'). Set it on every call — the primary's owner/repo for a primary-repo test, or a related repo's owner/repo for that repo's section of the run-scoped stateFile."),
|
|
483
|
-
},
|
|
1037
|
+
inputSchema,
|
|
484
1038
|
_meta: {
|
|
485
1039
|
keywords: ["run test", "execute test"],
|
|
486
1040
|
},
|
|
487
1041
|
}, async (params, extra) => {
|
|
1042
|
+
const sendProgress = makeProgressReporter(extra);
|
|
1043
|
+
// Progress callback adapter for TestExecutionService.
|
|
1044
|
+
const onExecutionProgress = async (progress) => {
|
|
1045
|
+
await sendProgress(progress.percent, 100, progress.message);
|
|
1046
|
+
};
|
|
1047
|
+
await sendProgress(0, 100, "Starting execution...");
|
|
488
1048
|
// `stateFile` may be omitted by the agent; the run's own state file (the
|
|
489
1049
|
// one skyramp_analyze_changes wrote for this run) is the fallback, so a
|
|
490
1050
|
// dropped argument cannot dodge the attempt cap or lose the result the
|
|
@@ -524,9 +1084,6 @@ export function registerExecuteSkyrampTestTool(server) {
|
|
|
524
1084
|
resolvedRunStatePath: resolveRunStatePath(),
|
|
525
1085
|
});
|
|
526
1086
|
}
|
|
527
|
-
const noRunNote = stateFilePath
|
|
528
|
-
? ""
|
|
529
|
-
: `\n\nNote: no run state file was found (none passed, and no active run), so this execution was not counted against an attempt cap and its result is not recorded for the report.`;
|
|
530
1087
|
// One execution at a time per run: everything below that touches the
|
|
531
1088
|
// state file — the cap read, the reservation, the retrofit staging, the
|
|
532
1089
|
// assertion proof, the result, the video record, the baseline verdict —
|
|
@@ -542,49 +1099,50 @@ export function registerExecuteSkyrampTestTool(server) {
|
|
|
542
1099
|
? stateFileKey(stateFilePath)
|
|
543
1100
|
: canonicalTestPath(params.testFile), async () => {
|
|
544
1101
|
let errorResult;
|
|
545
|
-
// Attempt-cap bookkeeping (SKYR-4460), visible to the outer catch so a
|
|
546
|
-
// throw after the reservation can still account for the attempt.
|
|
547
|
-
let maxFixAttempts = getMaxFixAttempts();
|
|
548
|
-
let priorAfterRuns = 0;
|
|
549
|
-
let reserved = false;
|
|
550
|
-
// Set when the tool re-ran a thrown transient connection error itself
|
|
551
|
-
// (see the executor call): both the result path and the outer catch
|
|
552
|
-
// must say so, or the agent is told it may re-run unchanged once more.
|
|
553
|
-
let transientRetried = false;
|
|
554
|
-
// The record the cap check read, and whether this run re-executes the
|
|
555
|
-
// same file contents it ran last time (SKYR-4460: the one unchanged
|
|
556
|
-
// re-run a file gets is then spent, and the texts say so).
|
|
557
|
-
let unchangedRerun = false;
|
|
558
|
-
// `repository` names the state section; `stateFilePath` (the argument, else
|
|
559
|
-
// this run's own file) says whether there is any state to write. A standalone
|
|
560
|
-
// execution — the local-dev workflow, an IDE call, no run — names its
|
|
561
|
-
// repository but writes nothing.
|
|
562
|
-
const stateSection = stateFilePath
|
|
563
|
-
? { file: stateFilePath, repo: params.repository }
|
|
564
|
-
: undefined;
|
|
565
|
-
// Helper to send progress notifications to the MCP client.
|
|
566
|
-
const sendProgress = makeProgressReporter(extra);
|
|
567
|
-
// Send immediate acknowledgment
|
|
568
|
-
await sendProgress(0, 100, "Starting execution...");
|
|
569
|
-
// Progress callback adapter for TestExecutionService
|
|
570
|
-
const onExecutionProgress = async (progress) => {
|
|
571
|
-
await sendProgress(progress.percent, 100, progress.message);
|
|
572
|
-
};
|
|
573
|
-
const previousBaseUrl = process.env.SKYRAMP_TEST_BASE_URL;
|
|
574
|
-
let didSetSkyrampBaseUrl = false;
|
|
575
|
-
let dockerNetwork;
|
|
576
1102
|
try {
|
|
577
1103
|
if (!path.isAbsolute(params.testFile)) {
|
|
578
1104
|
errorResult = toolError(`testFile must be an absolute path, got: ${params.testFile}`);
|
|
579
1105
|
return errorResult;
|
|
580
1106
|
}
|
|
581
|
-
// A missing test file is an argument error: refuse it before any
|
|
582
|
-
//
|
|
583
|
-
//
|
|
1107
|
+
// A missing test file is an argument error: refuse it before any side effect
|
|
1108
|
+
// (retrofit staging, the assertion proof record, the attempt reservation) has
|
|
1109
|
+
// recorded an execution that never ran.
|
|
584
1110
|
if (!fs.existsSync(params.testFile)) {
|
|
585
1111
|
errorResult = toolError(`Test file does not exist: ${params.testFile}`);
|
|
586
1112
|
return errorResult;
|
|
587
1113
|
}
|
|
1114
|
+
// Validate the inputs of whichever mode this call is in, before anything
|
|
1115
|
+
// below writes state: a call that cannot run must not record that it did.
|
|
1116
|
+
const command = params.commandOverride;
|
|
1117
|
+
const cwd = params.cwd;
|
|
1118
|
+
if (command === undefined) {
|
|
1119
|
+
if (params.workspacePath === undefined) {
|
|
1120
|
+
errorResult = toolError("workspacePath is required when commandOverride is omitted: the executor resolves the service and its base URL from the workspace. Pass commandOverride to run the test on this host instead.");
|
|
1121
|
+
return errorResult;
|
|
1122
|
+
}
|
|
1123
|
+
if (!path.isAbsolute(params.workspacePath)) {
|
|
1124
|
+
errorResult = toolError(`workspacePath must be an absolute path, got: ${params.workspacePath}`);
|
|
1125
|
+
return errorResult;
|
|
1126
|
+
}
|
|
1127
|
+
}
|
|
1128
|
+
else {
|
|
1129
|
+
if (cwd === undefined) {
|
|
1130
|
+
errorResult = toolError("cwd is required with commandOverride.");
|
|
1131
|
+
return errorResult;
|
|
1132
|
+
}
|
|
1133
|
+
if (!path.isAbsolute(cwd)) {
|
|
1134
|
+
errorResult = toolError(`cwd must be an absolute path, got: ${cwd}`);
|
|
1135
|
+
return errorResult;
|
|
1136
|
+
}
|
|
1137
|
+
if (!fs.statSync(cwd, { throwIfNoEntry: false })?.isDirectory()) {
|
|
1138
|
+
errorResult = toolError(`cwd ${cwd} is not an existing directory. The command was not run.`);
|
|
1139
|
+
return errorResult;
|
|
1140
|
+
}
|
|
1141
|
+
if (!commandNamesTestFile(command, params.testFile)) {
|
|
1142
|
+
errorResult = toolError(`commandOverride does not contain ${testFileNames(params.testFile).join(" or ")}, the name of testFile. The command was not run.`);
|
|
1143
|
+
return errorResult;
|
|
1144
|
+
}
|
|
1145
|
+
}
|
|
588
1146
|
// Hashed once, before any state write: recorded with the reservation
|
|
589
1147
|
// so the NEXT run can tell an unchanged re-run from a fixed file.
|
|
590
1148
|
const fileHash = sha256Of(fs.readFileSync(params.testFile, "utf8"));
|
|
@@ -593,63 +1151,49 @@ export function registerExecuteSkyrampTestTool(server) {
|
|
|
593
1151
|
errorResult = toolError(rebaseline.error);
|
|
594
1152
|
return errorResult;
|
|
595
1153
|
}
|
|
596
|
-
const
|
|
597
|
-
if (
|
|
598
|
-
if (!
|
|
1154
|
+
const snapshots = rebaseline.snapshots;
|
|
1155
|
+
if (snapshots.length > 0) {
|
|
1156
|
+
if (!stateFilePath) {
|
|
599
1157
|
errorResult = toolError("rebaselineSnapshots requires stateFile: the refresh is authorized against the UPDATE verdict skyramp_actions persisted there.");
|
|
600
1158
|
return errorResult;
|
|
601
1159
|
}
|
|
602
|
-
const authState = await StateManager.fromStatePath(
|
|
603
|
-
const auth = authorizeRebaseline(authState, params.testFile,
|
|
1160
|
+
const authState = await StateManager.fromStatePath(stateFilePath).readRepoData(params.repository);
|
|
1161
|
+
const auth = authorizeRebaseline(authState, params.testFile, snapshots);
|
|
604
1162
|
if (auth.error) {
|
|
605
1163
|
errorResult = toolError(auth.error);
|
|
606
1164
|
return errorResult;
|
|
607
1165
|
}
|
|
608
1166
|
}
|
|
609
|
-
//
|
|
610
|
-
//
|
|
611
|
-
//
|
|
612
|
-
//
|
|
613
|
-
|
|
614
|
-
|
|
615
|
-
|
|
616
|
-
|
|
617
|
-
|
|
618
|
-
|
|
619
|
-
const stateData = await stateManager.readRepoData(stateSection.repo);
|
|
620
|
-
const entry = stateData?.existingTests?.find((t) => canonicalTestPath(t.testFile) ===
|
|
621
|
-
canonicalTestPath(params.testFile));
|
|
622
|
-
if (entry?.source === TestSource.External) {
|
|
623
|
-
logger.info(`Skipping execution of external test ${params.testFile} — native suites are not run by ${TOOL_NAME}`);
|
|
624
|
-
return {
|
|
625
|
-
content: [
|
|
626
|
-
{
|
|
627
|
-
type: "text",
|
|
628
|
-
text: `Skipped execution: ${params.testFile} is marked external in the state file. ${TOOL_NAME} runs Skyramp-generated tests only; external (native) suites are not executed here.`,
|
|
629
|
-
},
|
|
630
|
-
],
|
|
631
|
-
};
|
|
632
|
-
}
|
|
633
|
-
}
|
|
634
|
-
catch (err) {
|
|
635
|
-
logger.warning(`External-test guard could not read stateFile (${err.message}); proceeding with execution`);
|
|
636
|
-
}
|
|
637
|
-
}
|
|
1167
|
+
// `stateFile` says whether there is any run state to count against and write
|
|
1168
|
+
// into; `repository` names the section inside it. A standalone execution —
|
|
1169
|
+
// the local-dev workflow, an IDE call, no run — names its repository but
|
|
1170
|
+
// writes nothing, and the tool says so in its answer.
|
|
1171
|
+
const stateSection = stateFilePath
|
|
1172
|
+
? { file: stateFilePath, repo: params.repository }
|
|
1173
|
+
: undefined;
|
|
1174
|
+
const noRunNote = stateSection
|
|
1175
|
+
? ""
|
|
1176
|
+
: `\n\nNote: no run state file was found (none passed, and no active run), so this execution was not counted against an attempt cap and its result is not recorded for the report.`;
|
|
638
1177
|
// Fix-and-rerun attempt cap (SKYR-4460). With a stateFile the cap is the one
|
|
639
|
-
// pinned in the run's state (the first call pins the prompt's value), and
|
|
640
|
-
//
|
|
641
|
-
//
|
|
642
|
-
//
|
|
643
|
-
//
|
|
644
|
-
maxFixAttempts = getMaxFixAttempts();
|
|
1178
|
+
// pinned in the run's state (the first call pins the prompt's value), and the
|
|
1179
|
+
// check fails CLOSED: this is the enforcement boundary, so a state that cannot
|
|
1180
|
+
// be read refuses the run instead of counting as zero prior attempts. A call
|
|
1181
|
+
// without any run state resolves from the environment and the default and
|
|
1182
|
+
// enforces nothing — there is no run to count against.
|
|
1183
|
+
let maxFixAttempts = getMaxFixAttempts();
|
|
1184
|
+
// Whether this run re-executes the same file contents as the last
|
|
1185
|
+
// one (SKYR-4460): the single unchanged re-run a file gets is then
|
|
1186
|
+
// spent, and the failure texts say so instead of offering another.
|
|
1187
|
+
let unchangedRerun = false;
|
|
1188
|
+
let priorAfterRuns = 0;
|
|
645
1189
|
if (stateSection) {
|
|
646
1190
|
let record;
|
|
647
1191
|
try {
|
|
648
1192
|
const stateData = await readRepoSectionOrThrow(stateSection.file, stateSection.repo);
|
|
649
|
-
// The pin is run-wide and lives at the ROOT, whichever section
|
|
650
|
-
//
|
|
651
|
-
//
|
|
652
|
-
//
|
|
1193
|
+
// The pin is run-wide and lives at the ROOT, whichever section this
|
|
1194
|
+
// file's record is in. A pinned value is re-validated: a hand-edited or
|
|
1195
|
+
// half-written state must not turn into "cap 0, nothing ever runs" or
|
|
1196
|
+
// "no cap".
|
|
653
1197
|
const pinned = await readPinnedMaxFixAttempts(stateSection.file);
|
|
654
1198
|
const current = getMaxFixAttempts();
|
|
655
1199
|
maxFixAttempts = pinned ?? current;
|
|
@@ -677,112 +1221,26 @@ export function registerExecuteSkyrampTestTool(server) {
|
|
|
677
1221
|
return errorResult;
|
|
678
1222
|
}
|
|
679
1223
|
}
|
|
680
|
-
// SKYR-4220:
|
|
681
|
-
// the last step before delivery, and a utils file left unstaged here ships a
|
|
682
|
-
// test importing a module the PR lacks. After the external-test skip: a native
|
|
683
|
-
// suite this tool will not run gets no scan and no staging.
|
|
1224
|
+
// SKYR-4220: a utils file left unstaged ships a test importing a module the PR lacks.
|
|
684
1225
|
await stageAndRecordRetrofits(params.testFile, undefined, stateFilePath);
|
|
685
|
-
|
|
686
|
-
|
|
687
|
-
|
|
688
|
-
// CALL needs no retry budget.
|
|
689
|
-
//
|
|
690
|
-
// Pass the run's state path through: without it the state path resolves
|
|
691
|
-
// from the CI/Testbot anchor alone and the check returns early — failing
|
|
692
|
-
// OPEN — for a caller that supplies a valid stateFile outside CI. It is the
|
|
693
|
-
// canonical path (explicit argument → this run's own file → undefined),
|
|
694
|
-
// the same one the external-test guard above and every write below use,
|
|
695
|
-
// so a symlinked argument cannot fork the state into a second file.
|
|
696
|
-
const owedReuseVerification = await pendingReuseDebt(params.testFile, stateFilePath, params.testType);
|
|
697
|
-
if (owedReuseVerification) {
|
|
698
|
-
errorResult = toolError(owedReuseVerification);
|
|
1226
|
+
const owedReuse = await pendingReuseDebt(params.testFile, stateFilePath, params.testType);
|
|
1227
|
+
if (owedReuse) {
|
|
1228
|
+
errorResult = toolError(owedReuse);
|
|
699
1229
|
return errorResult;
|
|
700
1230
|
}
|
|
701
|
-
// Deterministic assertion-enhancement check: the server verifies the file
|
|
702
|
-
// itself here (never relying on the agent to call `verify: true` — prose
|
|
703
|
-
// can be ignored, this cannot). Insufficient assertions return feedback
|
|
704
|
-
// instead of executing; a fixed file passes on the next execute call.
|
|
705
|
-
// Sits BELOW the external-test guard: an external test is skipped, not
|
|
706
|
-
// executed, so deferring it for assertion work would demand fixes to a
|
|
707
|
-
// file this tool will never run.
|
|
708
1231
|
const assertionFeedback = await assertionFeedbackForExecution(params.testFile, stateFilePath);
|
|
709
1232
|
if (assertionFeedback) {
|
|
710
1233
|
errorResult = toolError(assertionFeedback);
|
|
711
1234
|
return errorResult;
|
|
712
1235
|
}
|
|
713
|
-
// Proof-of-work substrate (SKYR-4262 follow-up): count this execution
|
|
714
|
-
// server-side so the report can cross-check generated vs executed —
|
|
715
|
-
// the narrative is LLM-authored, this number is not. Best-effort.
|
|
716
1236
|
await recordAssertionExecution(params.testFile, params.testType, stateFilePath);
|
|
717
|
-
//
|
|
718
|
-
|
|
719
|
-
//
|
|
720
|
-
//
|
|
721
|
-
//
|
|
722
|
-
//
|
|
723
|
-
|
|
724
|
-
// Resolve workspace config for base URL injection and Docker network
|
|
725
|
-
// attachment. A compose dockerNetwork is not host networking: on macOS
|
|
726
|
-
// the executor still needs localhost rewritten to host.docker.internal.
|
|
727
|
-
if (params.workspacePath) {
|
|
728
|
-
const workspaceConfig = await getWorkspaceBaseUrl(params.workspacePath, params.testFile, params.language, testRepoPath);
|
|
729
|
-
const { baseUrl, candidates } = workspaceConfig;
|
|
730
|
-
dockerNetwork = workspaceConfig.dockerNetwork;
|
|
731
|
-
const shouldInjectBaseUrl = shouldInjectSkyrampBaseUrl(params.testType, params.contractMode);
|
|
732
|
-
if (shouldInjectBaseUrl && !process.env.SKYRAMP_TEST_BASE_URL) {
|
|
733
|
-
if (baseUrl) {
|
|
734
|
-
process.env.SKYRAMP_TEST_BASE_URL = baseUrl;
|
|
735
|
-
didSetSkyrampBaseUrl = true;
|
|
736
|
-
}
|
|
737
|
-
else if (candidates.length > 0) {
|
|
738
|
-
errorResult = toolError([
|
|
739
|
-
`Cannot determine SKYRAMP_TEST_BASE_URL — test file matches multiple services:`,
|
|
740
|
-
...candidates.map((c) => ` • ${c.serviceName}: ${c.baseUrl}`),
|
|
741
|
-
``,
|
|
742
|
-
`Re-invoke with SKYRAMP_TEST_BASE_URL set to the correct service URL, or make each service's testDirectory unique in .skyramp/workspace.yml.`,
|
|
743
|
-
].join("\n"));
|
|
744
|
-
return errorResult;
|
|
745
|
-
}
|
|
746
|
-
}
|
|
747
|
-
}
|
|
748
|
-
const executionService = new TestExecutionService();
|
|
749
|
-
const effectiveToken = resolveEffectiveToken(params.unauthenticated, params.token, process.env.SKYRAMP_TEST_TOKEN);
|
|
750
|
-
if (!effectiveToken &&
|
|
751
|
-
!params.unauthenticated &&
|
|
752
|
-
params.token === undefined) {
|
|
753
|
-
logger.warning("No auth token available — authenticated endpoints will likely return 401. Set SKYRAMP_TEST_TOKEN or pass token/unauthenticated.");
|
|
754
|
-
}
|
|
755
|
-
// Execute test with progress callback - reports Docker cache/pull status.
|
|
756
|
-
// Retry once on transient connection errors (EOF, connection refused) —
|
|
757
|
-
// these indicate the SUT isn't fully ready yet, not a test failure.
|
|
758
|
-
// Retrying inside the tool call saves agent turns vs failing and
|
|
759
|
-
// requiring the agent to re-invoke. This covers a connection error the
|
|
760
|
-
// executor THROWS, under the one attempt already reserved; a connection
|
|
761
|
-
// error that surfaces inside the test's own output reaches the agent as
|
|
762
|
-
// a normal failure, and the prompt's unchanged-re-run allowance is for
|
|
763
|
-
// that one.
|
|
764
|
-
const execOptions = {
|
|
765
|
-
testFile: params.testFile,
|
|
766
|
-
workspacePath: params.workspacePath,
|
|
767
|
-
testRepoPath,
|
|
768
|
-
language: params.language,
|
|
769
|
-
testType: params.testType,
|
|
770
|
-
token: effectiveToken,
|
|
771
|
-
playwrightSaveStoragePath: params.playwrightSaveStoragePath,
|
|
772
|
-
dockerNetwork,
|
|
773
|
-
useHostNetwork: false,
|
|
774
|
-
...(rebaselineSnapshots.length > 0
|
|
775
|
-
? { rebaselineSnapshots }
|
|
776
|
-
: {}),
|
|
777
|
-
};
|
|
778
|
-
// Identity of the requested baselines before the run, to report afterwards
|
|
779
|
-
// which ones SmartPlaywright actually rewrote (SKYR-4298).
|
|
780
|
-
const baselinesBefore = rebaselineSnapshots.length > 0
|
|
781
|
-
? readBaselineState(params.testFile, rebaselineSnapshots)
|
|
782
|
-
: {};
|
|
783
|
-
// Reserve the attempt before the run (SKYR-4460): the counters advance now
|
|
784
|
-
// and the cap is pinned, so a state write that fails after the run cannot
|
|
785
|
-
// hand this file a free attempt. Fails closed, like the check above.
|
|
1237
|
+
// Reserve the attempt before the run (SKYR-4460). It sits ABOVE the mode
|
|
1238
|
+
// dispatch on purpose: the host path spawns the command itself and never
|
|
1239
|
+
// reaches TestExecutionService, so a reservation taken around the executor
|
|
1240
|
+
// call would count nothing on the path this tool now prefers. The counters
|
|
1241
|
+
// advance now and the cap is pinned, so a state write that fails after the
|
|
1242
|
+
// run cannot hand this file a free attempt. Fails closed, like the check above.
|
|
1243
|
+
let reserved = false;
|
|
786
1244
|
if (stateSection) {
|
|
787
1245
|
try {
|
|
788
1246
|
const reservation = await reserveTestExecutionAttempt(stateSection.file, stateSection.repo, params.testType, params.phase, params.testFile, maxFixAttempts, fileHash);
|
|
@@ -795,173 +1253,41 @@ export function registerExecuteSkyrampTestTool(server) {
|
|
|
795
1253
|
return errorResult;
|
|
796
1254
|
}
|
|
797
1255
|
}
|
|
798
|
-
|
|
799
|
-
|
|
800
|
-
|
|
801
|
-
|
|
802
|
-
|
|
803
|
-
|
|
804
|
-
|
|
805
|
-
|
|
806
|
-
|
|
1256
|
+
const budget = {
|
|
1257
|
+
// The budget line only means something when the run counts attempts, i.e.
|
|
1258
|
+
// when a stateFile records them and this is a final-phase run.
|
|
1259
|
+
attemptLine: stateSection && params.phase === "after"
|
|
1260
|
+
? `\n\n${describeAttempt(priorAfterRuns, maxFixAttempts, { unchangedRerun })}`
|
|
1261
|
+
: "",
|
|
1262
|
+
reserved,
|
|
1263
|
+
noRunNote,
|
|
1264
|
+
unchangedRerun,
|
|
1265
|
+
};
|
|
1266
|
+
await sendProgress(5, 100, "Starting test execution...");
|
|
1267
|
+
// Both modes owe the checks above. Only now does the path diverge.
|
|
1268
|
+
// Both take the RESOLVED state path, not the argument: everything they
|
|
1269
|
+
// hand it to — the result write, the video record, the verdict
|
|
1270
|
+
// reconciliation — must name the same file every check above read.
|
|
1271
|
+
const runParams = {
|
|
1272
|
+
...params,
|
|
1273
|
+
stateFile: stateFilePath,
|
|
1274
|
+
};
|
|
807
1275
|
try {
|
|
808
|
-
|
|
809
|
-
|
|
810
|
-
|
|
811
|
-
const errMsg = firstErr instanceof Error ? firstErr.message : String(firstErr);
|
|
812
|
-
if (/\bEOF\b|connection refused|ECONNREFUSED|ECONNRESET/i.test(errMsg)) {
|
|
813
|
-
logger.info(`Test execution hit transient connection error, retrying after ${transientRetryDelayMs}ms...`, { error: errMsg });
|
|
814
|
-
await sendProgress(50, 100, "SUT connection error — retrying in 10s...");
|
|
815
|
-
await new Promise((r) => setTimeout(r, transientRetryDelayMs));
|
|
816
|
-
transientRetried = true;
|
|
817
|
-
result = await executionService.executeTest(execOptions, onExecutionProgress);
|
|
818
|
-
}
|
|
819
|
-
else {
|
|
820
|
-
throw firstErr;
|
|
821
|
-
}
|
|
1276
|
+
return command === undefined
|
|
1277
|
+
? await runInExecutor(runParams, snapshots, budget, onExecutionProgress, sendProgress)
|
|
1278
|
+
: await runOnHost(runParams, command, cwd, snapshots, budget);
|
|
822
1279
|
}
|
|
823
|
-
|
|
824
|
-
|
|
825
|
-
: "";
|
|
826
|
-
// Update stateFile with execution results if provided. Multi-repo: write
|
|
827
|
-
// into the section for `repository` of the run-scoped file.
|
|
828
|
-
let persistWarning = noRunNote;
|
|
829
|
-
if (stateSection) {
|
|
830
|
-
try {
|
|
831
|
-
await persistTestExecutionResult(stateSection.file, stateSection.repo, params.testType, params.phase, result, { reserved: true });
|
|
832
|
-
logger.info(`Updated stateFile with execution results for ${params.testFile}`);
|
|
833
|
-
}
|
|
834
|
-
catch (err) {
|
|
835
|
-
// The attempt was reserved before the run, so the count is right.
|
|
836
|
-
// The RESULT is what the report reads for afterStatus, though, so
|
|
837
|
-
// an unrecorded one must reach the agent, not only the log.
|
|
838
|
-
logger.error(`Failed to update stateFile: ${err.message}`);
|
|
839
|
-
persistWarning = persistFailureWarning(stateSection.file, err.message);
|
|
840
|
-
}
|
|
841
|
-
}
|
|
842
|
-
// Record the recording for the report before returning, so it happens on the
|
|
843
|
-
// failure path too (SKYR-4156). skyramp_submit_report reads these records to
|
|
844
|
-
// populate testResults[].videoPath — testbot uploads only the video
|
|
845
|
-
// directories the report references, so an unrecorded video is never seen.
|
|
846
|
-
await recordExecutionVideo(result, stateFilePath);
|
|
847
|
-
// Which requested baselines were actually rewritten (SKYR-4298). Reported on
|
|
848
|
-
// pass and fail alike; a refreshed PNG is a deliverable like a generated
|
|
849
|
-
// spec, so stage the spec's snapshot directory whenever one was rewritten —
|
|
850
|
-
// even on a failing run, since the report gate checks the PNG, not the
|
|
851
|
-
// status — so the eval harness commit and the artifact collector see it
|
|
852
|
-
// (production delivery adds the whole test directory anyway). No-op outside
|
|
853
|
-
// a testbot run, like every other stageGeneratedPaths call; never fails the
|
|
854
|
-
// execution.
|
|
855
|
-
let refreshOutcomeText = "";
|
|
856
|
-
if (rebaselineSnapshots.length > 0) {
|
|
857
|
-
const after = readBaselineState(params.testFile, rebaselineSnapshots);
|
|
858
|
-
const outcome = diffBaselineState(baselinesBefore, after);
|
|
859
|
-
refreshOutcomeText = describeRefreshOutcome(outcome);
|
|
860
|
-
// Stage the rewritten files themselves, never the directory: `git add` on
|
|
861
|
-
// the directory would ship anything else sitting there under one
|
|
862
|
-
// authorized baseline's authority.
|
|
863
|
-
for (const name of outcome.refreshed) {
|
|
864
|
-
for (const file of outcome.refreshedFiles[name] ?? []) {
|
|
865
|
-
try {
|
|
866
|
-
await stageGeneratedPaths(path.join(snapshotDirFor(params.testFile), file));
|
|
867
|
-
}
|
|
868
|
-
catch (err) {
|
|
869
|
-
logger.warning(`Could not stage refreshed visual baseline ${file} for ${params.testFile}: ${err.message}`);
|
|
870
|
-
}
|
|
871
|
-
}
|
|
872
|
-
}
|
|
873
|
-
if (outcome.notRefreshed.length > 0 && stateSection) {
|
|
874
|
-
try {
|
|
875
|
-
const stateManager = StateManager.fromStatePath(stateSection.file);
|
|
876
|
-
const stateData = await stateManager.readRepoData(stateSection.repo);
|
|
877
|
-
if (stateData?.maintenanceVerdicts) {
|
|
878
|
-
const reconciled = applyRefreshOutcomeToVerdicts(stateData.maintenanceVerdicts, params.testFile, outcome, EXECUTOR_DOCKER_IMAGE);
|
|
879
|
-
await stateManager.updateRepoData({
|
|
880
|
-
...stateData,
|
|
881
|
-
maintenanceVerdicts: reconciled.verdicts,
|
|
882
|
-
}, { repo: stateSection.repo });
|
|
883
|
-
if (reconciled.note)
|
|
884
|
-
refreshOutcomeText += ` ${reconciled.note}`;
|
|
885
|
-
}
|
|
886
|
-
}
|
|
887
|
-
catch (err) {
|
|
888
|
-
logger.warning(`Could not reconcile the maintenance verdict with the refresh outcome: ${err.message}`);
|
|
889
|
-
}
|
|
890
|
-
}
|
|
1280
|
+
catch (err) {
|
|
1281
|
+
return accountForThrow(runParams, err, budget);
|
|
891
1282
|
}
|
|
892
|
-
const withRefreshOutcome = (text) => (refreshOutcomeText ? `${text}\n\n${refreshOutcomeText}` : text) +
|
|
893
|
-
transientRetryNote +
|
|
894
|
-
persistWarning;
|
|
895
|
-
// Progress is already reported by TestExecutionService
|
|
896
|
-
// Only report final status if not already at 100%
|
|
897
|
-
if (result.status !== TestExecutionStatus.Pass) {
|
|
898
|
-
// The budget line only means something when the run counts attempts,
|
|
899
|
-
// i.e. when a stateFile records them and this is a final-phase run.
|
|
900
|
-
const attemptLine = stateSection && params.phase === "after"
|
|
901
|
-
? `\n\n${describeAttempt(priorAfterRuns, maxFixAttempts, { unchangedRerun })}`
|
|
902
|
-
: "";
|
|
903
|
-
errorResult = toolError(withRefreshOutcome(withVideoInfo(buildExecutionFailureText(result, { unchangedRerun }) +
|
|
904
|
-
attemptLine, result.videoPath)));
|
|
905
|
-
return errorResult;
|
|
906
|
-
}
|
|
907
|
-
// Success - progress already reported by TestExecutionService
|
|
908
|
-
return {
|
|
909
|
-
content: [
|
|
910
|
-
{
|
|
911
|
-
type: "text",
|
|
912
|
-
text: withRefreshOutcome(withVideoInfo(`Test execution result: ${stripVTControlCharacters(result.output || "")}`, result.videoPath)),
|
|
913
|
-
},
|
|
914
|
-
],
|
|
915
|
-
};
|
|
916
1283
|
}
|
|
917
1284
|
catch (err) {
|
|
918
|
-
|
|
919
|
-
let attemptNote = "";
|
|
920
|
-
let persistNote = "";
|
|
921
|
-
// Both executions threw: the retry happened and the agent must still
|
|
922
|
-
// hear it, or the prompt lets it spend a third unchanged run.
|
|
923
|
-
const retryNote = transientRetried ? TRANSIENT_RETRY_NOTE : "";
|
|
924
|
-
// A throw after the reservation (Docker image setup, a workspace check,
|
|
925
|
-
// the rethrow from the transient-retry path) has spent an attempt; say
|
|
926
|
-
// so, and land an Error result so the record is not left as the
|
|
927
|
-
// reservation placeholder.
|
|
928
|
-
if (reserved && stateSection) {
|
|
929
|
-
try {
|
|
930
|
-
await persistTestExecutionResult(stateSection.file, stateSection.repo, params.testType, params.phase, {
|
|
931
|
-
testFile: params.testFile,
|
|
932
|
-
status: TestExecutionStatus.Error,
|
|
933
|
-
executedAt: new Date().toISOString(),
|
|
934
|
-
duration: 0,
|
|
935
|
-
errors: [message],
|
|
936
|
-
warnings: [],
|
|
937
|
-
output: "",
|
|
938
|
-
}, { reserved: true });
|
|
939
|
-
}
|
|
940
|
-
catch (persistErr) {
|
|
941
|
-
// Same as the main path: the count is right (reserved), the
|
|
942
|
-
// record is not, and the agent must hear it, not only the log.
|
|
943
|
-
logger.error(`Failed to record the execution error in stateFile: ${persistErr.message}`);
|
|
944
|
-
persistNote = persistFailureWarning(stateSection.file, persistErr.message);
|
|
945
|
-
}
|
|
946
|
-
if (params.phase === "after") {
|
|
947
|
-
attemptNote = `\n\n${describeAttempt(priorAfterRuns, maxFixAttempts, { unchangedRerun })}`;
|
|
948
|
-
}
|
|
949
|
-
}
|
|
950
|
-
errorResult = toolError(`Test execution failed: ${message}${retryNote}${attemptNote}${persistNote}${noRunNote}`);
|
|
1285
|
+
errorResult = toolError(`Test execution failed: ${err.message}`);
|
|
951
1286
|
return errorResult;
|
|
952
1287
|
}
|
|
953
1288
|
finally {
|
|
954
|
-
if (didSetSkyrampBaseUrl) {
|
|
955
|
-
if (previousBaseUrl === undefined) {
|
|
956
|
-
delete process.env.SKYRAMP_TEST_BASE_URL;
|
|
957
|
-
}
|
|
958
|
-
else {
|
|
959
|
-
process.env.SKYRAMP_TEST_BASE_URL = previousBaseUrl;
|
|
960
|
-
}
|
|
961
|
-
}
|
|
962
1289
|
AnalyticsService.pushMCPToolEvent(TOOL_NAME, errorResult, {
|
|
963
1290
|
testFile: params.testFile,
|
|
964
|
-
workspacePath: params.workspacePath,
|
|
965
1291
|
language: params.language,
|
|
966
1292
|
testType: params.testType,
|
|
967
1293
|
}).catch((err) => {
|