@skyramp/mcp 0.4.2-rc.1 → 0.4.2-rc.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/build/commands/localDevTestChangesCommand.js +1 -1
- package/build/commands/recommendTestsAndExecuteCommand.js +10 -1
- package/build/commands/testThisEndpointCommand.js +19 -2
- package/build/execution/wrapperConfig.d.ts +56 -0
- package/build/execution/wrapperConfig.js +155 -0
- package/build/index.js +6 -6
- package/build/playwright/registerPlaywrightTools.js +14 -0
- package/build/playwright/traceExportStore.d.ts +22 -0
- package/build/playwright/traceExportStore.js +81 -0
- package/build/playwright/traceRecordingPrompt.js +2 -1
- package/build/prompts/code-reuse.js +24 -21
- package/build/prompts/local-dev/local-dev-plan.js +6 -23
- package/build/prompts/local-dev/local-dev-prompts.js +1 -1
- package/build/prompts/shared-helper-policy.d.ts +36 -0
- package/build/prompts/shared-helper-policy.js +33 -1
- package/build/prompts/startTraceCollectionPrompts.js +1 -1
- package/build/prompts/sut-setup/modes/adaptWorkflowPrompt.js +6 -7
- package/build/prompts/sut-setup/shared.d.ts +1 -1
- package/build/prompts/sut-setup/shared.js +5 -3
- package/build/prompts/test-maintenance/drift-analysis-prompt.d.ts +16 -8
- package/build/prompts/test-maintenance/drift-analysis-prompt.js +90 -36
- package/build/prompts/test-maintenance/uiDriftAnalysisSections.js +1 -1
- package/build/prompts/test-recommendation/recommendationShared.d.ts +1 -1
- package/build/prompts/test-recommendation/recommendationShared.js +0 -1
- package/build/prompts/testbot/testbot-prompts.js +11 -9
- package/build/services/TestDiscoveryService.js +32 -4
- package/build/skills/runTestSkill.d.ts +6 -0
- package/build/skills/runTestSkill.js +17 -0
- package/build/tool-phases.js +0 -1
- package/build/tools/budgetExcuse.d.ts +15 -0
- package/build/tools/budgetExcuse.js +113 -0
- package/build/tools/code-refactor/utils-verify-gates.js +17 -3
- package/build/tools/executeSkyrampTestTool.d.ts +97 -48
- package/build/tools/executeSkyrampTestTool.js +775 -449
- package/build/tools/submitReportTool.js +128 -0
- package/build/tools/test-management/actionsTool.js +31 -0
- package/build/tools/test-management/analyzeChangesTool.d.ts +4 -4
- package/build/tools/test-management/analyzeTestHealthTool.d.ts +0 -11
- package/build/tools/test-management/analyzeTestHealthTool.js +7 -63
- package/build/tools/test-management/testsOwedBeforeRun.d.ts +28 -0
- package/build/tools/test-management/testsOwedBeforeRun.js +53 -0
- package/build/tools/trace/stopTraceCollectionTool.js +1 -1
- package/build/types/RepositoryAnalysis.d.ts +32 -32
- package/build/types/ReuseOutcome.d.ts +4 -3
- package/build/types/TestExecution.d.ts +2 -2
- package/build/types/TestTypes.d.ts +3 -0
- package/build/types/TestTypes.js +6 -0
- package/build/utils/AnalysisStateManager.d.ts +0 -7
- package/build/utils/AnalysisStateManager.js +1 -1
- package/build/utils/connectionErrors.d.ts +10 -0
- package/build/utils/connectionErrors.js +10 -0
- package/build/utils/language-helper.js +24 -3
- package/build/utils/progress.d.ts +1 -1
- package/build/utils/progress.js +1 -1
- package/build/utils/rebaselineSnapshots.d.ts +1 -1
- package/build/utils/rebaselineSnapshots.js +6 -16
- package/build/utils/reuseRouting.d.ts +10 -0
- package/build/utils/reuseRouting.js +15 -0
- package/build/utils/runContextGauge.d.ts +27 -0
- package/build/utils/runContextGauge.js +181 -0
- package/build/utils/skyrampMdContent.d.ts +1 -1
- package/build/utils/skyrampMdContent.js +1 -1
- package/build/utils/skyrampSdkVersion.d.ts +9 -0
- package/build/utils/skyrampSdkVersion.js +16 -0
- package/build/utils/testDependencyPolicy.js +21 -0
- package/build/utils/testExecutionRecord.d.ts +5 -1
- package/build/utils/testExecutionRecord.js +3 -1
- package/build/utils/testFileClassification.d.ts +8 -0
- package/build/utils/testFileClassification.js +36 -3
- package/build/utils/utils-verify/action-key.d.ts +42 -0
- package/build/utils/utils-verify/action-key.js +118 -36
- package/build/utils/utils-verify/action-sites.d.ts +32 -0
- package/build/utils/utils-verify/action-sites.js +202 -0
- package/build/utils/utils-verify/body-reach.js +2 -4
- package/build/utils/utils-verify/call-sites.d.ts +25 -6
- package/build/utils/utils-verify/call-sites.js +8 -5
- package/build/utils/utils-verify/index.d.ts +1 -0
- package/build/utils/utils-verify/index.js +1 -0
- package/build/utils/utils-verify/language-spec.d.ts +25 -0
- package/build/utils/utils-verify/language-spec.js +16 -2
- package/build/utils/utils-verify/parse.d.ts +10 -1
- package/build/utils/utils-verify/parse.js +19 -2
- package/build/utils/utils-verify/verify.d.ts +3 -2
- package/build/utils/utils-verify/verify.js +16 -18
- package/build/workspace/workspace.d.ts +72 -52
- package/build/workspace/workspace.js +12 -8
- package/package.json +1 -1
- package/plugin/prompts/testbot-task1.md +0 -2
- package/plugin/skills/fix-test-import-errors/SKILL.md +2 -1
- package/plugin/skills/run-test/SKILL.md +16 -0
- package/build/adapters/jestAdapter.d.ts +0 -14
- package/build/adapters/jestAdapter.js +0 -131
- package/build/adapters/mochaAdapter.d.ts +0 -13
- package/build/adapters/mochaAdapter.js +0 -93
- package/build/adapters/playwrightAdapter.d.ts +0 -17
- package/build/adapters/playwrightAdapter.js +0 -184
- package/build/adapters/pytestAdapter.d.ts +0 -15
- package/build/adapters/pytestAdapter.js +0 -119
- package/build/tools/runExistingTestsTool.d.ts +0 -138
- package/build/tools/runExistingTestsTool.js +0 -666
- package/build/types/ExternalTestExecution.d.ts +0 -67
- package/build/types/ExternalTestExecution.js +0 -8
- package/build/workspace/testSuites.d.ts +0 -20
- package/build/workspace/testSuites.js +0 -17
|
@@ -154,9 +154,10 @@ export interface HelperReuseOutcome {
|
|
|
154
154
|
* TODO: remove `helpersImported` once the testbot renderer reads this field. The
|
|
155
155
|
* two carry one fact; the count stays only because testbot#349 reads it today. */
|
|
156
156
|
helperNames: string[];
|
|
157
|
-
/** Inline request calls
|
|
158
|
-
*
|
|
159
|
-
* Omitted at zero. Informational,
|
|
157
|
+
/** Inline request calls (integration) or inline action sequences (UI) in OTHER
|
|
158
|
+
* Skyramp-generated tests beside this one that a helper in `utilsFile` already
|
|
159
|
+
* wraps — reuse that was available and not taken. Omitted at zero. Informational,
|
|
160
|
+
* like {@link ReuseOutcome.missedReuse}. */
|
|
160
161
|
siblingInlineCallSites?: number;
|
|
161
162
|
/** Whether the delivered utils file holds the shared-helper invariants: one helper
|
|
162
163
|
* per method+path, status-code-only assertions, method+resource names — with
|
|
@@ -41,8 +41,8 @@ export interface TestExecutionRecord {
|
|
|
41
41
|
executedAt: string;
|
|
42
42
|
duration: number;
|
|
43
43
|
/** The process's real exit code, or -1 when none was produced — a timeout
|
|
44
|
-
* kill or a
|
|
45
|
-
* see
|
|
44
|
+
* kill or a spawn error before the process exited (an Error-status run;
|
|
45
|
+
* see runTest in skyramp_execute_test). Never a real exit
|
|
46
46
|
* code, which is always 0-255. */
|
|
47
47
|
exitCode: number;
|
|
48
48
|
errors: string[];
|
|
@@ -7,6 +7,9 @@ export declare enum ProgrammingLanguage {
|
|
|
7
7
|
JAVASCRIPT = "javascript",
|
|
8
8
|
JAVA = "java"
|
|
9
9
|
}
|
|
10
|
+
/** The languages skyramp_execute_test accepts, as prose for a prompt. Rendered from
|
|
11
|
+
* the enum so the text and the gate that reads it cannot disagree. */
|
|
12
|
+
export declare function runnableLanguagesText(): string;
|
|
10
13
|
export declare enum TestType {
|
|
11
14
|
SMOKE = "smoke",
|
|
12
15
|
FUZZ = "fuzz",
|
package/build/types/TestTypes.js
CHANGED
|
@@ -9,6 +9,12 @@ export var ProgrammingLanguage;
|
|
|
9
9
|
ProgrammingLanguage["JAVASCRIPT"] = "javascript";
|
|
10
10
|
ProgrammingLanguage["JAVA"] = "java";
|
|
11
11
|
})(ProgrammingLanguage || (ProgrammingLanguage = {}));
|
|
12
|
+
/** The languages skyramp_execute_test accepts, as prose for a prompt. Rendered from
|
|
13
|
+
* the enum so the text and the gate that reads it cannot disagree. */
|
|
14
|
+
export function runnableLanguagesText() {
|
|
15
|
+
const names = Object.values(ProgrammingLanguage);
|
|
16
|
+
return `${names.slice(0, -1).join(", ")} or ${names[names.length - 1]}`;
|
|
17
|
+
}
|
|
12
18
|
export var TestType;
|
|
13
19
|
(function (TestType) {
|
|
14
20
|
TestType["SMOKE"] = "smoke";
|
|
@@ -5,7 +5,6 @@ import { TestAnalysisResult, MaintenanceActionCore } from "../types/TestAnalysis
|
|
|
5
5
|
import { RepositoryAnalysis, AnalysisScope } from "../types/RepositoryAnalysis.js";
|
|
6
6
|
import { PRTestContext } from "./pr-comment-parser.js";
|
|
7
7
|
import type { RemovedUiElement } from "./removedUiElements.js";
|
|
8
|
-
import type { ExternalTestRunRecord } from "../types/ExternalTestExecution.js";
|
|
9
8
|
import type { TestExecutionRecord, VideoRecord } from "../types/TestExecution.js";
|
|
10
9
|
import type { RepoCheckout } from "./reportVerification.js";
|
|
11
10
|
import type { Plan } from "../recommendation/registerPlan.js";
|
|
@@ -209,12 +208,6 @@ export interface UnifiedAnalysisState {
|
|
|
209
208
|
/** Maintenance actions applied via skyramp_actions — skyramp_submit_report derives
|
|
210
209
|
* testMaintenance from this. */
|
|
211
210
|
maintenanceVerdicts?: MaintenanceActionCore[];
|
|
212
|
-
/**
|
|
213
|
-
* External (repo-owned) test runs recorded by skyramp_run_existing_tests, one
|
|
214
|
-
* record per confirm/verify invocation. Drift analysis and the report fold in
|
|
215
|
-
* CONFIRMED failures from here instead of guessing. Absent until the tool runs.
|
|
216
|
-
*/
|
|
217
|
-
externalTestResults?: ExternalTestRunRecord[];
|
|
218
211
|
/** POM code-reuse outcome per generated UI spec, keyed by test-file BASENAME
|
|
219
212
|
* (what `newTestsCreated[].fileName` carries). Written by skyramp_reuse_code,
|
|
220
213
|
* which computes every value in-process; skyramp_submit_report merges it into
|
|
@@ -603,7 +603,7 @@ export class StateManager {
|
|
|
603
603
|
updatedAt: new Date().toISOString(),
|
|
604
604
|
// Preserve the prior step when the caller omits one (matches
|
|
605
605
|
// repositoryPath/repository/createdAt above) so a metadata-agnostic
|
|
606
|
-
// write — e.g.
|
|
606
|
+
// write — e.g. skyramp_execute_test recording an execution —
|
|
607
607
|
// does not wipe the pipeline step recorded by an earlier phase.
|
|
608
608
|
step: options?.step ?? existingMetadata?.step,
|
|
609
609
|
},
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The one place a connection-level failure is spelled. Three callers match these
|
|
3
|
+
* and each reads them for its own purpose — a transient MCP error, an executor
|
|
4
|
+
* retry, and the drift prompt's "the application was unreachable" verdict — so
|
|
5
|
+
* they are kept here rather than written out three times and drifting apart.
|
|
6
|
+
*/
|
|
7
|
+
/** Refused, reset, or the socket dropped mid-request. */
|
|
8
|
+
export declare const CONNECTION_DROPPED: RegExp;
|
|
9
|
+
/** The application never answered in time. */
|
|
10
|
+
export declare const CONNECTION_TIMED_OUT: RegExp;
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The one place a connection-level failure is spelled. Three callers match these
|
|
3
|
+
* and each reads them for its own purpose — a transient MCP error, an executor
|
|
4
|
+
* retry, and the drift prompt's "the application was unreachable" verdict — so
|
|
5
|
+
* they are kept here rather than written out three times and drifting apart.
|
|
6
|
+
*/
|
|
7
|
+
/** Refused, reset, or the socket dropped mid-request. */
|
|
8
|
+
export const CONNECTION_DROPPED = /\bECONNREFUSED\b|\bECONNRESET\b|\bconnection refused\b|\bconnection reset\b|\bsocket hang up\b/i;
|
|
9
|
+
/** The application never answered in time. */
|
|
10
|
+
export const CONNECTION_TIMED_OUT = /\bETIMEDOUT\b|\bDEADLINE_EXCEEDED\b|\bconnect(?:ion)? timed? ?out\b/i;
|
|
@@ -1,9 +1,13 @@
|
|
|
1
|
+
import { SKYRAMP_SDK_VERSION } from "./skyrampSdkVersion.js";
|
|
2
|
+
/** The SDK spec the generated JavaScript and TypeScript tests need. */
|
|
3
|
+
const npmSdkSpec = `@skyramp/skyramp@${SKYRAMP_SDK_VERSION}`;
|
|
1
4
|
const pythonDependencyGuidance = `
|
|
2
5
|
Persist each missing approved package in the development manifest:
|
|
3
6
|
- Use \`uv add --dev <package>\` for a project using \`[dependency-groups].dev\`.
|
|
4
7
|
- Use \`poetry add --group dev <package>\` for a Poetry project.
|
|
5
8
|
- Add it to \`requirements-dev.txt\`, then run \`python3 -m pip install -r requirements-dev.txt\`.
|
|
6
9
|
If no Python manifest exists at the test project root, create requirements-dev.txt and seed it with the missing approved package. Do not create it alongside pyproject.toml.
|
|
10
|
+
If the repository commits poetry.lock or uv.lock, update it in the same change. Use \`poetry lock\` or \`uv lock\`; both keep the versions already locked.
|
|
7
11
|
`;
|
|
8
12
|
const javascriptDependencyGuidance = `
|
|
9
13
|
Inspect package.json before installing anything.
|
|
@@ -11,19 +15,36 @@ Inspect package.json before installing anything.
|
|
|
11
15
|
- If no JavaScript/TypeScript manifest exists at the test project root, create package.json with only \`{ "devDependencies": {} }\` before adding the missing approved package.
|
|
12
16
|
- If an approved package is missing, add it as a devDependency with the repository's package manager. The npm equivalent is \`npm install --save-dev --legacy-peer-deps <package>\`.
|
|
13
17
|
- Never add, upgrade, or downgrade Playwright or @playwright/test when either is already declared.
|
|
18
|
+
- If the repository commits a lockfile, update it in the same change. Use \`npm install --package-lock-only\`, \`yarn install --mode update-lockfile\`, \`pnpm install --lockfile-only\`, or \`bun install --lockfile-only\` for the package manager the repository uses.
|
|
19
|
+
- Do not change \`allowScripts\`, \`pnpm.onlyBuiltDependencies\`, or \`trustedDependencies\`.
|
|
14
20
|
`;
|
|
15
21
|
const javaDependencyGuidance = `
|
|
16
22
|
Inspect existing Java build signals before adding a dependency. Use an existing manifest when present.
|
|
17
23
|
- For Maven wrapper or settings signals, create a minimal pom.xml only when it is absent, and add approved dependencies with <scope>test</scope>.
|
|
18
24
|
- For Gradle wrapper or settings signals, create the matching build.gradle or build.gradle.kts only when it is absent, and use testImplementation or testRuntimeOnly.
|
|
19
25
|
- Do not introduce a competing build system. If neither build system is identified, keep the standalone runner instead of inventing one.
|
|
26
|
+
- If the repository commits gradle.lockfile, update it in the same change with the repository's Gradle wrapper (\`./gradlew dependencies --write-locks\`), or \`gradle\` only when there is no wrapper.
|
|
20
27
|
`;
|
|
21
28
|
function getJavascriptSteps(extension) {
|
|
22
|
-
return `${javascriptDependencyGuidance}
|
|
29
|
+
return `${javascriptDependencyGuidance}${sdkDeclaration("javascript")}
|
|
23
30
|
# Execute the test file
|
|
24
31
|
npx playwright test <test-file>.spec.${extension} --reporter=list
|
|
25
32
|
`;
|
|
26
33
|
}
|
|
34
|
+
/** How to declare the Skyramp SDK, per language. Kept out of the shared
|
|
35
|
+
* dependency guidance because that guidance is also returned for `mock`, and
|
|
36
|
+
* a mock is deployed rather than run as a test — it needs no SDK. */
|
|
37
|
+
const SDK_DECLARATION = {
|
|
38
|
+
python: "Declare `skyramp` in the same development manifest.",
|
|
39
|
+
javascript: `If @skyramp/skyramp is not already declared, add ${npmSdkSpec} as a devDependency. If it is declared, leave its version alone and report the mismatch if the verifier refuses it.`,
|
|
40
|
+
java: "Declare `dev.skyramp:skyramp-library` with test scope. An existing manifest resolves the version through its own dependency management; a manifest you created has none, so give a concrete version there.",
|
|
41
|
+
};
|
|
42
|
+
/** The SDK line for a language, or nothing when the language has none. */
|
|
43
|
+
function sdkDeclaration(language) {
|
|
44
|
+
const key = language === "typescript" ? "javascript" : language;
|
|
45
|
+
const line = SDK_DECLARATION[key];
|
|
46
|
+
return line ? `\n${line}\n` : "";
|
|
47
|
+
}
|
|
27
48
|
export function getLanguageSteps(params) {
|
|
28
49
|
const { language, testType } = params;
|
|
29
50
|
if (testType === "mock") {
|
|
@@ -50,7 +71,7 @@ node <mock-file>.js
|
|
|
50
71
|
let steps = "";
|
|
51
72
|
switch (language) {
|
|
52
73
|
case "python":
|
|
53
|
-
steps = pythonDependencyGuidance;
|
|
74
|
+
steps = pythonDependencyGuidance + sdkDeclaration("python");
|
|
54
75
|
// e2e and ui tests
|
|
55
76
|
if (testType === "e2e" || testType === "ui") {
|
|
56
77
|
steps += `
|
|
@@ -80,7 +101,7 @@ python3 -m robot <robot-file>.robot
|
|
|
80
101
|
steps = getJavascriptSteps("ts");
|
|
81
102
|
break;
|
|
82
103
|
case "java":
|
|
83
|
-
steps = `${javaDependencyGuidance}
|
|
104
|
+
steps = `${javaDependencyGuidance}${sdkDeclaration("java")}
|
|
84
105
|
# Prerequisites
|
|
85
106
|
# Compile the test file
|
|
86
107
|
javac -cp "./lib/*" -d target <test-file>.java
|
|
@@ -11,6 +11,6 @@ export type ProgressReporter = (progress: number, total: number, message: string
|
|
|
11
11
|
* otherwise silently no-ops.
|
|
12
12
|
*
|
|
13
13
|
* Consolidates the identical closure that previously lived inline in multiple
|
|
14
|
-
* progress-emitting tools (analyzeChanges,
|
|
14
|
+
* progress-emitting tools (analyzeChanges, startTraceCollection).
|
|
15
15
|
*/
|
|
16
16
|
export declare function makeProgressReporter(extra: RequestHandlerExtra<ServerRequest, ServerNotification>): ProgressReporter;
|
package/build/utils/progress.js
CHANGED
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
* otherwise silently no-ops.
|
|
5
5
|
*
|
|
6
6
|
* Consolidates the identical closure that previously lived inline in multiple
|
|
7
|
-
* progress-emitting tools (analyzeChanges,
|
|
7
|
+
* progress-emitting tools (analyzeChanges, startTraceCollection).
|
|
8
8
|
*/
|
|
9
9
|
export function makeProgressReporter(extra) {
|
|
10
10
|
return async (progress, total, message) => {
|
|
@@ -6,7 +6,7 @@ import { z } from "zod";
|
|
|
6
6
|
*
|
|
7
7
|
* Bare filename only: no path separators, and no commas or whitespace, because the
|
|
8
8
|
* list is joined with commas into `SKYRAMP_UPDATE_SNAPSHOTS` and split again in the
|
|
9
|
-
*
|
|
9
|
+
* test runtime, where `a,b.png` would read as two baselines. One definition shared by
|
|
10
10
|
* both tools so they can never drift apart.
|
|
11
11
|
*/
|
|
12
12
|
export declare const REBASELINE_SNAPSHOT_NAME_RE: RegExp;
|
|
@@ -7,7 +7,7 @@ import { z } from "zod";
|
|
|
7
7
|
*
|
|
8
8
|
* Bare filename only: no path separators, and no commas or whitespace, because the
|
|
9
9
|
* list is joined with commas into `SKYRAMP_UPDATE_SNAPSHOTS` and split again in the
|
|
10
|
-
*
|
|
10
|
+
* test runtime, where `a,b.png` would read as two baselines. One definition shared by
|
|
11
11
|
* both tools so they can never drift apart.
|
|
12
12
|
*/
|
|
13
13
|
export const REBASELINE_SNAPSHOT_NAME_RE = /^[^/\\,\s]+\.png$/i;
|
|
@@ -22,21 +22,11 @@ export function baselineStem(name) {
|
|
|
22
22
|
return path.basename(name).replace(/\.png$/i, "");
|
|
23
23
|
}
|
|
24
24
|
/**
|
|
25
|
-
* Playwright's default snapshot layout
|
|
26
|
-
*
|
|
27
|
-
*
|
|
28
|
-
* `snapshotPathTemplate`
|
|
29
|
-
*
|
|
30
|
-
* relocate its snapshots out from under this rule.
|
|
31
|
-
*
|
|
32
|
-
* The executor's generated config declares no `projects`, but runner.sh passes
|
|
33
|
-
* `--browser=chromium`, so Playwright names the implicit project "chromium" and a
|
|
34
|
-
* refresh writes `<stem>-chromium-linux.png` (observed end to end on demoshop).
|
|
35
|
-
* Baselines the Skyramp executor produced therefore round-trip exactly. A baseline
|
|
36
|
-
* a repo's own CI wrote under another project name (`<stem>-Desktop-Chrome-linux.png`)
|
|
37
|
-
* is a different file: the executor's refresh adds its own and leaves that one as
|
|
38
|
-
* it was — the tool reports what it rewrote, and the repo's own suite is out of
|
|
39
|
-
* scope for skyramp_execute_test (which never runs it).
|
|
25
|
+
* Playwright's default snapshot layout: the baseline for `<stem>.png` lives in
|
|
26
|
+
* `<spec>-snapshots/` as `<stem>-<project>-<platform>.png` (or `<stem>-<platform>.png`
|
|
27
|
+
* with no project, or `<stem>.png` under a custom `snapshotPathTemplate`). A repo
|
|
28
|
+
* config that sets another `snapshotPathTemplate` moves its baselines out from under
|
|
29
|
+
* this rule; the tool then reports them as not refreshed rather than claim a refresh.
|
|
40
30
|
*
|
|
41
31
|
* Anchored so a longer baseline cannot vouch for a shorter one: `page-001-wide-chromium-linux.png`
|
|
42
32
|
* belongs to `page-001-wide.png` and must not satisfy a claim on `page-001.png`.
|
|
@@ -1,3 +1,13 @@
|
|
|
1
1
|
export declare function isBrowserTestType(testType?: string): boolean;
|
|
2
2
|
export declare function isModularizeFirstTarget(testType?: string, language?: string): boolean;
|
|
3
3
|
export declare function isPomAwareTarget(language: string, framework?: string, testType?: string): boolean;
|
|
4
|
+
/**
|
|
5
|
+
* Which family's inline call-site rule and sibling scan a test gets — the ONE gate
|
|
6
|
+
* for both. `api` for integration, whatever the flags (the rule is stated in API
|
|
7
|
+
* terms and predates the flag); `browser` for the seeded UI flow, where a Skyramp
|
|
8
|
+
* browser module exists to call into; `undefined` everywhere else (`contract`, `load`,
|
|
9
|
+
* e2e, an absent type, UI with the flag off or page-object reuse on), where an
|
|
10
|
+
* advisory would describe reuse the feature could not have taken. The prompt renders
|
|
11
|
+
* the rule and STEP 5c on it; the verify pass runs the scan on it.
|
|
12
|
+
*/
|
|
13
|
+
export declare function inlineCallSiteFamily(testType?: string, language?: string): "api" | "browser" | undefined;
|
|
@@ -52,3 +52,18 @@ export function isPomAwareTarget(language, framework, testType) {
|
|
|
52
52
|
return ((lang === "typescript" || lang === "javascript") &&
|
|
53
53
|
framework?.toLowerCase() === "playwright");
|
|
54
54
|
}
|
|
55
|
+
/**
|
|
56
|
+
* Which family's inline call-site rule and sibling scan a test gets — the ONE gate
|
|
57
|
+
* for both. `api` for integration, whatever the flags (the rule is stated in API
|
|
58
|
+
* terms and predates the flag); `browser` for the seeded UI flow, where a Skyramp
|
|
59
|
+
* browser module exists to call into; `undefined` everywhere else (`contract`, `load`,
|
|
60
|
+
* e2e, an absent type, UI with the flag off or page-object reuse on), where an
|
|
61
|
+
* advisory would describe reuse the feature could not have taken. The prompt renders
|
|
62
|
+
* the rule and STEP 5c on it; the verify pass runs the scan on it.
|
|
63
|
+
*/
|
|
64
|
+
export function inlineCallSiteFamily(testType, language) {
|
|
65
|
+
if (testType?.toLowerCase() === "integration")
|
|
66
|
+
return "api";
|
|
67
|
+
// A modularize-first target that is not integration is the seeded UI flow.
|
|
68
|
+
return isModularizeFirstTarget(testType, language) ? "browser" : undefined;
|
|
69
|
+
}
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
export interface ContextReading {
|
|
2
|
+
/** Tokens the last request actually carried: fresh input, cache reads and cache
|
|
3
|
+
* writes together. This is the number that fills the window. */
|
|
4
|
+
tokens: number;
|
|
5
|
+
window: number;
|
|
6
|
+
/** `tokens` as a whole percentage of `window`, for printing. */
|
|
7
|
+
percent: number;
|
|
8
|
+
}
|
|
9
|
+
/** This run's context usage, or undefined when it cannot be read.
|
|
10
|
+
*
|
|
11
|
+
* Undefined is ORDINARY, not an error: a local or IDE run has no `RUNNER_TEMP`
|
|
12
|
+
* and no agent log, and an agent whose CLI writes no NDJSON log never produces
|
|
13
|
+
* one. Every caller prints nothing and carries on. */
|
|
14
|
+
export declare function readRunContext(): ContextReading | undefined;
|
|
15
|
+
/** One sentence naming the measurement, for a refusal that has already told the
|
|
16
|
+
* agent not to stop on a context it believes it is short of. Empty when there is
|
|
17
|
+
* no reading — the refusal reads correctly without it.
|
|
18
|
+
*
|
|
19
|
+
* It states the reading and nothing else. It does not say the run has room: the
|
|
20
|
+
* reading can be 95% and the refusal still stands on its own grounds, and a
|
|
21
|
+
* sentence that argued headroom would be wrong exactly when it mattered most.
|
|
22
|
+
*
|
|
23
|
+
* CARRIES ITS OWN TRAILING SPACE, so a caller concatenates it unconditionally
|
|
24
|
+
* and an empty reading leaves no gap. */
|
|
25
|
+
export declare function contextGaugeSentence(): string;
|
|
26
|
+
/** Test seam: drop the per-run window cache. */
|
|
27
|
+
export declare function resetRunContextGaugeForTests(): void;
|
|
@@ -0,0 +1,181 @@
|
|
|
1
|
+
import * as fs from "fs";
|
|
2
|
+
import * as path from "path";
|
|
3
|
+
import { runArtifactDir } from "./AnalysisStateManager.js";
|
|
4
|
+
import { logger } from "./logger.js";
|
|
5
|
+
/** What the agent's own context usage is, read from the agent log the Testbot
|
|
6
|
+
* action streams beside this run's artifacts.
|
|
7
|
+
*
|
|
8
|
+
* WHY THIS EXISTS. Two runs cut planned work citing a "context budget" they
|
|
9
|
+
* could not measure: one declared itself out at 357k of a 1M window, the next at
|
|
10
|
+
* 457k, and neither had a warning from anywhere. The agent has no reading of its
|
|
11
|
+
* own context and the prompt cannot give it one, so a refusal that says "do not
|
|
12
|
+
* stop because you believe you are low on context" is arguing with a feeling.
|
|
13
|
+
* This turns it into a number the server can put in the refusal.
|
|
14
|
+
*
|
|
15
|
+
* It is a MEASUREMENT, never a limit. Nothing here refuses anything or decides
|
|
16
|
+
* anything; the callers do, on grounds that hold without it, and every one of
|
|
17
|
+
* them still works when this returns undefined. */
|
|
18
|
+
/** The Testbot action streams the agent CLI's `--output-format stream-json` here,
|
|
19
|
+
* live, in the same directory this server already writes `testbot-result.txt` to. */
|
|
20
|
+
const AGENT_LOG_NAME = "agent-log.ndjson";
|
|
21
|
+
/** Enough of the tail to hold the last assistant record with a usage block. These
|
|
22
|
+
* records carry whole tool results, so a fixed slice can miss one — 512 KiB covers
|
|
23
|
+
* it with room, and a miss costs a sentence, not a refusal. */
|
|
24
|
+
const TAIL_BYTES = 512 * 1024;
|
|
25
|
+
/** Enough of the head to hold the `init` record, which is the FIRST line. */
|
|
26
|
+
const HEAD_BYTES = 64 * 1024;
|
|
27
|
+
/** Windows for models with no `[1m]`-style suffix on the init record, first
|
|
28
|
+
* match wins. Opus 5 and Sonnet 5 are 1M by default — reading them as
|
|
29
|
+
* 200k is the same error the suffix rule below guards against, one level up.
|
|
30
|
+
* Bedrock prefixes the id (`us.anthropic.claude-opus-5`), so these match
|
|
31
|
+
* anywhere in the string rather than at the start. */
|
|
32
|
+
const MODEL_WINDOWS = [
|
|
33
|
+
[/claude-(opus|sonnet)-5/i, 1_000_000],
|
|
34
|
+
[/claude-haiku/i, 200_000],
|
|
35
|
+
];
|
|
36
|
+
/** A model none of the above names. ERRS LARGE on purpose: too small a window
|
|
37
|
+
* overstates the percentage, which is how this file manufactures the scarcity
|
|
38
|
+
* it exists to disprove, and no caller acts on the number — the refusals stand
|
|
39
|
+
* on their own grounds. Too large only understates it. */
|
|
40
|
+
const DEFAULT_CONTEXT_WINDOW = 1_000_000;
|
|
41
|
+
function windowForModel(model) {
|
|
42
|
+
for (const [pattern, window] of MODEL_WINDOWS) {
|
|
43
|
+
if (pattern.test(model))
|
|
44
|
+
return window;
|
|
45
|
+
}
|
|
46
|
+
return DEFAULT_CONTEXT_WINDOW;
|
|
47
|
+
}
|
|
48
|
+
/** Window and log path do not change within a run, so they are read once and kept
|
|
49
|
+
* against the directory they came from — a new run is a new directory. The token
|
|
50
|
+
* count is NOT cached: it is the thing that moves. */
|
|
51
|
+
let cachedForDir;
|
|
52
|
+
let cachedWindow;
|
|
53
|
+
function readSlice(file, bytes, fromEnd) {
|
|
54
|
+
const fd = fs.openSync(file, "r");
|
|
55
|
+
try {
|
|
56
|
+
const size = fs.fstatSync(fd).size;
|
|
57
|
+
const length = Math.min(size, bytes);
|
|
58
|
+
const buffer = Buffer.alloc(length);
|
|
59
|
+
fs.readSync(fd, buffer, 0, length, fromEnd ? size - length : 0);
|
|
60
|
+
return buffer.toString("utf8");
|
|
61
|
+
}
|
|
62
|
+
finally {
|
|
63
|
+
fs.closeSync(fd);
|
|
64
|
+
}
|
|
65
|
+
}
|
|
66
|
+
/** The context window this run's model was started with.
|
|
67
|
+
*
|
|
68
|
+
* READ FROM THE INIT RECORD, not from the per-message model. The two disagree:
|
|
69
|
+
* `message.model` is `claude-opus-5` while the init record says
|
|
70
|
+
* `claude-opus-5[1m]`, and taking the former turns 46% of a 1M window into 228%
|
|
71
|
+
* of a 200k one — a gauge that manufactures the panic it exists to prevent. */
|
|
72
|
+
function contextWindowFrom(head) {
|
|
73
|
+
for (const line of head.split("\n")) {
|
|
74
|
+
const trimmed = line.trim();
|
|
75
|
+
if (!trimmed.startsWith("{"))
|
|
76
|
+
continue;
|
|
77
|
+
let record;
|
|
78
|
+
try {
|
|
79
|
+
record = JSON.parse(trimmed);
|
|
80
|
+
}
|
|
81
|
+
catch {
|
|
82
|
+
// The head slice ends mid-record on any file longer than it. Only the first
|
|
83
|
+
// line matters here and it is whole.
|
|
84
|
+
break;
|
|
85
|
+
}
|
|
86
|
+
if (record.type !== "system" || record.subtype !== "init")
|
|
87
|
+
break;
|
|
88
|
+
const model = typeof record.model === "string" ? record.model : "";
|
|
89
|
+
const suffix = /\[(\d+)(m|k)\]/i.exec(model);
|
|
90
|
+
if (!suffix)
|
|
91
|
+
return windowForModel(model);
|
|
92
|
+
const scale = suffix[2].toLowerCase() === "m" ? 1_000_000 : 1_000;
|
|
93
|
+
return Number(suffix[1]) * scale;
|
|
94
|
+
}
|
|
95
|
+
return DEFAULT_CONTEXT_WINDOW;
|
|
96
|
+
}
|
|
97
|
+
/** Tokens on the most recent assistant request in the log, or undefined.
|
|
98
|
+
*
|
|
99
|
+
* Walks BACKWARDS to the first record that has them: the tail ends with the turn
|
|
100
|
+
* that called the tool now running, and every later line is that turn's own
|
|
101
|
+
* output. Records with no usage — heartbeats, progress, tool results — are
|
|
102
|
+
* skipped, so the reading is always the newest real request. */
|
|
103
|
+
function tokensFrom(tail) {
|
|
104
|
+
const lines = tail.split("\n");
|
|
105
|
+
for (let index = lines.length - 1; index >= 0; index--) {
|
|
106
|
+
const line = lines[index].trim();
|
|
107
|
+
// Cheap rejects first: this runs over a few thousand lines per call.
|
|
108
|
+
if (!line.startsWith("{") || !line.includes('"usage"'))
|
|
109
|
+
continue;
|
|
110
|
+
let record;
|
|
111
|
+
try {
|
|
112
|
+
record = JSON.parse(line);
|
|
113
|
+
}
|
|
114
|
+
catch {
|
|
115
|
+
continue;
|
|
116
|
+
}
|
|
117
|
+
const usage = record.message?.usage;
|
|
118
|
+
if (!usage)
|
|
119
|
+
continue;
|
|
120
|
+
const count = (key) => typeof usage[key] === "number" ? usage[key] : 0;
|
|
121
|
+
const tokens = count("input_tokens") +
|
|
122
|
+
count("cache_read_input_tokens") +
|
|
123
|
+
count("cache_creation_input_tokens");
|
|
124
|
+
if (tokens > 0)
|
|
125
|
+
return tokens;
|
|
126
|
+
}
|
|
127
|
+
return undefined;
|
|
128
|
+
}
|
|
129
|
+
/** This run's context usage, or undefined when it cannot be read.
|
|
130
|
+
*
|
|
131
|
+
* Undefined is ORDINARY, not an error: a local or IDE run has no `RUNNER_TEMP`
|
|
132
|
+
* and no agent log, and an agent whose CLI writes no NDJSON log never produces
|
|
133
|
+
* one. Every caller prints nothing and carries on. */
|
|
134
|
+
export function readRunContext() {
|
|
135
|
+
try {
|
|
136
|
+
const directory = runArtifactDir();
|
|
137
|
+
if (!directory)
|
|
138
|
+
return undefined;
|
|
139
|
+
const file = path.join(directory, AGENT_LOG_NAME);
|
|
140
|
+
if (!fs.existsSync(file))
|
|
141
|
+
return undefined;
|
|
142
|
+
if (directory !== cachedForDir) {
|
|
143
|
+
cachedForDir = directory;
|
|
144
|
+
cachedWindow = contextWindowFrom(readSlice(file, HEAD_BYTES, false));
|
|
145
|
+
}
|
|
146
|
+
const window = cachedWindow ?? DEFAULT_CONTEXT_WINDOW;
|
|
147
|
+
const tokens = tokensFrom(readSlice(file, TAIL_BYTES, true));
|
|
148
|
+
if (tokens === undefined || tokens <= 0)
|
|
149
|
+
return undefined;
|
|
150
|
+
return { tokens, window, percent: Math.round((tokens / window) * 100) };
|
|
151
|
+
}
|
|
152
|
+
catch (error) {
|
|
153
|
+
const detail = error instanceof Error ? error.message : String(error);
|
|
154
|
+
logger.warning(`readRunContext failed: ${detail}`);
|
|
155
|
+
return undefined;
|
|
156
|
+
}
|
|
157
|
+
}
|
|
158
|
+
/** One sentence naming the measurement, for a refusal that has already told the
|
|
159
|
+
* agent not to stop on a context it believes it is short of. Empty when there is
|
|
160
|
+
* no reading — the refusal reads correctly without it.
|
|
161
|
+
*
|
|
162
|
+
* It states the reading and nothing else. It does not say the run has room: the
|
|
163
|
+
* reading can be 95% and the refusal still stands on its own grounds, and a
|
|
164
|
+
* sentence that argued headroom would be wrong exactly when it mattered most.
|
|
165
|
+
*
|
|
166
|
+
* CARRIES ITS OWN TRAILING SPACE, so a caller concatenates it unconditionally
|
|
167
|
+
* and an empty reading leaves no gap. */
|
|
168
|
+
export function contextGaugeSentence() {
|
|
169
|
+
const reading = readRunContext();
|
|
170
|
+
if (!reading)
|
|
171
|
+
return "";
|
|
172
|
+
const k = (value) => `${Math.round(value / 1000)}k`;
|
|
173
|
+
return (`Measured, not estimated: this run's last request carried ${k(reading.tokens)} of a ` +
|
|
174
|
+
`${k(reading.window)} token context window (${reading.percent}%). ` +
|
|
175
|
+
"Use that number rather than an impression of it. ");
|
|
176
|
+
}
|
|
177
|
+
/** Test seam: drop the per-run window cache. */
|
|
178
|
+
export function resetRunContextGaugeForTests() {
|
|
179
|
+
cachedForDir = undefined;
|
|
180
|
+
cachedWindow = undefined;
|
|
181
|
+
}
|
|
@@ -2,4 +2,4 @@
|
|
|
2
2
|
* Skill content for skyramp.md — installed at auto-discovery paths: ~/.claude/skills/skyramp/SKILLS.md, ~/.cursor/skills/skyramp/SKILLS.md, ~/.github/skills/skyramp.md
|
|
3
3
|
* Follows the SKILL.md specification: https://agentskills.io/what-are-skills#the-skill-md-file
|
|
4
4
|
*/
|
|
5
|
-
export declare const SKYRAMP_MD_CONTENT = "---\nname: skyramp\ndescription: Generate, execute, and maintain API tests (smoke, contract, fuzz, load, integration, E2E, UI) using Skyramp MCP tools.\n---\n\n# Skyramp\n\n## When to use this skill\nUse this skill whenever the user asks to generate, run, or maintain API tests. Always read `<workspace-root>/.skyramp/workspace.yml` first to get `language`, `framework`, `outputDir`, and `api.baseUrl` \u2014 do not ask the user for values already present.\n\n---\n\n## Tools\n\n### Workspace Setup\n1. **`skyramp_initialize_workspace`** \u2014 Create or update `.skyramp/workspace.yml` in a git repository workspace. Scan the repo for all services before calling. Required before any other Skyramp tool.\n\n### Test Generation\n2. **`skyramp_smoke_test_generation`** \u2014 Verify an endpoint is reachable and returns a valid response.\n3. **`skyramp_contract_test_generation`** \u2014 Validate implementation matches OpenAPI/Swagger schema.\n4. **`skyramp_fuzz_test_generation`** \u2014 Send malformed or boundary inputs to find edge cases and security issues.\n5. **`skyramp_load_test_generation`** \u2014 Test performance under concurrent load. Optional: `loadDuration`, `loadNumThreads`. Accepts `trace` instead of `apiSchema`/`endpointURL`.\n6. **`skyramp_integration_test_generation`** \u2014 Multi-step workflows across one or more services. Supply one of: `apiSchema`+`endpointURL`, `trace`, or `scenarioFile`. Do not combine them.\n7. **`skyramp_e2e_test_generation`** \u2014 Full user journey covering UI and backend. Requires `trace` and `playwrightInput` zip. Do not pass `apiSchema` or `endpointURL`.\n8. **`skyramp_ui_test_generation`** \u2014 UI-only tests from Playwright recordings. Requires `playwrightInput` zip.\n\n### Trace Generation\n9. **`skyramp_start_trace_collection`** \u2014 Start capturing backend traffic. Set `playwright: true` for UI or E2E tests. Use an absolute path for `outputDir`.\n10. **`skyramp_stop_trace_collection`** \u2014 Stop capture and save the trace. Use the same `outputDir` and `playwrightEnabled` values as start.\n11. **`skyramp_scenario_test_generation`** \u2014 Scenario trace generation. Describe a multi-step flow in natural language to produce a scenario file. Pass output to `skyramp_integration_test_generation` via `scenarioFile`.\n\n### Test Execution\n12. **`skyramp_execute_test`** \u2014 Run
|
|
5
|
+
export declare const SKYRAMP_MD_CONTENT = "---\nname: skyramp\ndescription: Generate, execute, and maintain API tests (smoke, contract, fuzz, load, integration, E2E, UI) using Skyramp MCP tools.\n---\n\n# Skyramp\n\n## When to use this skill\nUse this skill whenever the user asks to generate, run, or maintain API tests. Always read `<workspace-root>/.skyramp/workspace.yml` first to get `language`, `framework`, `outputDir`, and `api.baseUrl` \u2014 do not ask the user for values already present.\n\n---\n\n## Tools\n\n### Workspace Setup\n1. **`skyramp_initialize_workspace`** \u2014 Create or update `.skyramp/workspace.yml` in a git repository workspace. Scan the repo for all services before calling. Required before any other Skyramp tool.\n\n### Test Generation\n2. **`skyramp_smoke_test_generation`** \u2014 Verify an endpoint is reachable and returns a valid response.\n3. **`skyramp_contract_test_generation`** \u2014 Validate implementation matches OpenAPI/Swagger schema.\n4. **`skyramp_fuzz_test_generation`** \u2014 Send malformed or boundary inputs to find edge cases and security issues.\n5. **`skyramp_load_test_generation`** \u2014 Test performance under concurrent load. Optional: `loadDuration`, `loadNumThreads`. Accepts `trace` instead of `apiSchema`/`endpointURL`.\n6. **`skyramp_integration_test_generation`** \u2014 Multi-step workflows across one or more services. Supply one of: `apiSchema`+`endpointURL`, `trace`, or `scenarioFile`. Do not combine them.\n7. **`skyramp_e2e_test_generation`** \u2014 Full user journey covering UI and backend. Requires `trace` and `playwrightInput` zip. Do not pass `apiSchema` or `endpointURL`.\n8. **`skyramp_ui_test_generation`** \u2014 UI-only tests from Playwright recordings. Requires `playwrightInput` zip.\n\n### Trace Generation\n9. **`skyramp_start_trace_collection`** \u2014 Start capturing backend traffic. Set `playwright: true` for UI or E2E tests. Use an absolute path for `outputDir`.\n10. **`skyramp_stop_trace_collection`** \u2014 Stop capture and save the trace. Use the same `outputDir` and `playwrightEnabled` values as start.\n11. **`skyramp_scenario_test_generation`** \u2014 Scenario trace generation. Describe a multi-step flow in natural language to produce a scenario file. Pass output to `skyramp_integration_test_generation` via `scenarioFile`.\n\n### Test Execution\n12. **`skyramp_execute_test`** \u2014 Run one test file and record the verdict from the exit code. Required: `testFile`, `language`, `testType`, `repository`. To run on this host, pass `commandOverride` (the command for that one file) with `cwd`. Omit `commandOverride` to run in the Skyramp executor instead, which needs `workspacePath`. Optional: `stateFile` (writes execution results back for health analysis). Output is capped at the last 200,000 characters.\n\n### Test Analysis & Maintenance\n13. **`skyramp_analyze_changes`** \u2014 Unified entry point: scans endpoints, discovers tests, computes diff. Takes `repositoryPath` and `scope`. Returns `stateFile` + recommendations.\n14. **`skyramp_analyze_test_health`** \u2014 Drift and health assessment for existing tests. Takes `stateFile`. Returns LLM prompt for scoring.\n15. **`skyramp_actions`** \u2014 Execute UPDATE / REGENERATE / VERIFY / DELETE actions. Takes `stateFile`. Call after analyze_test_health.\n\n### Code Quality\n16. **`skyramp_fix_errors`** \u2014 Fix compilation or runtime errors in a generated test file.\n17. **`skyramp_modularization`** \u2014 Refactor a test file into reusable modules. Set `isTraceBased: true` for trace-based tests.\n18. **`skyramp_reuse_code`** \u2014 Pull shared helpers from other Skyramp tests. Only when `code_reuse` was `true` at generation time.\n\n### Authentication\n19. **`skyramp_login`** \u2014 Log in to the Skyramp platform.\n20. **`skyramp_logout`** \u2014 Log out from the Skyramp platform.\n\n---\n\n## Prompts\n\nPrefer invoking a prompt over manually chaining tools \u2014 prompts run the full workflow automatically.\n\n- **`skyramp_trace_prompt`** \u2014 Trace collection setup and execution.\n- **`skyramp_test_health_analysis`** \u2014 Full maintenance flow (discover \u2192 drift \u2192 health \u2192 actions).\n- **`skyramp_testbot`** \u2014 PR-scoped recommendations + maintenance + report. Required: `prTitle`, `prDescription`, `repositoryPath`.\n\n---\n\n## Workflows\n\nUse these when a prompt is not available.\n\n**Generate and run a test**\nRead `.skyramp/workspace.yml` \u2192 Call appropriate generation tool \u2192 `skyramp_execute_test`\n\n**Trace-based test (integration / load / E2E / UI)**\n`skyramp_start_trace_collection` (set `playwright: true` for UI/E2E) \u2192 User exercises the app \u2192 `skyramp_stop_trace_collection` \u2192 Call target generation tool with `trace` (and `playwrightInput` for E2E/UI)\n\n**Scenario \u2192 integration test**\n`skyramp_scenario_test_generation` \u2192 `skyramp_integration_test_generation` with `scenarioFile`\n\n**Recommend tests for a PR**\n`skyramp_analyze_changes` (with `scope: \"branch_diff\"`) \u2192 follow enrichment steps \u2192 `skyramp_recommend_tests` \u2192 Generate recommended tests\n\n**Maintain existing tests**\n`skyramp_analyze_changes` \u2192 `skyramp_analyze_test_health` \u2192 *(optional)* execute tests via `skyramp_execute_test` with `stateFile` param (writes results back) \u2192 `skyramp_actions` (do not skip)\n\n---\n\n## Conventions\n\n- **`endpointURL`** \u2014 Full URL to a specific endpoint, not just the base URL. Build from `api.baseUrl` + path (e.g. `http://localhost:8000/api/v1/users`).\n- **`outputDir`** \u2014 Use absolute paths for both test output and trace collection.\n- **Mutually exclusive inputs** \u2014 `apiSchema`/`endpointURL`, `trace`, and `scenarioFile` are mutually exclusive for integration and load tests. Use exactly one.\n- **Sequential execution** \u2014 Execute tests sequentially (not in parallel) to avoid environment variable conflicts with `SKYRAMP_TEST_BASE_URL`.\n";
|
|
@@ -34,7 +34,7 @@ Use this skill whenever the user asks to generate, run, or maintain API tests. A
|
|
|
34
34
|
11. **\`skyramp_scenario_test_generation\`** — Scenario trace generation. Describe a multi-step flow in natural language to produce a scenario file. Pass output to \`skyramp_integration_test_generation\` via \`scenarioFile\`.
|
|
35
35
|
|
|
36
36
|
### Test Execution
|
|
37
|
-
12. **\`skyramp_execute_test\`** — Run
|
|
37
|
+
12. **\`skyramp_execute_test\`** — Run one test file and record the verdict from the exit code. Required: \`testFile\`, \`language\`, \`testType\`, \`repository\`. To run on this host, pass \`commandOverride\` (the command for that one file) with \`cwd\`. Omit \`commandOverride\` to run in the Skyramp executor instead, which needs \`workspacePath\`. Optional: \`stateFile\` (writes execution results back for health analysis). Output is capped at the last 200,000 characters.
|
|
38
38
|
|
|
39
39
|
### Test Analysis & Maintenance
|
|
40
40
|
13. **\`skyramp_analyze_changes\`** — Unified entry point: scans endpoints, discovers tests, computes diff. Takes \`repositoryPath\` and \`scope\`. Returns \`stateFile\` + recommendations.
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The Skyramp SDK version installed beside this server, which is the version the
|
|
3
|
+
* generated tests are written against.
|
|
4
|
+
*
|
|
5
|
+
* Read once: it cannot change while the process runs. The SDK is a direct
|
|
6
|
+
* dependency that this server imports at load time, so it is always installed
|
|
7
|
+
* here — a read failure is a broken install, not a state to degrade into.
|
|
8
|
+
*/
|
|
9
|
+
export declare const SKYRAMP_SDK_VERSION: string;
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
import { readFileSync } from "fs";
|
|
2
|
+
import { createRequire } from "module";
|
|
3
|
+
const require = createRequire(import.meta.url);
|
|
4
|
+
/**
|
|
5
|
+
* The Skyramp SDK version installed beside this server, which is the version the
|
|
6
|
+
* generated tests are written against.
|
|
7
|
+
*
|
|
8
|
+
* Read once: it cannot change while the process runs. The SDK is a direct
|
|
9
|
+
* dependency that this server imports at load time, so it is always installed
|
|
10
|
+
* here — a read failure is a broken install, not a state to degrade into.
|
|
11
|
+
*/
|
|
12
|
+
export const SKYRAMP_SDK_VERSION = readInstalledSdkVersion();
|
|
13
|
+
function readInstalledSdkVersion() {
|
|
14
|
+
const manifest = require.resolve("@skyramp/skyramp/package.json");
|
|
15
|
+
return JSON.parse(readFileSync(manifest, "utf8")).version;
|
|
16
|
+
}
|
|
@@ -6,6 +6,7 @@ import { promisify } from "util";
|
|
|
6
6
|
import fg from "fast-glob";
|
|
7
7
|
import { canonicalJson as canonical, isPlainObject as isObject, } from "./canonicalJson.js";
|
|
8
8
|
import { listChangedFiles } from "./reportVerification.js";
|
|
9
|
+
import { SKYRAMP_SDK_VERSION } from "./skyrampSdkVersion.js";
|
|
9
10
|
const execFileAsync = promisify(execFile);
|
|
10
11
|
export const NPM_TEST_PACKAGES = new Set([
|
|
11
12
|
"@playwright/test",
|
|
@@ -48,6 +49,7 @@ export const JAVA_TEST_PACKAGES = new Set([
|
|
|
48
49
|
"org.junit.platform:junit-platform-console-standalone",
|
|
49
50
|
"org.mockito:mockito-core",
|
|
50
51
|
"org.testng:testng",
|
|
52
|
+
"dev.skyramp:skyramp-library",
|
|
51
53
|
]);
|
|
52
54
|
const DEPENDENCY_FILES = {
|
|
53
55
|
javascript: {
|
|
@@ -57,6 +59,9 @@ const DEPENDENCY_FILES = {
|
|
|
57
59
|
"npm-shrinkwrap.json",
|
|
58
60
|
"yarn.lock",
|
|
59
61
|
"pnpm-lock.yaml",
|
|
62
|
+
// Named in the fix-errors guidance, so it must not bypass validation.
|
|
63
|
+
"bun.lock",
|
|
64
|
+
"bun.lockb",
|
|
60
65
|
],
|
|
61
66
|
},
|
|
62
67
|
python: {
|
|
@@ -277,10 +282,26 @@ function verifyPackageJson(file, beforeRaw, afterRaw, violations) {
|
|
|
277
282
|
else if (!isSafeNpmSpecifier(afterDev[name])) {
|
|
278
283
|
violations.push(`${file}: ${name} must use a registry version, not an alias, URL, git, or local package source.`);
|
|
279
284
|
}
|
|
285
|
+
else if (!declaresServerSdkVersion(name, afterDev[name])) {
|
|
286
|
+
violations.push(`${file}: ${name} is declared as ${afterDev[name]}, but the tests were generated by ${SKYRAMP_SDK_VERSION}. Declare exactly ${SKYRAMP_SDK_VERSION}, with no range operator: a range lets a clean install resolve a different SDK.`);
|
|
287
|
+
}
|
|
280
288
|
}
|
|
281
289
|
return (canonical(beforeRuntime) !== canonical(afterRuntime) ||
|
|
282
290
|
canonical(beforeDev) !== canonical(afterDev));
|
|
283
291
|
}
|
|
292
|
+
/** The SDK the tests were generated by is the one this server imports, so the
|
|
293
|
+
* repository must declare exactly that version. Only the npm spelling carries
|
|
294
|
+
* a version: the Python and Java manifests declare the SDK without one, so
|
|
295
|
+
* neither is checked here.
|
|
296
|
+
*
|
|
297
|
+
* The specifier must be the bare version. A range is refused, including
|
|
298
|
+
* `^1.2.3`: it permits 1.3.0, so a clean install elsewhere can resolve an SDK
|
|
299
|
+
* that did not write these tests, which is the whole thing this prevents. */
|
|
300
|
+
function declaresServerSdkVersion(name, specifier) {
|
|
301
|
+
if (name !== "@skyramp/skyramp")
|
|
302
|
+
return true;
|
|
303
|
+
return specifier.trim() === SKYRAMP_SDK_VERSION;
|
|
304
|
+
}
|
|
284
305
|
function verifyPyproject(file, beforeRaw, afterRaw, violations) {
|
|
285
306
|
const before = pythonDevView(beforeRaw);
|
|
286
307
|
const after = pythonDevView(afterRaw);
|
|
@@ -79,6 +79,7 @@ export declare function reserveTestExecutionAttempt(stateFile: string, repo: str
|
|
|
79
79
|
* per-file testExecutions record (every file, keyed by canonical path — see
|
|
80
80
|
* executionRecordFrom), plus the maintained-test executionBefore/After row
|
|
81
81
|
* when `testFile` has one. No-op when the repo section doesn't exist yet.
|
|
82
|
+
* Returns what was written, so the caller can say what was not.
|
|
82
83
|
*
|
|
83
84
|
* Extracted from skyramp_execute_test's own state-write block so the
|
|
84
85
|
* behavior that makes a file's result available to the post-execution
|
|
@@ -87,4 +88,7 @@ export declare function reserveTestExecutionAttempt(stateFile: string, repo: str
|
|
|
87
88
|
*/
|
|
88
89
|
export declare function persistTestExecutionResult(stateFile: string, repo: string, testType: TestType, phase: "before" | "after", result: TestExecutionResult, options?: {
|
|
89
90
|
reserved?: boolean;
|
|
90
|
-
}): Promise<
|
|
91
|
+
}): Promise<{
|
|
92
|
+
saved: boolean;
|
|
93
|
+
matchedExistingTest: boolean;
|
|
94
|
+
}>;
|