@skyramp/mcp 0.4.2-rc.1 → 0.4.2-rc.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (104) hide show
  1. package/build/commands/localDevTestChangesCommand.js +1 -1
  2. package/build/commands/recommendTestsAndExecuteCommand.js +10 -1
  3. package/build/commands/testThisEndpointCommand.js +19 -2
  4. package/build/execution/wrapperConfig.d.ts +56 -0
  5. package/build/execution/wrapperConfig.js +155 -0
  6. package/build/index.js +6 -6
  7. package/build/playwright/registerPlaywrightTools.js +14 -0
  8. package/build/playwright/traceExportStore.d.ts +22 -0
  9. package/build/playwright/traceExportStore.js +81 -0
  10. package/build/playwright/traceRecordingPrompt.js +2 -1
  11. package/build/prompts/code-reuse.js +24 -21
  12. package/build/prompts/local-dev/local-dev-plan.js +6 -23
  13. package/build/prompts/local-dev/local-dev-prompts.js +1 -1
  14. package/build/prompts/shared-helper-policy.d.ts +36 -0
  15. package/build/prompts/shared-helper-policy.js +33 -1
  16. package/build/prompts/startTraceCollectionPrompts.js +1 -1
  17. package/build/prompts/sut-setup/modes/adaptWorkflowPrompt.js +6 -7
  18. package/build/prompts/sut-setup/shared.d.ts +1 -1
  19. package/build/prompts/sut-setup/shared.js +5 -3
  20. package/build/prompts/test-maintenance/drift-analysis-prompt.d.ts +16 -8
  21. package/build/prompts/test-maintenance/drift-analysis-prompt.js +90 -36
  22. package/build/prompts/test-maintenance/uiDriftAnalysisSections.js +1 -1
  23. package/build/prompts/test-recommendation/recommendationShared.d.ts +1 -1
  24. package/build/prompts/test-recommendation/recommendationShared.js +0 -1
  25. package/build/prompts/testbot/testbot-prompts.js +11 -9
  26. package/build/services/TestDiscoveryService.js +32 -4
  27. package/build/skills/runTestSkill.d.ts +6 -0
  28. package/build/skills/runTestSkill.js +17 -0
  29. package/build/tool-phases.js +0 -1
  30. package/build/tools/budgetExcuse.d.ts +15 -0
  31. package/build/tools/budgetExcuse.js +113 -0
  32. package/build/tools/code-refactor/utils-verify-gates.js +17 -3
  33. package/build/tools/executeSkyrampTestTool.d.ts +97 -48
  34. package/build/tools/executeSkyrampTestTool.js +775 -449
  35. package/build/tools/submitReportTool.js +128 -0
  36. package/build/tools/test-management/actionsTool.js +31 -0
  37. package/build/tools/test-management/analyzeChangesTool.d.ts +4 -4
  38. package/build/tools/test-management/analyzeTestHealthTool.d.ts +0 -11
  39. package/build/tools/test-management/analyzeTestHealthTool.js +7 -63
  40. package/build/tools/test-management/testsOwedBeforeRun.d.ts +28 -0
  41. package/build/tools/test-management/testsOwedBeforeRun.js +53 -0
  42. package/build/tools/trace/stopTraceCollectionTool.js +1 -1
  43. package/build/types/RepositoryAnalysis.d.ts +32 -32
  44. package/build/types/ReuseOutcome.d.ts +4 -3
  45. package/build/types/TestExecution.d.ts +2 -2
  46. package/build/types/TestTypes.d.ts +3 -0
  47. package/build/types/TestTypes.js +6 -0
  48. package/build/utils/AnalysisStateManager.d.ts +0 -7
  49. package/build/utils/AnalysisStateManager.js +1 -1
  50. package/build/utils/connectionErrors.d.ts +10 -0
  51. package/build/utils/connectionErrors.js +10 -0
  52. package/build/utils/language-helper.js +24 -3
  53. package/build/utils/progress.d.ts +1 -1
  54. package/build/utils/progress.js +1 -1
  55. package/build/utils/rebaselineSnapshots.d.ts +1 -1
  56. package/build/utils/rebaselineSnapshots.js +6 -16
  57. package/build/utils/reuseRouting.d.ts +10 -0
  58. package/build/utils/reuseRouting.js +15 -0
  59. package/build/utils/runContextGauge.d.ts +27 -0
  60. package/build/utils/runContextGauge.js +181 -0
  61. package/build/utils/skyrampMdContent.d.ts +1 -1
  62. package/build/utils/skyrampMdContent.js +1 -1
  63. package/build/utils/skyrampSdkVersion.d.ts +9 -0
  64. package/build/utils/skyrampSdkVersion.js +16 -0
  65. package/build/utils/testDependencyPolicy.js +21 -0
  66. package/build/utils/testExecutionRecord.d.ts +5 -1
  67. package/build/utils/testExecutionRecord.js +3 -1
  68. package/build/utils/testFileClassification.d.ts +8 -0
  69. package/build/utils/testFileClassification.js +36 -3
  70. package/build/utils/utils-verify/action-key.d.ts +42 -0
  71. package/build/utils/utils-verify/action-key.js +118 -36
  72. package/build/utils/utils-verify/action-sites.d.ts +32 -0
  73. package/build/utils/utils-verify/action-sites.js +202 -0
  74. package/build/utils/utils-verify/body-reach.js +2 -4
  75. package/build/utils/utils-verify/call-sites.d.ts +25 -6
  76. package/build/utils/utils-verify/call-sites.js +8 -5
  77. package/build/utils/utils-verify/index.d.ts +1 -0
  78. package/build/utils/utils-verify/index.js +1 -0
  79. package/build/utils/utils-verify/language-spec.d.ts +25 -0
  80. package/build/utils/utils-verify/language-spec.js +16 -2
  81. package/build/utils/utils-verify/parse.d.ts +10 -1
  82. package/build/utils/utils-verify/parse.js +19 -2
  83. package/build/utils/utils-verify/verify.d.ts +3 -2
  84. package/build/utils/utils-verify/verify.js +16 -18
  85. package/build/workspace/workspace.d.ts +72 -52
  86. package/build/workspace/workspace.js +12 -8
  87. package/package.json +1 -1
  88. package/plugin/prompts/testbot-task1.md +0 -2
  89. package/plugin/skills/fix-test-import-errors/SKILL.md +2 -1
  90. package/plugin/skills/run-test/SKILL.md +16 -0
  91. package/build/adapters/jestAdapter.d.ts +0 -14
  92. package/build/adapters/jestAdapter.js +0 -131
  93. package/build/adapters/mochaAdapter.d.ts +0 -13
  94. package/build/adapters/mochaAdapter.js +0 -93
  95. package/build/adapters/playwrightAdapter.d.ts +0 -17
  96. package/build/adapters/playwrightAdapter.js +0 -184
  97. package/build/adapters/pytestAdapter.d.ts +0 -15
  98. package/build/adapters/pytestAdapter.js +0 -119
  99. package/build/tools/runExistingTestsTool.d.ts +0 -138
  100. package/build/tools/runExistingTestsTool.js +0 -666
  101. package/build/types/ExternalTestExecution.d.ts +0 -67
  102. package/build/types/ExternalTestExecution.js +0 -8
  103. package/build/workspace/testSuites.d.ts +0 -20
  104. package/build/workspace/testSuites.js +0 -17
@@ -154,9 +154,10 @@ export interface HelperReuseOutcome {
154
154
  * TODO: remove `helpersImported` once the testbot renderer reads this field. The
155
155
  * two carry one fact; the count stays only because testbot#349 reads it today. */
156
156
  helperNames: string[];
157
- /** Inline request calls in OTHER Skyramp-generated tests beside this one that a
158
- * helper in `utilsFile` already wraps — reuse that was available and not taken.
159
- * Omitted at zero. Informational, like {@link ReuseOutcome.missedReuse}. */
157
+ /** Inline request calls (integration) or inline action sequences (UI) in OTHER
158
+ * Skyramp-generated tests beside this one that a helper in `utilsFile` already
159
+ * wraps — reuse that was available and not taken. Omitted at zero. Informational,
160
+ * like {@link ReuseOutcome.missedReuse}. */
160
161
  siblingInlineCallSites?: number;
161
162
  /** Whether the delivered utils file holds the shared-helper invariants: one helper
162
163
  * per method+path, status-code-only assertions, method+resource names — with
@@ -41,8 +41,8 @@ export interface TestExecutionRecord {
41
41
  executedAt: string;
42
42
  duration: number;
43
43
  /** The process's real exit code, or -1 when none was produced — a timeout
44
- * kill or a thrown error before the process exited (an Error-status run;
45
- * see TestExecutionService's timeout and catch paths). Never a real exit
44
+ * kill or a spawn error before the process exited (an Error-status run;
45
+ * see runTest in skyramp_execute_test). Never a real exit
46
46
  * code, which is always 0-255. */
47
47
  exitCode: number;
48
48
  errors: string[];
@@ -7,6 +7,9 @@ export declare enum ProgrammingLanguage {
7
7
  JAVASCRIPT = "javascript",
8
8
  JAVA = "java"
9
9
  }
10
+ /** The languages skyramp_execute_test accepts, as prose for a prompt. Rendered from
11
+ * the enum so the text and the gate that reads it cannot disagree. */
12
+ export declare function runnableLanguagesText(): string;
10
13
  export declare enum TestType {
11
14
  SMOKE = "smoke",
12
15
  FUZZ = "fuzz",
@@ -9,6 +9,12 @@ export var ProgrammingLanguage;
9
9
  ProgrammingLanguage["JAVASCRIPT"] = "javascript";
10
10
  ProgrammingLanguage["JAVA"] = "java";
11
11
  })(ProgrammingLanguage || (ProgrammingLanguage = {}));
12
+ /** The languages skyramp_execute_test accepts, as prose for a prompt. Rendered from
13
+ * the enum so the text and the gate that reads it cannot disagree. */
14
+ export function runnableLanguagesText() {
15
+ const names = Object.values(ProgrammingLanguage);
16
+ return `${names.slice(0, -1).join(", ")} or ${names[names.length - 1]}`;
17
+ }
12
18
  export var TestType;
13
19
  (function (TestType) {
14
20
  TestType["SMOKE"] = "smoke";
@@ -5,7 +5,6 @@ import { TestAnalysisResult, MaintenanceActionCore } from "../types/TestAnalysis
5
5
  import { RepositoryAnalysis, AnalysisScope } from "../types/RepositoryAnalysis.js";
6
6
  import { PRTestContext } from "./pr-comment-parser.js";
7
7
  import type { RemovedUiElement } from "./removedUiElements.js";
8
- import type { ExternalTestRunRecord } from "../types/ExternalTestExecution.js";
9
8
  import type { TestExecutionRecord, VideoRecord } from "../types/TestExecution.js";
10
9
  import type { RepoCheckout } from "./reportVerification.js";
11
10
  import type { Plan } from "../recommendation/registerPlan.js";
@@ -209,12 +208,6 @@ export interface UnifiedAnalysisState {
209
208
  /** Maintenance actions applied via skyramp_actions — skyramp_submit_report derives
210
209
  * testMaintenance from this. */
211
210
  maintenanceVerdicts?: MaintenanceActionCore[];
212
- /**
213
- * External (repo-owned) test runs recorded by skyramp_run_existing_tests, one
214
- * record per confirm/verify invocation. Drift analysis and the report fold in
215
- * CONFIRMED failures from here instead of guessing. Absent until the tool runs.
216
- */
217
- externalTestResults?: ExternalTestRunRecord[];
218
211
  /** POM code-reuse outcome per generated UI spec, keyed by test-file BASENAME
219
212
  * (what `newTestsCreated[].fileName` carries). Written by skyramp_reuse_code,
220
213
  * which computes every value in-process; skyramp_submit_report merges it into
@@ -603,7 +603,7 @@ export class StateManager {
603
603
  updatedAt: new Date().toISOString(),
604
604
  // Preserve the prior step when the caller omits one (matches
605
605
  // repositoryPath/repository/createdAt above) so a metadata-agnostic
606
- // write — e.g. persistExternalRun appending externalTestResults —
606
+ // write — e.g. skyramp_execute_test recording an execution —
607
607
  // does not wipe the pipeline step recorded by an earlier phase.
608
608
  step: options?.step ?? existingMetadata?.step,
609
609
  },
@@ -0,0 +1,10 @@
1
+ /**
2
+ * The one place a connection-level failure is spelled. Three callers match these
3
+ * and each reads them for its own purpose — a transient MCP error, an executor
4
+ * retry, and the drift prompt's "the application was unreachable" verdict — so
5
+ * they are kept here rather than written out three times and drifting apart.
6
+ */
7
+ /** Refused, reset, or the socket dropped mid-request. */
8
+ export declare const CONNECTION_DROPPED: RegExp;
9
+ /** The application never answered in time. */
10
+ export declare const CONNECTION_TIMED_OUT: RegExp;
@@ -0,0 +1,10 @@
1
+ /**
2
+ * The one place a connection-level failure is spelled. Three callers match these
3
+ * and each reads them for its own purpose — a transient MCP error, an executor
4
+ * retry, and the drift prompt's "the application was unreachable" verdict — so
5
+ * they are kept here rather than written out three times and drifting apart.
6
+ */
7
+ /** Refused, reset, or the socket dropped mid-request. */
8
+ export const CONNECTION_DROPPED = /\bECONNREFUSED\b|\bECONNRESET\b|\bconnection refused\b|\bconnection reset\b|\bsocket hang up\b/i;
9
+ /** The application never answered in time. */
10
+ export const CONNECTION_TIMED_OUT = /\bETIMEDOUT\b|\bDEADLINE_EXCEEDED\b|\bconnect(?:ion)? timed? ?out\b/i;
@@ -1,9 +1,13 @@
1
+ import { SKYRAMP_SDK_VERSION } from "./skyrampSdkVersion.js";
2
+ /** The SDK spec the generated JavaScript and TypeScript tests need. */
3
+ const npmSdkSpec = `@skyramp/skyramp@${SKYRAMP_SDK_VERSION}`;
1
4
  const pythonDependencyGuidance = `
2
5
  Persist each missing approved package in the development manifest:
3
6
  - Use \`uv add --dev <package>\` for a project using \`[dependency-groups].dev\`.
4
7
  - Use \`poetry add --group dev <package>\` for a Poetry project.
5
8
  - Add it to \`requirements-dev.txt\`, then run \`python3 -m pip install -r requirements-dev.txt\`.
6
9
  If no Python manifest exists at the test project root, create requirements-dev.txt and seed it with the missing approved package. Do not create it alongside pyproject.toml.
10
+ If the repository commits poetry.lock or uv.lock, update it in the same change. Use \`poetry lock\` or \`uv lock\`; both keep the versions already locked.
7
11
  `;
8
12
  const javascriptDependencyGuidance = `
9
13
  Inspect package.json before installing anything.
@@ -11,19 +15,36 @@ Inspect package.json before installing anything.
11
15
  - If no JavaScript/TypeScript manifest exists at the test project root, create package.json with only \`{ "devDependencies": {} }\` before adding the missing approved package.
12
16
  - If an approved package is missing, add it as a devDependency with the repository's package manager. The npm equivalent is \`npm install --save-dev --legacy-peer-deps <package>\`.
13
17
  - Never add, upgrade, or downgrade Playwright or @playwright/test when either is already declared.
18
+ - If the repository commits a lockfile, update it in the same change. Use \`npm install --package-lock-only\`, \`yarn install --mode update-lockfile\`, \`pnpm install --lockfile-only\`, or \`bun install --lockfile-only\` for the package manager the repository uses.
19
+ - Do not change \`allowScripts\`, \`pnpm.onlyBuiltDependencies\`, or \`trustedDependencies\`.
14
20
  `;
15
21
  const javaDependencyGuidance = `
16
22
  Inspect existing Java build signals before adding a dependency. Use an existing manifest when present.
17
23
  - For Maven wrapper or settings signals, create a minimal pom.xml only when it is absent, and add approved dependencies with <scope>test</scope>.
18
24
  - For Gradle wrapper or settings signals, create the matching build.gradle or build.gradle.kts only when it is absent, and use testImplementation or testRuntimeOnly.
19
25
  - Do not introduce a competing build system. If neither build system is identified, keep the standalone runner instead of inventing one.
26
+ - If the repository commits gradle.lockfile, update it in the same change with the repository's Gradle wrapper (\`./gradlew dependencies --write-locks\`), or \`gradle\` only when there is no wrapper.
20
27
  `;
21
28
  function getJavascriptSteps(extension) {
22
- return `${javascriptDependencyGuidance}
29
+ return `${javascriptDependencyGuidance}${sdkDeclaration("javascript")}
23
30
  # Execute the test file
24
31
  npx playwright test <test-file>.spec.${extension} --reporter=list
25
32
  `;
26
33
  }
34
+ /** How to declare the Skyramp SDK, per language. Kept out of the shared
35
+ * dependency guidance because that guidance is also returned for `mock`, and
36
+ * a mock is deployed rather than run as a test — it needs no SDK. */
37
+ const SDK_DECLARATION = {
38
+ python: "Declare `skyramp` in the same development manifest.",
39
+ javascript: `If @skyramp/skyramp is not already declared, add ${npmSdkSpec} as a devDependency. If it is declared, leave its version alone and report the mismatch if the verifier refuses it.`,
40
+ java: "Declare `dev.skyramp:skyramp-library` with test scope. An existing manifest resolves the version through its own dependency management; a manifest you created has none, so give a concrete version there.",
41
+ };
42
+ /** The SDK line for a language, or nothing when the language has none. */
43
+ function sdkDeclaration(language) {
44
+ const key = language === "typescript" ? "javascript" : language;
45
+ const line = SDK_DECLARATION[key];
46
+ return line ? `\n${line}\n` : "";
47
+ }
27
48
  export function getLanguageSteps(params) {
28
49
  const { language, testType } = params;
29
50
  if (testType === "mock") {
@@ -50,7 +71,7 @@ node <mock-file>.js
50
71
  let steps = "";
51
72
  switch (language) {
52
73
  case "python":
53
- steps = pythonDependencyGuidance;
74
+ steps = pythonDependencyGuidance + sdkDeclaration("python");
54
75
  // e2e and ui tests
55
76
  if (testType === "e2e" || testType === "ui") {
56
77
  steps += `
@@ -80,7 +101,7 @@ python3 -m robot <robot-file>.robot
80
101
  steps = getJavascriptSteps("ts");
81
102
  break;
82
103
  case "java":
83
- steps = `${javaDependencyGuidance}
104
+ steps = `${javaDependencyGuidance}${sdkDeclaration("java")}
84
105
  # Prerequisites
85
106
  # Compile the test file
86
107
  javac -cp "./lib/*" -d target <test-file>.java
@@ -11,6 +11,6 @@ export type ProgressReporter = (progress: number, total: number, message: string
11
11
  * otherwise silently no-ops.
12
12
  *
13
13
  * Consolidates the identical closure that previously lived inline in multiple
14
- * progress-emitting tools (analyzeChanges, executeSkyrampTest, startTraceCollection).
14
+ * progress-emitting tools (analyzeChanges, startTraceCollection).
15
15
  */
16
16
  export declare function makeProgressReporter(extra: RequestHandlerExtra<ServerRequest, ServerNotification>): ProgressReporter;
@@ -4,7 +4,7 @@
4
4
  * otherwise silently no-ops.
5
5
  *
6
6
  * Consolidates the identical closure that previously lived inline in multiple
7
- * progress-emitting tools (analyzeChanges, executeSkyrampTest, startTraceCollection).
7
+ * progress-emitting tools (analyzeChanges, startTraceCollection).
8
8
  */
9
9
  export function makeProgressReporter(extra) {
10
10
  return async (progress, total, message) => {
@@ -6,7 +6,7 @@ import { z } from "zod";
6
6
  *
7
7
  * Bare filename only: no path separators, and no commas or whitespace, because the
8
8
  * list is joined with commas into `SKYRAMP_UPDATE_SNAPSHOTS` and split again in the
9
- * executor, where `a,b.png` would read as two baselines. One definition shared by
9
+ * test runtime, where `a,b.png` would read as two baselines. One definition shared by
10
10
  * both tools so they can never drift apart.
11
11
  */
12
12
  export declare const REBASELINE_SNAPSHOT_NAME_RE: RegExp;
@@ -7,7 +7,7 @@ import { z } from "zod";
7
7
  *
8
8
  * Bare filename only: no path separators, and no commas or whitespace, because the
9
9
  * list is joined with commas into `SKYRAMP_UPDATE_SNAPSHOTS` and split again in the
10
- * executor, where `a,b.png` would read as two baselines. One definition shared by
10
+ * test runtime, where `a,b.png` would read as two baselines. One definition shared by
11
11
  * both tools so they can never drift apart.
12
12
  */
13
13
  export const REBASELINE_SNAPSHOT_NAME_RE = /^[^/\\,\s]+\.png$/i;
@@ -22,21 +22,11 @@ export function baselineStem(name) {
22
22
  return path.basename(name).replace(/\.png$/i, "");
23
23
  }
24
24
  /**
25
- * Playwright's default snapshot layout, which is what the executor produces: the
26
- * baseline for `<stem>.png` lives in `<spec>-snapshots/` as `<stem>-<project>-<platform>.png`
27
- * (or `<stem>-<platform>.png` with no project, or `<stem>.png` under a custom
28
- * `snapshotPathTemplate`). Safe to assume here because TestExecutionService mounts
29
- * its own generated Playwright config over any the repo carries, so a repo cannot
30
- * relocate its snapshots out from under this rule.
31
- *
32
- * The executor's generated config declares no `projects`, but runner.sh passes
33
- * `--browser=chromium`, so Playwright names the implicit project "chromium" and a
34
- * refresh writes `<stem>-chromium-linux.png` (observed end to end on demoshop).
35
- * Baselines the Skyramp executor produced therefore round-trip exactly. A baseline
36
- * a repo's own CI wrote under another project name (`<stem>-Desktop-Chrome-linux.png`)
37
- * is a different file: the executor's refresh adds its own and leaves that one as
38
- * it was — the tool reports what it rewrote, and the repo's own suite is out of
39
- * scope for skyramp_execute_test (which never runs it).
25
+ * Playwright's default snapshot layout: the baseline for `<stem>.png` lives in
26
+ * `<spec>-snapshots/` as `<stem>-<project>-<platform>.png` (or `<stem>-<platform>.png`
27
+ * with no project, or `<stem>.png` under a custom `snapshotPathTemplate`). A repo
28
+ * config that sets another `snapshotPathTemplate` moves its baselines out from under
29
+ * this rule; the tool then reports them as not refreshed rather than claim a refresh.
40
30
  *
41
31
  * Anchored so a longer baseline cannot vouch for a shorter one: `page-001-wide-chromium-linux.png`
42
32
  * belongs to `page-001-wide.png` and must not satisfy a claim on `page-001.png`.
@@ -1,3 +1,13 @@
1
1
  export declare function isBrowserTestType(testType?: string): boolean;
2
2
  export declare function isModularizeFirstTarget(testType?: string, language?: string): boolean;
3
3
  export declare function isPomAwareTarget(language: string, framework?: string, testType?: string): boolean;
4
+ /**
5
+ * Which family's inline call-site rule and sibling scan a test gets — the ONE gate
6
+ * for both. `api` for integration, whatever the flags (the rule is stated in API
7
+ * terms and predates the flag); `browser` for the seeded UI flow, where a Skyramp
8
+ * browser module exists to call into; `undefined` everywhere else (`contract`, `load`,
9
+ * e2e, an absent type, UI with the flag off or page-object reuse on), where an
10
+ * advisory would describe reuse the feature could not have taken. The prompt renders
11
+ * the rule and STEP 5c on it; the verify pass runs the scan on it.
12
+ */
13
+ export declare function inlineCallSiteFamily(testType?: string, language?: string): "api" | "browser" | undefined;
@@ -52,3 +52,18 @@ export function isPomAwareTarget(language, framework, testType) {
52
52
  return ((lang === "typescript" || lang === "javascript") &&
53
53
  framework?.toLowerCase() === "playwright");
54
54
  }
55
+ /**
56
+ * Which family's inline call-site rule and sibling scan a test gets — the ONE gate
57
+ * for both. `api` for integration, whatever the flags (the rule is stated in API
58
+ * terms and predates the flag); `browser` for the seeded UI flow, where a Skyramp
59
+ * browser module exists to call into; `undefined` everywhere else (`contract`, `load`,
60
+ * e2e, an absent type, UI with the flag off or page-object reuse on), where an
61
+ * advisory would describe reuse the feature could not have taken. The prompt renders
62
+ * the rule and STEP 5c on it; the verify pass runs the scan on it.
63
+ */
64
+ export function inlineCallSiteFamily(testType, language) {
65
+ if (testType?.toLowerCase() === "integration")
66
+ return "api";
67
+ // A modularize-first target that is not integration is the seeded UI flow.
68
+ return isModularizeFirstTarget(testType, language) ? "browser" : undefined;
69
+ }
@@ -0,0 +1,27 @@
1
+ export interface ContextReading {
2
+ /** Tokens the last request actually carried: fresh input, cache reads and cache
3
+ * writes together. This is the number that fills the window. */
4
+ tokens: number;
5
+ window: number;
6
+ /** `tokens` as a whole percentage of `window`, for printing. */
7
+ percent: number;
8
+ }
9
+ /** This run's context usage, or undefined when it cannot be read.
10
+ *
11
+ * Undefined is ORDINARY, not an error: a local or IDE run has no `RUNNER_TEMP`
12
+ * and no agent log, and an agent whose CLI writes no NDJSON log never produces
13
+ * one. Every caller prints nothing and carries on. */
14
+ export declare function readRunContext(): ContextReading | undefined;
15
+ /** One sentence naming the measurement, for a refusal that has already told the
16
+ * agent not to stop on a context it believes it is short of. Empty when there is
17
+ * no reading — the refusal reads correctly without it.
18
+ *
19
+ * It states the reading and nothing else. It does not say the run has room: the
20
+ * reading can be 95% and the refusal still stands on its own grounds, and a
21
+ * sentence that argued headroom would be wrong exactly when it mattered most.
22
+ *
23
+ * CARRIES ITS OWN TRAILING SPACE, so a caller concatenates it unconditionally
24
+ * and an empty reading leaves no gap. */
25
+ export declare function contextGaugeSentence(): string;
26
+ /** Test seam: drop the per-run window cache. */
27
+ export declare function resetRunContextGaugeForTests(): void;
@@ -0,0 +1,181 @@
1
+ import * as fs from "fs";
2
+ import * as path from "path";
3
+ import { runArtifactDir } from "./AnalysisStateManager.js";
4
+ import { logger } from "./logger.js";
5
+ /** What the agent's own context usage is, read from the agent log the Testbot
6
+ * action streams beside this run's artifacts.
7
+ *
8
+ * WHY THIS EXISTS. Two runs cut planned work citing a "context budget" they
9
+ * could not measure: one declared itself out at 357k of a 1M window, the next at
10
+ * 457k, and neither had a warning from anywhere. The agent has no reading of its
11
+ * own context and the prompt cannot give it one, so a refusal that says "do not
12
+ * stop because you believe you are low on context" is arguing with a feeling.
13
+ * This turns it into a number the server can put in the refusal.
14
+ *
15
+ * It is a MEASUREMENT, never a limit. Nothing here refuses anything or decides
16
+ * anything; the callers do, on grounds that hold without it, and every one of
17
+ * them still works when this returns undefined. */
18
+ /** The Testbot action streams the agent CLI's `--output-format stream-json` here,
19
+ * live, in the same directory this server already writes `testbot-result.txt` to. */
20
+ const AGENT_LOG_NAME = "agent-log.ndjson";
21
+ /** Enough of the tail to hold the last assistant record with a usage block. These
22
+ * records carry whole tool results, so a fixed slice can miss one — 512 KiB covers
23
+ * it with room, and a miss costs a sentence, not a refusal. */
24
+ const TAIL_BYTES = 512 * 1024;
25
+ /** Enough of the head to hold the `init` record, which is the FIRST line. */
26
+ const HEAD_BYTES = 64 * 1024;
27
+ /** Windows for models with no `[1m]`-style suffix on the init record, first
28
+ * match wins. Opus 5 and Sonnet 5 are 1M by default — reading them as
29
+ * 200k is the same error the suffix rule below guards against, one level up.
30
+ * Bedrock prefixes the id (`us.anthropic.claude-opus-5`), so these match
31
+ * anywhere in the string rather than at the start. */
32
+ const MODEL_WINDOWS = [
33
+ [/claude-(opus|sonnet)-5/i, 1_000_000],
34
+ [/claude-haiku/i, 200_000],
35
+ ];
36
+ /** A model none of the above names. ERRS LARGE on purpose: too small a window
37
+ * overstates the percentage, which is how this file manufactures the scarcity
38
+ * it exists to disprove, and no caller acts on the number — the refusals stand
39
+ * on their own grounds. Too large only understates it. */
40
+ const DEFAULT_CONTEXT_WINDOW = 1_000_000;
41
+ function windowForModel(model) {
42
+ for (const [pattern, window] of MODEL_WINDOWS) {
43
+ if (pattern.test(model))
44
+ return window;
45
+ }
46
+ return DEFAULT_CONTEXT_WINDOW;
47
+ }
48
+ /** Window and log path do not change within a run, so they are read once and kept
49
+ * against the directory they came from — a new run is a new directory. The token
50
+ * count is NOT cached: it is the thing that moves. */
51
+ let cachedForDir;
52
+ let cachedWindow;
53
+ function readSlice(file, bytes, fromEnd) {
54
+ const fd = fs.openSync(file, "r");
55
+ try {
56
+ const size = fs.fstatSync(fd).size;
57
+ const length = Math.min(size, bytes);
58
+ const buffer = Buffer.alloc(length);
59
+ fs.readSync(fd, buffer, 0, length, fromEnd ? size - length : 0);
60
+ return buffer.toString("utf8");
61
+ }
62
+ finally {
63
+ fs.closeSync(fd);
64
+ }
65
+ }
66
+ /** The context window this run's model was started with.
67
+ *
68
+ * READ FROM THE INIT RECORD, not from the per-message model. The two disagree:
69
+ * `message.model` is `claude-opus-5` while the init record says
70
+ * `claude-opus-5[1m]`, and taking the former turns 46% of a 1M window into 228%
71
+ * of a 200k one — a gauge that manufactures the panic it exists to prevent. */
72
+ function contextWindowFrom(head) {
73
+ for (const line of head.split("\n")) {
74
+ const trimmed = line.trim();
75
+ if (!trimmed.startsWith("{"))
76
+ continue;
77
+ let record;
78
+ try {
79
+ record = JSON.parse(trimmed);
80
+ }
81
+ catch {
82
+ // The head slice ends mid-record on any file longer than it. Only the first
83
+ // line matters here and it is whole.
84
+ break;
85
+ }
86
+ if (record.type !== "system" || record.subtype !== "init")
87
+ break;
88
+ const model = typeof record.model === "string" ? record.model : "";
89
+ const suffix = /\[(\d+)(m|k)\]/i.exec(model);
90
+ if (!suffix)
91
+ return windowForModel(model);
92
+ const scale = suffix[2].toLowerCase() === "m" ? 1_000_000 : 1_000;
93
+ return Number(suffix[1]) * scale;
94
+ }
95
+ return DEFAULT_CONTEXT_WINDOW;
96
+ }
97
+ /** Tokens on the most recent assistant request in the log, or undefined.
98
+ *
99
+ * Walks BACKWARDS to the first record that has them: the tail ends with the turn
100
+ * that called the tool now running, and every later line is that turn's own
101
+ * output. Records with no usage — heartbeats, progress, tool results — are
102
+ * skipped, so the reading is always the newest real request. */
103
+ function tokensFrom(tail) {
104
+ const lines = tail.split("\n");
105
+ for (let index = lines.length - 1; index >= 0; index--) {
106
+ const line = lines[index].trim();
107
+ // Cheap rejects first: this runs over a few thousand lines per call.
108
+ if (!line.startsWith("{") || !line.includes('"usage"'))
109
+ continue;
110
+ let record;
111
+ try {
112
+ record = JSON.parse(line);
113
+ }
114
+ catch {
115
+ continue;
116
+ }
117
+ const usage = record.message?.usage;
118
+ if (!usage)
119
+ continue;
120
+ const count = (key) => typeof usage[key] === "number" ? usage[key] : 0;
121
+ const tokens = count("input_tokens") +
122
+ count("cache_read_input_tokens") +
123
+ count("cache_creation_input_tokens");
124
+ if (tokens > 0)
125
+ return tokens;
126
+ }
127
+ return undefined;
128
+ }
129
+ /** This run's context usage, or undefined when it cannot be read.
130
+ *
131
+ * Undefined is ORDINARY, not an error: a local or IDE run has no `RUNNER_TEMP`
132
+ * and no agent log, and an agent whose CLI writes no NDJSON log never produces
133
+ * one. Every caller prints nothing and carries on. */
134
+ export function readRunContext() {
135
+ try {
136
+ const directory = runArtifactDir();
137
+ if (!directory)
138
+ return undefined;
139
+ const file = path.join(directory, AGENT_LOG_NAME);
140
+ if (!fs.existsSync(file))
141
+ return undefined;
142
+ if (directory !== cachedForDir) {
143
+ cachedForDir = directory;
144
+ cachedWindow = contextWindowFrom(readSlice(file, HEAD_BYTES, false));
145
+ }
146
+ const window = cachedWindow ?? DEFAULT_CONTEXT_WINDOW;
147
+ const tokens = tokensFrom(readSlice(file, TAIL_BYTES, true));
148
+ if (tokens === undefined || tokens <= 0)
149
+ return undefined;
150
+ return { tokens, window, percent: Math.round((tokens / window) * 100) };
151
+ }
152
+ catch (error) {
153
+ const detail = error instanceof Error ? error.message : String(error);
154
+ logger.warning(`readRunContext failed: ${detail}`);
155
+ return undefined;
156
+ }
157
+ }
158
+ /** One sentence naming the measurement, for a refusal that has already told the
159
+ * agent not to stop on a context it believes it is short of. Empty when there is
160
+ * no reading — the refusal reads correctly without it.
161
+ *
162
+ * It states the reading and nothing else. It does not say the run has room: the
163
+ * reading can be 95% and the refusal still stands on its own grounds, and a
164
+ * sentence that argued headroom would be wrong exactly when it mattered most.
165
+ *
166
+ * CARRIES ITS OWN TRAILING SPACE, so a caller concatenates it unconditionally
167
+ * and an empty reading leaves no gap. */
168
+ export function contextGaugeSentence() {
169
+ const reading = readRunContext();
170
+ if (!reading)
171
+ return "";
172
+ const k = (value) => `${Math.round(value / 1000)}k`;
173
+ return (`Measured, not estimated: this run's last request carried ${k(reading.tokens)} of a ` +
174
+ `${k(reading.window)} token context window (${reading.percent}%). ` +
175
+ "Use that number rather than an impression of it. ");
176
+ }
177
+ /** Test seam: drop the per-run window cache. */
178
+ export function resetRunContextGaugeForTests() {
179
+ cachedForDir = undefined;
180
+ cachedWindow = undefined;
181
+ }
@@ -2,4 +2,4 @@
2
2
  * Skill content for skyramp.md — installed at auto-discovery paths: ~/.claude/skills/skyramp/SKILLS.md, ~/.cursor/skills/skyramp/SKILLS.md, ~/.github/skills/skyramp.md
3
3
  * Follows the SKILL.md specification: https://agentskills.io/what-are-skills#the-skill-md-file
4
4
  */
5
- export declare const SKYRAMP_MD_CONTENT = "---\nname: skyramp\ndescription: Generate, execute, and maintain API tests (smoke, contract, fuzz, load, integration, E2E, UI) using Skyramp MCP tools.\n---\n\n# Skyramp\n\n## When to use this skill\nUse this skill whenever the user asks to generate, run, or maintain API tests. Always read `<workspace-root>/.skyramp/workspace.yml` first to get `language`, `framework`, `outputDir`, and `api.baseUrl` \u2014 do not ask the user for values already present.\n\n---\n\n## Tools\n\n### Workspace Setup\n1. **`skyramp_initialize_workspace`** \u2014 Create or update `.skyramp/workspace.yml` in a git repository workspace. Scan the repo for all services before calling. Required before any other Skyramp tool.\n\n### Test Generation\n2. **`skyramp_smoke_test_generation`** \u2014 Verify an endpoint is reachable and returns a valid response.\n3. **`skyramp_contract_test_generation`** \u2014 Validate implementation matches OpenAPI/Swagger schema.\n4. **`skyramp_fuzz_test_generation`** \u2014 Send malformed or boundary inputs to find edge cases and security issues.\n5. **`skyramp_load_test_generation`** \u2014 Test performance under concurrent load. Optional: `loadDuration`, `loadNumThreads`. Accepts `trace` instead of `apiSchema`/`endpointURL`.\n6. **`skyramp_integration_test_generation`** \u2014 Multi-step workflows across one or more services. Supply one of: `apiSchema`+`endpointURL`, `trace`, or `scenarioFile`. Do not combine them.\n7. **`skyramp_e2e_test_generation`** \u2014 Full user journey covering UI and backend. Requires `trace` and `playwrightInput` zip. Do not pass `apiSchema` or `endpointURL`.\n8. **`skyramp_ui_test_generation`** \u2014 UI-only tests from Playwright recordings. Requires `playwrightInput` zip.\n\n### Trace Generation\n9. **`skyramp_start_trace_collection`** \u2014 Start capturing backend traffic. Set `playwright: true` for UI or E2E tests. Use an absolute path for `outputDir`.\n10. **`skyramp_stop_trace_collection`** \u2014 Stop capture and save the trace. Use the same `outputDir` and `playwrightEnabled` values as start.\n11. **`skyramp_scenario_test_generation`** \u2014 Scenario trace generation. Describe a multi-step flow in natural language to produce a scenario file. Pass output to `skyramp_integration_test_generation` via `scenarioFile`.\n\n### Test Execution\n12. **`skyramp_execute_test`** \u2014 Run a single Skyramp-generated test file. Required: `workspacePath`, `language`, `testType`, `testFile`. Optional: `stateFile` (writes execution results back for health analysis). For multiple tests, call sequentially to avoid env var conflicts.\n\n### Test Analysis & Maintenance\n13. **`skyramp_analyze_changes`** \u2014 Unified entry point: scans endpoints, discovers tests, computes diff. Takes `repositoryPath` and `scope`. Returns `stateFile` + recommendations.\n14. **`skyramp_analyze_test_health`** \u2014 Drift and health assessment for existing tests. Takes `stateFile`. Returns LLM prompt for scoring.\n15. **`skyramp_actions`** \u2014 Execute UPDATE / REGENERATE / VERIFY / DELETE actions. Takes `stateFile`. Call after analyze_test_health.\n\n### Code Quality\n16. **`skyramp_fix_errors`** \u2014 Fix compilation or runtime errors in a generated test file.\n17. **`skyramp_modularization`** \u2014 Refactor a test file into reusable modules. Set `isTraceBased: true` for trace-based tests.\n18. **`skyramp_reuse_code`** \u2014 Pull shared helpers from other Skyramp tests. Only when `code_reuse` was `true` at generation time.\n\n### Authentication\n19. **`skyramp_login`** \u2014 Log in to the Skyramp platform.\n20. **`skyramp_logout`** \u2014 Log out from the Skyramp platform.\n\n---\n\n## Prompts\n\nPrefer invoking a prompt over manually chaining tools \u2014 prompts run the full workflow automatically.\n\n- **`skyramp_trace_prompt`** \u2014 Trace collection setup and execution.\n- **`skyramp_test_health_analysis`** \u2014 Full maintenance flow (discover \u2192 drift \u2192 health \u2192 actions).\n- **`skyramp_testbot`** \u2014 PR-scoped recommendations + maintenance + report. Required: `prTitle`, `prDescription`, `repositoryPath`.\n\n---\n\n## Workflows\n\nUse these when a prompt is not available.\n\n**Generate and run a test**\nRead `.skyramp/workspace.yml` \u2192 Call appropriate generation tool \u2192 `skyramp_execute_test`\n\n**Trace-based test (integration / load / E2E / UI)**\n`skyramp_start_trace_collection` (set `playwright: true` for UI/E2E) \u2192 User exercises the app \u2192 `skyramp_stop_trace_collection` \u2192 Call target generation tool with `trace` (and `playwrightInput` for E2E/UI)\n\n**Scenario \u2192 integration test**\n`skyramp_scenario_test_generation` \u2192 `skyramp_integration_test_generation` with `scenarioFile`\n\n**Recommend tests for a PR**\n`skyramp_analyze_changes` (with `scope: \"branch_diff\"`) \u2192 follow enrichment steps \u2192 `skyramp_recommend_tests` \u2192 Generate recommended tests\n\n**Maintain existing tests**\n`skyramp_analyze_changes` \u2192 `skyramp_analyze_test_health` \u2192 *(optional)* execute tests via `skyramp_execute_test` with `stateFile` param (writes results back) \u2192 `skyramp_actions` (do not skip)\n\n---\n\n## Conventions\n\n- **`endpointURL`** \u2014 Full URL to a specific endpoint, not just the base URL. Build from `api.baseUrl` + path (e.g. `http://localhost:8000/api/v1/users`).\n- **`outputDir`** \u2014 Use absolute paths for both test output and trace collection.\n- **Mutually exclusive inputs** \u2014 `apiSchema`/`endpointURL`, `trace`, and `scenarioFile` are mutually exclusive for integration and load tests. Use exactly one.\n- **Sequential execution** \u2014 Execute tests sequentially (not in parallel) to avoid environment variable conflicts with `SKYRAMP_TEST_BASE_URL`.\n";
5
+ export declare const SKYRAMP_MD_CONTENT = "---\nname: skyramp\ndescription: Generate, execute, and maintain API tests (smoke, contract, fuzz, load, integration, E2E, UI) using Skyramp MCP tools.\n---\n\n# Skyramp\n\n## When to use this skill\nUse this skill whenever the user asks to generate, run, or maintain API tests. Always read `<workspace-root>/.skyramp/workspace.yml` first to get `language`, `framework`, `outputDir`, and `api.baseUrl` \u2014 do not ask the user for values already present.\n\n---\n\n## Tools\n\n### Workspace Setup\n1. **`skyramp_initialize_workspace`** \u2014 Create or update `.skyramp/workspace.yml` in a git repository workspace. Scan the repo for all services before calling. Required before any other Skyramp tool.\n\n### Test Generation\n2. **`skyramp_smoke_test_generation`** \u2014 Verify an endpoint is reachable and returns a valid response.\n3. **`skyramp_contract_test_generation`** \u2014 Validate implementation matches OpenAPI/Swagger schema.\n4. **`skyramp_fuzz_test_generation`** \u2014 Send malformed or boundary inputs to find edge cases and security issues.\n5. **`skyramp_load_test_generation`** \u2014 Test performance under concurrent load. Optional: `loadDuration`, `loadNumThreads`. Accepts `trace` instead of `apiSchema`/`endpointURL`.\n6. **`skyramp_integration_test_generation`** \u2014 Multi-step workflows across one or more services. Supply one of: `apiSchema`+`endpointURL`, `trace`, or `scenarioFile`. Do not combine them.\n7. **`skyramp_e2e_test_generation`** \u2014 Full user journey covering UI and backend. Requires `trace` and `playwrightInput` zip. Do not pass `apiSchema` or `endpointURL`.\n8. **`skyramp_ui_test_generation`** \u2014 UI-only tests from Playwright recordings. Requires `playwrightInput` zip.\n\n### Trace Generation\n9. **`skyramp_start_trace_collection`** \u2014 Start capturing backend traffic. Set `playwright: true` for UI or E2E tests. Use an absolute path for `outputDir`.\n10. **`skyramp_stop_trace_collection`** \u2014 Stop capture and save the trace. Use the same `outputDir` and `playwrightEnabled` values as start.\n11. **`skyramp_scenario_test_generation`** \u2014 Scenario trace generation. Describe a multi-step flow in natural language to produce a scenario file. Pass output to `skyramp_integration_test_generation` via `scenarioFile`.\n\n### Test Execution\n12. **`skyramp_execute_test`** \u2014 Run one test file and record the verdict from the exit code. Required: `testFile`, `language`, `testType`, `repository`. To run on this host, pass `commandOverride` (the command for that one file) with `cwd`. Omit `commandOverride` to run in the Skyramp executor instead, which needs `workspacePath`. Optional: `stateFile` (writes execution results back for health analysis). Output is capped at the last 200,000 characters.\n\n### Test Analysis & Maintenance\n13. **`skyramp_analyze_changes`** \u2014 Unified entry point: scans endpoints, discovers tests, computes diff. Takes `repositoryPath` and `scope`. Returns `stateFile` + recommendations.\n14. **`skyramp_analyze_test_health`** \u2014 Drift and health assessment for existing tests. Takes `stateFile`. Returns LLM prompt for scoring.\n15. **`skyramp_actions`** \u2014 Execute UPDATE / REGENERATE / VERIFY / DELETE actions. Takes `stateFile`. Call after analyze_test_health.\n\n### Code Quality\n16. **`skyramp_fix_errors`** \u2014 Fix compilation or runtime errors in a generated test file.\n17. **`skyramp_modularization`** \u2014 Refactor a test file into reusable modules. Set `isTraceBased: true` for trace-based tests.\n18. **`skyramp_reuse_code`** \u2014 Pull shared helpers from other Skyramp tests. Only when `code_reuse` was `true` at generation time.\n\n### Authentication\n19. **`skyramp_login`** \u2014 Log in to the Skyramp platform.\n20. **`skyramp_logout`** \u2014 Log out from the Skyramp platform.\n\n---\n\n## Prompts\n\nPrefer invoking a prompt over manually chaining tools \u2014 prompts run the full workflow automatically.\n\n- **`skyramp_trace_prompt`** \u2014 Trace collection setup and execution.\n- **`skyramp_test_health_analysis`** \u2014 Full maintenance flow (discover \u2192 drift \u2192 health \u2192 actions).\n- **`skyramp_testbot`** \u2014 PR-scoped recommendations + maintenance + report. Required: `prTitle`, `prDescription`, `repositoryPath`.\n\n---\n\n## Workflows\n\nUse these when a prompt is not available.\n\n**Generate and run a test**\nRead `.skyramp/workspace.yml` \u2192 Call appropriate generation tool \u2192 `skyramp_execute_test`\n\n**Trace-based test (integration / load / E2E / UI)**\n`skyramp_start_trace_collection` (set `playwright: true` for UI/E2E) \u2192 User exercises the app \u2192 `skyramp_stop_trace_collection` \u2192 Call target generation tool with `trace` (and `playwrightInput` for E2E/UI)\n\n**Scenario \u2192 integration test**\n`skyramp_scenario_test_generation` \u2192 `skyramp_integration_test_generation` with `scenarioFile`\n\n**Recommend tests for a PR**\n`skyramp_analyze_changes` (with `scope: \"branch_diff\"`) \u2192 follow enrichment steps \u2192 `skyramp_recommend_tests` \u2192 Generate recommended tests\n\n**Maintain existing tests**\n`skyramp_analyze_changes` \u2192 `skyramp_analyze_test_health` \u2192 *(optional)* execute tests via `skyramp_execute_test` with `stateFile` param (writes results back) \u2192 `skyramp_actions` (do not skip)\n\n---\n\n## Conventions\n\n- **`endpointURL`** \u2014 Full URL to a specific endpoint, not just the base URL. Build from `api.baseUrl` + path (e.g. `http://localhost:8000/api/v1/users`).\n- **`outputDir`** \u2014 Use absolute paths for both test output and trace collection.\n- **Mutually exclusive inputs** \u2014 `apiSchema`/`endpointURL`, `trace`, and `scenarioFile` are mutually exclusive for integration and load tests. Use exactly one.\n- **Sequential execution** \u2014 Execute tests sequentially (not in parallel) to avoid environment variable conflicts with `SKYRAMP_TEST_BASE_URL`.\n";
@@ -34,7 +34,7 @@ Use this skill whenever the user asks to generate, run, or maintain API tests. A
34
34
  11. **\`skyramp_scenario_test_generation\`** — Scenario trace generation. Describe a multi-step flow in natural language to produce a scenario file. Pass output to \`skyramp_integration_test_generation\` via \`scenarioFile\`.
35
35
 
36
36
  ### Test Execution
37
- 12. **\`skyramp_execute_test\`** — Run a single Skyramp-generated test file. Required: \`workspacePath\`, \`language\`, \`testType\`, \`testFile\`. Optional: \`stateFile\` (writes execution results back for health analysis). For multiple tests, call sequentially to avoid env var conflicts.
37
+ 12. **\`skyramp_execute_test\`** — Run one test file and record the verdict from the exit code. Required: \`testFile\`, \`language\`, \`testType\`, \`repository\`. To run on this host, pass \`commandOverride\` (the command for that one file) with \`cwd\`. Omit \`commandOverride\` to run in the Skyramp executor instead, which needs \`workspacePath\`. Optional: \`stateFile\` (writes execution results back for health analysis). Output is capped at the last 200,000 characters.
38
38
 
39
39
  ### Test Analysis & Maintenance
40
40
  13. **\`skyramp_analyze_changes\`** — Unified entry point: scans endpoints, discovers tests, computes diff. Takes \`repositoryPath\` and \`scope\`. Returns \`stateFile\` + recommendations.
@@ -0,0 +1,9 @@
1
+ /**
2
+ * The Skyramp SDK version installed beside this server, which is the version the
3
+ * generated tests are written against.
4
+ *
5
+ * Read once: it cannot change while the process runs. The SDK is a direct
6
+ * dependency that this server imports at load time, so it is always installed
7
+ * here — a read failure is a broken install, not a state to degrade into.
8
+ */
9
+ export declare const SKYRAMP_SDK_VERSION: string;
@@ -0,0 +1,16 @@
1
+ import { readFileSync } from "fs";
2
+ import { createRequire } from "module";
3
+ const require = createRequire(import.meta.url);
4
+ /**
5
+ * The Skyramp SDK version installed beside this server, which is the version the
6
+ * generated tests are written against.
7
+ *
8
+ * Read once: it cannot change while the process runs. The SDK is a direct
9
+ * dependency that this server imports at load time, so it is always installed
10
+ * here — a read failure is a broken install, not a state to degrade into.
11
+ */
12
+ export const SKYRAMP_SDK_VERSION = readInstalledSdkVersion();
13
+ function readInstalledSdkVersion() {
14
+ const manifest = require.resolve("@skyramp/skyramp/package.json");
15
+ return JSON.parse(readFileSync(manifest, "utf8")).version;
16
+ }
@@ -6,6 +6,7 @@ import { promisify } from "util";
6
6
  import fg from "fast-glob";
7
7
  import { canonicalJson as canonical, isPlainObject as isObject, } from "./canonicalJson.js";
8
8
  import { listChangedFiles } from "./reportVerification.js";
9
+ import { SKYRAMP_SDK_VERSION } from "./skyrampSdkVersion.js";
9
10
  const execFileAsync = promisify(execFile);
10
11
  export const NPM_TEST_PACKAGES = new Set([
11
12
  "@playwright/test",
@@ -48,6 +49,7 @@ export const JAVA_TEST_PACKAGES = new Set([
48
49
  "org.junit.platform:junit-platform-console-standalone",
49
50
  "org.mockito:mockito-core",
50
51
  "org.testng:testng",
52
+ "dev.skyramp:skyramp-library",
51
53
  ]);
52
54
  const DEPENDENCY_FILES = {
53
55
  javascript: {
@@ -57,6 +59,9 @@ const DEPENDENCY_FILES = {
57
59
  "npm-shrinkwrap.json",
58
60
  "yarn.lock",
59
61
  "pnpm-lock.yaml",
62
+ // Named in the fix-errors guidance, so it must not bypass validation.
63
+ "bun.lock",
64
+ "bun.lockb",
60
65
  ],
61
66
  },
62
67
  python: {
@@ -277,10 +282,26 @@ function verifyPackageJson(file, beforeRaw, afterRaw, violations) {
277
282
  else if (!isSafeNpmSpecifier(afterDev[name])) {
278
283
  violations.push(`${file}: ${name} must use a registry version, not an alias, URL, git, or local package source.`);
279
284
  }
285
+ else if (!declaresServerSdkVersion(name, afterDev[name])) {
286
+ violations.push(`${file}: ${name} is declared as ${afterDev[name]}, but the tests were generated by ${SKYRAMP_SDK_VERSION}. Declare exactly ${SKYRAMP_SDK_VERSION}, with no range operator: a range lets a clean install resolve a different SDK.`);
287
+ }
280
288
  }
281
289
  return (canonical(beforeRuntime) !== canonical(afterRuntime) ||
282
290
  canonical(beforeDev) !== canonical(afterDev));
283
291
  }
292
+ /** The SDK the tests were generated by is the one this server imports, so the
293
+ * repository must declare exactly that version. Only the npm spelling carries
294
+ * a version: the Python and Java manifests declare the SDK without one, so
295
+ * neither is checked here.
296
+ *
297
+ * The specifier must be the bare version. A range is refused, including
298
+ * `^1.2.3`: it permits 1.3.0, so a clean install elsewhere can resolve an SDK
299
+ * that did not write these tests, which is the whole thing this prevents. */
300
+ function declaresServerSdkVersion(name, specifier) {
301
+ if (name !== "@skyramp/skyramp")
302
+ return true;
303
+ return specifier.trim() === SKYRAMP_SDK_VERSION;
304
+ }
284
305
  function verifyPyproject(file, beforeRaw, afterRaw, violations) {
285
306
  const before = pythonDevView(beforeRaw);
286
307
  const after = pythonDevView(afterRaw);
@@ -79,6 +79,7 @@ export declare function reserveTestExecutionAttempt(stateFile: string, repo: str
79
79
  * per-file testExecutions record (every file, keyed by canonical path — see
80
80
  * executionRecordFrom), plus the maintained-test executionBefore/After row
81
81
  * when `testFile` has one. No-op when the repo section doesn't exist yet.
82
+ * Returns what was written, so the caller can say what was not.
82
83
  *
83
84
  * Extracted from skyramp_execute_test's own state-write block so the
84
85
  * behavior that makes a file's result available to the post-execution
@@ -87,4 +88,7 @@ export declare function reserveTestExecutionAttempt(stateFile: string, repo: str
87
88
  */
88
89
  export declare function persistTestExecutionResult(stateFile: string, repo: string, testType: TestType, phase: "before" | "after", result: TestExecutionResult, options?: {
89
90
  reserved?: boolean;
90
- }): Promise<void>;
91
+ }): Promise<{
92
+ saved: boolean;
93
+ matchedExistingTest: boolean;
94
+ }>;