@skyramp/mcp 0.4.2-rc.1 → 0.4.2-rc.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (110) hide show
  1. package/build/commands/localDevTestChangesCommand.js +1 -1
  2. package/build/commands/recommendTestsAndExecuteCommand.js +10 -1
  3. package/build/commands/testThisEndpointCommand.js +19 -2
  4. package/build/execution/wrapperConfig.d.ts +56 -0
  5. package/build/execution/wrapperConfig.js +155 -0
  6. package/build/index.js +6 -6
  7. package/build/playwright/registerPlaywrightTools.js +14 -0
  8. package/build/playwright/traceExportStore.d.ts +22 -0
  9. package/build/playwright/traceExportStore.js +81 -0
  10. package/build/playwright/traceRecordingPrompt.js +2 -1
  11. package/build/prompts/code-reuse.js +24 -21
  12. package/build/prompts/local-dev/local-dev-plan.js +6 -23
  13. package/build/prompts/local-dev/local-dev-prompts.js +1 -1
  14. package/build/prompts/shared-helper-policy.d.ts +36 -0
  15. package/build/prompts/shared-helper-policy.js +33 -1
  16. package/build/prompts/startTraceCollectionPrompts.js +1 -1
  17. package/build/prompts/sut-setup/modes/adaptWorkflowPrompt.js +6 -7
  18. package/build/prompts/sut-setup/shared.d.ts +1 -1
  19. package/build/prompts/sut-setup/shared.js +5 -3
  20. package/build/prompts/test-maintenance/drift-analysis-prompt.d.ts +16 -8
  21. package/build/prompts/test-maintenance/drift-analysis-prompt.js +90 -36
  22. package/build/prompts/test-maintenance/uiDriftAnalysisSections.js +1 -1
  23. package/build/prompts/test-recommendation/recommendationShared.d.ts +1 -1
  24. package/build/prompts/test-recommendation/recommendationShared.js +0 -1
  25. package/build/prompts/testbot/testbot-prompts.js +11 -9
  26. package/build/services/TestDiscoveryService.js +32 -4
  27. package/build/skills/runTestSkill.d.ts +6 -0
  28. package/build/skills/runTestSkill.js +17 -0
  29. package/build/tool-phases.js +0 -1
  30. package/build/tools/budgetExcuse.d.ts +15 -0
  31. package/build/tools/budgetExcuse.js +113 -0
  32. package/build/tools/code-refactor/utils-verify-gates.js +17 -3
  33. package/build/tools/executeSkyrampTestTool.d.ts +97 -48
  34. package/build/tools/executeSkyrampTestTool.js +775 -449
  35. package/build/tools/generate-tests/generateE2ERestTool.d.ts +0 -1
  36. package/build/tools/generate-tests/generateE2ERestTool.js +1 -9
  37. package/build/tools/generate-tests/generateUIRestTool.d.ts +0 -2
  38. package/build/tools/generate-tests/generateUIRestTool.js +1 -9
  39. package/build/tools/submitReportTool.js +128 -0
  40. package/build/tools/test-management/actionsTool.js +31 -0
  41. package/build/tools/test-management/analyzeChangesTool.d.ts +4 -4
  42. package/build/tools/test-management/analyzeTestHealthTool.d.ts +0 -11
  43. package/build/tools/test-management/analyzeTestHealthTool.js +7 -63
  44. package/build/tools/test-management/testsOwedBeforeRun.d.ts +28 -0
  45. package/build/tools/test-management/testsOwedBeforeRun.js +53 -0
  46. package/build/tools/trace/stopTraceCollectionTool.js +1 -1
  47. package/build/types/RepositoryAnalysis.d.ts +32 -32
  48. package/build/types/ReuseOutcome.d.ts +4 -3
  49. package/build/types/TestExecution.d.ts +2 -2
  50. package/build/types/TestTypes.d.ts +3 -7
  51. package/build/types/TestTypes.js +6 -20
  52. package/build/utils/AnalysisStateManager.d.ts +0 -7
  53. package/build/utils/AnalysisStateManager.js +1 -1
  54. package/build/utils/connectionErrors.d.ts +10 -0
  55. package/build/utils/connectionErrors.js +10 -0
  56. package/build/utils/language-helper.js +24 -3
  57. package/build/utils/progress.d.ts +1 -1
  58. package/build/utils/progress.js +1 -1
  59. package/build/utils/rebaselineSnapshots.d.ts +1 -1
  60. package/build/utils/rebaselineSnapshots.js +6 -16
  61. package/build/utils/reuseRouting.d.ts +10 -0
  62. package/build/utils/reuseRouting.js +15 -0
  63. package/build/utils/runContextGauge.d.ts +27 -0
  64. package/build/utils/runContextGauge.js +181 -0
  65. package/build/utils/skyrampMdContent.d.ts +1 -1
  66. package/build/utils/skyrampMdContent.js +1 -1
  67. package/build/utils/skyrampSdkVersion.d.ts +9 -0
  68. package/build/utils/skyrampSdkVersion.js +16 -0
  69. package/build/utils/testDependencyPolicy.js +21 -0
  70. package/build/utils/testExecutionRecord.d.ts +5 -1
  71. package/build/utils/testExecutionRecord.js +3 -1
  72. package/build/utils/testFileClassification.d.ts +8 -0
  73. package/build/utils/testFileClassification.js +36 -3
  74. package/build/utils/utils-verify/action-key.d.ts +42 -0
  75. package/build/utils/utils-verify/action-key.js +118 -36
  76. package/build/utils/utils-verify/action-sites.d.ts +32 -0
  77. package/build/utils/utils-verify/action-sites.js +202 -0
  78. package/build/utils/utils-verify/body-reach.js +2 -4
  79. package/build/utils/utils-verify/call-sites.d.ts +25 -6
  80. package/build/utils/utils-verify/call-sites.js +8 -5
  81. package/build/utils/utils-verify/index.d.ts +1 -0
  82. package/build/utils/utils-verify/index.js +1 -0
  83. package/build/utils/utils-verify/language-spec.d.ts +25 -0
  84. package/build/utils/utils-verify/language-spec.js +16 -2
  85. package/build/utils/utils-verify/parse.d.ts +10 -1
  86. package/build/utils/utils-verify/parse.js +19 -2
  87. package/build/utils/utils-verify/verify.d.ts +3 -2
  88. package/build/utils/utils-verify/verify.js +16 -18
  89. package/build/workspace/workspace.d.ts +72 -52
  90. package/build/workspace/workspace.js +12 -8
  91. package/package.json +1 -1
  92. package/plugin/prompts/testbot-task1.md +0 -2
  93. package/plugin/skills/enhance-assertions/reference/shared-rules.md +1 -1
  94. package/plugin/skills/enhance-assertions/reference/ui.md +1 -1
  95. package/plugin/skills/fix-test-import-errors/SKILL.md +2 -1
  96. package/plugin/skills/run-test/SKILL.md +16 -0
  97. package/build/adapters/jestAdapter.d.ts +0 -14
  98. package/build/adapters/jestAdapter.js +0 -131
  99. package/build/adapters/mochaAdapter.d.ts +0 -13
  100. package/build/adapters/mochaAdapter.js +0 -93
  101. package/build/adapters/playwrightAdapter.d.ts +0 -17
  102. package/build/adapters/playwrightAdapter.js +0 -184
  103. package/build/adapters/pytestAdapter.d.ts +0 -15
  104. package/build/adapters/pytestAdapter.js +0 -119
  105. package/build/tools/runExistingTestsTool.d.ts +0 -138
  106. package/build/tools/runExistingTestsTool.js +0 -666
  107. package/build/types/ExternalTestExecution.d.ts +0 -67
  108. package/build/types/ExternalTestExecution.js +0 -8
  109. package/build/workspace/testSuites.d.ts +0 -20
  110. package/build/workspace/testSuites.js +0 -17
@@ -0,0 +1,181 @@
1
+ import * as fs from "fs";
2
+ import * as path from "path";
3
+ import { runArtifactDir } from "./AnalysisStateManager.js";
4
+ import { logger } from "./logger.js";
5
+ /** What the agent's own context usage is, read from the agent log the Testbot
6
+ * action streams beside this run's artifacts.
7
+ *
8
+ * WHY THIS EXISTS. Two runs cut planned work citing a "context budget" they
9
+ * could not measure: one declared itself out at 357k of a 1M window, the next at
10
+ * 457k, and neither had a warning from anywhere. The agent has no reading of its
11
+ * own context and the prompt cannot give it one, so a refusal that says "do not
12
+ * stop because you believe you are low on context" is arguing with a feeling.
13
+ * This turns it into a number the server can put in the refusal.
14
+ *
15
+ * It is a MEASUREMENT, never a limit. Nothing here refuses anything or decides
16
+ * anything; the callers do, on grounds that hold without it, and every one of
17
+ * them still works when this returns undefined. */
18
+ /** The Testbot action streams the agent CLI's `--output-format stream-json` here,
19
+ * live, in the same directory this server already writes `testbot-result.txt` to. */
20
+ const AGENT_LOG_NAME = "agent-log.ndjson";
21
+ /** Enough of the tail to hold the last assistant record with a usage block. These
22
+ * records carry whole tool results, so a fixed slice can miss one — 512 KiB covers
23
+ * it with room, and a miss costs a sentence, not a refusal. */
24
+ const TAIL_BYTES = 512 * 1024;
25
+ /** Enough of the head to hold the `init` record, which is the FIRST line. */
26
+ const HEAD_BYTES = 64 * 1024;
27
+ /** Windows for models with no `[1m]`-style suffix on the init record, first
28
+ * match wins. Opus 5 and Sonnet 5 are 1M by default — reading them as
29
+ * 200k is the same error the suffix rule below guards against, one level up.
30
+ * Bedrock prefixes the id (`us.anthropic.claude-opus-5`), so these match
31
+ * anywhere in the string rather than at the start. */
32
+ const MODEL_WINDOWS = [
33
+ [/claude-(opus|sonnet)-5/i, 1_000_000],
34
+ [/claude-haiku/i, 200_000],
35
+ ];
36
+ /** A model none of the above names. ERRS LARGE on purpose: too small a window
37
+ * overstates the percentage, which is how this file manufactures the scarcity
38
+ * it exists to disprove, and no caller acts on the number — the refusals stand
39
+ * on their own grounds. Too large only understates it. */
40
+ const DEFAULT_CONTEXT_WINDOW = 1_000_000;
41
+ function windowForModel(model) {
42
+ for (const [pattern, window] of MODEL_WINDOWS) {
43
+ if (pattern.test(model))
44
+ return window;
45
+ }
46
+ return DEFAULT_CONTEXT_WINDOW;
47
+ }
48
+ /** Window and log path do not change within a run, so they are read once and kept
49
+ * against the directory they came from — a new run is a new directory. The token
50
+ * count is NOT cached: it is the thing that moves. */
51
+ let cachedForDir;
52
+ let cachedWindow;
53
+ function readSlice(file, bytes, fromEnd) {
54
+ const fd = fs.openSync(file, "r");
55
+ try {
56
+ const size = fs.fstatSync(fd).size;
57
+ const length = Math.min(size, bytes);
58
+ const buffer = Buffer.alloc(length);
59
+ fs.readSync(fd, buffer, 0, length, fromEnd ? size - length : 0);
60
+ return buffer.toString("utf8");
61
+ }
62
+ finally {
63
+ fs.closeSync(fd);
64
+ }
65
+ }
66
+ /** The context window this run's model was started with.
67
+ *
68
+ * READ FROM THE INIT RECORD, not from the per-message model. The two disagree:
69
+ * `message.model` is `claude-opus-5` while the init record says
70
+ * `claude-opus-5[1m]`, and taking the former turns 46% of a 1M window into 228%
71
+ * of a 200k one — a gauge that manufactures the panic it exists to prevent. */
72
+ function contextWindowFrom(head) {
73
+ for (const line of head.split("\n")) {
74
+ const trimmed = line.trim();
75
+ if (!trimmed.startsWith("{"))
76
+ continue;
77
+ let record;
78
+ try {
79
+ record = JSON.parse(trimmed);
80
+ }
81
+ catch {
82
+ // The head slice ends mid-record on any file longer than it. Only the first
83
+ // line matters here and it is whole.
84
+ break;
85
+ }
86
+ if (record.type !== "system" || record.subtype !== "init")
87
+ break;
88
+ const model = typeof record.model === "string" ? record.model : "";
89
+ const suffix = /\[(\d+)(m|k)\]/i.exec(model);
90
+ if (!suffix)
91
+ return windowForModel(model);
92
+ const scale = suffix[2].toLowerCase() === "m" ? 1_000_000 : 1_000;
93
+ return Number(suffix[1]) * scale;
94
+ }
95
+ return DEFAULT_CONTEXT_WINDOW;
96
+ }
97
+ /** Tokens on the most recent assistant request in the log, or undefined.
98
+ *
99
+ * Walks BACKWARDS to the first record that has them: the tail ends with the turn
100
+ * that called the tool now running, and every later line is that turn's own
101
+ * output. Records with no usage — heartbeats, progress, tool results — are
102
+ * skipped, so the reading is always the newest real request. */
103
+ function tokensFrom(tail) {
104
+ const lines = tail.split("\n");
105
+ for (let index = lines.length - 1; index >= 0; index--) {
106
+ const line = lines[index].trim();
107
+ // Cheap rejects first: this runs over a few thousand lines per call.
108
+ if (!line.startsWith("{") || !line.includes('"usage"'))
109
+ continue;
110
+ let record;
111
+ try {
112
+ record = JSON.parse(line);
113
+ }
114
+ catch {
115
+ continue;
116
+ }
117
+ const usage = record.message?.usage;
118
+ if (!usage)
119
+ continue;
120
+ const count = (key) => typeof usage[key] === "number" ? usage[key] : 0;
121
+ const tokens = count("input_tokens") +
122
+ count("cache_read_input_tokens") +
123
+ count("cache_creation_input_tokens");
124
+ if (tokens > 0)
125
+ return tokens;
126
+ }
127
+ return undefined;
128
+ }
129
+ /** This run's context usage, or undefined when it cannot be read.
130
+ *
131
+ * Undefined is ORDINARY, not an error: a local or IDE run has no `RUNNER_TEMP`
132
+ * and no agent log, and an agent whose CLI writes no NDJSON log never produces
133
+ * one. Every caller prints nothing and carries on. */
134
+ export function readRunContext() {
135
+ try {
136
+ const directory = runArtifactDir();
137
+ if (!directory)
138
+ return undefined;
139
+ const file = path.join(directory, AGENT_LOG_NAME);
140
+ if (!fs.existsSync(file))
141
+ return undefined;
142
+ if (directory !== cachedForDir) {
143
+ cachedForDir = directory;
144
+ cachedWindow = contextWindowFrom(readSlice(file, HEAD_BYTES, false));
145
+ }
146
+ const window = cachedWindow ?? DEFAULT_CONTEXT_WINDOW;
147
+ const tokens = tokensFrom(readSlice(file, TAIL_BYTES, true));
148
+ if (tokens === undefined || tokens <= 0)
149
+ return undefined;
150
+ return { tokens, window, percent: Math.round((tokens / window) * 100) };
151
+ }
152
+ catch (error) {
153
+ const detail = error instanceof Error ? error.message : String(error);
154
+ logger.warning(`readRunContext failed: ${detail}`);
155
+ return undefined;
156
+ }
157
+ }
158
+ /** One sentence naming the measurement, for a refusal that has already told the
159
+ * agent not to stop on a context it believes it is short of. Empty when there is
160
+ * no reading — the refusal reads correctly without it.
161
+ *
162
+ * It states the reading and nothing else. It does not say the run has room: the
163
+ * reading can be 95% and the refusal still stands on its own grounds, and a
164
+ * sentence that argued headroom would be wrong exactly when it mattered most.
165
+ *
166
+ * CARRIES ITS OWN TRAILING SPACE, so a caller concatenates it unconditionally
167
+ * and an empty reading leaves no gap. */
168
+ export function contextGaugeSentence() {
169
+ const reading = readRunContext();
170
+ if (!reading)
171
+ return "";
172
+ const k = (value) => `${Math.round(value / 1000)}k`;
173
+ return (`Measured, not estimated: this run's last request carried ${k(reading.tokens)} of a ` +
174
+ `${k(reading.window)} token context window (${reading.percent}%). ` +
175
+ "Use that number rather than an impression of it. ");
176
+ }
177
+ /** Test seam: drop the per-run window cache. */
178
+ export function resetRunContextGaugeForTests() {
179
+ cachedForDir = undefined;
180
+ cachedWindow = undefined;
181
+ }
@@ -2,4 +2,4 @@
2
2
  * Skill content for skyramp.md — installed at auto-discovery paths: ~/.claude/skills/skyramp/SKILLS.md, ~/.cursor/skills/skyramp/SKILLS.md, ~/.github/skills/skyramp.md
3
3
  * Follows the SKILL.md specification: https://agentskills.io/what-are-skills#the-skill-md-file
4
4
  */
5
- export declare const SKYRAMP_MD_CONTENT = "---\nname: skyramp\ndescription: Generate, execute, and maintain API tests (smoke, contract, fuzz, load, integration, E2E, UI) using Skyramp MCP tools.\n---\n\n# Skyramp\n\n## When to use this skill\nUse this skill whenever the user asks to generate, run, or maintain API tests. Always read `<workspace-root>/.skyramp/workspace.yml` first to get `language`, `framework`, `outputDir`, and `api.baseUrl` \u2014 do not ask the user for values already present.\n\n---\n\n## Tools\n\n### Workspace Setup\n1. **`skyramp_initialize_workspace`** \u2014 Create or update `.skyramp/workspace.yml` in a git repository workspace. Scan the repo for all services before calling. Required before any other Skyramp tool.\n\n### Test Generation\n2. **`skyramp_smoke_test_generation`** \u2014 Verify an endpoint is reachable and returns a valid response.\n3. **`skyramp_contract_test_generation`** \u2014 Validate implementation matches OpenAPI/Swagger schema.\n4. **`skyramp_fuzz_test_generation`** \u2014 Send malformed or boundary inputs to find edge cases and security issues.\n5. **`skyramp_load_test_generation`** \u2014 Test performance under concurrent load. Optional: `loadDuration`, `loadNumThreads`. Accepts `trace` instead of `apiSchema`/`endpointURL`.\n6. **`skyramp_integration_test_generation`** \u2014 Multi-step workflows across one or more services. Supply one of: `apiSchema`+`endpointURL`, `trace`, or `scenarioFile`. Do not combine them.\n7. **`skyramp_e2e_test_generation`** \u2014 Full user journey covering UI and backend. Requires `trace` and `playwrightInput` zip. Do not pass `apiSchema` or `endpointURL`.\n8. **`skyramp_ui_test_generation`** \u2014 UI-only tests from Playwright recordings. Requires `playwrightInput` zip.\n\n### Trace Generation\n9. **`skyramp_start_trace_collection`** \u2014 Start capturing backend traffic. Set `playwright: true` for UI or E2E tests. Use an absolute path for `outputDir`.\n10. **`skyramp_stop_trace_collection`** \u2014 Stop capture and save the trace. Use the same `outputDir` and `playwrightEnabled` values as start.\n11. **`skyramp_scenario_test_generation`** \u2014 Scenario trace generation. Describe a multi-step flow in natural language to produce a scenario file. Pass output to `skyramp_integration_test_generation` via `scenarioFile`.\n\n### Test Execution\n12. **`skyramp_execute_test`** \u2014 Run a single Skyramp-generated test file. Required: `workspacePath`, `language`, `testType`, `testFile`. Optional: `stateFile` (writes execution results back for health analysis). For multiple tests, call sequentially to avoid env var conflicts.\n\n### Test Analysis & Maintenance\n13. **`skyramp_analyze_changes`** \u2014 Unified entry point: scans endpoints, discovers tests, computes diff. Takes `repositoryPath` and `scope`. Returns `stateFile` + recommendations.\n14. **`skyramp_analyze_test_health`** \u2014 Drift and health assessment for existing tests. Takes `stateFile`. Returns LLM prompt for scoring.\n15. **`skyramp_actions`** \u2014 Execute UPDATE / REGENERATE / VERIFY / DELETE actions. Takes `stateFile`. Call after analyze_test_health.\n\n### Code Quality\n16. **`skyramp_fix_errors`** \u2014 Fix compilation or runtime errors in a generated test file.\n17. **`skyramp_modularization`** \u2014 Refactor a test file into reusable modules. Set `isTraceBased: true` for trace-based tests.\n18. **`skyramp_reuse_code`** \u2014 Pull shared helpers from other Skyramp tests. Only when `code_reuse` was `true` at generation time.\n\n### Authentication\n19. **`skyramp_login`** \u2014 Log in to the Skyramp platform.\n20. **`skyramp_logout`** \u2014 Log out from the Skyramp platform.\n\n---\n\n## Prompts\n\nPrefer invoking a prompt over manually chaining tools \u2014 prompts run the full workflow automatically.\n\n- **`skyramp_trace_prompt`** \u2014 Trace collection setup and execution.\n- **`skyramp_test_health_analysis`** \u2014 Full maintenance flow (discover \u2192 drift \u2192 health \u2192 actions).\n- **`skyramp_testbot`** \u2014 PR-scoped recommendations + maintenance + report. Required: `prTitle`, `prDescription`, `repositoryPath`.\n\n---\n\n## Workflows\n\nUse these when a prompt is not available.\n\n**Generate and run a test**\nRead `.skyramp/workspace.yml` \u2192 Call appropriate generation tool \u2192 `skyramp_execute_test`\n\n**Trace-based test (integration / load / E2E / UI)**\n`skyramp_start_trace_collection` (set `playwright: true` for UI/E2E) \u2192 User exercises the app \u2192 `skyramp_stop_trace_collection` \u2192 Call target generation tool with `trace` (and `playwrightInput` for E2E/UI)\n\n**Scenario \u2192 integration test**\n`skyramp_scenario_test_generation` \u2192 `skyramp_integration_test_generation` with `scenarioFile`\n\n**Recommend tests for a PR**\n`skyramp_analyze_changes` (with `scope: \"branch_diff\"`) \u2192 follow enrichment steps \u2192 `skyramp_recommend_tests` \u2192 Generate recommended tests\n\n**Maintain existing tests**\n`skyramp_analyze_changes` \u2192 `skyramp_analyze_test_health` \u2192 *(optional)* execute tests via `skyramp_execute_test` with `stateFile` param (writes results back) \u2192 `skyramp_actions` (do not skip)\n\n---\n\n## Conventions\n\n- **`endpointURL`** \u2014 Full URL to a specific endpoint, not just the base URL. Build from `api.baseUrl` + path (e.g. `http://localhost:8000/api/v1/users`).\n- **`outputDir`** \u2014 Use absolute paths for both test output and trace collection.\n- **Mutually exclusive inputs** \u2014 `apiSchema`/`endpointURL`, `trace`, and `scenarioFile` are mutually exclusive for integration and load tests. Use exactly one.\n- **Sequential execution** \u2014 Execute tests sequentially (not in parallel) to avoid environment variable conflicts with `SKYRAMP_TEST_BASE_URL`.\n";
5
+ export declare const SKYRAMP_MD_CONTENT = "---\nname: skyramp\ndescription: Generate, execute, and maintain API tests (smoke, contract, fuzz, load, integration, E2E, UI) using Skyramp MCP tools.\n---\n\n# Skyramp\n\n## When to use this skill\nUse this skill whenever the user asks to generate, run, or maintain API tests. Always read `<workspace-root>/.skyramp/workspace.yml` first to get `language`, `framework`, `outputDir`, and `api.baseUrl` \u2014 do not ask the user for values already present.\n\n---\n\n## Tools\n\n### Workspace Setup\n1. **`skyramp_initialize_workspace`** \u2014 Create or update `.skyramp/workspace.yml` in a git repository workspace. Scan the repo for all services before calling. Required before any other Skyramp tool.\n\n### Test Generation\n2. **`skyramp_smoke_test_generation`** \u2014 Verify an endpoint is reachable and returns a valid response.\n3. **`skyramp_contract_test_generation`** \u2014 Validate implementation matches OpenAPI/Swagger schema.\n4. **`skyramp_fuzz_test_generation`** \u2014 Send malformed or boundary inputs to find edge cases and security issues.\n5. **`skyramp_load_test_generation`** \u2014 Test performance under concurrent load. Optional: `loadDuration`, `loadNumThreads`. Accepts `trace` instead of `apiSchema`/`endpointURL`.\n6. **`skyramp_integration_test_generation`** \u2014 Multi-step workflows across one or more services. Supply one of: `apiSchema`+`endpointURL`, `trace`, or `scenarioFile`. Do not combine them.\n7. **`skyramp_e2e_test_generation`** \u2014 Full user journey covering UI and backend. Requires `trace` and `playwrightInput` zip. Do not pass `apiSchema` or `endpointURL`.\n8. **`skyramp_ui_test_generation`** \u2014 UI-only tests from Playwright recordings. Requires `playwrightInput` zip.\n\n### Trace Generation\n9. **`skyramp_start_trace_collection`** \u2014 Start capturing backend traffic. Set `playwright: true` for UI or E2E tests. Use an absolute path for `outputDir`.\n10. **`skyramp_stop_trace_collection`** \u2014 Stop capture and save the trace. Use the same `outputDir` and `playwrightEnabled` values as start.\n11. **`skyramp_scenario_test_generation`** \u2014 Scenario trace generation. Describe a multi-step flow in natural language to produce a scenario file. Pass output to `skyramp_integration_test_generation` via `scenarioFile`.\n\n### Test Execution\n12. **`skyramp_execute_test`** \u2014 Run one test file and record the verdict from the exit code. Required: `testFile`, `language`, `testType`, `repository`. To run on this host, pass `commandOverride` (the command for that one file) with `cwd`. Omit `commandOverride` to run in the Skyramp executor instead, which needs `workspacePath`. Optional: `stateFile` (writes execution results back for health analysis). Output is capped at the last 200,000 characters.\n\n### Test Analysis & Maintenance\n13. **`skyramp_analyze_changes`** \u2014 Unified entry point: scans endpoints, discovers tests, computes diff. Takes `repositoryPath` and `scope`. Returns `stateFile` + recommendations.\n14. **`skyramp_analyze_test_health`** \u2014 Drift and health assessment for existing tests. Takes `stateFile`. Returns LLM prompt for scoring.\n15. **`skyramp_actions`** \u2014 Execute UPDATE / REGENERATE / VERIFY / DELETE actions. Takes `stateFile`. Call after analyze_test_health.\n\n### Code Quality\n16. **`skyramp_fix_errors`** \u2014 Fix compilation or runtime errors in a generated test file.\n17. **`skyramp_modularization`** \u2014 Refactor a test file into reusable modules. Set `isTraceBased: true` for trace-based tests.\n18. **`skyramp_reuse_code`** \u2014 Pull shared helpers from other Skyramp tests. Only when `code_reuse` was `true` at generation time.\n\n### Authentication\n19. **`skyramp_login`** \u2014 Log in to the Skyramp platform.\n20. **`skyramp_logout`** \u2014 Log out from the Skyramp platform.\n\n---\n\n## Prompts\n\nPrefer invoking a prompt over manually chaining tools \u2014 prompts run the full workflow automatically.\n\n- **`skyramp_trace_prompt`** \u2014 Trace collection setup and execution.\n- **`skyramp_test_health_analysis`** \u2014 Full maintenance flow (discover \u2192 drift \u2192 health \u2192 actions).\n- **`skyramp_testbot`** \u2014 PR-scoped recommendations + maintenance + report. Required: `prTitle`, `prDescription`, `repositoryPath`.\n\n---\n\n## Workflows\n\nUse these when a prompt is not available.\n\n**Generate and run a test**\nRead `.skyramp/workspace.yml` \u2192 Call appropriate generation tool \u2192 `skyramp_execute_test`\n\n**Trace-based test (integration / load / E2E / UI)**\n`skyramp_start_trace_collection` (set `playwright: true` for UI/E2E) \u2192 User exercises the app \u2192 `skyramp_stop_trace_collection` \u2192 Call target generation tool with `trace` (and `playwrightInput` for E2E/UI)\n\n**Scenario \u2192 integration test**\n`skyramp_scenario_test_generation` \u2192 `skyramp_integration_test_generation` with `scenarioFile`\n\n**Recommend tests for a PR**\n`skyramp_analyze_changes` (with `scope: \"branch_diff\"`) \u2192 follow enrichment steps \u2192 `skyramp_recommend_tests` \u2192 Generate recommended tests\n\n**Maintain existing tests**\n`skyramp_analyze_changes` \u2192 `skyramp_analyze_test_health` \u2192 *(optional)* execute tests via `skyramp_execute_test` with `stateFile` param (writes results back) \u2192 `skyramp_actions` (do not skip)\n\n---\n\n## Conventions\n\n- **`endpointURL`** \u2014 Full URL to a specific endpoint, not just the base URL. Build from `api.baseUrl` + path (e.g. `http://localhost:8000/api/v1/users`).\n- **`outputDir`** \u2014 Use absolute paths for both test output and trace collection.\n- **Mutually exclusive inputs** \u2014 `apiSchema`/`endpointURL`, `trace`, and `scenarioFile` are mutually exclusive for integration and load tests. Use exactly one.\n- **Sequential execution** \u2014 Execute tests sequentially (not in parallel) to avoid environment variable conflicts with `SKYRAMP_TEST_BASE_URL`.\n";
@@ -34,7 +34,7 @@ Use this skill whenever the user asks to generate, run, or maintain API tests. A
34
34
  11. **\`skyramp_scenario_test_generation\`** — Scenario trace generation. Describe a multi-step flow in natural language to produce a scenario file. Pass output to \`skyramp_integration_test_generation\` via \`scenarioFile\`.
35
35
 
36
36
  ### Test Execution
37
- 12. **\`skyramp_execute_test\`** — Run a single Skyramp-generated test file. Required: \`workspacePath\`, \`language\`, \`testType\`, \`testFile\`. Optional: \`stateFile\` (writes execution results back for health analysis). For multiple tests, call sequentially to avoid env var conflicts.
37
+ 12. **\`skyramp_execute_test\`** — Run one test file and record the verdict from the exit code. Required: \`testFile\`, \`language\`, \`testType\`, \`repository\`. To run on this host, pass \`commandOverride\` (the command for that one file) with \`cwd\`. Omit \`commandOverride\` to run in the Skyramp executor instead, which needs \`workspacePath\`. Optional: \`stateFile\` (writes execution results back for health analysis). Output is capped at the last 200,000 characters.
38
38
 
39
39
  ### Test Analysis & Maintenance
40
40
  13. **\`skyramp_analyze_changes\`** — Unified entry point: scans endpoints, discovers tests, computes diff. Takes \`repositoryPath\` and \`scope\`. Returns \`stateFile\` + recommendations.
@@ -0,0 +1,9 @@
1
+ /**
2
+ * The Skyramp SDK version installed beside this server, which is the version the
3
+ * generated tests are written against.
4
+ *
5
+ * Read once: it cannot change while the process runs. The SDK is a direct
6
+ * dependency that this server imports at load time, so it is always installed
7
+ * here — a read failure is a broken install, not a state to degrade into.
8
+ */
9
+ export declare const SKYRAMP_SDK_VERSION: string;
@@ -0,0 +1,16 @@
1
+ import { readFileSync } from "fs";
2
+ import { createRequire } from "module";
3
+ const require = createRequire(import.meta.url);
4
+ /**
5
+ * The Skyramp SDK version installed beside this server, which is the version the
6
+ * generated tests are written against.
7
+ *
8
+ * Read once: it cannot change while the process runs. The SDK is a direct
9
+ * dependency that this server imports at load time, so it is always installed
10
+ * here — a read failure is a broken install, not a state to degrade into.
11
+ */
12
+ export const SKYRAMP_SDK_VERSION = readInstalledSdkVersion();
13
+ function readInstalledSdkVersion() {
14
+ const manifest = require.resolve("@skyramp/skyramp/package.json");
15
+ return JSON.parse(readFileSync(manifest, "utf8")).version;
16
+ }
@@ -6,6 +6,7 @@ import { promisify } from "util";
6
6
  import fg from "fast-glob";
7
7
  import { canonicalJson as canonical, isPlainObject as isObject, } from "./canonicalJson.js";
8
8
  import { listChangedFiles } from "./reportVerification.js";
9
+ import { SKYRAMP_SDK_VERSION } from "./skyrampSdkVersion.js";
9
10
  const execFileAsync = promisify(execFile);
10
11
  export const NPM_TEST_PACKAGES = new Set([
11
12
  "@playwright/test",
@@ -48,6 +49,7 @@ export const JAVA_TEST_PACKAGES = new Set([
48
49
  "org.junit.platform:junit-platform-console-standalone",
49
50
  "org.mockito:mockito-core",
50
51
  "org.testng:testng",
52
+ "dev.skyramp:skyramp-library",
51
53
  ]);
52
54
  const DEPENDENCY_FILES = {
53
55
  javascript: {
@@ -57,6 +59,9 @@ const DEPENDENCY_FILES = {
57
59
  "npm-shrinkwrap.json",
58
60
  "yarn.lock",
59
61
  "pnpm-lock.yaml",
62
+ // Named in the fix-errors guidance, so it must not bypass validation.
63
+ "bun.lock",
64
+ "bun.lockb",
60
65
  ],
61
66
  },
62
67
  python: {
@@ -277,10 +282,26 @@ function verifyPackageJson(file, beforeRaw, afterRaw, violations) {
277
282
  else if (!isSafeNpmSpecifier(afterDev[name])) {
278
283
  violations.push(`${file}: ${name} must use a registry version, not an alias, URL, git, or local package source.`);
279
284
  }
285
+ else if (!declaresServerSdkVersion(name, afterDev[name])) {
286
+ violations.push(`${file}: ${name} is declared as ${afterDev[name]}, but the tests were generated by ${SKYRAMP_SDK_VERSION}. Declare exactly ${SKYRAMP_SDK_VERSION}, with no range operator: a range lets a clean install resolve a different SDK.`);
287
+ }
280
288
  }
281
289
  return (canonical(beforeRuntime) !== canonical(afterRuntime) ||
282
290
  canonical(beforeDev) !== canonical(afterDev));
283
291
  }
292
+ /** The SDK the tests were generated by is the one this server imports, so the
293
+ * repository must declare exactly that version. Only the npm spelling carries
294
+ * a version: the Python and Java manifests declare the SDK without one, so
295
+ * neither is checked here.
296
+ *
297
+ * The specifier must be the bare version. A range is refused, including
298
+ * `^1.2.3`: it permits 1.3.0, so a clean install elsewhere can resolve an SDK
299
+ * that did not write these tests, which is the whole thing this prevents. */
300
+ function declaresServerSdkVersion(name, specifier) {
301
+ if (name !== "@skyramp/skyramp")
302
+ return true;
303
+ return specifier.trim() === SKYRAMP_SDK_VERSION;
304
+ }
284
305
  function verifyPyproject(file, beforeRaw, afterRaw, violations) {
285
306
  const before = pythonDevView(beforeRaw);
286
307
  const after = pythonDevView(afterRaw);
@@ -79,6 +79,7 @@ export declare function reserveTestExecutionAttempt(stateFile: string, repo: str
79
79
  * per-file testExecutions record (every file, keyed by canonical path — see
80
80
  * executionRecordFrom), plus the maintained-test executionBefore/After row
81
81
  * when `testFile` has one. No-op when the repo section doesn't exist yet.
82
+ * Returns what was written, so the caller can say what was not.
82
83
  *
83
84
  * Extracted from skyramp_execute_test's own state-write block so the
84
85
  * behavior that makes a file's result available to the post-execution
@@ -87,4 +88,7 @@ export declare function reserveTestExecutionAttempt(stateFile: string, repo: str
87
88
  */
88
89
  export declare function persistTestExecutionResult(stateFile: string, repo: string, testType: TestType, phase: "before" | "after", result: TestExecutionResult, options?: {
89
90
  reserved?: boolean;
90
- }): Promise<void>;
91
+ }): Promise<{
92
+ saved: boolean;
93
+ matchedExistingTest: boolean;
94
+ }>;
@@ -231,6 +231,7 @@ fileHash) {
231
231
  * per-file testExecutions record (every file, keyed by canonical path — see
232
232
  * executionRecordFrom), plus the maintained-test executionBefore/After row
233
233
  * when `testFile` has one. No-op when the repo section doesn't exist yet.
234
+ * Returns what was written, so the caller can say what was not.
234
235
  *
235
236
  * Extracted from skyramp_execute_test's own state-write block so the
236
237
  * behavior that makes a file's result available to the post-execution
@@ -245,7 +246,7 @@ export async function persistTestExecutionResult(stateFile, repo, testType, phas
245
246
  const stateManager = StateManager.fromStatePath(stateFile);
246
247
  const stateData = await stateManager.readRepoData(repo);
247
248
  if (!stateData)
248
- return;
249
+ return { saved: false, matchedExistingTest: false };
249
250
  const testIndex = stateData.existingTests?.findIndex((t) => canonicalTestPath(t.testFile) === canonicalTestPath(testFile)) ?? -1;
250
251
  if (testIndex >= 0) {
251
252
  if (phase === "before") {
@@ -263,5 +264,6 @@ export async function persistTestExecutionResult(stateFile, repo, testType, phas
263
264
  !options.reserved),
264
265
  };
265
266
  await stateManager.updateRepoData(stateData, { repo });
267
+ return { saved: true, matchedExistingTest: testIndex >= 0 };
266
268
  });
267
269
  }
@@ -14,6 +14,14 @@ export declare function isDiscoveredTestFile(filePath: string): boolean;
14
14
  * `isDiscoveredTestFile`'s rule, which also counts `tests/`, a page-object
15
15
  * directory and `test_orders.py`. The maintenance tool wants this narrow one. */
16
16
  export declare function isTestFile(filePath: string): boolean;
17
+ /** The file's own NAME says it is a test — `login.spec.ts`, `orders.test.js`,
18
+ * `test_orders.py`, `UserDAOTest.java`. Almost entirely a NAME rule: a helper,
19
+ * a fixture, a page object or a config sitting in `e2e/` or `cypress/support/`
20
+ * is not a test and cannot be run as one. The one exception is `__tests__/`,
21
+ * Jest's convention that everything inside it is a test whatever it is called.
22
+ * Exported for callers that must not act on a file merely because of where it
23
+ * lives. */
24
+ export declare function hasTestFileName(filePath: string): boolean;
17
25
  /** Whether a path is a test file, by directory or by name. Read by the route
18
26
  * checks: a route declared in a mock server or a spec fixture is not a surface
19
27
  * anyone serves. */
@@ -76,10 +76,43 @@ const TEST_ROOT_DIR = /(^|[\\/])(tests?|__tests__|__mocks__|e2e|cypress|testing)
76
76
  * `TEST_FILE_PATTERNS` rows it resembles — any extension, and `.`, `_` or `-`
77
77
  * before `test` — so no row above can stand in for it. */
78
78
  const TEST_FILE_NAME = /[._-](test|spec)\.[a-z]+$|(^|[\\/])test_[^\\/]*$/i;
79
+ /** The file's own NAME says it is a test — `login.spec.ts`, `orders.test.js`,
80
+ * `test_orders.py`, `UserDAOTest.java`. Almost entirely a NAME rule: a helper,
81
+ * a fixture, a page object or a config sitting in `e2e/` or `cypress/support/`
82
+ * is not a test and cannot be run as one. The one exception is `__tests__/`,
83
+ * Jest's convention that everything inside it is a test whatever it is called.
84
+ * Exported for callers that must not act on a file merely because of where it
85
+ * lives. */
86
+ export function hasTestFileName(filePath) {
87
+ return (
88
+ // Discovery's own name rule, applied to the basename so it admits no
89
+ // directory. It carries the framework spellings the regexes below miss:
90
+ // `UserDAOTest.java`, `orders_smoke.py`, `checkout_e2e.ts`.
91
+ TEST_FILE_PATTERNS.some((p) => p.test(path.basename(filePath))) ||
92
+ TEST_FILE_NAME.test(filePath) ||
93
+ CAMEL_TEST_FILE_NAME.test(filePath) ||
94
+ // Surefire's default includes are `Test*.java` as well as `*Test.java`, and
95
+ // Failsafe's are `*IT.java`. CAMEL_TEST_FILE_NAME needs a lowercase
96
+ // character before the suffix, so the acronym spelling (`APIIT.java`) and
97
+ // the prefix spelling (`TestOrder.java`) both fall through it.
98
+ /(^|[\\/])Test[A-Z][A-Za-z0-9]*\.(java|kt|kts|scala)$/.test(filePath) ||
99
+ /[A-Za-z0-9]IT\.(java|kt|kts|scala)$/.test(filePath) ||
100
+ // mocha's default spec glob is `./test/*.js` — one directory level, and no
101
+ // name rule at all, so `test/app.js` is a test to mocha.
102
+ /(^|[\\/])test[\\/][^\\/]+\.(js|cjs|mjs)$/.test(filePath) ||
103
+ // `login.e2e.ts` and Cypress's own `login.cy.ts` — the dot forms
104
+ // TEST_FILE_NAME does not carry. `.cy.` is to Cypress what `.spec.` is to
105
+ // Playwright.
106
+ /\.(cy|e2e)\.[a-z]+$/i.test(filePath) ||
107
+ // Jest's convention: everything under `__tests__/` is a test, whatever it
108
+ // is called. The only directory that means that on its own.
109
+ /(^|[\\/])__tests__[\\/]/.test(filePath));
110
+ }
79
111
  /** The CamelCase spellings, which carry no separator before `Test`.
80
- * Case-SENSITIVE and must stay so: with `i` it also matches `Latest.java` and
81
- * `Manifest.cs`, ordinary source files. That is why the case-insensitive
82
- * `*Test.java` / `*Tests.cs` rows above cannot serve here. */
112
+ * Case-SENSITIVE and must stay so: with `i` it also matches `Latest.java`,
113
+ * `Audit.java`, `Edit.cs` and `Digit.ts`, ordinary source files. That is why
114
+ * the case-insensitive `*Test.java` / `*Tests.cs` rows above cannot serve
115
+ * here. */
83
116
  const CAMEL_TEST_FILE_NAME = /[a-z0-9](Tests?|IT)\.[A-Za-z0-9]+$/;
84
117
  /** Whether a path is a test file, by directory or by name. Read by the route
85
118
  * checks: a route declared in a mock server or a spec fixture is not a surface
@@ -25,7 +25,49 @@ import type { UtilsLanguage } from "./language-spec.js";
25
25
  * helper in either language gets the same key.
26
26
  */
27
27
  export declare function actionKeyOf(body: string, language?: UtilsLanguage): string | undefined;
28
+ /** The keys of every step, in order, or `undefined` when there is no step or a
29
+ * step has no key — the sequence `actionKeyOf` joins, kept as steps for a caller
30
+ * that matches step by step (a key is not split back: a selector may carry the
31
+ * separator). */
32
+ export declare function actionStepKeys(steps: ActionStep[]): string[] | undefined;
33
+ /** One Playwright action in a body, in source order — the unit `actionKeyOf` joins. */
34
+ export interface ActionStep {
35
+ /** The step's key (`verb chain [target]`), or `undefined` when this action cannot
36
+ * be read: its locator is not built from plain string literals, or its arguments
37
+ * do not balance. A helper with one such step has no key; a spec's other steps
38
+ * are still read, so an unreadable step is a boundary no matched run crosses. */
39
+ key?: string;
40
+ /** 1-based line of the action verb. */
41
+ line: number;
42
+ /** The source between the previous action's closing `)` (or the start of the
43
+ * body) and this action's chain: what stands between two consecutive actions.
44
+ * Comments are as the caller left them; string contents are kept. Empty for
45
+ * the first step. See `betweenActionsIsTransparent`. */
46
+ gapBefore: string;
47
+ /** The source after this action's closing `)` up to the next action's chain, or
48
+ * to the end of the body for the last step — the next step's `gapBefore`. */
49
+ gapAfter: string;
50
+ }
51
+ /**
52
+ * Every action in `body`, in order, each keyed as `actionKeyOf` keys it. Where
53
+ * `actionKeyOf` answers "what does this helper do" with one string, this answers
54
+ * "which actions, where, and what sits between them" — what a scan for an inline
55
+ * copy of a helper's sequence needs. Lines survive the Python and alias rewrites:
56
+ * neither adds or removes a newline.
57
+ */
58
+ export declare function actionStepsOf(body: string, language?: UtilsLanguage): ActionStep[];
59
+ /** `click empty-cart-btn; fill qty` → `click \`empty-cart-btn\`; fill \`qty\`` — the
60
+ * key as the agent reads it, selectors quoted so a `#id` never reads as prose. */
61
+ export declare function describeActionKey(key: string): string;
28
62
  /** Index of the `)` that closes the `(` at `open`, on a string-blanked copy. */
29
63
  export declare function matchingClose(structure: string, open: number): number;
64
+ /** Whitespace and braces dropped outside string literals, `'` normalised to `"` —
65
+ * the spelling a locator chain has inside a key. */
66
+ export declare function normalizeChain(text: string): string;
30
67
  /** String contents replaced by spaces, delimiters kept, same length. */
31
68
  export declare function blankStringContents(text: string): string;
69
+ /** The statements of a fragment as `[start, end)` spans on `structure` (a
70
+ * string-blanked copy): each cut where valueEnd cuts — a `;` at depth 0, or a line
71
+ * end at depth 0 that the next line does not continue with `.` — so a
72
+ * Prettier-wrapped chain is one statement, not one per line. */
73
+ export declare function statementSpans(structure: string): [number, number][];