@skyramp/mcp 0.4.2-rc.1 → 0.4.2-rc.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (110) hide show
  1. package/build/commands/localDevTestChangesCommand.js +1 -1
  2. package/build/commands/recommendTestsAndExecuteCommand.js +10 -1
  3. package/build/commands/testThisEndpointCommand.js +19 -2
  4. package/build/execution/wrapperConfig.d.ts +56 -0
  5. package/build/execution/wrapperConfig.js +155 -0
  6. package/build/index.js +6 -6
  7. package/build/playwright/registerPlaywrightTools.js +14 -0
  8. package/build/playwright/traceExportStore.d.ts +22 -0
  9. package/build/playwright/traceExportStore.js +81 -0
  10. package/build/playwright/traceRecordingPrompt.js +2 -1
  11. package/build/prompts/code-reuse.js +24 -21
  12. package/build/prompts/local-dev/local-dev-plan.js +6 -23
  13. package/build/prompts/local-dev/local-dev-prompts.js +1 -1
  14. package/build/prompts/shared-helper-policy.d.ts +36 -0
  15. package/build/prompts/shared-helper-policy.js +33 -1
  16. package/build/prompts/startTraceCollectionPrompts.js +1 -1
  17. package/build/prompts/sut-setup/modes/adaptWorkflowPrompt.js +6 -7
  18. package/build/prompts/sut-setup/shared.d.ts +1 -1
  19. package/build/prompts/sut-setup/shared.js +5 -3
  20. package/build/prompts/test-maintenance/drift-analysis-prompt.d.ts +16 -8
  21. package/build/prompts/test-maintenance/drift-analysis-prompt.js +90 -36
  22. package/build/prompts/test-maintenance/uiDriftAnalysisSections.js +1 -1
  23. package/build/prompts/test-recommendation/recommendationShared.d.ts +1 -1
  24. package/build/prompts/test-recommendation/recommendationShared.js +0 -1
  25. package/build/prompts/testbot/testbot-prompts.js +11 -9
  26. package/build/services/TestDiscoveryService.js +32 -4
  27. package/build/skills/runTestSkill.d.ts +6 -0
  28. package/build/skills/runTestSkill.js +17 -0
  29. package/build/tool-phases.js +0 -1
  30. package/build/tools/budgetExcuse.d.ts +15 -0
  31. package/build/tools/budgetExcuse.js +113 -0
  32. package/build/tools/code-refactor/utils-verify-gates.js +17 -3
  33. package/build/tools/executeSkyrampTestTool.d.ts +97 -48
  34. package/build/tools/executeSkyrampTestTool.js +775 -449
  35. package/build/tools/generate-tests/generateE2ERestTool.d.ts +0 -1
  36. package/build/tools/generate-tests/generateE2ERestTool.js +1 -9
  37. package/build/tools/generate-tests/generateUIRestTool.d.ts +0 -2
  38. package/build/tools/generate-tests/generateUIRestTool.js +1 -9
  39. package/build/tools/submitReportTool.js +128 -0
  40. package/build/tools/test-management/actionsTool.js +31 -0
  41. package/build/tools/test-management/analyzeChangesTool.d.ts +4 -4
  42. package/build/tools/test-management/analyzeTestHealthTool.d.ts +0 -11
  43. package/build/tools/test-management/analyzeTestHealthTool.js +7 -63
  44. package/build/tools/test-management/testsOwedBeforeRun.d.ts +28 -0
  45. package/build/tools/test-management/testsOwedBeforeRun.js +53 -0
  46. package/build/tools/trace/stopTraceCollectionTool.js +1 -1
  47. package/build/types/RepositoryAnalysis.d.ts +32 -32
  48. package/build/types/ReuseOutcome.d.ts +4 -3
  49. package/build/types/TestExecution.d.ts +2 -2
  50. package/build/types/TestTypes.d.ts +3 -7
  51. package/build/types/TestTypes.js +6 -20
  52. package/build/utils/AnalysisStateManager.d.ts +0 -7
  53. package/build/utils/AnalysisStateManager.js +1 -1
  54. package/build/utils/connectionErrors.d.ts +10 -0
  55. package/build/utils/connectionErrors.js +10 -0
  56. package/build/utils/language-helper.js +24 -3
  57. package/build/utils/progress.d.ts +1 -1
  58. package/build/utils/progress.js +1 -1
  59. package/build/utils/rebaselineSnapshots.d.ts +1 -1
  60. package/build/utils/rebaselineSnapshots.js +6 -16
  61. package/build/utils/reuseRouting.d.ts +10 -0
  62. package/build/utils/reuseRouting.js +15 -0
  63. package/build/utils/runContextGauge.d.ts +27 -0
  64. package/build/utils/runContextGauge.js +181 -0
  65. package/build/utils/skyrampMdContent.d.ts +1 -1
  66. package/build/utils/skyrampMdContent.js +1 -1
  67. package/build/utils/skyrampSdkVersion.d.ts +9 -0
  68. package/build/utils/skyrampSdkVersion.js +16 -0
  69. package/build/utils/testDependencyPolicy.js +21 -0
  70. package/build/utils/testExecutionRecord.d.ts +5 -1
  71. package/build/utils/testExecutionRecord.js +3 -1
  72. package/build/utils/testFileClassification.d.ts +8 -0
  73. package/build/utils/testFileClassification.js +36 -3
  74. package/build/utils/utils-verify/action-key.d.ts +42 -0
  75. package/build/utils/utils-verify/action-key.js +118 -36
  76. package/build/utils/utils-verify/action-sites.d.ts +32 -0
  77. package/build/utils/utils-verify/action-sites.js +202 -0
  78. package/build/utils/utils-verify/body-reach.js +2 -4
  79. package/build/utils/utils-verify/call-sites.d.ts +25 -6
  80. package/build/utils/utils-verify/call-sites.js +8 -5
  81. package/build/utils/utils-verify/index.d.ts +1 -0
  82. package/build/utils/utils-verify/index.js +1 -0
  83. package/build/utils/utils-verify/language-spec.d.ts +25 -0
  84. package/build/utils/utils-verify/language-spec.js +16 -2
  85. package/build/utils/utils-verify/parse.d.ts +10 -1
  86. package/build/utils/utils-verify/parse.js +19 -2
  87. package/build/utils/utils-verify/verify.d.ts +3 -2
  88. package/build/utils/utils-verify/verify.js +16 -18
  89. package/build/workspace/workspace.d.ts +72 -52
  90. package/build/workspace/workspace.js +12 -8
  91. package/package.json +1 -1
  92. package/plugin/prompts/testbot-task1.md +0 -2
  93. package/plugin/skills/enhance-assertions/reference/shared-rules.md +1 -1
  94. package/plugin/skills/enhance-assertions/reference/ui.md +1 -1
  95. package/plugin/skills/fix-test-import-errors/SKILL.md +2 -1
  96. package/plugin/skills/run-test/SKILL.md +16 -0
  97. package/build/adapters/jestAdapter.d.ts +0 -14
  98. package/build/adapters/jestAdapter.js +0 -131
  99. package/build/adapters/mochaAdapter.d.ts +0 -13
  100. package/build/adapters/mochaAdapter.js +0 -93
  101. package/build/adapters/playwrightAdapter.d.ts +0 -17
  102. package/build/adapters/playwrightAdapter.js +0 -184
  103. package/build/adapters/pytestAdapter.d.ts +0 -15
  104. package/build/adapters/pytestAdapter.js +0 -119
  105. package/build/tools/runExistingTestsTool.d.ts +0 -138
  106. package/build/tools/runExistingTestsTool.js +0 -666
  107. package/build/types/ExternalTestExecution.d.ts +0 -67
  108. package/build/types/ExternalTestExecution.js +0 -8
  109. package/build/workspace/testSuites.d.ts +0 -20
  110. package/build/workspace/testSuites.js +0 -17
@@ -12,6 +12,7 @@ import { TASK_ANALYZE_MAINTAIN, TASK_GENERATE, TASK_SUBMIT, TESTBOT_TASK1_MAINTA
12
12
  import { getTraceRecordingPromptText } from "../../playwright/traceRecordingPrompt.js";
13
13
  import { isContractConsumerModeEnabled, isPomReuseEnabled, isUtilsReuseEnabled, isSkillsLoaded, } from "../../utils/featureFlags.js";
14
14
  import { fixErrorsInstruction } from "../../skills/fixTestImportErrorsSkill.js";
15
+ import { runTestCall } from "../../skills/runTestSkill.js";
15
16
  import { alignAssertionsStep } from "../../skills/validateAssertionAlignmentSkill.js";
16
17
  import { ASSERTION_ENHANCEABLE_TEST_TYPES } from "../../types/TestTypes.js";
17
18
  import { enhanceAssertionsCall } from "../../skills/enhanceAssertionsSkill.js";
@@ -307,16 +308,19 @@ export function getTestbotPrompt(opts) {
307
308
 
308
309
  `;
309
310
  }
310
- // The Task 1 maintenance baseline sub-step branches in plan-only eval runs (SKYR-3879
311
- // plan-only lane): the SUT is not running, so the pre-edit baseline
312
- // execution is replaced by static-analysis-only verdicts. Defined before
313
- // task1Section, which interpolates it.
311
+ // The Task 1 maintenance sub-steps that run tests branch in plan-only eval runs
312
+ // (SKYR-3879 plan-only lane): the SUT is not running, so those runs are replaced
313
+ // by static-analysis-only verdicts. Defined before task1Section, which
314
+ // interpolates them.
314
315
  let maintenanceBeforeExecStep;
316
+ let verifyExternalStep;
315
317
  if (planOnly) {
318
+ verifyExternalStep = ` ${MAINTAIN.VERIFY_EXTERNAL}. Plan-only run: do not run the external tests you edited.`;
316
319
  maintenanceBeforeExecStep = ` ${MAINTAIN.BASELINE}. Plan-only run: skip the pre-edit baseline execution — the application is not running. Record every maintenance verdict from static drift analysis alone; execution statuses simply remain unrecorded.`;
317
320
  }
318
321
  else {
319
- maintenanceBeforeExecStep = ` ${MAINTAIN.BASELINE}. Call \`skyramp_execute_test\` with \`phase: "before"\` and \`stateFile\` for every UPDATE/REGENERATE/DELETE test. Exclude tests marked \`[external]\` — those are baselined in ${stepSubRef(TASK1.MAINTAIN, MAINTAIN.CONFIRM_EXTERNAL)} via \`skyramp_run_existing_tests\`. Run them sequentially, not in parallel. This captures the pre-edit baseline — do not skip even if you expect the test to fail. Never pass \`rebaselineSnapshots\` here: a \`Screenshot comparison failed\` on this run is the evidence that a visual baseline is stale, and the refresh belongs to the final execution.`;
322
+ verifyExternalStep = ` ${MAINTAIN.VERIFY_EXTERNAL}. Run the external tests you edited again. Do this step right after ${stepSubRef(TASK1.MAINTAIN, MAINTAIN.APPLY_ACTIONS)}. It applies to each \`[external]\` test file that you edited in ${stepSubRef(TASK1.MAINTAIN, MAINTAIN.APPLY_ACTIONS)} and that ${stepSubRef(TASK1.MAINTAIN, MAINTAIN.BASELINE)} ran. Run each such file with \`skyramp_execute_test\`, \`phase: "after"\` and \`stateFile\`, and use the same command as in ${stepSubRef(TASK1.MAINTAIN, MAINTAIN.BASELINE)}. Record each result as the file's \`afterStatus\`. The final execution excludes \`[external]\` files, so this step is where they get their after-run. If no file applies, say so in your report. If a file still fails, report it — do not run it again.`;
323
+ maintenanceBeforeExecStep = ` ${MAINTAIN.BASELINE}. Call \`skyramp_execute_test\` with \`phase: "before"\` and \`stateFile\` for every Skyramp test you will change, and for every test the repository owns that you will mark UPDATE: ${runTestCall()}. Run them one at a time, not in parallel. Run the command even when you expect the runner or a package to be missing: the failure output is what names what is missing, and a check you make before the run reaches no repair. On this run repair the environment only — install what is missing and run the file again — and never edit the test file, because the result of the file as the repository wrote it IS the baseline. This captures the pre-edit baseline — do not skip it even when you expect the test to fail. \`skyramp_actions\` refuses an UPDATE of the repository's own test until that test has this run. Nothing else owes one: a VERIFY changes no file, and a REGENERATE or a DELETE of the repository's own test is a recommendation you report rather than apply. Never pass \`rebaselineSnapshots\` on this run: a \`Screenshot comparison failed\` here is the evidence that a visual baseline is stale, and the refresh belongs to the final execution.`;
320
324
  }
321
325
  // For follow-up requests: emit the @skyramp-testbot header + guardrails, both stop-early.
322
326
  // For first-run prompts: emit the full Task 1 analysis + maintenance section.
@@ -346,8 +350,6 @@ ${analyzeBlock}
346
350
 
347
351
  ${TASK1.MAINTAIN}. **Maintain existing tests:**
348
352
 
349
- ${MAINTAIN.CONFIRM_EXTERNAL}. Confirm external-test breakage before assessment. If any service in \`${repositoryPath}/.skyramp/workspace.yml\` declares test-env config (\`runtimeDetails.test*\`), call \`skyramp_run_existing_tests\` (\`mode: "confirm"\`, \`stateFile\`) on the tests \`skyramp_analyze_changes\` marked \`[external]\` — the repo's OWN suite, not the Skyramp-generated tests — before \`skyramp_analyze_test_health\`, so the run-confirmed failures fold into its assessment. Do NOT read the \`[external]\` test files to *guess* the change's impact — RUN them with \`skyramp_run_existing_tests\` to get real pass/fail; reading them is not a substitute for running them. These \`[external]\` suites also do not run through \`skyramp_execute_test\`, so skipping this leaves their status \`Unknown\`. Selector, health-gate, and self-skip behavior are in the tool description. (Skyramp-generated tests are handled in ${stepSubRef(TASK1.MAINTAIN, MAINTAIN.BASELINE)}, not here.)
350
-
351
353
  ${MAINTAIN.TEST_HEALTH}. Call \`skyramp_analyze_test_health\` with \`stateFile\` (from \`skyramp_analyze_changes\` output). Pass \`blueprintCaptured: true\` when \`browser_blueprint\` was called successfully earlier in this session — see the parameter description for when this applies. **Do NOT read application source files** (routes, models, controllers) — all change information you need is in the \`skyramp_analyze_changes\` output and the diff. Exception: the UI drift pre-scan (\`UI_SYMBOL_PRESCAN\`) may instruct you to read changed frontend files to extract exported symbols — follow those instructions when present.
352
354
 
353
355
  ${MAINTAIN.UPDATE_INSTRUCTIONS}. Write \`updateInstructions\` for each UPDATE or REGENERATE test before calling \`skyramp_actions\` — articulating the change first prevents file content from overriding diff-based reasoning.
@@ -356,7 +358,7 @@ ${maintenanceBeforeExecStep}
356
358
 
357
359
  ${MAINTAIN.APPLY_ACTIONS}. Call \`skyramp_actions\` with \`stateFile\` (from \`skyramp_analyze_changes\` output) and apply the edits it returns.
358
360
 
359
- ${MAINTAIN.VERIFY_EXTERNAL}. Verify external-test fixes. **This step is not optional and it is the easiest one to forget — you have just edited files in ${stepSubRef(TASK1.MAINTAIN, MAINTAIN.APPLY_ACTIONS)}, so come back here before you move on to anything else.** It applies whenever ${stepSubRef(TASK1.MAINTAIN, MAINTAIN.CONFIRM_EXTERNAL)} reported a real pass/fail result for a file you then edited. It does NOT apply when ${stepSubRef(TASK1.MAINTAIN, MAINTAIN.CONFIRM_EXTERNAL)} returned \`skipped: true\` or \`ran: 0\` for every suite — there is no baseline to compare against, so say so in your report instead of re-running. When it applies: re-run those \`[external]\` files with \`skyramp_run_existing_tests\` (\`mode: "verify"\`, \`stateFile\`) and record each file's result as its \`afterStatus\`. Editing an \`[external]\` file that ${stepSubRef(TASK1.MAINTAIN, MAINTAIN.CONFIRM_EXTERNAL)} confirmed failing and NOT re-running it leaves your own fix unverified — you would be reporting a repair you never saw work. A still-failing verify is surfaced in the report — do not loop.
361
+ ${verifyExternalStep}
360
362
 
361
363
  ${codeReviewBlock}
362
364
 
@@ -624,7 +626,7 @@ ${FINISH_CHECKS_BLOCK}
624
626
  **Execution timing:**
625
627
  - **beforeStatus** (maintained tests only): execute each maintained test file **once at the start** (before any edits) to capture \`beforeStatus\`. This is the only execution allowed before edits.
626
628
  - **Probe residue**: before the final execution, delete through the API the records your own probes and recordings created. Find each one by the id its create response returned, or by the exact value you typed for it during a recording. Do not delete anything else, and do not delete by a pattern. Do not use the database or a container. Do not add the deletion to a test. List in the report the records the API could not remove.
627
- - **Final execution**: Do NOT call \`skyramp_execute_test\` again until ALL maintenance edits AND ALL new test generation/enhancement are complete. Then execute every test file once — maintained files (for \`afterStatus\`) and new files together — passing \`stateFile\` and \`repository\` on every call: the attempt cap and the \`afterStatus\` the report reads both live in that file, and \`repository\` names the section the result lands in. Omitting \`stateFile\` does not skip the cap — the tool falls back to this run's own state file — but a wrong or missing \`repository\` fails the call. For a maintained spec whose \`skyramp_actions\` entry carried \`rebaseline_snapshots\`, pass that exact list as \`rebaselineSnapshots\` on this run and only this run. **Execute tests SEQUENTIALLY (one at a time)** — do NOT send multiple \`skyramp_execute_test\` calls in the same tool call batch, as concurrent execution overwhelms the stdio transport and causes MCP disconnection. Exclude tests marked \`[external]\`.
629
+ - **Final execution**: Apart from ${stepSubRef(TASK1.MAINTAIN, MAINTAIN.VERIFY_EXTERNAL)}, do NOT call \`skyramp_execute_test\` again until ALL maintenance edits AND ALL new test generation/enhancement are complete. Then execute every test file once — maintained files (for \`afterStatus\`) and new files together — passing \`stateFile\` and \`repository\` on every call: the attempt cap and the \`afterStatus\` the report reads both live in that file, and \`repository\` names the section the result lands in. Omitting \`stateFile\` does not skip the cap — the tool falls back to this run's own state file — but a wrong or missing \`repository\` fails the call. For a maintained spec whose \`skyramp_actions\` entry carried \`rebaseline_snapshots\`, pass that exact list as \`rebaselineSnapshots\` on this run and only this run. **Execute tests SEQUENTIALLY (one at a time)** — do NOT send multiple \`skyramp_execute_test\` calls in the same tool call batch, as concurrent execution overwhelms the stdio transport and causes MCP disconnection. Exclude tests marked \`[external]\`.
628
630
  - Only report test results for files you actually ran.
629
631
 
630
632
  ### Post-Execution: Assertion Alignment
@@ -4,7 +4,15 @@ import { logger } from "../utils/logger.js";
4
4
  import { TestSource } from "../types/TestAnalysis.js";
5
5
  import { TestType } from "../types/TestTypes.js";
6
6
  import fg from "fast-glob";
7
- import { isDiscoveredTestFile } from "../utils/testFileClassification.js";
7
+ import { hasTestFileName, isDiscoveredTestFile, } from "../utils/testFileClassification.js";
8
+ /**
9
+ * A Jest/Vitest snapshot in the colocated CRA layout. Its name never identifies
10
+ * it as a test (`Button.test.tsx.snap` cannot match a `.<ext>` tail across the
11
+ * `.tsx.` dot), but a render change leaves it stale, so it is maintenance
12
+ * material. It never enters the before-run gate: `detectLanguage` on `.snap`
13
+ * returns "unknown".
14
+ */
15
+ const COLOCATED_SNAPSHOT = /[\\/]__snapshots__[\\/][^\\/]+\.snap$/;
8
16
  export class TestDiscoveryService {
9
17
  EXCLUDED_DIRS = [
10
18
  "node_modules",
@@ -143,6 +151,7 @@ export class TestDiscoveryService {
143
151
  const result = { skyramp: [], external: [] };
144
152
  const contentCache = new Map();
145
153
  const marker = this.SKYRAMP_MARKER;
154
+ const dropped = [];
146
155
  const batchSize = 100;
147
156
  for (let i = 0; i < testCandidates.length; i += batchSize) {
148
157
  const batch = testCandidates.slice(i, i + batchSize);
@@ -153,16 +162,27 @@ export class TestDiscoveryService {
153
162
  result.skyramp.push(file);
154
163
  contentCache.set(file, content);
155
164
  }
156
- else {
165
+ else if (hasTestFileName(file) || COLOCATED_SNAPSHOT.test(file)) {
157
166
  result.external.push(file);
158
167
  contentCache.set(file, content);
159
168
  }
169
+ else {
170
+ // A candidate without the marker and without a test file name is a
171
+ // helper, fixture, page object, setup file or config that only
172
+ // qualified by sitting in a test directory. It cannot be run as a
173
+ // test, so it is not external coverage. Skyramp's own files keep the
174
+ // wide rule above -- the marker is what identifies them.
175
+ dropped.push(file);
176
+ }
160
177
  }
161
178
  catch (error) {
162
- logger.debug(`Skipping file ${file}: ${error}`);
179
+ logger.warning(`Skipping file ${file}: ${error}`);
163
180
  }
164
181
  }
165
182
  }
183
+ if (dropped.length > 0) {
184
+ logger.debug(`Dropped ${dropped.length} test-directory candidates with no test file name: ${dropped.slice(0, 10).join(", ")}`);
185
+ }
166
186
  logger.debug(`Classified ${result.skyramp.length} Skyramp files, ${result.external.length} external test files`);
167
187
  return { ...result, contentCache };
168
188
  }
@@ -171,7 +191,11 @@ export class TestDiscoveryService {
171
191
  * or directory placement.
172
192
  */
173
193
  isExternalTestFile(filePath) {
174
- return isDiscoveredTestFile(filePath);
194
+ // Either rule may admit a candidate: the directory rule for a test that
195
+ // sits in a test tree, the name rule for one that does not. Without the
196
+ // second, `auth.e2e.ts` or `OrdersIT.java` outside a test directory never
197
+ // reaches the classification below.
198
+ return isDiscoveredTestFile(filePath) || hasTestFileName(filePath);
175
199
  }
176
200
  /**
177
201
  * Extract metadata from a test file
@@ -473,7 +497,11 @@ export class TestDiscoveryService {
473
497
  const languageMap = {
474
498
  ".py": "python",
475
499
  ".js": "javascript",
500
+ ".mjs": "javascript",
501
+ ".cjs": "javascript",
476
502
  ".ts": "typescript",
503
+ ".mts": "typescript",
504
+ ".cts": "typescript",
477
505
  ".tsx": "typescript",
478
506
  ".jsx": "javascript",
479
507
  ".java": "java",
@@ -0,0 +1,6 @@
1
+ /** The `run-test` skill ships in `plugin/skills/` at the package root. */
2
+ export declare const RUN_TEST_SKILL = "run-test";
3
+ /** How to shape a command for skyramp_execute_test without the skill. */
4
+ export declare const RUN_TEST_COMMAND_RULES: string;
5
+ /** How a prompt tells the agent to run one test file. */
6
+ export declare function runTestCall(): string;
@@ -0,0 +1,17 @@
1
+ import { isSkillsLoaded } from "../utils/featureFlags.js";
2
+ import { fixErrorsInstruction } from "./fixTestImportErrorsSkill.js";
3
+ /** The `run-test` skill ships in `plugin/skills/` at the package root. */
4
+ export const RUN_TEST_SKILL = "run-test";
5
+ /** How to shape a command for skyramp_execute_test without the skill. */
6
+ export const RUN_TEST_COMMAND_RULES = `for a TypeScript or JavaScript Playwright UI or end-to-end test, add \`--config "$SKYRAMP_PLAYWRIGHT_CONFIG"\`; ` +
7
+ `find the environment variable the test reads for its base URL — \`SKYRAMP_TEST_BASE_URL\` in a generated test, or the name the repository's own config reads in an existing one — and when the run exports no value for that name, put \`<NAME>=$SKYRAMP_TEST_SERVICE_URL_<SERVICENAME>\` in front of the command, for the service the test targets`;
8
+ /** How a prompt tells the agent to run one test file. */
9
+ export function runTestCall() {
10
+ if (isSkillsLoaded()) {
11
+ return (`invoke the \`${RUN_TEST_SKILL}\` skill with arguments ` +
12
+ `\`<absolute path of the test file> <language> <framework> <test type>\``);
13
+ }
14
+ return (`call \`skyramp_execute_test\` with the repository's own command for that one test file, ` +
15
+ `or the framework's usual command when the repository has none (${RUN_TEST_COMMAND_RULES}). ` +
16
+ `If the output shows that a package is missing — the runner itself, or a module the test or its config imports — ${fixErrorsInstruction()} and run the test again`);
17
+ }
@@ -15,7 +15,6 @@ export const TOOL_PHASE_MAP = {
15
15
  skyramp_generate_enriched_integration_test: "generating",
16
16
  skyramp_enrich_test_with_mocks: "generating",
17
17
  skyramp_execute_test: { before: "maintaining", after: "executing" },
18
- skyramp_run_existing_tests: { before: "maintaining", after: "executing" },
19
18
  skyramp_analyze_test_health: "maintaining",
20
19
  skyramp_submit_report: "reporting",
21
20
  };
@@ -0,0 +1,15 @@
1
+ /** True when this answer's reason for dropping a test is that the run was running
2
+ * out. Blank answers are NOT this: an unanswered objection is published open,
3
+ * which is already honest. */
4
+ export declare function citesBudget(answer: string | undefined): boolean;
5
+ /** How many times the report may be refused for this. The refusal is meant to
6
+ * cost the agent the tests it dropped, not the whole run: past this the answer
7
+ * stands and is published with the report, where a reader can weigh it. Same
8
+ * bound and same reasoning as the unrecorded-UI gate. */
9
+ export declare const MAX_BUDGET_EXCUSE_REFUSALS = 3;
10
+ /** Count one refusal and return the new total. */
11
+ export declare function countBudgetExcuseRefusal(): number;
12
+ /** Refusals so far, without counting one. */
13
+ export declare function budgetExcuseRefusals(): number;
14
+ /** Test seam. */
15
+ export declare function resetBudgetExcuseStoreForTests(): void;
@@ -0,0 +1,113 @@
1
+ import { currentRunStateFile } from "../utils/AnalysisStateManager.js";
2
+ import { logger } from "../utils/logger.js";
3
+ /** "The run ran out of budget" as a reason a planned test was dropped, and how
4
+ * many times the report has been refused for giving it.
5
+ *
6
+ * WHY THIS EXISTS. `deliveredMatchesPlan` raises "planned but not delivered" and
7
+ * suggests "add the test to the report, or record in the report why it was
8
+ * dropped". Run 9dfc8405 recorded the reason seven times — "this run exhausted
9
+ * its working budget" — and shipped 2 of 9 planned tests with the report
10
+ * accepted. It wrote that at 46% of a 1M context window, with no warning from
11
+ * anywhere, and the word appears NOWHERE in its reasoning: it is a justification
12
+ * written at submit time, not a decision reached from a reading.
13
+ *
14
+ * So this is not a budget check. It is a check on ONE SENTENCE the run is not
15
+ * entitled to write. A test blocked by something outside the run — a service
16
+ * that is not running, a branch that is gone, a credential the run lacks — still
17
+ * answers and still ships, exactly as before. */
18
+ /** Phrases that say "I stopped because I was running out", and nothing else.
19
+ *
20
+ * Deliberately narrow, and narrowed further after review: `budget` and `token
21
+ * limit` are ORDINARY APPLICATION NOUNS. An unscoped `\bbudget\b` refuses "the
22
+ * budget page does not render: /budgets returns 500", and an unscoped `token
23
+ * limit` refuses "the API rejects the request past its token limit" — real
24
+ * blockers on a customer whose product has budgets, unreachable on any eval
25
+ * repo, and nothing here is gated on the gauge, so the collision is the whole
26
+ * test.
27
+ *
28
+ * Every answer run 9dfc8405 sent is run-scoped — "this run exhausted its
29
+ * working budget", "for budget reasons" — so scoping costs nothing. A miss
30
+ * costs a refusal that would have helped; a false positive costs a run that
31
+ * cannot report a genuine blocker, which is far worse. */
32
+ const BUDGET_EXCUSE = [
33
+ // "for budget reasons" / "for budgetary reasons" — no other reading.
34
+ /\bbudget(ary)? reasons?\b/i,
35
+ // A spending verb beside the budget: "exhausted its working budget", "the
36
+ // remaining budget". SINGULAR ONLY — a product's own budgets are a plural
37
+ // list ("the remaining budgets are empty") and are not this.
38
+ /\b(exhausted|spent|ran out of|out of|low on|remaining|no more)\b[^.;]{0,24}\bbudget\b/i,
39
+ // "no budget left for", "not enough context left". The trailing word is what
40
+ // keeps "the page has no time zone selector" out.
41
+ /\b(no|insufficient|not enough) (time|context|tokens?|budget) (left|remaining|available|to spare|for)\b/i,
42
+ // The run's own context. Not a domain noun in any application we test.
43
+ /\bcontext (window|limit|budget)\b/i,
44
+ /\b(low on|out of|ran out of|exhausted (its|the|my)?) ?(context|time)\b/i,
45
+ /\btime (ran out|constraints?|pressure)\b/i,
46
+ // A token BUDGET is the agent talking about itself; a token LIMIT is a real
47
+ // API concept and stays out.
48
+ /\btokens? budget\b/i,
49
+ // Tokens only with the run as the subject: a rate limiter runs out of tokens
50
+ // too, and says so in a sentence a blocker needs.
51
+ /\b(this run|the run|I|we)\b[^.;]{0,30}\b(ran out of|out of|low on|exhausted)\b[^.;]{0,12}\btokens?\b/i,
52
+ ];
53
+ /** True when this answer's reason for dropping a test is that the run was running
54
+ * out. Blank answers are NOT this: an unanswered objection is published open,
55
+ * which is already honest. */
56
+ export function citesBudget(answer) {
57
+ const text = typeof answer === "string" ? answer.trim() : "";
58
+ if (!text)
59
+ return false;
60
+ return BUDGET_EXCUSE.some((pattern) => pattern.test(text));
61
+ }
62
+ /** How many times the report may be refused for this. The refusal is meant to
63
+ * cost the agent the tests it dropped, not the whole run: past this the answer
64
+ * stands and is published with the report, where a reader can weigh it. Same
65
+ * bound and same reasoning as the unrecorded-UI gate. */
66
+ export const MAX_BUDGET_EXCUSE_REFUSALS = 3;
67
+ /** In process and keyed by the run, for the reason `traceExportStore` gives: the
68
+ * only reader shares a process with the tool, and `NO_RUN` is a slot like any
69
+ * other so a run whose state path will not resolve is still bounded. */
70
+ const NO_RUN = "<no-run>";
71
+ let countedRunStateFile;
72
+ let refusals = 0;
73
+ function reseatRun() {
74
+ let runStateFile;
75
+ try {
76
+ runStateFile = currentRunStateFile() ?? NO_RUN;
77
+ }
78
+ catch {
79
+ runStateFile = NO_RUN;
80
+ }
81
+ if (runStateFile !== countedRunStateFile) {
82
+ countedRunStateFile = runStateFile;
83
+ refusals = 0;
84
+ }
85
+ }
86
+ /** Count one refusal and return the new total. */
87
+ export function countBudgetExcuseRefusal() {
88
+ try {
89
+ reseatRun();
90
+ refusals += 1;
91
+ return refusals;
92
+ }
93
+ catch (error) {
94
+ const detail = error instanceof Error ? error.message : String(error);
95
+ logger.warning(`countBudgetExcuseRefusal failed: ${detail}`);
96
+ return MAX_BUDGET_EXCUSE_REFUSALS;
97
+ }
98
+ }
99
+ /** Refusals so far, without counting one. */
100
+ export function budgetExcuseRefusals() {
101
+ try {
102
+ reseatRun();
103
+ return refusals;
104
+ }
105
+ catch {
106
+ return MAX_BUDGET_EXCUSE_REFUSALS;
107
+ }
108
+ }
109
+ /** Test seam. */
110
+ export function resetBudgetExcuseStoreForTests() {
111
+ countedRunStateFile = undefined;
112
+ refusals = 0;
113
+ }
@@ -1,5 +1,6 @@
1
1
  import * as path from "path";
2
2
  import { allowMarkerLine, describeAssertionCount, MARKERABLE_UTILS_KINDS, HELPER_FAMILIES, } from "../../utils/utils-verify/index.js";
3
+ import { describeActionKey } from "../../utils/utils-verify/action-key.js";
3
4
  import { utilsFileLabel } from "./reuse-state.js";
4
5
  import { STATUS_ONCE_RULE } from "../../prompts/shared-helper-policy.js";
5
6
  /** How many verify rounds the agent may spend before it must document what remains.
@@ -218,6 +219,9 @@ function moduleLabel(r) {
218
219
  ? utilsFileLabel(r.utilsFiles)
219
220
  : "the shared utils file";
220
221
  }
222
+ /** The half of the sibling call-site advisory both families share: a site inside
223
+ * the sibling's own helper, the execution owed, and that nothing blocks. */
224
+ const inlineSiteTail = (localCopy) => `For a site in a local helper, replace ${localCopy} inside the local helper with a call to the shared helper and keep the local helper's other statements; delete the local helper only if it then does nothing but forward, moving the import and the differing literals to its call sites. Then re-verify AND run the edited file with skyramp_execute_test — an edited pre-existing test without a recorded execution cannot be reported. This does not block execution.`;
221
225
  /** SKYR-4219: the sibling call-site pass the prompt asks for and the agent skips (0/2
222
226
  * seeded eval runs). Stated here because the verify call is the one point every run
223
227
  * reaches. Advisory — the sibling is a pre-existing test the maintenance flow owns, so
@@ -230,12 +234,22 @@ function inlineCallSitesAdvisory(r) {
230
234
  const labels = new Map(r.utilsFiles.map((f) => [f, utilsFileLabel([f])]));
231
235
  const labelOf = (f) => labels.get(f) ?? utilsFileLabel([f]);
232
236
  const n = r.inlineCallSites.length;
237
+ // One family per pass: the sites' kind decides the family half of the text.
238
+ const family = r.inlineCallSites[0].kind === "sequence"
239
+ ? HELPER_FAMILIES.browser
240
+ : HELPER_FAMILIES.api;
241
+ const lines = (c) => c.kind === "sequence" && c.endLine !== c.line
242
+ ? `${c.line}-${c.endLine}`
243
+ : `${c.line}`;
244
+ const what = (c) => c.kind === "sequence"
245
+ ? describeActionKey(c.actionKey)
246
+ : `${c.method} ${c.path}`;
233
247
  return [
234
248
  "",
235
249
  "",
236
- `ADVISORY — ${n} inline request call${n === 1 ? "" : "s"} in other Skyramp-generated tests that a shared helper already wraps:`,
237
- ...r.inlineCallSites.map((c) => `- ${show(c.file, r)}:${c.line}${c.localHelper ? ` (in local helper ${c.localHelper})` : ""} · ${c.method} ${c.path} · ${labelOf(c.utilsFile)}.${c.helper}`),
238
- `Each is a candidate: the request's method, path and call shape match the helper's. Replace the block with a call to the helper only if the remaining differences are literals (they become the arguments; keep the block's response variable; import the helper into that file; change nothing else there). For a site in a local helper, replace that request inside the local helper with a call to the shared helper and keep the local helper's other statements; delete the local helper only if it then does nothing but forward, moving the import and the differing literals to its call sites. Then re-verify AND run the edited file with skyramp_execute_test — an edited pre-existing test without a recorded execution cannot be reported. This does not block execution.`,
250
+ `ADVISORY — ${n} ${family.inlineSite.noun}${n === 1 ? "" : "s"} in other Skyramp-generated tests that a shared helper already ${family.inlineSite.already}:`,
251
+ ...r.inlineCallSites.map((c) => `- ${show(c.file, r)}:${lines(c)}${c.localHelper ? ` (in local helper ${c.localHelper})` : ""} · ${what(c)} · ${labelOf(c.utilsFile)}.${c.helper}`),
252
+ `${family.inlineSite.instruction} ${inlineSiteTail(family.inlineSite.localCopy)}`,
239
253
  ].join("\n");
240
254
  }
241
255
  /** Operations written in two or more sibling generated tests with no shared helper —
@@ -1,18 +1,18 @@
1
+ import { z } from "zod";
1
2
  import { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js";
2
- import { TestExecutionResult } from "../types/TestExecution.js";
3
- import { TestType } from "../types/TestTypes.js";
3
+ import { TestExecutionStatus, TestExecutionResult, type TestExecutionRecord } from "../types/TestExecution.js";
4
+ import { ProgrammingLanguage, TestType } from "../types/TestTypes.js";
4
5
  import { MaintenanceActionCore, TestAnalysisResult } from "../types/TestAnalysis.js";
5
- import type { TestExecutionRecord } from "../types/TestExecution.js";
6
- export declare const CONTRACT_EXECUTION_MODES: readonly ["provider", "consumer"];
7
- export type ContractExecutionMode = (typeof CONTRACT_EXECUTION_MODES)[number];
6
+ export declare const TOOL_NAME = "skyramp_execute_test";
7
+ /** Output kept per run, from its end: it is held in memory, saved to the state file, and returned to the agent. */
8
+ export declare const MAX_OUTPUT_CHARS = 200000;
8
9
  /**
9
- * Resolve the effective auth token for test execution.
10
- * `unauthenticated: true` forces no token (empty string) so the container
11
- * env omits SKYRAMP_TEST_TOKEN entirely — unauthenticated endpoints won't
12
- * receive an empty Authorization header that triggers encoding errors (E7).
10
+ * `unauthenticated: true` forces no token, so the child env carries no
11
+ * SKYRAMP_TEST_TOKEN at all — an empty Authorization header triggers encoding
12
+ * errors on unauthenticated endpoints (E7). An empty `token` means "use the
13
+ * server's environment", as the local-dev prompt passes it.
13
14
  */
14
15
  export declare function resolveEffectiveToken(unauthenticated?: boolean, paramToken?: string, envToken?: string): string;
15
- export declare function shouldInjectSkyrampBaseUrl(testType: TestType, contractMode?: ContractExecutionMode): boolean;
16
16
  /**
17
17
  * Append the recorded video path to execution output.
18
18
  *
@@ -22,6 +22,67 @@ export declare function shouldInjectSkyrampBaseUrl(testType: TestType, contractM
22
22
  * its own attachment line.
23
23
  */
24
24
  export declare function withVideoInfo(output: string, videoPath?: string): string;
25
+ /**
26
+ * One directory per run: a unique name leaves nothing to clean up between runs, and
27
+ * the random suffix keeps two runs in the same millisecond apart.
28
+ */
29
+ export declare function videoSubdirName(testFile: string): string;
30
+ /** The first `video.webm` under a run's video directory, if one was recorded. */
31
+ export declare function collectVideoPath(videoDir: string): string | undefined;
32
+ /**
33
+ * The directory whose `.skyramp/workspace.yml` describes this run: the nearest
34
+ * ancestor of `cwd` that has one, else the run's primary checkout (a tests repo
35
+ * delivered apart from the SUT has no workspace.yml of its own), else `cwd`.
36
+ */
37
+ export declare function resolveWorkspaceRoot(cwd: string): string;
38
+ /** pytest splits PYTEST_ADDOPTS with shlex, so a path with whitespace needs quotes. */
39
+ export declare function appendPytestVideoOpts(existing: string | undefined, videoDir: string): string;
40
+ /**
41
+ * The names by which a command can select `testFile`: its basename, and for Java the
42
+ * class name, because Maven selects a test with `-Dtest=FooTest`.
43
+ */
44
+ export declare function testFileNames(testFile: string): string[];
45
+ /**
46
+ * Backslashes are ignored: Playwright and Jest read the path as a regular
47
+ * expression, so the agent escapes it (`a\.spec\.ts`).
48
+ *
49
+ * A trailing shell comment is removed first. `npm test # a_test.py` names the
50
+ * file only in a comment and would run the whole suite, and the check exists
51
+ * to stop exactly that. This is a name check, not a parse: a command that
52
+ * mentions the file in some other way it does not run still passes.
53
+ */
54
+ /** The config an explicit `--config <path>` names, resolved against `cwd`, or
55
+ * undefined when the command carries none. `$SKYRAMP_PLAYWRIGHT_CONFIG` is
56
+ * ours and is not a repository config. */
57
+ export declare function explicitPlaywrightConfig(command: string, cwd: string): string | undefined;
58
+ /** The command with its `--config <repo config>` pointed at the wrapper. The
59
+ * wrapper imports that same config, so the run keeps the repository's testDir
60
+ * and projects and gains the video overlay. Without this swap a command that
61
+ * names a config runs the config directly and records nothing. */
62
+ export declare function pointConfigAtWrapper(command: string, wrapperPath: string): string;
63
+ export declare function commandNamesTestFile(command: string, testFile: string): boolean;
64
+ /** pytest exits 4 on a usage error; without pytest-playwright `--video` is one. */
65
+ export declare function pytestRejectedVideoOptions(run: SpawnedRun): boolean;
66
+ /** The line proving no test ran, or undefined when the output does not say so. */
67
+ export declare function noTestRanReason(output: string): string | undefined;
68
+ export declare function verdictForExit(exitCode: number | null, timedOut: boolean, output?: string): TestExecutionStatus;
69
+ export interface SpawnedRun {
70
+ exitCode: number | null;
71
+ output: string;
72
+ timedOut: boolean;
73
+ spawnError?: string;
74
+ duration: number;
75
+ }
76
+ /**
77
+ * Runs `command` through the shell in its own process group, so a timeout kills
78
+ * the runner's children too. stdout and stderr share one buffer in arrival order.
79
+ */
80
+ export declare function spawnTestCommand(opts: {
81
+ command: string;
82
+ cwd: string;
83
+ env: NodeJS.ProcessEnv;
84
+ timeoutMs: number;
85
+ }): Promise<SpawnedRun>;
25
86
  /**
26
87
  * Normalize the `rebaselineSnapshots` request (SKYR-4298): dedupe and drop blanks,
27
88
  * and refuse it on the pre-edit run. A refresh there would overwrite the very
@@ -32,23 +93,6 @@ export declare function resolveRebaselineSnapshots(requested: string[] | undefin
32
93
  snapshots: string[];
33
94
  error?: string;
34
95
  };
35
- /**
36
- * The failure text the agent receives. Everything it needs has to be in here:
37
- * only a tool's return value reaches the transcript, and Claude Code does not
38
- * capture an MCP server's stderr, so anything this omits is unrecoverable once
39
- * the run ends.
40
- *
41
- * `output` alone was not enough. When the executor produced none, the message
42
- * read "Test execution failed:" and stopped — so the agent inferred a cause and
43
- * reported it as fact. Measured in eval run 32220788735: it blamed an
44
- * unreachable backend while the app was answering 200 on both the host and the
45
- * docker bridge, and that invented cause reached `issuesFound`.
46
- *
47
- * The 401 hint is emitted only when the output carries a 401 in a status-shaped
48
- * position — see HTTP_401_SHAPES. It used to be unconditional, which pointed the
49
- * agent at authentication on a run whose output was empty, and then it keyed off
50
- * the bare number, which pointed it there on a test count or a line number.
51
- */
52
96
  /** The on-disk files a requested baseline name resolves to; empty when none matches. */
53
97
  export type BaselineFileState = Array<{
54
98
  file: string;
@@ -56,20 +100,12 @@ export type BaselineFileState = Array<{
56
100
  mtimeMs: number;
57
101
  }>;
58
102
  /**
59
- * Snapshot of the requested baselines under `<spec>-snapshots/` (SKYR-4298): for
60
- * each requested name, every PNG whose name matches the stem (a spec may hold both
61
- * `<stem>-linux.png` and `<stem>-chromium-linux.png`; latching onto one of them would
62
- * misreport the other) with its size and mtime. Taken before and after the run so
63
- * the tool can tell the agent which baselines were actually rewritten — SmartPlaywright
64
- * is the only party that knows the exact filename, and an executor image that lacks
65
- * SKYRAMP_UPDATE_SNAPSHOTS (or a name matching no toHaveScreenshot call) leaves
66
- * every file untouched.
103
+ * For each requested name, every PNG under `<spec>-snapshots/` whose name matches the
104
+ * stem, with size and mtime. A spec may hold both `<stem>-linux.png` and
105
+ * `<stem>-chromium-linux.png`; latching onto one would misreport the other. Taken
106
+ * before and after the run to tell which baselines were actually rewritten.
67
107
  */
68
108
  export declare function readBaselineState(specFile: string, requested: string[]): Record<string, BaselineFileState>;
69
- /**
70
- * Which requested baselines changed on disk between two readBaselineState calls, and
71
- * which files carried the change (the ones to stage).
72
- */
73
109
  export declare function diffBaselineState(before: Record<string, BaselineFileState>, after: Record<string, BaselineFileState>): {
74
110
  refreshed: string[];
75
111
  notRefreshed: string[];
@@ -92,19 +128,16 @@ export declare function authorizeRebaseline(stateData: {
92
128
  error?: string;
93
129
  };
94
130
  /**
95
- * Reconcile the persisted verdict with what the executor actually did (SKYR-4298).
96
- * An executor image whose @skyramp/skyramp predates SKYRAMP_UPDATE_SNAPSHOTS rewrites
97
- * nothing; left alone, the verdict would still promise a refresh, the report gate
98
- * would refuse the report, and nothing in the prompt makes the agent's way out
99
- * deterministic. So: names that were not refreshed are dropped from the verdict; a
100
- * rebaseline-only UPDATE with nothing left becomes VERIFY (the test stays red, the
101
- * rationale says why), and an UPDATE that also carried edits keeps UPDATE and is
102
- * held to its edit. The report then reflects what happened, not what was asked.
131
+ * Reconcile the persisted verdict with what the run actually did (SKYR-4298). A
132
+ * @skyramp/skyramp that predates SKYRAMP_UPDATE_SNAPSHOTS rewrites nothing; left
133
+ * alone, the verdict would still promise a refresh and the report gate would refuse
134
+ * the report. Names not refreshed are dropped; a rebaseline-only UPDATE with nothing
135
+ * left becomes VERIFY, and an UPDATE that also carried edits is held to its edit.
103
136
  */
104
137
  export declare function applyRefreshOutcomeToVerdicts(verdicts: MaintenanceActionCore[], testFile: string, outcome: {
105
138
  refreshed: string[];
106
139
  notRefreshed: string[];
107
- }, executorImage: string): {
140
+ }): {
108
141
  verdicts: MaintenanceActionCore[];
109
142
  note?: string;
110
143
  };
@@ -118,6 +151,7 @@ export declare function describeRefreshOutcome(outcome: {
118
151
  refreshed: string[];
119
152
  notRefreshed: string[];
120
153
  }): string;
154
+ /**
121
155
  /**
122
156
  * The fix-and-rerun attempt cap (SKYR-4460), enforced where the prompt's prose
123
157
  * cannot be: a file that has already received `cap` runs in the final phase
@@ -139,4 +173,19 @@ export declare function describeAttempt(priorAfterRuns: number, cap: number, opt
139
173
  export declare function buildExecutionFailureText(result: TestExecutionResult, opts?: {
140
174
  unchangedRerun?: boolean;
141
175
  }): string;
176
+ export declare const inputSchema: {
177
+ commandOverride: z.ZodOptional<z.ZodString>;
178
+ cwd: z.ZodOptional<z.ZodString>;
179
+ workspacePath: z.ZodOptional<z.ZodString>;
180
+ testFile: z.ZodString;
181
+ language: z.ZodNativeEnum<typeof ProgrammingLanguage>;
182
+ testType: z.ZodNativeEnum<typeof TestType>;
183
+ phase: z.ZodDefault<z.ZodEnum<["before", "after"]>>;
184
+ stateFile: z.ZodOptional<z.ZodString>;
185
+ repository: z.ZodString;
186
+ token: z.ZodOptional<z.ZodString>;
187
+ unauthenticated: z.ZodOptional<z.ZodBoolean>;
188
+ rebaselineSnapshots: z.ZodOptional<z.ZodArray<z.ZodString, "many">>;
189
+ timeout: z.ZodOptional<z.ZodNumber>;
190
+ };
142
191
  export declare function registerExecuteSkyrampTestTool(server: McpServer): void;