@skyramp/mcp 0.4.2-rc.1 → 0.4.2-rc.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (104) hide show
  1. package/build/commands/localDevTestChangesCommand.js +1 -1
  2. package/build/commands/recommendTestsAndExecuteCommand.js +10 -1
  3. package/build/commands/testThisEndpointCommand.js +19 -2
  4. package/build/execution/wrapperConfig.d.ts +56 -0
  5. package/build/execution/wrapperConfig.js +155 -0
  6. package/build/index.js +6 -6
  7. package/build/playwright/registerPlaywrightTools.js +14 -0
  8. package/build/playwright/traceExportStore.d.ts +22 -0
  9. package/build/playwright/traceExportStore.js +81 -0
  10. package/build/playwright/traceRecordingPrompt.js +2 -1
  11. package/build/prompts/code-reuse.js +24 -21
  12. package/build/prompts/local-dev/local-dev-plan.js +6 -23
  13. package/build/prompts/local-dev/local-dev-prompts.js +1 -1
  14. package/build/prompts/shared-helper-policy.d.ts +36 -0
  15. package/build/prompts/shared-helper-policy.js +33 -1
  16. package/build/prompts/startTraceCollectionPrompts.js +1 -1
  17. package/build/prompts/sut-setup/modes/adaptWorkflowPrompt.js +6 -7
  18. package/build/prompts/sut-setup/shared.d.ts +1 -1
  19. package/build/prompts/sut-setup/shared.js +5 -3
  20. package/build/prompts/test-maintenance/drift-analysis-prompt.d.ts +16 -8
  21. package/build/prompts/test-maintenance/drift-analysis-prompt.js +90 -36
  22. package/build/prompts/test-maintenance/uiDriftAnalysisSections.js +1 -1
  23. package/build/prompts/test-recommendation/recommendationShared.d.ts +1 -1
  24. package/build/prompts/test-recommendation/recommendationShared.js +0 -1
  25. package/build/prompts/testbot/testbot-prompts.js +11 -9
  26. package/build/services/TestDiscoveryService.js +32 -4
  27. package/build/skills/runTestSkill.d.ts +6 -0
  28. package/build/skills/runTestSkill.js +17 -0
  29. package/build/tool-phases.js +0 -1
  30. package/build/tools/budgetExcuse.d.ts +15 -0
  31. package/build/tools/budgetExcuse.js +113 -0
  32. package/build/tools/code-refactor/utils-verify-gates.js +17 -3
  33. package/build/tools/executeSkyrampTestTool.d.ts +97 -48
  34. package/build/tools/executeSkyrampTestTool.js +775 -449
  35. package/build/tools/submitReportTool.js +128 -0
  36. package/build/tools/test-management/actionsTool.js +31 -0
  37. package/build/tools/test-management/analyzeChangesTool.d.ts +4 -4
  38. package/build/tools/test-management/analyzeTestHealthTool.d.ts +0 -11
  39. package/build/tools/test-management/analyzeTestHealthTool.js +7 -63
  40. package/build/tools/test-management/testsOwedBeforeRun.d.ts +28 -0
  41. package/build/tools/test-management/testsOwedBeforeRun.js +53 -0
  42. package/build/tools/trace/stopTraceCollectionTool.js +1 -1
  43. package/build/types/RepositoryAnalysis.d.ts +32 -32
  44. package/build/types/ReuseOutcome.d.ts +4 -3
  45. package/build/types/TestExecution.d.ts +2 -2
  46. package/build/types/TestTypes.d.ts +3 -0
  47. package/build/types/TestTypes.js +6 -0
  48. package/build/utils/AnalysisStateManager.d.ts +0 -7
  49. package/build/utils/AnalysisStateManager.js +1 -1
  50. package/build/utils/connectionErrors.d.ts +10 -0
  51. package/build/utils/connectionErrors.js +10 -0
  52. package/build/utils/language-helper.js +24 -3
  53. package/build/utils/progress.d.ts +1 -1
  54. package/build/utils/progress.js +1 -1
  55. package/build/utils/rebaselineSnapshots.d.ts +1 -1
  56. package/build/utils/rebaselineSnapshots.js +6 -16
  57. package/build/utils/reuseRouting.d.ts +10 -0
  58. package/build/utils/reuseRouting.js +15 -0
  59. package/build/utils/runContextGauge.d.ts +27 -0
  60. package/build/utils/runContextGauge.js +181 -0
  61. package/build/utils/skyrampMdContent.d.ts +1 -1
  62. package/build/utils/skyrampMdContent.js +1 -1
  63. package/build/utils/skyrampSdkVersion.d.ts +9 -0
  64. package/build/utils/skyrampSdkVersion.js +16 -0
  65. package/build/utils/testDependencyPolicy.js +21 -0
  66. package/build/utils/testExecutionRecord.d.ts +5 -1
  67. package/build/utils/testExecutionRecord.js +3 -1
  68. package/build/utils/testFileClassification.d.ts +8 -0
  69. package/build/utils/testFileClassification.js +36 -3
  70. package/build/utils/utils-verify/action-key.d.ts +42 -0
  71. package/build/utils/utils-verify/action-key.js +118 -36
  72. package/build/utils/utils-verify/action-sites.d.ts +32 -0
  73. package/build/utils/utils-verify/action-sites.js +202 -0
  74. package/build/utils/utils-verify/body-reach.js +2 -4
  75. package/build/utils/utils-verify/call-sites.d.ts +25 -6
  76. package/build/utils/utils-verify/call-sites.js +8 -5
  77. package/build/utils/utils-verify/index.d.ts +1 -0
  78. package/build/utils/utils-verify/index.js +1 -0
  79. package/build/utils/utils-verify/language-spec.d.ts +25 -0
  80. package/build/utils/utils-verify/language-spec.js +16 -2
  81. package/build/utils/utils-verify/parse.d.ts +10 -1
  82. package/build/utils/utils-verify/parse.js +19 -2
  83. package/build/utils/utils-verify/verify.d.ts +3 -2
  84. package/build/utils/utils-verify/verify.js +16 -18
  85. package/build/workspace/workspace.d.ts +72 -52
  86. package/build/workspace/workspace.js +12 -8
  87. package/package.json +1 -1
  88. package/plugin/prompts/testbot-task1.md +0 -2
  89. package/plugin/skills/fix-test-import-errors/SKILL.md +2 -1
  90. package/plugin/skills/run-test/SKILL.md +16 -0
  91. package/build/adapters/jestAdapter.d.ts +0 -14
  92. package/build/adapters/jestAdapter.js +0 -131
  93. package/build/adapters/mochaAdapter.d.ts +0 -13
  94. package/build/adapters/mochaAdapter.js +0 -93
  95. package/build/adapters/playwrightAdapter.d.ts +0 -17
  96. package/build/adapters/playwrightAdapter.js +0 -184
  97. package/build/adapters/pytestAdapter.d.ts +0 -15
  98. package/build/adapters/pytestAdapter.js +0 -119
  99. package/build/tools/runExistingTestsTool.d.ts +0 -138
  100. package/build/tools/runExistingTestsTool.js +0 -666
  101. package/build/types/ExternalTestExecution.d.ts +0 -67
  102. package/build/types/ExternalTestExecution.js +0 -8
  103. package/build/workspace/testSuites.d.ts +0 -20
  104. package/build/workspace/testSuites.js +0 -17
@@ -20,6 +20,9 @@ import { checkDefectsReported, checkIssueTraceability, } from "../recommendation
20
20
  import { checkExpectedOutcomeAfterExecution, } from "../recommendation/verifiers/expectedOutcome.js";
21
21
  import { findInvalidSourceCitations, findUnchangedFileClaims, listChangedFiles, listChangedFilesAcross, listChangedFilesAbs, } from "../utils/reportVerification.js";
22
22
  import { isPlanOnlyMode } from "../utils/planOnlyMode.js";
23
+ import { contextGaugeSentence } from "../utils/runContextGauge.js";
24
+ import { budgetExcuseRefusals, citesBudget, countBudgetExcuseRefusal, MAX_BUDGET_EXCUSE_REFUSALS, } from "./budgetExcuse.js";
25
+ import { countUiDeliveryRefusal, MAX_UI_DELIVERY_REFUSALS, traceExportAttempts, uiDeliveryRefusals, } from "../playwright/traceExportStore.js";
23
26
  import { getReportLanguage, isEnforcedReportLanguage, findLanguageViolations, findLanguageNearMisses, reportLanguageDisplayName, } from "../utils/reportLanguage.js";
24
27
  import { canonicalTestPath, findAssertionRecordByFileName, rederiveAssertionOutcome, } from "./code-refactor/assertion-state.js";
25
28
  import { rederiveReuse, deriveReuseFromDelivered, REUSE_SUBMIT_MAX_REFUSALS, reuseChainSkipped, samePath, } from "./code-refactor/reuse-state.js";
@@ -1124,6 +1127,29 @@ function skippedDeliveredTests(delivered, results) {
1124
1127
  !matching.some((row) => row.status !== "Skipped"));
1125
1128
  });
1126
1129
  }
1130
+ /** A planned test the plan typed `ui`. The plan's `testType` is free-form text —
1131
+ * the unapproved-entry check above says so — so it is trimmed and lower-cased
1132
+ * rather than compared against the enum. */
1133
+ function plannedUiTests(plannedTests) {
1134
+ return plannedTests.filter((test) => !!test?.plannedTestId?.trim() &&
1135
+ String(test?.scenario?.testType ?? "")
1136
+ .trim()
1137
+ .toLowerCase() === TestType.UI);
1138
+ }
1139
+ /** Planned UI tests with no entry in the report: the recordings that never
1140
+ * happened, named by the ids the agent would otherwise only have to answer for. */
1141
+ /** Every planned test with no entry in the report, whatever its type. The UI
1142
+ * filter below narrows this; the budget-excuse gate does not, because a dropped
1143
+ * contract test is dropped the same way a dropped recording is. */
1144
+ function undeliveredPlannedTests(plannedTests, delivered) {
1145
+ const shipped = new Set(delivered
1146
+ .map((test) => test.plannedTestId?.trim())
1147
+ .filter((id) => !!id));
1148
+ return plannedTests.filter((test) => !!test?.plannedTestId?.trim() && !shipped.has(test.plannedTestId.trim()));
1149
+ }
1150
+ function undeliveredPlannedUiTests(plannedTests, delivered) {
1151
+ return undeliveredPlannedTests(plannedUiTests(plannedTests), delivered);
1152
+ }
1127
1153
  /** Whether the file can be read. The written record outlives the file, and a
1128
1154
  * path that exists but cannot be opened owes no run either — the objection
1129
1155
  * would name a file nobody can look at. */
@@ -2166,9 +2192,111 @@ export function registerSubmitReportTool(server) {
2166
2192
  .map((test) => `${test.plannedTestId?.trim() || test.testId}`)
2167
2193
  .join(", ")} ${skippedTests.length === 1 ? "has" : "have"} no non-skipped execution result. ` +
2168
2194
  "Do not skip because you believe you are low on context or time, or because you speculate that another attempt will not help. " +
2195
+ contextGaugeSentence() +
2169
2196
  "Run each remaining test with skyramp_execute_test, include its Pass or Fail result in testResults, then call skyramp_submit_report again.");
2170
2197
  return errorResult;
2171
2198
  }
2199
+ // A planned UI test with no report entry is a recording that never happened.
2200
+ // `deliveredMatchesPlan` already raises it and already accepts an answer, and
2201
+ // that is how one run shipped a 14-test plan with ZERO UI tests: a single
2202
+ // skyramp_export_zip, a trace that came out polluted, and an answer blaming
2203
+ // the recorder. Refuse while the run has recorded fewer times than it planned
2204
+ // UI tests; past that it has genuinely tried each one and the answer stands.
2205
+ //
2206
+ // WHY THE RETRY IS NOT FUTILE, and why the message says so: a successful
2207
+ // skyramp_export_zip DRAINS the recorder and closes the browser, so the next
2208
+ // recording starts from an empty buffer. Nothing the agent can read says that,
2209
+ // and that run reasoned the opposite and stopped one call short of a clean
2210
+ // trace.
2211
+ const undeliveredUi = isPlanOnlyMode() || planUnreadable
2212
+ ? []
2213
+ : undeliveredPlannedUiTests(stateData.plan?.plannedTests ?? [], dedupedNewTests);
2214
+ const plannedUiCount = planUnreadable
2215
+ ? 0
2216
+ : plannedUiTests(stateData.plan?.plannedTests ?? []).length;
2217
+ // ATTEMPTS, so one more recording always answers a refusal and the gate opens
2218
+ // after at most `plannedUiCount` of them. The refusal cap is the second bound,
2219
+ // for the agent that resubmits having recorded nothing: this check is meant to
2220
+ // cost a recording, never the whole run.
2221
+ const recordings = traceExportAttempts();
2222
+ if (undeliveredUi.length > 0 &&
2223
+ recordings < plannedUiCount &&
2224
+ uiDeliveryRefusals() < MAX_UI_DELIVERY_REFUSALS) {
2225
+ const refusal = countUiDeliveryRefusal();
2226
+ const ids = undeliveredUi
2227
+ .map((test) => test.plannedTestId.trim())
2228
+ .join(", ");
2229
+ errorResult = toolError(`Cannot submit the report: ${undeliveredUi.length} registered UI plan test${undeliveredUi.length === 1 ? "" : "s"} ` +
2230
+ `${ids} ${undeliveredUi.length === 1 ? "has" : "have"} no entry in newTestsCreated, and this run recorded ` +
2231
+ `${recordings} trace${recordings === 1 ? "" : "s"} for ${plannedUiCount} planned UI test${plannedUiCount === 1 ? "" : "s"}. ` +
2232
+ "Do not skip because you believe you are low on context or time, or because you speculate that another attempt will not help. " +
2233
+ contextGaugeSentence() +
2234
+ "A successful skyramp_export_zip clears the recording buffer and closes the browser, so the next recording starts clean — " +
2235
+ "a trace that came out polluted is a reason to record again, not a reason to stop. " +
2236
+ "Record each remaining test in its own pass (browser_navigate to the start URL, the flow, skyramp_export_zip, skyramp_ui_test_generation), " +
2237
+ "add it to newTestsCreated with its plannedTestId, then call skyramp_submit_report again. " +
2238
+ `The plan and every answer you have given are in the state file and survive this refusal (refusal ${refusal} of ${MAX_UI_DELIVERY_REFUSALS}).`);
2239
+ return errorResult;
2240
+ }
2241
+ // A planned test dropped because the run says it was running out.
2242
+ //
2243
+ // `deliveredMatchesPlan` raises "planned but not delivered" and suggests
2244
+ // "add the test to the report, or record in the report why it was dropped".
2245
+ // Run 9dfc8405 recorded the reason seven times — "this run exhausted its
2246
+ // working budget" — and shipped 2 of 9 planned tests with the report
2247
+ // ACCEPTED. It wrote that at 46% of a 1M context window with no warning from
2248
+ // anywhere, and the word appears nowhere in its reasoning: the budget was not
2249
+ // a finding, it was a sentence that closed an objection.
2250
+ //
2251
+ // This refuses that ONE sentence, not a budget. A test blocked by something
2252
+ // outside the run answers and ships exactly as before, which is why the
2253
+ // message says what a reason that stands looks like.
2254
+ //
2255
+ // Answers from earlier calls count too: one that closed the objection on
2256
+ // submit N still closes it on N+1 without being re-sent, and a gate reading
2257
+ // only this call would be walked past by not repeating it.
2258
+ const answerText = new Map();
2259
+ const carriedAnswers = Array.isArray(stateData.reportObjections?.answeredObjections)
2260
+ ? stateData.reportObjections.answeredObjections
2261
+ : [];
2262
+ for (const entry of carriedAnswers) {
2263
+ const id = entry
2264
+ ?.objection?.objectionId;
2265
+ const answer = entry?.answer;
2266
+ if (typeof id === "string" && typeof answer === "string")
2267
+ answerText.set(id, answer);
2268
+ }
2269
+ for (const entry of params.answers ?? []) {
2270
+ if (typeof entry?.objectionId === "string" &&
2271
+ typeof entry?.answer === "string")
2272
+ answerText.set(entry.objectionId, entry.answer);
2273
+ }
2274
+ const droppedForBudget = isPlanOnlyMode() || planUnreadable
2275
+ ? []
2276
+ : undeliveredPlannedTests(stateData.plan?.plannedTests ?? [], dedupedNewTests).filter((test) => citesBudget(answerText.get(`deliveredMatchesPlan:${test.plannedTestId.trim()}`)));
2277
+ if (droppedForBudget.length > 0 &&
2278
+ budgetExcuseRefusals() < MAX_BUDGET_EXCUSE_REFUSALS) {
2279
+ const refusal = countBudgetExcuseRefusal();
2280
+ const ids = droppedForBudget
2281
+ .map((test) => test.plannedTestId.trim())
2282
+ .join(", ");
2283
+ const plural = droppedForBudget.length === 1;
2284
+ errorResult = toolError(`Cannot submit the report: ${droppedForBudget.length} planned test${plural ? "" : "s"} ` +
2285
+ `${ids} ${plural ? "is" : "are"} missing from newTestsCreated, and the answer given for ` +
2286
+ `${plural ? "it" : "each"} is that the run was short of budget, context or time. ` +
2287
+ "That is not a reason this report accepts. " +
2288
+ contextGaugeSentence() +
2289
+ "A reason that stands names something outside the run that stopped the test: a service that is not " +
2290
+ "running, a branch that no longer exists, a credential this run does not hold, a behaviour the " +
2291
+ "application will not reach. Running low is not one of those — it is a reason to write the test now, " +
2292
+ "with the rest of the report already built. " +
2293
+ `Write the missing test${plural ? "" : "s"}, add ${plural ? "it" : "them"} to newTestsCreated with the ` +
2294
+ "plannedTestId, and call skyramp_submit_report again — or replace the answer with what actually " +
2295
+ "stopped it. " +
2296
+ `The plan, the report and every answer you have given are in the state file and survive this refusal ` +
2297
+ `(refusal ${refusal} of ${MAX_BUDGET_EXCUSE_REFUSALS}).`);
2298
+ return errorResult;
2299
+ }
2172
2300
  // Report-time checks on what the customer will read. They need no plan, so a
2173
2301
  // run without one still gets them.
2174
2302
  const reportChecks = [
@@ -13,6 +13,9 @@ import { REBASELINE_SNAPSHOT_NAME_RE, looksLikeOnDiskBaselineName, } from "../..
13
13
  import { buildRenameStrategy, buildFileRenameStrategy, buildUpdateStrategy, buildRebaselineStrategy, buildRegenerateStrategy, buildDeleteStrategy, buildUpdateFileInstruction, buildRegenerateFileInstruction, } from "../../prompts/test-maintenance/actionsInstructions.js";
14
14
  import { buildModularizeFirstNextSteps, isMaintenanceRegenerable, isModularizeFirstHandOffTarget, regenerateCodeReuse, } from "../../prompts/reuse-hand-off.js";
15
15
  import { recordReuseHandOff } from "../code-refactor/reuse-state.js";
16
+ import { isPlanOnlyMode } from "../../utils/planOnlyMode.js";
17
+ import { runTestCall } from "../../skills/runTestSkill.js";
18
+ import { formatOwedFiles, testsOwedBeforeRun } from "./testsOwedBeforeRun.js";
16
19
  /**
17
20
  * Compute a suggested new filename when an endpoint is renamed.
18
21
  */
@@ -442,6 +445,34 @@ export function registerActionsTool(server) {
442
445
  if (renamed)
443
446
  r._suggestedNewFile = renamed;
444
447
  }
448
+ // ── Before-run gate on the tests this call will edit ──────────────
449
+ // Drift on the repository's own test must rest on a run, not on a
450
+ // reading of its source: without the pre-edit result the report cannot
451
+ // say what the edit changed, and the test lands at afterStatus=Unknown.
452
+ // The subject is what this call edits, and nothing wider: the gate used
453
+ // to sit in skyramp_analyze_test_health and demand a run for every
454
+ // external test in state, which is the first hundred files discovery
455
+ // found in glob order.
456
+ //
457
+ // A DELETE or a REGENERATE on an external test is report-only (the run
458
+ // edits nothing), and a related repo's entry is already a VERIFY by
459
+ // here, so neither owes a run.
460
+ const { owed, exemptLanguage } = testsOwedBeforeRun(recommendations, catalogByFile);
461
+ if (exemptLanguage > 0) {
462
+ logger.info(`Before-run gate: ${exemptLanguage} recommended external test(s) owe no run — skyramp_execute_test cannot run their language.`);
463
+ }
464
+ // A plan-only run has no application to run the tests against.
465
+ if (owed.length > 0 && !isPlanOnlyMode()) {
466
+ const filesList = formatOwedFiles(owed);
467
+ errorResult = toolError(`${owed.length} test${owed.length === 1 ? "" : "s"} you recommend editing ${owed.length === 1 ? "has" : "have"} no phase: "before" result recorded — never run, or a run was started and recorded no result: ${filesList}. ` +
468
+ `Run each file through skyramp_execute_test with phase: "before", stateFile: "${args.stateFile}"${args.repository ? `, repository: "${args.repository}"` : ""}: ${runTestCall()}. Then call ${TOOL_NAME} again. Nothing was recorded by this call.`);
469
+ return errorResult;
470
+ }
471
+ else if (owed.length > 0) {
472
+ // planOnlyMode is process-global and captured at prompt-render time, so
473
+ // a skip must be visible: no run means no evidence behind these edits.
474
+ logger.warning(`Before-run gate skipped in a plan-only run: ${owed.length} test(s) you recommend editing have no phase: "before" result recorded.`);
475
+ }
445
476
  const recommendedPaths = new Set(recommendations.map((r) => r.testFilePath));
446
477
  const ignoredTestFiles = testAnalysisResults
447
478
  .map((t) => t.testFile)
@@ -131,8 +131,8 @@ export declare const analyzeChangesOutputSchema: {
131
131
  analysedWorkspaceFile: z.ZodString;
132
132
  }, "strip", z.ZodTypeAny, {
133
133
  message: string;
134
- kind: "workspace-disagreement";
135
134
  serviceName: string;
135
+ kind: "workspace-disagreement";
136
136
  field: "authHeader" | "authScheme" | "authType";
137
137
  primaryWorkspaceFile: string;
138
138
  analysedWorkspaceFile: string;
@@ -140,8 +140,8 @@ export declare const analyzeChangesOutputSchema: {
140
140
  analysed: string;
141
141
  }, {
142
142
  message: string;
143
- kind: "workspace-disagreement";
144
143
  serviceName: string;
144
+ kind: "workspace-disagreement";
145
145
  field: "authHeader" | "authScheme" | "authType";
146
146
  primaryWorkspaceFile: string;
147
147
  analysedWorkspaceFile: string;
@@ -156,8 +156,8 @@ export declare const analyzeChangesOutputSchema: {
156
156
  openApiSpecPath?: string | undefined;
157
157
  authFindings?: {
158
158
  message: string;
159
- kind: "workspace-disagreement";
160
159
  serviceName: string;
160
+ kind: "workspace-disagreement";
161
161
  field: "authHeader" | "authScheme" | "authType";
162
162
  primaryWorkspaceFile: string;
163
163
  analysedWorkspaceFile: string;
@@ -172,8 +172,8 @@ export declare const analyzeChangesOutputSchema: {
172
172
  openApiSpecPath?: string | undefined;
173
173
  authFindings?: {
174
174
  message: string;
175
- kind: "workspace-disagreement";
176
175
  serviceName: string;
176
+ kind: "workspace-disagreement";
177
177
  field: "authHeader" | "authScheme" | "authType";
178
178
  primaryWorkspaceFile: string;
179
179
  analysedWorkspaceFile: string;
@@ -1,13 +1,2 @@
1
1
  import { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js";
2
- /**
3
- * Names of services whose workspace.yml records a runnable external-test suite
4
- * (a reader-backed framework — playwright/pytest/jest/vitest/mocha — +
5
- * a `runtimeDetails.testSuites` entry with a testRunCommand) — i.e. what skyramp_run_existing_tests can
6
- * actually run. Used to scope the confirm gate so it only fires on repos that
7
- * went through Test Environment Setup. Checks each service's suites via
8
- * `resolveSuites` (rather than calling resolveRunConfig) so every
9
- * `runtimeDetails.testSuites` entry is recognized. Returns [] on any read/parse
10
- * error so a config problem can never block drift analysis.
11
- */
12
- export declare function runnableTestEnvServices(repositoryPath: string): Promise<string[]>;
13
2
  export declare function registerAnalyzeTestHealthTool(server: McpServer): void;
@@ -5,9 +5,6 @@ import { AnalyticsService } from "../../services/AnalyticsService.js";
5
5
  import { buildDriftAnalysisPrompt } from "../../prompts/test-maintenance/drift-analysis-prompt.js";
6
6
  import { TestSource } from "../../types/TestAnalysis.js";
7
7
  import { toolError } from "../../utils/utils.js";
8
- import { WorkspaceConfigManager } from "../../workspace/workspace.js";
9
- import { resolveSuites } from "../../workspace/testSuites.js";
10
- import { frameworkHasReader } from "../runExistingTestsTool.js";
11
8
  const TOOL_NAME = "skyramp_analyze_test_health";
12
9
  // UI test frameworks. "playwright-ui" and "cypress-ui" are the browser-interaction
13
10
  // variants detected when the file uses the page fixture (not just the request fixture).
@@ -38,27 +35,6 @@ function isUiTest(t) {
38
35
  UI_TEST_EXTENSIONS.test(t.testFile) ||
39
36
  isUiSnap);
40
37
  }
41
- /**
42
- * Names of services whose workspace.yml records a runnable external-test suite
43
- * (a reader-backed framework — playwright/pytest/jest/vitest/mocha — +
44
- * a `runtimeDetails.testSuites` entry with a testRunCommand) — i.e. what skyramp_run_existing_tests can
45
- * actually run. Used to scope the confirm gate so it only fires on repos that
46
- * went through Test Environment Setup. Checks each service's suites via
47
- * `resolveSuites` (rather than calling resolveRunConfig) so every
48
- * `runtimeDetails.testSuites` entry is recognized. Returns [] on any read/parse
49
- * error so a config problem can never block drift analysis.
50
- */
51
- export async function runnableTestEnvServices(repositoryPath) {
52
- try {
53
- const config = await new WorkspaceConfigManager(repositoryPath).read();
54
- return (config.services ?? [])
55
- .filter((s) => resolveSuites(s).some((suite) => frameworkHasReader(suite.framework) && !!suite.testRunCommand))
56
- .map((s) => s.serviceName);
57
- }
58
- catch {
59
- return [];
60
- }
61
- }
62
38
  export function registerAnalyzeTestHealthTool(server) {
63
39
  server.registerTool(TOOL_NAME, {
64
40
  annotations: {
@@ -111,44 +87,12 @@ export function registerAnalyzeTestHealthTool(server) {
111
87
  const skyrampCount = existingTests.filter((t) => t.source !== TestSource.External).length;
112
88
  const externalCount = existingTests.length - skyrampCount;
113
89
  logger.info(`Loaded ${skyrampCount} Skyramp + ${externalCount} relevant external tests from state file`);
114
- // ── External-test confirm gate ────────────────────────────────────
115
- // When the diff touches the repo's OWN [external] tests AND the workspace
116
- // has a runnable test-env config (runtimeDetails.test*), the drift
117
- // assessment must be grounded in a real run of those tests
118
- // (skyramp_run_existing_tests mode:confirm), not the agent's guess from
119
- // source. Prompt prose alone proved model-fragile — agents read the config
120
- // and still skipped the confirm, leaving the repo's own tests at
121
- // afterStatus=Unknown — so enforce ordering here: refuse to produce the
122
- // drift prompt until a confirm run has been recorded. Loop-safe:
123
- // skyramp_run_existing_tests persists an externalTestResults record for
124
- // EVERY outcome (pass/fail/skip/unhealthy), so one confirm call clears the
125
- // gate. Only a confirm-mode record counts — a verify run (or any future
126
- // record type) must not bypass the required confirm pass.
127
- // Scoped by runnableTestEnvServices so it never fires on repos without
128
- // test-env setup, and fail-open on any config error.
129
- const confirmAttempted = (stateData.externalTestResults ?? []).some((r) => r.mode === "confirm");
130
- const runnableServices = externalCount > 0 && !confirmAttempted
131
- ? await runnableTestEnvServices(repositoryPath)
132
- : [];
133
- if (runnableServices.length > 0) {
134
- const externalFiles = existingTests
135
- .filter((t) => t.source === TestSource.External)
136
- .map((t) => t.testFile);
137
- // Cap the enumerated examples — a broad diff can mark dozens of external
138
- // tests, and the agent already has the full [external] list from
139
- // skyramp_analyze_changes.
140
- const shown = externalFiles.slice(0, 8);
141
- const filesList = shown.map((f) => `"${f}"`).join(", ") +
142
- (externalFiles.length > shown.length
143
- ? `, …(+${externalFiles.length - shown.length} more)`
144
- : "");
145
- // With >1 runnable service the tool needs an explicit `service`.
146
- const serviceHint = runnableServices.length > 1
147
- ? ` Multiple runnable test services are configured — pass service: "${runnableServices[0]}".`
148
- : "";
149
- return toolError(`This repository has a configured test environment (runtimeDetails.test*) and ${externalCount} external test${externalCount === 1 ? "" : "s"} relevant to this diff, but skyramp_run_existing_tests has not run yet. Run the repo's OWN tests BEFORE assessing drift so the assessment is grounded in real pass/fail (otherwise these tests are left at status Unknown). ` +
150
- `Call skyramp_run_existing_tests with mode: "confirm", stateFile: "${stateManager.getStatePath()}"${args.repository ? `, repository: "${args.repository}"` : ""}, and testSelectors set to the [external] test files skyramp_analyze_changes marked (e.g. ${filesList}).${serviceHint} Then call skyramp_analyze_test_health again.`);
151
- }
90
+ // The repository's own tests, for the run results the drift prompt
91
+ // renders. A file with no `phase: "before"` run reaches the prompt as
92
+ // not run: the gate that demanded a run for every one of them moved to
93
+ // skyramp_actions, where the recommendations say which tests this run
94
+ // will edit.
95
+ const externalTests = existingTests.filter((t) => t.source === TestSource.External);
152
96
  // Delete stale diff files only — state files must remain for the caller's skyramp_actions call.
153
97
  try {
154
98
  await StateManager.cleanupOldFiles(24, undefined, []);
@@ -186,7 +130,7 @@ export function registerAnalyzeTestHealthTool(server) {
186
130
  const allRepoPaths = [
187
131
  ...new Set((await stateManager.listRepoCheckouts()).map((c) => c.root)),
188
132
  ];
189
- const promptText = buildDriftAnalysisPrompt(stateManager.getStatePath(), apiTests.map((t) => ({ testFile: t.testFile, source: t.source })), uiDriftParams, allRepoPaths, stateData?.externalTestResults);
133
+ const promptText = buildDriftAnalysisPrompt(stateManager.getStatePath(), apiTests.map((t) => ({ testFile: t.testFile, source: t.source })), uiDriftParams, allRepoPaths, externalTests.flatMap((t) => t.executionBefore ? [t.executionBefore] : []));
190
134
  return {
191
135
  structuredContent: { prompt: promptText },
192
136
  content: [
@@ -0,0 +1,28 @@
1
+ import { type DriftRecommendation, type TestAnalysisResult } from "../../types/TestAnalysis.js";
2
+ /** One catalogued test, narrowed to what the gate reads. */
3
+ export type GateCatalogRow = Pick<TestAnalysisResult, "testFile" | "language" | "source" | "executionBefore">;
4
+ /** One recommendation, narrowed to what the gate reads. */
5
+ export type GateRecommendation = Pick<DriftRecommendation, "testFilePath" | "action" | "reportOnly">;
6
+ export interface TestsOwedBeforeRun {
7
+ /** Files this call would edit that have no usable pre-edit run. */
8
+ owed: string[];
9
+ /** Owed no run because skyramp_execute_test cannot run that language. */
10
+ exemptLanguage: number;
11
+ }
12
+ /**
13
+ * The tests this call edits that have no `phase: "before"` result.
14
+ *
15
+ * Only UPDATE reaches here. An external REGENERATE or DELETE is turned into a
16
+ * report-only recommendation before the gate runs, and a report-only row edits
17
+ * nothing, so neither owes a run.
18
+ *
19
+ * A recommendation with no catalog row is deliberately NOT gated. `skyramp_actions`
20
+ * accepts a real test file discovery missed and records a verdict for it rather than
21
+ * dropping it (SKYR-3906/SKYR-3938, the B1 test). Refusing it here for want of a
22
+ * before-run would reverse that: it has no catalog row, so it never has a recorded
23
+ * run, and the gate would refuse it on every call. Gating it is a change to B1's
24
+ * contract, not to this one.
25
+ */
26
+ export declare function testsOwedBeforeRun(recommendations: readonly GateRecommendation[], catalogByFile: ReadonlyMap<string, GateCatalogRow>): TestsOwedBeforeRun;
27
+ /** Every file, quoted, one list — the agent must run all of them. */
28
+ export declare function formatOwedFiles(owed: readonly string[]): string;
@@ -0,0 +1,53 @@
1
+ import { DriftAction, TestSource, } from "../../types/TestAnalysis.js";
2
+ import { ProgrammingLanguage } from "../../types/TestTypes.js";
3
+ import { TestExecutionStatus } from "../../types/TestExecution.js";
4
+ /**
5
+ * The tests this call edits that have no `phase: "before"` result.
6
+ *
7
+ * Only UPDATE reaches here. An external REGENERATE or DELETE is turned into a
8
+ * report-only recommendation before the gate runs, and a report-only row edits
9
+ * nothing, so neither owes a run.
10
+ *
11
+ * A recommendation with no catalog row is deliberately NOT gated. `skyramp_actions`
12
+ * accepts a real test file discovery missed and records a verdict for it rather than
13
+ * dropping it (SKYR-3906/SKYR-3938, the B1 test). Refusing it here for want of a
14
+ * before-run would reverse that: it has no catalog row, so it never has a recorded
15
+ * run, and the gate would refuse it on every call. Gating it is a change to B1's
16
+ * contract, not to this one.
17
+ */
18
+ export function testsOwedBeforeRun(recommendations, catalogByFile) {
19
+ const runnableLanguages = Object.values(ProgrammingLanguage);
20
+ let exemptLanguage = 0;
21
+ const owed = recommendations
22
+ .filter((rec) => {
23
+ if (rec.reportOnly)
24
+ return false;
25
+ if (rec.action !== DriftAction.Update)
26
+ return false;
27
+ const row = catalogByFile.get(rec.testFilePath);
28
+ if (!row || row.source !== TestSource.External)
29
+ return false;
30
+ // Unknown is the placeholder skyramp_execute_test reserves before it
31
+ // spawns the runner. A run that died there left no evidence, so the
32
+ // placeholder must not read as a recorded baseline.
33
+ if (row.executionBefore &&
34
+ row.executionBefore.status !== TestExecutionStatus.Unknown)
35
+ return false;
36
+ // A language skyramp_execute_test does not accept can never record a
37
+ // before-run. No name check follows it: discovery admits an external file
38
+ // only when its name identifies a test or it is a colocated snapshot, and
39
+ // a .snap is exempt by language here, so a second rule would be a copy
40
+ // that can go out of sync.
41
+ if (!runnableLanguages.includes(row.language)) {
42
+ exemptLanguage++;
43
+ return false;
44
+ }
45
+ return true;
46
+ })
47
+ .map((rec) => rec.testFilePath);
48
+ return { owed, exemptLanguage };
49
+ }
50
+ /** Every file, quoted, one list — the agent must run all of them. */
51
+ export function formatOwedFiles(owed) {
52
+ return owed.map((f) => `"${f}"`).join(", ");
53
+ }
@@ -142,7 +142,7 @@ For detailed documentation visit: https://www.skyramp.dev/docs/load-test/advance
142
142
  const sessionAppendix = savedSession
143
143
  ? `\n\nPlaywright session storage saved to: ${savedSession}\nRe-use it by:\n` +
144
144
  `• Pass \`playwrightStoragePath: "${savedSession}"\` to skyramp_start_trace_collection for future recordings (skips login).\n` +
145
- `• Generated tests that reference \`storageState: "${savedSession}"\` will auto-mount the file when run via skyramp_execute_test.`
145
+ `• Generated tests that reference \`storageState: "${savedSession}"\` read it from that path when run via skyramp_execute_test.`
146
146
  : "";
147
147
  errorResult = {
148
148
  content: [
@@ -927,15 +927,15 @@ export declare const chainingRefSchema: z.ZodObject<{
927
927
  }, "strip", z.ZodTypeAny, {
928
928
  sourceStep: number;
929
929
  sourceField: string;
930
- sourceLocation: "body" | "header" | "cookie";
930
+ sourceLocation: "cookie" | "body" | "header";
931
931
  targetParam: string;
932
- targetLocation: "path" | "body" | "header" | "cookie" | "query";
932
+ targetLocation: "path" | "cookie" | "body" | "header" | "query";
933
933
  }, {
934
934
  sourceStep: number;
935
935
  sourceField: string;
936
- sourceLocation: "body" | "header" | "cookie";
936
+ sourceLocation: "cookie" | "body" | "header";
937
937
  targetParam: string;
938
- targetLocation: "path" | "body" | "header" | "cookie" | "query";
938
+ targetLocation: "path" | "cookie" | "body" | "header" | "query";
939
939
  }>;
940
940
  export declare const scenarioStepSchema: z.ZodObject<{
941
941
  order: z.ZodNumber;
@@ -957,15 +957,15 @@ export declare const scenarioStepSchema: z.ZodObject<{
957
957
  }, "strip", z.ZodTypeAny, {
958
958
  sourceStep: number;
959
959
  sourceField: string;
960
- sourceLocation: "body" | "header" | "cookie";
960
+ sourceLocation: "cookie" | "body" | "header";
961
961
  targetParam: string;
962
- targetLocation: "path" | "body" | "header" | "cookie" | "query";
962
+ targetLocation: "path" | "cookie" | "body" | "header" | "query";
963
963
  }, {
964
964
  sourceStep: number;
965
965
  sourceField: string;
966
- sourceLocation: "body" | "header" | "cookie";
966
+ sourceLocation: "cookie" | "body" | "header";
967
967
  targetParam: string;
968
- targetLocation: "path" | "body" | "header" | "cookie" | "query";
968
+ targetLocation: "path" | "cookie" | "body" | "header" | "query";
969
969
  }>, z.ZodArray<z.ZodObject<{
970
970
  sourceStep: z.ZodNumber;
971
971
  sourceField: z.ZodString;
@@ -975,15 +975,15 @@ export declare const scenarioStepSchema: z.ZodObject<{
975
975
  }, "strip", z.ZodTypeAny, {
976
976
  sourceStep: number;
977
977
  sourceField: string;
978
- sourceLocation: "body" | "header" | "cookie";
978
+ sourceLocation: "cookie" | "body" | "header";
979
979
  targetParam: string;
980
- targetLocation: "path" | "body" | "header" | "cookie" | "query";
980
+ targetLocation: "path" | "cookie" | "body" | "header" | "query";
981
981
  }, {
982
982
  sourceStep: number;
983
983
  sourceField: string;
984
- sourceLocation: "body" | "header" | "cookie";
984
+ sourceLocation: "cookie" | "body" | "header";
985
985
  targetParam: string;
986
- targetLocation: "path" | "body" | "header" | "cookie" | "query";
986
+ targetLocation: "path" | "cookie" | "body" | "header" | "query";
987
987
  }>, "many">]>>;
988
988
  bodyMustInclude: z.ZodOptional<z.ZodArray<z.ZodString, "many">>;
989
989
  }, "strip", z.ZodTypeAny, {
@@ -1000,15 +1000,15 @@ export declare const scenarioStepSchema: z.ZodObject<{
1000
1000
  chainsFrom?: {
1001
1001
  sourceStep: number;
1002
1002
  sourceField: string;
1003
- sourceLocation: "body" | "header" | "cookie";
1003
+ sourceLocation: "cookie" | "body" | "header";
1004
1004
  targetParam: string;
1005
- targetLocation: "path" | "body" | "header" | "cookie" | "query";
1005
+ targetLocation: "path" | "cookie" | "body" | "header" | "query";
1006
1006
  } | {
1007
1007
  sourceStep: number;
1008
1008
  sourceField: string;
1009
- sourceLocation: "body" | "header" | "cookie";
1009
+ sourceLocation: "cookie" | "body" | "header";
1010
1010
  targetParam: string;
1011
- targetLocation: "path" | "body" | "header" | "cookie" | "query";
1011
+ targetLocation: "path" | "cookie" | "body" | "header" | "query";
1012
1012
  }[] | undefined;
1013
1013
  bodyMustInclude?: string[] | undefined;
1014
1014
  }, {
@@ -1025,15 +1025,15 @@ export declare const scenarioStepSchema: z.ZodObject<{
1025
1025
  chainsFrom?: {
1026
1026
  sourceStep: number;
1027
1027
  sourceField: string;
1028
- sourceLocation: "body" | "header" | "cookie";
1028
+ sourceLocation: "cookie" | "body" | "header";
1029
1029
  targetParam: string;
1030
- targetLocation: "path" | "body" | "header" | "cookie" | "query";
1030
+ targetLocation: "path" | "cookie" | "body" | "header" | "query";
1031
1031
  } | {
1032
1032
  sourceStep: number;
1033
1033
  sourceField: string;
1034
- sourceLocation: "body" | "header" | "cookie";
1034
+ sourceLocation: "cookie" | "body" | "header";
1035
1035
  targetParam: string;
1036
- targetLocation: "path" | "body" | "header" | "cookie" | "query";
1036
+ targetLocation: "path" | "cookie" | "body" | "header" | "query";
1037
1037
  }[] | undefined;
1038
1038
  bodyMustInclude?: string[] | undefined;
1039
1039
  }>;
@@ -1184,14 +1184,14 @@ export declare const repositoryAnalysisSchema: z.ZodObject<{
1184
1184
  path: string;
1185
1185
  version: string;
1186
1186
  authType: string;
1187
- endpointCount: number;
1188
1187
  baseUrl: string;
1188
+ endpointCount: number;
1189
1189
  }, {
1190
1190
  path: string;
1191
1191
  version: string;
1192
1192
  authType: string;
1193
- endpointCount: number;
1194
1193
  baseUrl: string;
1194
+ endpointCount: number;
1195
1195
  }>, "many">;
1196
1196
  playwrightRecordings: z.ZodArray<z.ZodObject<{
1197
1197
  path: z.ZodString;
@@ -1225,8 +1225,8 @@ export declare const repositoryAnalysisSchema: z.ZodObject<{
1225
1225
  path: string;
1226
1226
  version: string;
1227
1227
  authType: string;
1228
- endpointCount: number;
1229
1228
  baseUrl: string;
1229
+ endpointCount: number;
1230
1230
  }[];
1231
1231
  playwrightRecordings: {
1232
1232
  path: string;
@@ -1244,8 +1244,8 @@ export declare const repositoryAnalysisSchema: z.ZodObject<{
1244
1244
  path: string;
1245
1245
  version: string;
1246
1246
  authType: string;
1247
- endpointCount: number;
1248
1247
  baseUrl: string;
1248
+ endpointCount: number;
1249
1249
  }[];
1250
1250
  playwrightRecordings: {
1251
1251
  path: string;
@@ -1389,6 +1389,9 @@ export declare const repositoryAnalysisSchema: z.ZodObject<{
1389
1389
  scanDepth: "quick" | "full";
1390
1390
  analysisScope?: AnalysisScope | undefined;
1391
1391
  };
1392
+ workspace: {
1393
+ baseUrl: string;
1394
+ };
1392
1395
  projectClassification: {
1393
1396
  projectType: "rest-api" | "frontend" | "full-stack" | "microservices" | "library" | "cli" | "other";
1394
1397
  primaryLanguage: string;
@@ -1416,8 +1419,8 @@ export declare const repositoryAnalysisSchema: z.ZodObject<{
1416
1419
  path: string;
1417
1420
  version: string;
1418
1421
  authType: string;
1419
- endpointCount: number;
1420
1422
  baseUrl: string;
1423
+ endpointCount: number;
1421
1424
  }[];
1422
1425
  playwrightRecordings: {
1423
1426
  path: string;
@@ -1431,9 +1434,6 @@ export declare const repositoryAnalysisSchema: z.ZodObject<{
1431
1434
  }[];
1432
1435
  notFound: string[];
1433
1436
  };
1434
- workspace: {
1435
- baseUrl: string;
1436
- };
1437
1437
  authentication: {
1438
1438
  method: string;
1439
1439
  configLocation: string;
@@ -1477,6 +1477,9 @@ export declare const repositoryAnalysisSchema: z.ZodObject<{
1477
1477
  scanDepth: "quick" | "full";
1478
1478
  analysisScope?: AnalysisScope | undefined;
1479
1479
  };
1480
+ workspace: {
1481
+ baseUrl: string;
1482
+ };
1480
1483
  projectClassification: {
1481
1484
  projectType: "rest-api" | "frontend" | "full-stack" | "microservices" | "library" | "cli" | "other";
1482
1485
  primaryLanguage: string;
@@ -1504,8 +1507,8 @@ export declare const repositoryAnalysisSchema: z.ZodObject<{
1504
1507
  path: string;
1505
1508
  version: string;
1506
1509
  authType: string;
1507
- endpointCount: number;
1508
1510
  baseUrl: string;
1511
+ endpointCount: number;
1509
1512
  }[];
1510
1513
  playwrightRecordings: {
1511
1514
  path: string;
@@ -1519,9 +1522,6 @@ export declare const repositoryAnalysisSchema: z.ZodObject<{
1519
1522
  }[];
1520
1523
  notFound: string[];
1521
1524
  };
1522
- workspace: {
1523
- baseUrl: string;
1524
- };
1525
1525
  authentication: {
1526
1526
  method: string;
1527
1527
  configLocation: string;