@skyramp/mcp 0.3.8 → 0.4.0-rc.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (229) hide show
  1. package/build/commands/commandLibrary.d.ts +1 -1
  2. package/build/commands/commandLibrary.js +3 -3
  3. package/build/commands/recommendTestsAndExecuteCommand.d.ts +1 -1
  4. package/build/commands/recommendTestsAndExecuteCommand.js +35 -20
  5. package/build/commands/testThisEndpointCommand.js +35 -19
  6. package/build/index.js +9 -3
  7. package/build/playwright/blueprintDigest.d.ts +15 -0
  8. package/build/playwright/blueprintDigest.js +152 -0
  9. package/build/playwright/blueprintDigestStore.d.ts +31 -0
  10. package/build/playwright/blueprintDigestStore.js +117 -0
  11. package/build/playwright/registerPlaywrightTools.js +60 -12
  12. package/build/playwright/traceRecordingPrompt.js +8 -7
  13. package/build/prompts/enhance-assertions/sharedAssertionRules.js +9 -8
  14. package/build/prompts/enhance-assertions/uiAssertionsPrompt.js +24 -2
  15. package/build/prompts/promptAssets.d.ts +20 -0
  16. package/build/prompts/promptAssets.js +55 -0
  17. package/build/prompts/sut-setup/modes/dockerComposePrompt.js +19 -5
  18. package/build/prompts/test-maintenance/actionsInstructions.d.ts +4 -0
  19. package/build/prompts/test-maintenance/actionsInstructions.js +14 -2
  20. package/build/prompts/test-maintenance/drift-analysis-prompt.d.ts +0 -10
  21. package/build/prompts/test-maintenance/drift-analysis-prompt.js +2 -11
  22. package/build/prompts/test-maintenance/uiDriftAnalysisSections.js +8 -4
  23. package/build/prompts/test-recommendation/diffExecutionPlan.d.ts +5 -22
  24. package/build/prompts/test-recommendation/diffExecutionPlan.js +37 -465
  25. package/build/prompts/test-recommendation/recommendationSections.d.ts +7 -17
  26. package/build/prompts/test-recommendation/recommendationSections.js +67 -309
  27. package/build/prompts/test-recommendation/recommendationShared.d.ts +19 -47
  28. package/build/prompts/test-recommendation/recommendationShared.js +49 -155
  29. package/build/prompts/test-recommendation/registerRecommendTestsPrompt.d.ts +0 -5
  30. package/build/prompts/test-recommendation/registerRecommendTestsPrompt.js +10 -153
  31. package/build/prompts/test-recommendation/test-recommendation-prompt.d.ts +2 -29
  32. package/build/prompts/test-recommendation/test-recommendation-prompt.js +32 -457
  33. package/build/prompts/testbot/planDeclarations.d.ts +6 -0
  34. package/build/prompts/testbot/planDeclarations.js +9 -0
  35. package/build/prompts/testbot/testbot-prompts.d.ts +8 -0
  36. package/build/prompts/testbot/testbot-prompts.js +256 -381
  37. package/build/recommendation/answers.d.ts +35 -0
  38. package/build/recommendation/answers.js +96 -0
  39. package/build/recommendation/registerPlan.d.ts +49 -0
  40. package/build/recommendation/registerPlan.js +117 -0
  41. package/build/recommendation/runVerifiers.d.ts +10 -0
  42. package/build/recommendation/runVerifiers.js +49 -0
  43. package/build/recommendation/subjectStep.d.ts +42 -0
  44. package/build/recommendation/subjectStep.js +86 -0
  45. package/build/recommendation/types.d.ts +163 -0
  46. package/build/recommendation/types.js +20 -0
  47. package/build/recommendation/verifierContracts.d.ts +382 -0
  48. package/build/recommendation/verifierContracts.js +263 -0
  49. package/build/recommendation/verifiers/changedFile.d.ts +2 -0
  50. package/build/recommendation/verifiers/changedFile.js +82 -0
  51. package/build/recommendation/verifiers/citedPath.d.ts +12 -0
  52. package/build/recommendation/verifiers/citedPath.js +35 -0
  53. package/build/recommendation/verifiers/coverage.d.ts +7 -0
  54. package/build/recommendation/verifiers/coverage.js +617 -0
  55. package/build/recommendation/verifiers/deliveredMatchesPlan.d.ts +11 -0
  56. package/build/recommendation/verifiers/deliveredMatchesPlan.js +33 -0
  57. package/build/recommendation/verifiers/endpointGrounded.d.ts +17 -0
  58. package/build/recommendation/verifiers/endpointGrounded.js +128 -0
  59. package/build/recommendation/verifiers/existingCoverage.d.ts +6 -0
  60. package/build/recommendation/verifiers/existingCoverage.js +51 -0
  61. package/build/recommendation/verifiers/expectedOutcome.d.ts +31 -0
  62. package/build/recommendation/verifiers/expectedOutcome.js +105 -0
  63. package/build/recommendation/verifiers/removedElementGuarded.d.ts +2 -0
  64. package/build/recommendation/verifiers/removedElementGuarded.js +57 -0
  65. package/build/recommendation/verifiers/reportedCategory.d.ts +26 -0
  66. package/build/recommendation/verifiers/reportedCategory.js +84 -0
  67. package/build/recommendation/verifiers/screenRoute.d.ts +10 -0
  68. package/build/recommendation/verifiers/screenRoute.js +118 -0
  69. package/build/recommendation/verifiers/statedDifference.d.ts +6 -0
  70. package/build/recommendation/verifiers/statedDifference.js +140 -0
  71. package/build/recommendation/verifiers/uiElementGrounded.d.ts +7 -0
  72. package/build/recommendation/verifiers/uiElementGrounded.js +318 -0
  73. package/build/resources/analysisResources.js +1 -114
  74. package/build/resources/testbotResource.js +23 -13
  75. package/build/services/ModularizationService.js +2 -1
  76. package/build/services/TestDiscoveryService.d.ts +3 -72
  77. package/build/services/TestDiscoveryService.js +10 -303
  78. package/build/services/containerEnv.d.ts +1 -1
  79. package/build/services/containerEnv.js +12 -0
  80. package/build/skills/fixTestImportErrorsSkill.d.ts +13 -0
  81. package/build/skills/fixTestImportErrorsSkill.js +20 -0
  82. package/build/toolNames.d.ts +1 -0
  83. package/build/toolNames.js +1 -0
  84. package/build/tools/code-refactor/enhanceAssertionsTool.js +3 -3
  85. package/build/tools/code-refactor/modularizationTool.js +2 -1
  86. package/build/tools/executeSkyrampTestTool.d.ts +80 -0
  87. package/build/tools/executeSkyrampTestTool.js +246 -19
  88. package/build/tools/generate-tests/generateBatchScenarioRestTool.js +6 -0
  89. package/build/tools/generate-tests/generateContractRestTool.js +3 -3
  90. package/build/tools/generate-tests/planGuard.d.ts +2 -2
  91. package/build/tools/generate-tests/planGuard.js +78 -18
  92. package/build/tools/one-click/oneClickTool.d.ts +0 -1
  93. package/build/tools/one-click/oneClickTool.js +0 -5
  94. package/build/tools/submitReportTool.d.ts +48 -42
  95. package/build/tools/submitReportTool.js +576 -193
  96. package/build/tools/test-management/actionsTool.js +72 -4
  97. package/build/tools/test-management/analyzeChangesTool.d.ts +144 -48
  98. package/build/tools/test-management/analyzeChangesTool.js +212 -1219
  99. package/build/tools/test-management/analyzeTestHealthTool.js +13 -24
  100. package/build/tools/test-management/index.d.ts +1 -0
  101. package/build/tools/test-management/index.js +1 -0
  102. package/build/tools/test-management/registerTestPlanTool.d.ts +795 -172
  103. package/build/tools/test-management/registerTestPlanTool.js +609 -542
  104. package/build/tools/test-management/resolveScreenTool.d.ts +75 -0
  105. package/build/tools/test-management/resolveScreenTool.js +289 -0
  106. package/build/types/BlueprintDigest.d.ts +34 -0
  107. package/build/types/BlueprintDigest.js +1 -0
  108. package/build/types/RepositoryAnalysis.d.ts +20 -1559
  109. package/build/types/RepositoryAnalysis.js +2 -58
  110. package/build/types/StepMethod.d.ts +40 -0
  111. package/build/types/StepMethod.js +77 -0
  112. package/build/types/TestAnalysis.d.ts +12 -0
  113. package/build/types/TestExecution.d.ts +4 -0
  114. package/build/types/TestRecommendation.d.ts +24 -24
  115. package/build/types/TestRecommendation.js +91 -89
  116. package/build/types/TestbotPromptOptions.d.ts +0 -4
  117. package/build/types/TestbotReport.d.ts +64 -2
  118. package/build/utils/AnalysisStateManager.d.ts +79 -113
  119. package/build/utils/AnalysisStateManager.js +147 -57
  120. package/build/utils/assertion-verify/api-shared-lints.js +1 -1
  121. package/build/utils/assertion-verify/metrics.js +85 -36
  122. package/build/utils/assertion-verify/ui-lints.d.ts +0 -5
  123. package/build/utils/assertion-verify/ui-lints.js +32 -0
  124. package/build/utils/branchDiff.d.ts +63 -31
  125. package/build/utils/branchDiff.js +242 -94
  126. package/build/utils/containedPath.d.ts +18 -0
  127. package/build/utils/containedPath.js +73 -0
  128. package/build/utils/dartRouteExtractor.d.ts +18 -34
  129. package/build/utils/dartRouteExtractor.js +101 -173
  130. package/build/utils/featureFlags.d.ts +12 -0
  131. package/build/utils/featureFlags.js +14 -0
  132. package/build/utils/frontendSelectors.d.ts +48 -27
  133. package/build/utils/frontendSelectors.js +241 -80
  134. package/build/utils/pathMatching.d.ts +2 -4
  135. package/build/utils/pathMatching.js +2 -4
  136. package/build/utils/planMatchKeys.d.ts +38 -47
  137. package/build/utils/planMatchKeys.js +143 -81
  138. package/build/utils/rebaselineSnapshots.d.ts +24 -0
  139. package/build/utils/rebaselineSnapshots.js +65 -0
  140. package/build/utils/removedUiElements.d.ts +22 -0
  141. package/build/utils/removedUiElements.js +106 -0
  142. package/build/utils/reportVerification.d.ts +2 -6
  143. package/build/utils/reportVerification.js +61 -2
  144. package/build/utils/screenRoutes.d.ts +66 -0
  145. package/build/utils/screenRoutes.js +727 -0
  146. package/build/utils/sourceRouteExtractor.js +320 -112
  147. package/build/utils/testFileClassification.d.ts +11 -2
  148. package/build/utils/testFileClassification.js +44 -2
  149. package/build/utils/testFixtures.d.ts +5 -0
  150. package/build/utils/testFixtures.js +13 -0
  151. package/build/utils/utils.d.ts +0 -1
  152. package/build/utils/utils.js +0 -11
  153. package/build/utils/versions.d.ts +3 -3
  154. package/build/utils/versions.js +1 -1
  155. package/build/workspace/workspace.d.ts +12 -12
  156. package/node_modules/playwright/lib/mcp/skyramp/assertHiddenTool.js +56 -0
  157. package/node_modules/playwright/lib/mcp/skyramp/assertTool.js +2 -1
  158. package/node_modules/playwright/lib/mcp/skyramp/loadTraceTool.js +10 -0
  159. package/node_modules/playwright/lib/mcp/skyramp/skyRampImport.js +4 -1
  160. package/node_modules/playwright/lib/mcp/skyramp/traceRecordingBackend.js +160 -1
  161. package/node_modules/playwright/lib/mcp/test/skyRampExport.js +4 -2
  162. package/node_modules/playwright/node_modules/playwright-core/lib/server/codegen/skyramp/jsonlReader.js +1 -0
  163. package/node_modules/playwright/node_modules/playwright-core/lib/server/recorder/recorderSignalProcessor.js +2 -0
  164. package/node_modules/playwright/node_modules/playwright-core/lib/server/recorder.js +5 -1
  165. package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/{index.-Id052Lr.js → index.B7KbSQcC.js} +1 -1
  166. package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/index.html +1 -1
  167. package/node_modules/playwright/node_modules/playwright-core/package.json +1 -1
  168. package/node_modules/playwright/node_modules/playwright-core/src/server/codegen/skyramp/jsonlReader.ts +1 -1
  169. package/node_modules/playwright/node_modules/playwright-core/src/server/recorder/recorderSignalProcessor.ts +7 -0
  170. package/node_modules/playwright/node_modules/playwright-core/src/server/recorder.ts +6 -1
  171. package/node_modules/playwright/package.json +1 -1
  172. package/package.json +4 -3
  173. package/plugin/.claude-plugin/plugin.json +8 -0
  174. package/plugin/plugin.json +6 -0
  175. package/plugin/prompts/declaring-a-plan.md +20 -0
  176. package/plugin/prompts/generate-tests/context-fetching.md +4 -0
  177. package/plugin/prompts/generate-tests/execution-plan.md +63 -0
  178. package/plugin/prompts/generate-tests/generation.md +108 -0
  179. package/plugin/prompts/generate-tests/path-parameters.md +1 -0
  180. package/plugin/prompts/generate-tests/reasoning-protocol.md +17 -0
  181. package/plugin/prompts/generate-tests/tool-workflow-variants.md +61 -0
  182. package/plugin/prompts/generate-tests/tool-workflows.md +65 -0
  183. package/plugin/prompts/plan-tests.md +42 -0
  184. package/plugin/prompts/testbot-task1.md +82 -0
  185. package/plugin/skills/fix-test-import-errors/SKILL.md +98 -0
  186. package/build/prompts/test-recommendation/analysisOutputPrompt.d.ts +0 -84
  187. package/build/prompts/test-recommendation/analysisOutputPrompt.js +0 -369
  188. package/build/prompts/test-recommendation/fullRepoCatalog.d.ts +0 -7
  189. package/build/prompts/test-recommendation/fullRepoCatalog.js +0 -283
  190. package/build/prompts/test-recommendation/scopeAssessment.d.ts +0 -81
  191. package/build/prompts/test-recommendation/scopeAssessment.js +0 -359
  192. package/build/recommendation/budgeters/diversityBalancedBudgeter.d.ts +0 -7
  193. package/build/recommendation/budgeters/diversityBalancedBudgeter.js +0 -105
  194. package/build/recommendation/budgeters/fixedNBudgeter.d.ts +0 -7
  195. package/build/recommendation/budgeters/fixedNBudgeter.js +0 -11
  196. package/build/recommendation/budgeters/shared.d.ts +0 -32
  197. package/build/recommendation/budgeters/shared.js +0 -246
  198. package/build/recommendation/discriminators.d.ts +0 -37
  199. package/build/recommendation/discriminators.js +0 -379
  200. package/build/recommendation/diversity.d.ts +0 -47
  201. package/build/recommendation/diversity.js +0 -101
  202. package/build/recommendation/planRanker.d.ts +0 -65
  203. package/build/recommendation/planRanker.js +0 -83
  204. package/build/recommendation/testFixtures.d.ts +0 -25
  205. package/build/recommendation/testFixtures.js +0 -45
  206. package/build/types/FrontendIntegration.d.ts +0 -28
  207. package/build/types/FrontendIntegration.js +0 -22
  208. package/build/types/Recommendation.d.ts +0 -146
  209. package/build/types/Recommendation.js +0 -74
  210. package/build/utils/changedRoutes.d.ts +0 -29
  211. package/build/utils/changedRoutes.js +0 -87
  212. package/build/utils/frontendIntegration.d.ts +0 -9
  213. package/build/utils/frontendIntegration.js +0 -243
  214. package/build/utils/importerHop.d.ts +0 -135
  215. package/build/utils/importerHop.js +0 -489
  216. package/build/utils/pathAffinityClassification.d.ts +0 -49
  217. package/build/utils/pathAffinityClassification.js +0 -180
  218. package/build/utils/pythonMountPrefixes.d.ts +0 -25
  219. package/build/utils/pythonMountPrefixes.js +0 -347
  220. package/build/utils/repoScanner.d.ts +0 -34
  221. package/build/utils/repoScanner.js +0 -300
  222. package/build/utils/routeParsers.d.ts +0 -95
  223. package/build/utils/routeParsers.js +0 -951
  224. package/build/utils/scenarioDrafting.d.ts +0 -92
  225. package/build/utils/scenarioDrafting.js +0 -951
  226. package/build/utils/subjectEndpoints.d.ts +0 -19
  227. package/build/utils/subjectEndpoints.js +0 -98
  228. package/build/utils/uiPageEnumerator.d.ts +0 -172
  229. package/build/utils/uiPageEnumerator.js +0 -474
@@ -1,6 +1,4 @@
1
1
  import { z } from "zod";
2
- import { SCENARIO_CATEGORIES } from "./TestRecommendation.js";
3
- import { TestType } from "./TestTypes.js";
4
2
  /**
5
3
  * Repository Analysis Types
6
4
  * Comprehensive structure for analyzing code repositories
@@ -95,63 +93,13 @@ export const scenarioStepSchema = z.object({
95
93
  .optional(),
96
94
  bodyMustInclude: z.array(z.string()).optional(),
97
95
  });
98
- export const draftedScenarioSchema = z.object({
99
- scenarioName: z.string(),
100
- description: z.string(),
101
- category: z.enum(SCENARIO_CATEGORIES),
102
- priority: z.enum(["high", "medium", "low"]),
103
- steps: z.array(scenarioStepSchema),
104
- chainingKeys: z.array(z.string()),
105
- requiresAuth: z.boolean(),
106
- estimatedComplexity: z.enum(["simple", "moderate", "complex"]),
107
- source: z.nativeEnum(ScenarioSource).optional(),
108
- testType: z.nativeEnum(TestType).optional(),
109
- bugCatchingTarget: z.string().optional(),
110
- });
111
96
  export const branchDiffContextSchema = z.object({
112
97
  currentBranch: z.string(),
113
98
  baseBranch: z.string(),
114
99
  changedFiles: z.array(z.string()),
115
- newEndpoints: z.array(z.object({
116
- path: z.string(),
117
- methods: z.array(z.object({
118
- method: z.string(),
119
- sourceFile: z.string(),
120
- interactionCount: z.number(),
121
- })),
122
- })),
123
- modifiedEndpoints: z.array(z.object({
124
- path: z.string(),
125
- methods: z.array(z.object({
126
- method: z.string(),
127
- sourceFile: z.string(),
128
- changeType: z.enum(["added", "modified", "removed"]),
129
- })),
130
- })),
131
- removedEndpoints: z
132
- .array(z.object({
133
- path: z.string(),
134
- methods: z.array(z.object({
135
- method: z.string(),
136
- sourceFile: z.string(),
137
- changeType: z.literal("removed"),
138
- })),
139
- }))
140
- .optional(),
141
- affectedServices: z.array(z.string()),
100
+ affectedServices: z.array(z.string()).optional(),
142
101
  summary: z.string().optional(),
143
102
  });
144
- export const routeDiscoveryInfoSchema = z.object({
145
- candidateFiles: z.array(z.string()),
146
- staticHints: z.array(z.object({
147
- path: z.string(),
148
- methods: z.array(z.string()),
149
- sourceFile: z.string(),
150
- })),
151
- openApiPaths: z.array(z.string()),
152
- routerMountContext: z.array(z.string()),
153
- diffFilePath: z.string().optional(),
154
- });
155
103
  export const analysisMetadataSchema = z.object({
156
104
  repositoryName: z.string(),
157
105
  analysisDate: z.string(),
@@ -196,7 +144,6 @@ export const repositoryAnalysisSchema = z.object({
196
144
  userFlows: z.array(z.string()),
197
145
  dataFlows: z.array(z.string()),
198
146
  integrationPatterns: z.array(z.string()),
199
- draftedScenarios: z.array(draftedScenarioSchema),
200
147
  }),
201
148
  artifacts: z.object({
202
149
  openApiSpecs: z.array(z.object({
@@ -218,10 +165,8 @@ export const repositoryAnalysisSchema = z.object({
218
165
  })),
219
166
  notFound: z.array(z.string()),
220
167
  }),
221
- apiEndpoints: z.object({
222
- totalCount: z.number(),
168
+ workspace: z.object({
223
169
  baseUrl: z.string(),
224
- endpoints: z.array(enrichedEndpointSchema),
225
170
  }),
226
171
  authentication: z.object({
227
172
  method: z.string(),
@@ -252,6 +197,5 @@ export const repositoryAnalysisSchema = z.object({
252
197
  estimatedCoverage: z.number().optional(),
253
198
  relevantExternalTestPaths: z.array(z.string()).optional(),
254
199
  }),
255
- routeDiscovery: routeDiscoveryInfoSchema.optional(),
256
200
  branchDiffContext: branchDiffContextSchema.optional(),
257
201
  });
@@ -0,0 +1,40 @@
1
+ import { z } from "zod";
2
+ /** What a scenario step's `method` may say, as three closed sets rather than free
3
+ * text. `HttpMethod` alone refused a UI recommendation, and `HttpMethod | UiVerb`
4
+ * would refuse an operation name — so the third member is what makes the union
5
+ * work: a call with no verb writes `OPERATION` and its name goes in `path`. */
6
+ /** Every method RFC 9110 defines. A verb missing here reads as a page
7
+ * interaction and skips four checks at once. */
8
+ export declare const HTTP_STEP_METHODS: readonly ["GET", "POST", "PUT", "PATCH", "DELETE", "HEAD", "OPTIONS", "TRACE", "CONNECT"];
9
+ /** The page interactions a step may name. TAKEN FROM WHAT RUNS HAVE WRITTEN: eleven
10
+ * are every distinct `method` on a `ui` or `e2e` candidate across the 1,858
11
+ * plan-state files in eval-logs, and `tap` is the twelfth because the schema and
12
+ * the skill offer it. A verb missing here is a REFUSED submission. */
13
+ export declare const UI_STEP_VERBS: readonly ["assert", "click", "drag", "drag-hold", "hover", "inspect", "navigate", "press", "release", "tap", "type", "wait"];
14
+ /** A call to something with no HTTP verb — a GraphQL operation, a Thrift or gRPC
15
+ * method. The operation's own name goes in `path`, and in `routes.operation`. */
16
+ export declare const OPERATION_STEP_METHOD = "OPERATION";
17
+ export declare const STEP_METHODS: readonly ["GET", "POST", "PUT", "PATCH", "DELETE", "HEAD", "OPTIONS", "TRACE", "CONNECT", "assert", "click", "drag", "drag-hold", "hover", "inspect", "navigate", "press", "release", "tap", "type", "wait", "OPERATION"];
18
+ export type HttpStepMethod = (typeof HTTP_STEP_METHODS)[number];
19
+ export type UiStepVerb = (typeof UI_STEP_VERBS)[number];
20
+ export type StepMethod = (typeof STEP_METHODS)[number];
21
+ /** One spelling of a method reduced to the canonical one, or `undefined` when it is
22
+ * none of the three sets. Case and surrounding space are ignored on input and
23
+ * fixed on store, so `" post"` and `"Click"` are the same step as `POST` and
24
+ * `click`. Without the trim, one trailing space decided which checks ran. */
25
+ export declare function normalizeStepMethod(raw: unknown): StepMethod | undefined;
26
+ /** Whether a method names a page interaction. Read off the canonical spelling,
27
+ * so a step that never went through the schema is judged the same way. */
28
+ export declare function isUiStepVerb(raw: unknown): boolean;
29
+ /** Whether a method names an HTTP request. */
30
+ export declare function isHttpStepMethod(raw: unknown): boolean;
31
+ /** The description every `method` field shares, so the three schemas cannot
32
+ * describe the same field differently. It says what the field IS; the accepted
33
+ * vocabulary is in the error the enum raises, which is where a wrong value gets
34
+ * told about it. */
35
+ export declare const STEP_METHOD_DESCRIPTION: string;
36
+ /**
37
+ * The field itself: normalised first, then checked, so a rejection is about the
38
+ * verb and never about its capitalisation.
39
+ */
40
+ export declare const stepMethodSchema: z.ZodEffects<z.ZodEnum<["GET", "POST", "PUT", "PATCH", "DELETE", "HEAD", "OPTIONS", "TRACE", "CONNECT", "assert", "click", "drag", "drag-hold", "hover", "inspect", "navigate", "press", "release", "tap", "type", "wait", "OPERATION"]>, "GET" | "POST" | "PUT" | "DELETE" | "PATCH" | "HEAD" | "OPTIONS" | "type" | "TRACE" | "CONNECT" | "assert" | "click" | "drag" | "drag-hold" | "hover" | "inspect" | "navigate" | "press" | "release" | "tap" | "wait" | "OPERATION", unknown>;
@@ -0,0 +1,77 @@
1
+ import { z } from "zod";
2
+ /** What a scenario step's `method` may say, as three closed sets rather than free
3
+ * text. `HttpMethod` alone refused a UI recommendation, and `HttpMethod | UiVerb`
4
+ * would refuse an operation name — so the third member is what makes the union
5
+ * work: a call with no verb writes `OPERATION` and its name goes in `path`. */
6
+ /** Every method RFC 9110 defines. A verb missing here reads as a page
7
+ * interaction and skips four checks at once. */
8
+ export const HTTP_STEP_METHODS = [
9
+ "GET",
10
+ "POST",
11
+ "PUT",
12
+ "PATCH",
13
+ "DELETE",
14
+ "HEAD",
15
+ "OPTIONS",
16
+ "TRACE",
17
+ "CONNECT",
18
+ ];
19
+ /** The page interactions a step may name. TAKEN FROM WHAT RUNS HAVE WRITTEN: eleven
20
+ * are every distinct `method` on a `ui` or `e2e` candidate across the 1,858
21
+ * plan-state files in eval-logs, and `tap` is the twelfth because the schema and
22
+ * the skill offer it. A verb missing here is a REFUSED submission. */
23
+ export const UI_STEP_VERBS = [
24
+ "assert",
25
+ "click",
26
+ "drag",
27
+ "drag-hold",
28
+ "hover",
29
+ "inspect",
30
+ "navigate",
31
+ "press",
32
+ "release",
33
+ "tap",
34
+ "type",
35
+ "wait",
36
+ ];
37
+ /** A call to something with no HTTP verb — a GraphQL operation, a Thrift or gRPC
38
+ * method. The operation's own name goes in `path`, and in `routes.operation`. */
39
+ export const OPERATION_STEP_METHOD = "OPERATION";
40
+ export const STEP_METHODS = [...HTTP_STEP_METHODS, ...UI_STEP_VERBS, OPERATION_STEP_METHOD];
41
+ const HTTP_BY_SPELLING = new Map(HTTP_STEP_METHODS.map((m) => [m.toLowerCase(), m]));
42
+ const UI_BY_SPELLING = new Map(UI_STEP_VERBS.map((v) => [v, v]));
43
+ /** One spelling of a method reduced to the canonical one, or `undefined` when it is
44
+ * none of the three sets. Case and surrounding space are ignored on input and
45
+ * fixed on store, so `" post"` and `"Click"` are the same step as `POST` and
46
+ * `click`. Without the trim, one trailing space decided which checks ran. */
47
+ export function normalizeStepMethod(raw) {
48
+ const spelling = String(raw ?? "").trim().toLowerCase();
49
+ if (!spelling)
50
+ return undefined;
51
+ return HTTP_BY_SPELLING.get(spelling) ?? UI_BY_SPELLING.get(spelling) ?? (spelling === "operation" ? OPERATION_STEP_METHOD : undefined);
52
+ }
53
+ /** Whether a method names a page interaction. Read off the canonical spelling,
54
+ * so a step that never went through the schema is judged the same way. */
55
+ export function isUiStepVerb(raw) {
56
+ const method = normalizeStepMethod(raw);
57
+ return method !== undefined && UI_STEP_VERBS.includes(method);
58
+ }
59
+ /** Whether a method names an HTTP request. */
60
+ export function isHttpStepMethod(raw) {
61
+ const method = normalizeStepMethod(raw);
62
+ return method !== undefined && HTTP_STEP_METHODS.includes(method);
63
+ }
64
+ const STEP_METHOD_HELP = `must be an HTTP method (${HTTP_STEP_METHODS.join(", ")}), a page interaction ` +
65
+ `(${UI_STEP_VERBS.join(", ")}), or "${OPERATION_STEP_METHOD}" for a call with no HTTP verb — ` +
66
+ `write the operation's own name in \`path\`, not here. Case does not matter.`;
67
+ /** The description every `method` field shares, so the three schemas cannot
68
+ * describe the same field differently. It says what the field IS; the accepted
69
+ * vocabulary is in the error the enum raises, which is where a wrong value gets
70
+ * told about it. */
71
+ export const STEP_METHOD_DESCRIPTION = "What this step does to the system: an HTTP method, a page interaction, or `OPERATION` for a call " +
72
+ "with no HTTP verb — for that one, write the operation's own name in `path`.";
73
+ /**
74
+ * The field itself: normalised first, then checked, so a rejection is about the
75
+ * verb and never about its capitalisation.
76
+ */
77
+ export const stepMethodSchema = z.preprocess((value) => (typeof value === "string" ? (normalizeStepMethod(value) ?? value.trim()) : value), z.enum(STEP_METHODS, { errorMap: () => ({ message: `method ${STEP_METHOD_HELP}` }) }));
@@ -41,6 +41,18 @@ export interface MaintenanceActionCore {
41
41
  * downstream stage that reasons about the edit needs both — notably the report-time
42
42
  * working-tree check, which otherwise calls a POM-backed UPDATE unbacked (SKYR-4129). */
43
43
  pomFile?: string;
44
+ /** Visual-snapshot baselines (toHaveScreenshot filenames, e.g. "page-001.png") an UPDATE
45
+ * refreshes because the diff changed how the captured page/element looks (SKYR-4298).
46
+ * The edit lands in the PNG under `<spec>-snapshots/`, not in the spec, so the
47
+ * report-time working-tree check must accept that PNG as the UPDATE's backing, and
48
+ * the final `skyramp_execute_test` must receive the list as `rebaselineSnapshots`. */
49
+ rebaselineSnapshots?: string[];
50
+ /** True when the UPDATE carries rebaselineSnapshots and no updateInstructions: the
51
+ * refresh is its whole maintenance, so no edit to the spec/POM is expected. When false
52
+ * (or absent) with baselines listed, the report-time check requires BOTH the edit and
53
+ * the rewritten PNG — a listed baseline must never exempt the selector edit the same
54
+ * verdict claimed (SKYR-3883 stays in force). */
55
+ rebaselineOnly?: boolean;
44
56
  }
45
57
  /** Normalized internal recommendation built from LLM-supplied args.recommendations. */
46
58
  export interface DriftRecommendation extends MaintenanceActionCore {
@@ -60,6 +60,10 @@ export interface TestExecutionOptions {
60
60
  playwrightSaveStoragePath?: string;
61
61
  dockerNetwork?: string;
62
62
  useHostNetwork?: boolean;
63
+ /** Visual-snapshot baselines (toHaveScreenshot filenames, e.g. "page-001.png") this run
64
+ * replaces instead of comparing against — forwarded to SmartPlaywright as
65
+ * SKYRAMP_UPDATE_SNAPSHOTS (SKYR-4298). Only for an intended UI change the diff explains. */
66
+ rebaselineSnapshots?: string[];
63
67
  }
64
68
  /**
65
69
  * Progress callback for reporting execution status
@@ -1,33 +1,33 @@
1
- /** Internal scenario-scoring tier — UPPERCASE, includes CRITICAL. */
2
- export declare enum PriorityTier {
3
- CRITICAL = "CRITICAL",
4
- HIGH = "HIGH",
5
- MEDIUM = "MEDIUM",
6
- LOW = "LOW"
7
- }
8
- /** How a scenario relates to the diff — tie-breaker within a priority tier. */
1
+ /** How a scenario relates to the diff. */
9
2
  export declare enum Novelty {
10
3
  NEW = "new",
11
4
  MODIFIED = "modified",
12
5
  EXISTING = "existing"
13
6
  }
14
- /** All categories including internal ones. */
7
+ /** Every category, and the only list there is. The plan and the report validate
8
+ * against this one: the report used to take a shorter list and rename the three
9
+ * above into it, which meant one test carried two category names. */
15
10
  export declare const SCENARIO_CATEGORIES: readonly ["new_endpoint", "bug_caught", "requirement_conflict", "business_rule", "security_boundary", "data_integrity", "breaking_change", "auth", "error_handling", "workflow", "data_validation", "crud"];
16
11
  export type ScenarioCategory = typeof SCENARIO_CATEGORIES[number];
17
- /** Categories valid for tool submissions (excludes internal-only categories). */
18
- export declare const TEST_CATEGORIES: readonly ["business_rule", "security_boundary", "data_integrity", "breaking_change", "auth", "error_handling", "workflow", "data_validation", "crud"];
19
- export type TestCategory = typeof TEST_CATEGORIES[number];
20
- /** Priority assignment for each category. */
21
- export declare const CATEGORY_PRIORITY: Record<ScenarioCategory, PriorityTier>;
22
- /** Map internal-only categories to their external equivalent for tool submission. */
23
- export declare function externalCategory(cat: ScenarioCategory): TestCategory;
12
+ /** What each category means, and whether it expects a failing test. The plan
13
+ * tool renders its `category` description from this, so the agent reads these
14
+ * sentences as it picks one. `expectsRed` is true only for the two naming a
15
+ * specific defect: a wider set would fire verifier 4 on ordinary candidates. */
16
+ export declare const CATEGORY_MEANINGS: Record<ScenarioCategory, {
17
+ meaning: string;
18
+ expectsRed: boolean;
19
+ }>;
20
+ /** The category list as the plan tool shows it, one line each, table order. */
21
+ export declare function categoryMenu(): string;
24
22
  /**
25
- * Categories whose scenarios target a specific identified defect — a code flaw
26
- * (`bug_caught`) or a stated requirement the code contradicts
27
- * (`requirement_conflict`). They survive external-test dedup: an existing test on
28
- * the same endpoint exercises the surface, not the flaw, so removing them would
29
- * drop the only test that fails on the defect.
23
+ * `category` is a free string whose schema description is an "e.g." list, so the
24
+ * agent picks the spelling: `Bug_Caught`, `bug-caught` and a trailing space each
25
+ * skipped verifier 4. Case, surrounding space and the `-`/`_`/space separator
26
+ * only — no alias list, since a category this misses is one nobody declared.
30
27
  */
31
- export declare const FLAW_TARGETING_CATEGORIES: readonly ["bug_caught", "requirement_conflict"];
32
- /** Whether `category` targets a specific identified defect (see {@link FLAW_TARGETING_CATEGORIES}). */
33
- export declare function isFlawTargetingCategory(category: ScenarioCategory | undefined): boolean;
28
+ export declare function normalizeCategory(raw: unknown): string;
29
+ /** Whether a candidate's `category` names a test expected to fail on the current
30
+ * code, in any spelling. Verifier 4 and the report's `expectedToFail` derivation
31
+ * read this one answer: comparing the field two ways made `Bug_Caught`
32
+ * bug-catching in one and ordinary in the other. */
33
+ export declare function categoryExpectsRed(raw: unknown): boolean;
@@ -1,113 +1,115 @@
1
- /** Internal scenario-scoring tier — UPPERCASE, includes CRITICAL. */
2
- export var PriorityTier;
3
- (function (PriorityTier) {
4
- PriorityTier["CRITICAL"] = "CRITICAL";
5
- PriorityTier["HIGH"] = "HIGH";
6
- PriorityTier["MEDIUM"] = "MEDIUM";
7
- PriorityTier["LOW"] = "LOW";
8
- })(PriorityTier || (PriorityTier = {}));
9
- /** How a scenario relates to the diff — tie-breaker within a priority tier. */
1
+ /** How a scenario relates to the diff. */
10
2
  export var Novelty;
11
3
  (function (Novelty) {
12
4
  Novelty["NEW"] = "new";
13
5
  Novelty["MODIFIED"] = "modified";
14
6
  Novelty["EXISTING"] = "existing";
15
7
  })(Novelty || (Novelty = {}));
16
- /** Internal-only categories (not submitted to tools). */
8
+ /** The three categories that say where a test came from or what defect it names,
9
+ * rather than what it proves. The server assigns `new_endpoint`; the agent
10
+ * declares the other two. */
17
11
  const INTERNAL_CATEGORIES = [
18
- "new_endpoint", // MEDIUM - diff-direct scenario; where a test came from, not a guarantee of a slot
19
- "bug_caught", // CRITICAL - tests targeting a specific <bug_found> flaw identified during enrichment
20
- // CRITICAL - tests asserting a requirement the PR title/description (or a
21
- // requirements file it references) states, which the implemented behavior
22
- // contradicts. Separate from bug_caught deliberately (SKYR-4291): labelled
23
- // bug_caught, a requirement-vs-code mismatch competed with the code-review
24
- // flaws for the same promotion and lost it on severity. Its own category means
25
- // its own carve-out here, first place in the promotion order SKYR-4275's bound
26
- // hands out, and its own coverage gate.
12
+ "new_endpoint", // a diff-direct draft: where it came from, not what it proves
13
+ "bug_caught", // targets a specific <bug_found> flaw identified during enrichment
14
+ // Its own category rather than a bug_caught: the two are found by different
15
+ // reads, and calling a contradicted requirement a code flaw sends the author to
16
+ // the wrong place to fix it.
27
17
  "requirement_conflict",
28
18
  ];
29
- /** External categories valid for tool submissions, ordered by priority. */
19
+ /** The categories that say what a test proves. */
30
20
  const CATEGORIES = [
31
- // HIGH priority
32
- "business_rule", // formula bugs, unique constraints, state machines — most common production failures
21
+ "business_rule", // formula bugs, unique constraints, state machines
33
22
  "security_boundary", // auth, permission, cross-user isolation, idempotency
34
23
  "data_integrity", // cascade deletes, orphan prevention, referential integrity
35
24
  "breaking_change", // route renames, auth migration, response shape changes
36
25
  "auth", // authentication and authorization flows
37
- "error_handling", // missing 404/422 guards — silent failures are real bugs
38
- // MEDIUM priority
26
+ "error_handling", // missing 404/422 guards
39
27
  "workflow", // cross-resource integration, user journeys
40
28
  "data_validation", // input validation and schema enforcement
41
- // LOW priority
42
29
  "crud", // basic create/read/update/delete operations
43
30
  ];
44
- /** All categories including internal ones. */
31
+ /** Every category, and the only list there is. The plan and the report validate
32
+ * against this one: the report used to take a shorter list and rename the three
33
+ * above into it, which meant one test carried two category names. */
45
34
  export const SCENARIO_CATEGORIES = [...INTERNAL_CATEGORIES, ...CATEGORIES];
46
- /** Categories valid for tool submissions (excludes internal-only categories). */
47
- export const TEST_CATEGORIES = CATEGORIES;
48
- /** Priority assignment for each category. */
49
- export const CATEGORY_PRIORITY = {
50
- // CRITICAL means "this test targets a specific identified flaw" — only bug_caught
51
- // asserts that. new_endpoint says where a test came from (a template fired on an
52
- // endpoint the diff added), not that it earns a guaranteed slot. Reserving the
53
- // top tier for it let server-drafted templates outrank anything an agent said,
54
- // by label rather than merit. (The registration enum is SCENARIO_CATEGORIES, so
55
- // an agent CAN submit new_endpoint; externalCategory maps it to crud/LOW for
56
- // report and merge purposes only.)
57
- //
58
- // It sits BELOW the categories that state what a test proves, not level with
59
- // them. Level was still too high: the last rank key is the candidateId, so among
60
- // same-tier candidates the ALPHABET decided. Template names start with the
61
- // resource (billing-, bookings-, export-, members-) and an agent names a scenario
62
- // for what it checks, so on cc15-org-reviewer-role all 8 templates outranked all
63
- // 12 candidates the agent wrote, and 2 of the 3 GENERATE slots went to the letter
64
- // "b" beating "o". MEDIUM, not LOW: a template on an endpoint the diff just added
65
- // is still worth more than a crud test on an endpoint that was always there. The
66
- // floor survives — when the agent offers nothing above MEDIUM, templates still
67
- // fill GENERATE.
68
- new_endpoint: PriorityTier.MEDIUM,
69
- bug_caught: PriorityTier.CRITICAL, // tests targeting a <bug_found> flaw — always in GENERATE
70
- // A stated requirement the implementation contradicts is the point of the PR,
71
- // so it sits in the same top tier as bug_caught. Being its own category, it is
72
- // carved out separately in planRanker and takes the FIRST slot of the promotion
73
- // bound SKYR-4275 sets — the code-review flaws promote into what is left, so a
74
- // requirement conflict is never the finding that loses on severity.
75
- requirement_conflict: PriorityTier.CRITICAL,
76
- business_rule: PriorityTier.HIGH, // formula/business-logic bugs are high priority
77
- security_boundary: PriorityTier.HIGH,
78
- data_integrity: PriorityTier.HIGH,
79
- breaking_change: PriorityTier.HIGH,
80
- auth: PriorityTier.HIGH,
81
- error_handling: PriorityTier.HIGH,
82
- workflow: PriorityTier.MEDIUM,
83
- data_validation: PriorityTier.MEDIUM,
84
- crud: PriorityTier.LOW,
35
+ /** What each category means, and whether it expects a failing test. The plan
36
+ * tool renders its `category` description from this, so the agent reads these
37
+ * sentences as it picks one. `expectsRed` is true only for the two naming a
38
+ * specific defect: a wider set would fire verifier 4 on ordinary candidates. */
39
+ export const CATEGORY_MEANINGS = {
40
+ new_endpoint: {
41
+ meaning: "set by the server's own drafts; do not assign it",
42
+ expectsRed: false,
43
+ },
44
+ bug_caught: {
45
+ meaning: "a flaw you found in the changed code; the test is expected to fail until it is fixed",
46
+ expectsRed: true,
47
+ },
48
+ requirement_conflict: {
49
+ meaning: "the code contradicts a requirement the PR states; expected to fail; reported as a critical or high issue",
50
+ expectsRed: true,
51
+ },
52
+ business_rule: {
53
+ meaning: "a formula, unique constraint or state machine the code is supposed to enforce",
54
+ expectsRed: false,
55
+ },
56
+ security_boundary: {
57
+ meaning: "auth, permissions, cross-user isolation or idempotency at a boundary",
58
+ expectsRed: false,
59
+ },
60
+ data_integrity: {
61
+ meaning: "cascade deletes, orphan prevention and referential integrity between resources",
62
+ expectsRed: false,
63
+ },
64
+ breaking_change: {
65
+ meaning: "a route rename, auth migration or response shape change that callers depend on",
66
+ expectsRed: false,
67
+ },
68
+ auth: {
69
+ meaning: "an authentication or authorization flow",
70
+ expectsRed: false,
71
+ },
72
+ error_handling: {
73
+ meaning: "a missing 404 or 422 guard on bad input",
74
+ expectsRed: false,
75
+ },
76
+ workflow: {
77
+ meaning: "a user journey across more than one resource",
78
+ expectsRed: false,
79
+ },
80
+ data_validation: {
81
+ meaning: "input validation and schema enforcement on a request",
82
+ expectsRed: false,
83
+ },
84
+ crud: {
85
+ meaning: "basic create, read, update or delete on one resource",
86
+ expectsRed: false,
87
+ },
85
88
  };
86
- /** Map internal-only categories to their external equivalent for tool submission. */
87
- export function externalCategory(cat) {
88
- if (cat === "new_endpoint")
89
- return "crud";
90
- if (cat === "bug_caught")
91
- return "business_rule";
92
- // The stated requirement IS the business rule the test asserts — same landing
93
- // spot as bug_caught, so a requirement-conflict test reads as a rule check in
94
- // the customer-facing report rather than an unrecognised label.
95
- if (cat === "requirement_conflict")
96
- return "business_rule";
97
- return cat;
89
+ /** The category list as the plan tool shows it, one line each, table order. */
90
+ export function categoryMenu() {
91
+ return SCENARIO_CATEGORIES.map((category) => {
92
+ const row = CATEGORY_MEANINGS[category];
93
+ return `- \`${category}\` — ${row.meaning}${row.expectsRed ? " (expects a failing test)" : ""}`;
94
+ }).join("\n");
98
95
  }
99
96
  /**
100
- * Categories whose scenarios target a specific identified defect — a code flaw
101
- * (`bug_caught`) or a stated requirement the code contradicts
102
- * (`requirement_conflict`). They survive external-test dedup: an existing test on
103
- * the same endpoint exercises the surface, not the flaw, so removing them would
104
- * drop the only test that fails on the defect.
97
+ * `category` is a free string whose schema description is an "e.g." list, so the
98
+ * agent picks the spelling: `Bug_Caught`, `bug-caught` and a trailing space each
99
+ * skipped verifier 4. Case, surrounding space and the `-`/`_`/space separator
100
+ * only — no alias list, since a category this misses is one nobody declared.
105
101
  */
106
- export const FLAW_TARGETING_CATEGORIES = [
107
- "bug_caught",
108
- "requirement_conflict",
109
- ];
110
- /** Whether `category` targets a specific identified defect (see {@link FLAW_TARGETING_CATEGORIES}). */
111
- export function isFlawTargetingCategory(category) {
112
- return category !== undefined && FLAW_TARGETING_CATEGORIES.includes(category);
102
+ export function normalizeCategory(raw) {
103
+ return String(raw ?? "")
104
+ .trim()
105
+ .toLowerCase()
106
+ .replace(/[-\s]+/g, "_");
107
+ }
108
+ /** Whether a candidate's `category` names a test expected to fail on the current
109
+ * code, in any spelling. Verifier 4 and the report's `expectedToFail` derivation
110
+ * read this one answer: comparing the field two ways made `Bug_Caught`
111
+ * bug-catching in one and ordinary in the other. */
112
+ export function categoryExpectsRed(raw) {
113
+ const row = CATEGORY_MEANINGS[normalizeCategory(raw)];
114
+ return row?.expectsRed === true;
113
115
  }
@@ -18,10 +18,6 @@ export interface TestbotPromptOptions {
18
18
  prDescription: string;
19
19
  repositoryPath: string;
20
20
  baseBranch?: string;
21
- maxRecommendations?: number;
22
- maxGenerate?: number;
23
- /** Reserved — accepted for API compat but not yet wired into the prompt. */
24
- maxCritical?: number;
25
21
  prNumber?: number;
26
22
  userPrompt?: string;
27
23
  services?: Service[];