@skyramp/mcp 0.3.1 → 0.3.2-rc.pom-2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (177) hide show
  1. package/build/index.js +2 -1
  2. package/build/prompts/code-reuse.d.ts +7 -1
  3. package/build/prompts/code-reuse.js +8 -5
  4. package/build/prompts/code-reuse.test.d.ts +1 -0
  5. package/build/prompts/code-reuse.test.js +62 -0
  6. package/build/prompts/pom-aware-code-reuse.d.ts +6 -1
  7. package/build/prompts/pom-aware-code-reuse.js +100 -53
  8. package/build/prompts/pom-aware-code-reuse.test.d.ts +1 -0
  9. package/build/prompts/pom-aware-code-reuse.test.js +11 -0
  10. package/build/prompts/test-recommendation/diffExecutionPlan.d.ts +4 -2
  11. package/build/prompts/test-recommendation/diffExecutionPlan.js +11 -65
  12. package/build/prompts/test-recommendation/fullRepoCatalog.d.ts +2 -2
  13. package/build/prompts/test-recommendation/recommendationSections.js +5 -2
  14. package/build/prompts/test-recommendation/scopeAssessment.js +1 -1
  15. package/build/prompts/test-recommendation/test-recommendation-prompt.d.ts +26 -1
  16. package/build/prompts/test-recommendation/test-recommendation-prompt.js +68 -56
  17. package/build/prompts/test-recommendation/test-recommendation-prompt.test.js +48 -11
  18. package/build/prompts/testbot/testbot-prompts.d.ts +1 -1
  19. package/build/prompts/testbot/testbot-prompts.js +67 -21
  20. package/build/prompts/testbot/testbot-prompts.test.js +44 -0
  21. package/build/recommendation/budgeters/diversityBalancedBudgeter.d.ts +7 -0
  22. package/build/recommendation/budgeters/diversityBalancedBudgeter.js +71 -0
  23. package/build/recommendation/budgeters/diversityBalancedBudgeter.test.d.ts +1 -0
  24. package/build/recommendation/budgeters/diversityBalancedBudgeter.test.js +75 -0
  25. package/build/recommendation/budgeters/fixedNBudgeter.d.ts +7 -0
  26. package/build/recommendation/budgeters/fixedNBudgeter.js +11 -0
  27. package/build/recommendation/budgeters/fixedNBudgeter.test.d.ts +1 -0
  28. package/build/recommendation/budgeters/fixedNBudgeter.test.js +66 -0
  29. package/build/recommendation/budgeters/shared.d.ts +19 -0
  30. package/build/recommendation/budgeters/shared.js +66 -0
  31. package/build/recommendation/discriminators.d.ts +31 -0
  32. package/build/recommendation/discriminators.js +355 -0
  33. package/build/recommendation/discriminators.test.d.ts +1 -0
  34. package/build/recommendation/discriminators.test.js +324 -0
  35. package/build/recommendation/diversity.d.ts +47 -0
  36. package/build/recommendation/diversity.js +101 -0
  37. package/build/recommendation/diversity.test.d.ts +1 -0
  38. package/build/recommendation/diversity.test.js +77 -0
  39. package/build/recommendation/planRanker.d.ts +50 -0
  40. package/build/recommendation/planRanker.js +67 -0
  41. package/build/recommendation/planRanker.test.d.ts +1 -0
  42. package/build/recommendation/planRanker.test.js +110 -0
  43. package/build/recommendation/testFixtures.d.ts +25 -0
  44. package/build/recommendation/testFixtures.js +45 -0
  45. package/build/resources/testbotResource.js +4 -1
  46. package/build/services/ScenarioGenerationService.d.ts +5 -0
  47. package/build/services/ScenarioGenerationService.js +16 -1
  48. package/build/services/ScenarioGenerationService.test.js +44 -0
  49. package/build/services/TestExecutionService.d.ts +15 -1
  50. package/build/services/TestExecutionService.js +210 -55
  51. package/build/services/TestExecutionService.test.js +397 -0
  52. package/build/services/TestGenerationService.js +19 -1
  53. package/build/services/TestGenerationService.test.js +58 -0
  54. package/build/tool-phases.js +1 -0
  55. package/build/toolNames.d.ts +19 -0
  56. package/build/toolNames.js +19 -0
  57. package/build/tools/code-refactor/codeReuseTool.d.ts +7 -0
  58. package/build/tools/code-refactor/codeReuseTool.js +130 -4
  59. package/build/tools/code-refactor/codeReuseTool.test.d.ts +1 -0
  60. package/build/tools/code-refactor/codeReuseTool.test.js +290 -0
  61. package/build/tools/executeSkyrampTestTool.js +8 -2
  62. package/build/tools/generate-tests/generateBatchScenarioRestTool.d.ts +6 -1
  63. package/build/tools/generate-tests/generateBatchScenarioRestTool.js +110 -17
  64. package/build/tools/generate-tests/generateBatchScenarioRestTool.test.js +147 -0
  65. package/build/tools/generate-tests/generateContractRestTool.js +11 -1
  66. package/build/tools/generate-tests/generateIntegrationRestTool.js +24 -1
  67. package/build/tools/generate-tests/generateIntegrationRestTool.test.d.ts +1 -0
  68. package/build/tools/generate-tests/generateIntegrationRestTool.test.js +159 -0
  69. package/build/tools/generate-tests/planGuard.d.ts +13 -0
  70. package/build/tools/generate-tests/planGuard.js +78 -0
  71. package/build/tools/generate-tests/planGuard.test.d.ts +1 -0
  72. package/build/tools/generate-tests/planGuard.test.js +185 -0
  73. package/build/tools/generate-tests/scenarioFileIdentity.d.ts +10 -0
  74. package/build/tools/generate-tests/scenarioFileIdentity.js +46 -0
  75. package/build/tools/generate-tests/scenarioLint.d.ts +30 -0
  76. package/build/tools/generate-tests/scenarioLint.js +150 -0
  77. package/build/tools/generate-tests/scenarioLint.test.d.ts +1 -0
  78. package/build/tools/generate-tests/scenarioLint.test.js +100 -0
  79. package/build/tools/submitReportTool.js +78 -0
  80. package/build/tools/submitReportTool.test.js +255 -0
  81. package/build/tools/test-management/analyzeChangesTool.js +55 -2
  82. package/build/tools/test-management/analyzeChangesTool.test.js +12 -0
  83. package/build/tools/test-management/index.d.ts +1 -0
  84. package/build/tools/test-management/index.js +1 -0
  85. package/build/tools/test-management/registerTestPlanTool.d.ts +2 -0
  86. package/build/tools/test-management/registerTestPlanTool.js +329 -0
  87. package/build/tools/test-management/registerTestPlanTool.test.d.ts +1 -0
  88. package/build/tools/test-management/registerTestPlanTool.test.js +296 -0
  89. package/build/types/Recommendation.d.ts +97 -0
  90. package/build/types/Recommendation.js +48 -0
  91. package/build/types/RepositoryAnalysis.d.ts +14 -14
  92. package/build/types/TestExecution.d.ts +2 -0
  93. package/build/types/TestRecommendation.d.ts +12 -1
  94. package/build/types/TestRecommendation.js +26 -11
  95. package/build/types/TestTypes.js +1 -1
  96. package/build/utils/AnalysisStateManager.d.ts +47 -0
  97. package/build/utils/docker.test.js +1 -1
  98. package/build/utils/planMatchKeys.d.ts +61 -0
  99. package/build/utils/planMatchKeys.js +125 -0
  100. package/build/utils/pom-scope/import-expansion.d.ts +5 -0
  101. package/build/utils/pom-scope/import-expansion.js +32 -0
  102. package/build/utils/pom-scope/index.d.ts +39 -0
  103. package/build/utils/pom-scope/index.js +120 -0
  104. package/build/utils/pom-scope/index.test.d.ts +1 -0
  105. package/build/utils/pom-scope/index.test.js +239 -0
  106. package/build/utils/pom-scope/pom-files.d.ts +3 -0
  107. package/build/utils/pom-scope/pom-files.js +48 -0
  108. package/build/utils/pom-scope/pom-files.test.d.ts +1 -0
  109. package/build/utils/pom-scope/pom-files.test.js +29 -0
  110. package/build/utils/pom-scope/scoring.d.ts +20 -0
  111. package/build/utils/pom-scope/scoring.js +45 -0
  112. package/build/utils/pom-scope/scoring.test.d.ts +1 -0
  113. package/build/utils/pom-scope/scoring.test.js +39 -0
  114. package/build/utils/pom-scope/selector-extractor.d.ts +7 -0
  115. package/build/utils/pom-scope/selector-extractor.js +57 -0
  116. package/build/utils/pom-scope/selector-extractor.test.d.ts +1 -0
  117. package/build/utils/pom-scope/selector-extractor.test.js +67 -0
  118. package/build/utils/pom-verify/__fixtures__/af-style/asset-list.page.d.ts +5 -0
  119. package/build/utils/pom-verify/__fixtures__/af-style/asset-list.page.js +5 -0
  120. package/build/utils/pom-verify/__fixtures__/af-style/pageobjects/asset-list-page.d.ts +5 -0
  121. package/build/utils/pom-verify/__fixtures__/af-style/pageobjects/asset-list-page.js +9 -0
  122. package/build/utils/pom-verify/__fixtures__/af-style/report.iframe.page.d.ts +4 -0
  123. package/build/utils/pom-verify/__fixtures__/af-style/report.iframe.page.js +4 -0
  124. package/build/utils/pom-verify/__fixtures__/af-style/workflow-footer.page.d.ts +4 -0
  125. package/build/utils/pom-verify/__fixtures__/af-style/workflow-footer.page.js +4 -0
  126. package/build/utils/pom-verify/bindings.d.ts +19 -0
  127. package/build/utils/pom-verify/bindings.js +161 -0
  128. package/build/utils/pom-verify/bindings.test.d.ts +1 -0
  129. package/build/utils/pom-verify/bindings.test.js +164 -0
  130. package/build/utils/pom-verify/calls.d.ts +16 -0
  131. package/build/utils/pom-verify/calls.js +42 -0
  132. package/build/utils/pom-verify/calls.test.d.ts +1 -0
  133. package/build/utils/pom-verify/calls.test.js +61 -0
  134. package/build/utils/pom-verify/index.d.ts +4 -0
  135. package/build/utils/pom-verify/index.js +4 -0
  136. package/build/utils/pom-verify/resolve.d.ts +7 -0
  137. package/build/utils/pom-verify/resolve.js +27 -0
  138. package/build/utils/pom-verify/resolve.test.d.ts +1 -0
  139. package/build/utils/pom-verify/resolve.test.js +68 -0
  140. package/build/utils/pom-verify/strip.d.ts +9 -0
  141. package/build/utils/pom-verify/strip.js +89 -0
  142. package/build/utils/pom-verify/verify.d.ts +14 -0
  143. package/build/utils/pom-verify/verify.js +158 -0
  144. package/build/utils/pom-verify/verify.test.d.ts +1 -0
  145. package/build/utils/pom-verify/verify.test.js +325 -0
  146. package/build/utils/reportVerification.d.ts +61 -0
  147. package/build/utils/reportVerification.js +104 -0
  148. package/build/utils/reportVerification.test.d.ts +1 -0
  149. package/build/utils/reportVerification.test.js +185 -0
  150. package/build/utils/scenarioDrafting.js +5 -5
  151. package/build/utils/versions.d.ts +3 -3
  152. package/build/utils/versions.js +1 -1
  153. package/build/utils/workspaceAuth.d.ts +9 -1
  154. package/build/utils/workspaceAuth.js +25 -5
  155. package/build/utils/workspaceAuth.test.js +48 -0
  156. package/build/workspace/workspace.d.ts +20 -0
  157. package/build/workspace/workspace.js +4 -0
  158. package/build/workspace/workspace.test.js +10 -0
  159. package/node_modules/playwright/lib/mcp/skyramp/traceRecordingBackend.js +77 -8
  160. package/node_modules/playwright/node_modules/playwright-core/lib/generated/injectedScriptSource.js +1 -1
  161. package/node_modules/playwright/node_modules/playwright-core/lib/generated/pollingRecorderSource.js +1 -1
  162. package/node_modules/playwright/node_modules/playwright-core/lib/utils/isomorphic/volatileDate.js +101 -0
  163. package/node_modules/playwright/node_modules/playwright-core/lib/utils.js +2 -0
  164. package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/assets/{codeMirrorModule-aszq5EdG.js → codeMirrorModule-Bzd72-bG.js} +1 -1
  165. package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/assets/{defaultSettingsView-BxS7Jm4s.js → defaultSettingsView-DzxTioTK.js} +101 -101
  166. package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/{index.D4JTTy4R.js → index.BGc30U3S.js} +1 -1
  167. package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/index.html +2 -2
  168. package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/{uiMode.DaRMQKOI.js → uiMode.IaDrb29A.js} +1 -1
  169. package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/uiMode.html +2 -2
  170. package/node_modules/playwright/node_modules/playwright-core/package.json +1 -1
  171. package/node_modules/playwright/node_modules/playwright-core/src/generated/injectedScriptSource.ts +1 -1
  172. package/node_modules/playwright/node_modules/playwright-core/src/generated/pollingRecorderSource.ts +1 -1
  173. package/node_modules/playwright/node_modules/playwright-core/src/utils/isomorphic/volatileDate.ts +131 -0
  174. package/node_modules/playwright/node_modules/playwright-core/src/utils.ts +1 -0
  175. package/node_modules/playwright/package.json +1 -1
  176. package/package.json +3 -3
  177. package/node_modules/playwright/node_modules/playwright-core/.DS_Store +0 -0
@@ -1,3 +1,4 @@
1
+ import { roundRobinByType } from "../../recommendation/diversity.js";
1
2
  import { AUTH_MIDDLEWARE_PATTERNS_STR } from "../../utils/workspaceAuth.js";
2
3
  import { resolveServiceDetailsRef } from "../../utils/utils.js";
3
4
  import { logger } from "../../utils/logger.js";
@@ -21,7 +22,7 @@ The highest-severity \`<bug_found>\` block from this code review triggers a mand
21
22
  }
22
23
  function _execCoverageBody(ctx) {
23
24
  return `${ctx.externalTestFilesList}For every GENERATE item below, check its endpoint path and test type against the Existing Tests list (further down in the prompt).
24
- - **\`[external]\` tests**: If the endpoint is already covered by an \`[external]\` test of the same type → skip the resource entirely (do NOT create or update), **except for \`bug_caught\` and attack-surface \`security_boundary\` items**. A protected item may be skipped only if the external test directly asserts the same flaw/bypass and would fail/pass based on that same bug; same endpoint/resource coverage alone is not enough. Backfill from ADDITIONAL using the priority order below:
25
+ - **\`[external]\` tests**: If the endpoint is already covered by an \`[external]\` test of the same type **that exercises the behavior this PR changes** → skip the resource entirely (do NOT create or update). Resource-level overlap alone is NOT coverage: when the PR tightens or adds a constraint (e.g. a max-items/max-length/enum bound), an external test that never crosses the new bound does not cover it — generate the boundary test. This matters most for single-resource APIs (e.g. one CRD served by the kube-apiserver), where resource-level dedup would permanently block all generation. \`bug_caught\` and attack-surface \`security_boundary\` items get the strictest reading: they may be skipped only if the external test directly asserts the same flaw/bypass and would fail/pass based on that same bug. Backfill from ADDITIONAL using the priority order below:
25
26
  1. **BUG-CATCHING TESTS FIRST (CRITICAL)**: If source code analysis revealed a bug, logic error, or incorrect formula (e.g. discount math adding instead of subtracting, off-by-one errors, missing validation), CREATE A TEST THAT EXPOSES IT. The test SHOULD FAIL — that's the point. Document the bug. Example: if discount formula is wrong, test with discount=20% and assert correct math. If no bug found, skip to #2.
26
27
  2. **PR-endpoint edge cases**: Look for integration test candidates covering error paths, boundary values, or alternative scenarios for the SAME endpoints changed in the PR diff. If no suitable candidate exists in ADDITIONAL, derive one from your source-code enrichment findings.
27
28
  3. **Same-resource other scenarios**: Other HTTP methods or flows on the same resource group touched by the PR.
@@ -94,6 +95,11 @@ For each pair of GENERATE items, ask: same HTTP method + path + step sequence +
94
95
 
95
96
  Same step sequence with only payload differences (e.g. 10% vs 5% discount both returning 200) = same code path = duplicate. Different scenario names do not make duplicate tests distinct.`;
96
97
  }
98
+ function _execRegisterBody(_ctx) {
99
+ return `Register your complete candidate list — every test you would generate OR recommend — via \`skyramp_register_test_plan\` (\`stateFile\` required). Include a discriminator claim (\`discriminator\` field — valid kinds and anchor rules are in the tool schema) for candidates probing the changed logic identified in Step ${EXEC_STEP_CODE_REVIEW}/Step ${EXEC_STEP_ENRICH}.
100
+
101
+ The returned GENERATE list is mandatory and final — generation tools reject unregistered scenarios. If the tool demotes a discriminator claim (returned in \`demotions\` with a reason), either strengthen the claim — a step that actually exercises the declared \`kind\`, or a verbatim anchor that occurs in the diff — or drop it; the candidate itself stays in the plan either way.`;
102
+ }
97
103
  function _execExecuteBody(ctx) {
98
104
  return `Replace any scenario that pairs unrelated resources with one reflecting actual foreign-key relationships in the codebase.
99
105
  Use the field names and values from the \`<source_evidence>\` blocks you quoted in Step ${EXEC_STEP_ENRICH} to fill all tool call parameters. Prefer reusing Step ${EXEC_STEP_ENRICH} evidence when it already resolves a placeholder, but if a placeholder cannot be replaced with concrete values from files already read, you may read the specific schema, model, or handler file needed to resolve it. Assert response field values, not just status codes.
@@ -127,6 +133,7 @@ const _execPlan = new PromptPlan({ startFrom: 0 })
127
133
  .step("ENRICH", "Parameter Grounding & Priority Assignment", _execEnrichBody)
128
134
  .step("DIVERSITY", (ctx) => `Diversity check (using enriched knowledge from Step ${ctx.enrichStepLabel})`, _execDiversityBody)
129
135
  .step("EXECUTE", "Execute merged plan in rank order", _execExecuteBody)
136
+ .step("REGISTER", "Register your test plan", _execRegisterBody)
130
137
  .done();
131
138
  // ── Exported step label constants ─────────────────────────────────────────────
132
139
  /** "0" — Code Review: correctness analysis */
@@ -139,6 +146,8 @@ export const EXEC_STEP_ENRICH = _execPlan.labels.ENRICH; // "2"
139
146
  export const EXEC_STEP_DIVERSITY = _execPlan.labels.DIVERSITY; // "3"
140
147
  /** "4" — Execute merged plan */
141
148
  export const EXEC_STEP_EXECUTE = _execPlan.labels.EXECUTE; // "4"
149
+ /** "5" — Register test plan (SKYR-3879 Path B checkpoint) */
150
+ export const EXEC_STEP_REGISTER = _execPlan.labels.REGISTER; // "5"
142
151
  const SERVICE_REFS = resolveServiceDetailsRef();
143
152
  function prioritizeAttackSurfaceBundles(items) {
144
153
  const reordered = [];
@@ -154,69 +163,6 @@ function prioritizeAttackSurfaceBundles(items) {
154
163
  }
155
164
  return reordered;
156
165
  }
157
- /**
158
- * Select `count` items from a rank-ordered list, distributing GENERATE slots
159
- * EVENLY across the test types present, with spillover.
160
- *
161
- * Policy (kept identical to the multi-repo prose in testbot-prompts.ts's
162
- * "Cross-repo test generation" block — change both together):
163
- * - Protected items first: CRITICAL-priority and attack-surface security_boundary
164
- * scenarios always take a slot before round-robin (they must stay in GENERATE).
165
- * - Bucket the rest by inferred test type (contract vs integration — the same
166
- * inference used when rendering: `testType ?? (steps===1 ? contract : integration)`).
167
- * - Round-robin one item per non-empty bucket per round, in the buckets' order of
168
- * first appearance in the rank-ordered list (so the highest-ranked type wins
169
- * round 1), preserving rank order within each bucket.
170
- * - Spillover: an exhausted bucket is skipped on later rounds, so its freed slots
171
- * go to the next type's next-highest item.
172
- *
173
- * Degenerate cases match the previous pure rank-order slice exactly: a single type
174
- * present, or `count >= items.length`, returns the same items in the same order —
175
- * so backend-only / single-type runs are unchanged (no regression).
176
- */
177
- function roundRobinByType(rankOrdered, count) {
178
- if (count <= 0)
179
- return [];
180
- // Everything fits → no need to bucket; identical to the old slice.
181
- if (count >= rankOrdered.length)
182
- return rankOrdered.slice(0, count);
183
- const inferType = (s) => s.testType ?? (s.steps.length === 1 ? "contract" : "integration");
184
- // Protected items occupy GENERATE slots first, in rank order.
185
- const selected = [];
186
- const remaining = [];
187
- for (const item of rankOrdered) {
188
- if (selected.length < count &&
189
- (item.priority === "CRITICAL" || isAttackSurfaceSecurityBoundary(item.scenario))) {
190
- selected.push(item);
191
- }
192
- else {
193
- remaining.push(item);
194
- }
195
- }
196
- // Bucket the remainder by inferred type, preserving rank order and first-appearance
197
- // bucket order.
198
- const order = [];
199
- const buckets = new Map();
200
- for (const item of remaining) {
201
- const t = inferType(item.scenario);
202
- if (!buckets.has(t)) {
203
- buckets.set(t, []);
204
- order.push(t);
205
- }
206
- buckets.get(t).push(item);
207
- }
208
- // Round-robin one per non-empty bucket per round until full.
209
- while (selected.length < count && order.some((t) => buckets.get(t).length > 0)) {
210
- for (const t of order) {
211
- if (selected.length >= count)
212
- break;
213
- const bucket = buckets.get(t);
214
- if (bucket.length > 0)
215
- selected.push(bucket.shift());
216
- }
217
- }
218
- return selected;
219
- }
220
166
  export function buildExecutionPlan(scored, maxGen, topN, baseUrl, authHeaderValue, authSchemeSnippet, authTypeValue, seed, endpointCount, isUIOnlyPR, hasFrontendChanges = false, hasTraces = false, externalCoverage = new Set(), relevantExternalTestPaths = [],
221
167
  /**
222
168
  * Whether the diff classified at least one new/modified/removed endpoint.
@@ -473,7 +419,7 @@ ${buildScopeAssessmentSection(topN, maxGen, isUIOnlyPR, isUIOnlyPR ? 100 : hasFr
473
419
 
474
420
  ${_execPlan.render(_ctx)}
475
421
 
476
- ### GENERATE (after completing Steps ${EXEC_STEP_CODE_REVIEW}–${EXEC_STEP_EXECUTE} above) — generate exactly these items in order; add variations to ADDITIONAL instead. If Step ${EXEC_STEP_COVERAGE} converts an item to UPDATE, backfill from ADDITIONAL (priority order in Step ${EXEC_STEP_COVERAGE})
422
+ ### GENERATE (after completing Steps ${EXEC_STEP_CODE_REVIEW}–${EXEC_STEP_EXECUTE} above and registering via Step ${EXEC_STEP_REGISTER}) — the list below is a starting point; \`skyramp_register_test_plan\`'s returned GENERATE list is the final, mandatory one. Generate exactly those items in order; add variations to ADDITIONAL instead. If Step ${EXEC_STEP_COVERAGE} converts an item to UPDATE, backfill from ADDITIONAL (priority order in Step ${EXEC_STEP_COVERAGE})
477
423
 
478
424
  ${isUIOnlyPR
479
425
  ? uiGenerateBlocks ||
@@ -1,7 +1,7 @@
1
1
  import { DraftedScenario } from "../../types/RepositoryAnalysis.js";
2
- import { PriorityTier } from "../../types/TestRecommendation.js";
2
+ import { Novelty, PriorityTier } from "../../types/TestRecommendation.js";
3
3
  export declare function buildFullRepoRecommendations(scored: Array<{
4
4
  scenario: DraftedScenario;
5
5
  priority: PriorityTier;
6
- novelty: string;
6
+ novelty: Novelty;
7
7
  }>, topN: number, baseUrl: string, authHeaderValue: string, authSchemeSnippet: string, authTypeValue: string, isFrontendProject?: boolean, isFrontendOnlyProject?: boolean, externalCoverage?: Set<string>): string;
@@ -51,7 +51,7 @@ Before each GENERATE tool call, confirm WHERE each key value comes from:
51
51
  - **endpointURL** → workspace \`baseUrl\` + endpoint path (both required — never path alone)
52
52
  - **authHeader / authScheme** → workspace config or OpenAPI \`securitySchemes\`
53
53
  - **Foreign-key path params** → chained from a prior step's response — never invented or hardcoded. Common field names: \`id\`, \`uuid\`, \`_id\`, \`*_id\`; use whatever identifier field the server returns for this resource. The chaining source can be a response body (POST or GET), a response header (e.g. \`Location\`), or a cookie.
54
- - **Names / string values** → realistic; append timestamp suffix to avoid re-run conflicts
54
+ - **Names / string values** → realistic. Do NOT hardcode a timestamp/uuid suffix. Instead, for create fields that carry a UNIQUE constraint (e.g. \`name\`, \`slug\`, \`email\` — confirm from the source schema: \`unique=True\`, SQL \`UNIQUE\`, DTO), list their field paths in the step's \`uniqueFields\` (gjson notation) — the generator injects a run-unique value so re-runs don't 409.
55
55
 
56
56
  ## Ranking Rule
57
57
  For each GENERATE item, include one sentence in your output (before the tool calls) stating the specific bug or failure it targets — derived from \`bugCatchingTarget\` or your source-code reading. Example: "Targets: order total miscalculation — total_amount = sum(item.price × item.quantity) should recompute when items array changes."
@@ -332,7 +332,10 @@ ${authGuidance}
332
332
  collection array (e.g. \`"items": [{"product_id": <chained from prior POST>, "quantity": 2}]\`).
333
333
  Never send a PATCH that only modifies metadata (discount, status) without also including the
334
334
  items/products collection — such a test will not catch collection-level or total-recalculation bugs.
335
- Use unique names with timestamp suffix to avoid conflicts on re-runs.
335
+ For UNIQUE-constrained create fields, set the step's \`uniqueFields\` (gjson paths) rather than
336
+ hardcoding timestamp suffixes — the generator makes them run-unique so re-runs don't 409.
337
+ For every resource-creating step (POST/PUT), also add a matching DELETE step for the created
338
+ resource (path ID chained from the create response) so the scenario cleans up after itself.
336
339
  For GET/PUT/DELETE with path IDs, use a placeholder — chaining resolves the real ID.
337
340
  2. Produces a \`scenario_<name>.json\` in the same \`outputDir\` as the test files (not \`.skyramp/\`).
338
341
  3. Call \`skyramp_integration_test_generation\` with \`scenarioFile\`: ${integrationAuthNote}
@@ -200,7 +200,7 @@ Budget Plan: 0 total — no new, modified, or removed endpoints were classified
200
200
 
201
201
  With a 0-total Budget Plan: generate zero tests, recommend zero tests, and follow the zero-test report path. Do NOT draft baseline or generic tests for unchanged endpoints to fill a budget — an empty diff surface is a valid, expected outcome.
202
202
 
203
- **Exception — claim the ceiling only with evidence:** if your code review of the changed files shows an observable API behavior change the classifier missed (e.g. a DTO/serializer/service change that alters a response shape, or a deployment/config change that newly exposes or removes endpoints), raise your Budget Plan to cover exactly those affected endpoints, up to ${maxTotal} total (${effectiveGenerate} generate + ${additional} additional), 0% UI/E2E. Every test must name the changed file that justifies it. State your raised plan now in the canonical format — \`Budget Plan: <total> total (<generate> generate + <additional> additional), 0% UI/E2E\` — and use those exact numbers throughout the rest of the prompt; the raised generate count is your committed generate count.`;
203
+ **Exception — claim the ceiling only with evidence:** if your code review of the changed files shows an observable API behavior change the classifier missed (e.g. a DTO/serializer/service change that alters a response shape, a deployment/config change that newly exposes or removes endpoints, or a schema-defined API contract change — a CRD type/kubebuilder validation marker, GraphQL schema, or gRPC proto edit that adds, removes, or re-validates what the server accepts or returns), raise your Budget Plan to cover exactly those affected endpoints, up to ${maxTotal} total (${effectiveGenerate} generate + ${additional} additional), 0% UI/E2E. Note: repositories whose entire API surface is schema-defined (e.g. a Kubernetes operator serving CRDs through the kube-apiserver) ALWAYS classify zero endpoints — for these, a schema change in the diff IS the endpoint change; evaluate this exception against the schema files instead of concluding there is nothing to test. Every test must name the changed file that justifies it. State your raised plan now in the canonical format — \`Budget Plan: <total> total (<generate> generate + <additional> additional), 0% UI/E2E\` — and use those exact numbers throughout the rest of the prompt; the raised generate count is your committed generate count.`;
204
204
  }
205
205
  // Unambiguous backend-only or UI-only: emit a single Budget Plan line — no LLM counting needed.
206
206
  if (precomputedUIPct !== undefined) {
@@ -1,6 +1,31 @@
1
- import { RepositoryAnalysis, AnalysisScope } from "../../types/RepositoryAnalysis.js";
1
+ import { RepositoryAnalysis, AnalysisScope, DraftedScenario } from "../../types/RepositoryAnalysis.js";
2
2
  import { WorkspaceAuthType } from "../../utils/workspaceAuth.js";
3
3
  import { PRTestContext } from "../../utils/pr-comment-parser.js";
4
+ import { Novelty, PriorityTier } from "../../types/TestRecommendation.js";
4
5
  import { buildExternalCoverageSet, externalDedupKey } from "./recommendationShared.js";
5
6
  export { buildExternalCoverageSet, externalDedupKey };
7
+ /** Result of {@link computeScoredCandidates} — the scoring/classification
8
+ * inputs shared between prompt rendering and the SKYR-3879 register-plan
9
+ * pre-seed (analyzeChangesTool.ts), so both derive the exact same numbers. */
10
+ export interface ScoredCandidatesResult {
11
+ scored: Array<{
12
+ scenario: DraftedScenario;
13
+ priority: PriorityTier;
14
+ novelty: Novelty;
15
+ }>;
16
+ filteredChangedFiles: string[];
17
+ isUIOnlyPR: boolean;
18
+ hasFrontendChanges: boolean;
19
+ hasApiChanges: boolean;
20
+ maxGen: number;
21
+ seed: string;
22
+ }
23
+ /**
24
+ * Compute the pre-ranked scenario scoring, the UI/API classification flags,
25
+ * and the effective GENERATE cap — extracted from buildRecommendationPrompt
26
+ * so skyramp_analyze_changes (SKYR-3879 Path B pre-seed) can derive the exact
27
+ * same candidate set the Execution Plan prompt is built from, without
28
+ * duplicating (and risking drift from) this scoring logic.
29
+ */
30
+ export declare function computeScoredCandidates(analysis: RepositoryAnalysis, analysisScope?: AnalysisScope, topN?: number, maxGenerateOverride?: number): ScoredCandidatesResult;
6
31
  export declare function buildRecommendationPrompt(analysis: RepositoryAnalysis, analysisScope?: AnalysisScope, topN?: number, prContext?: PRTestContext, workspaceAuthHeader?: string, workspaceAuthType?: WorkspaceAuthType, workspaceAuthScheme?: string, maxGenerateOverride?: number, sessionId?: string): string;
@@ -3,7 +3,7 @@ import { AnalysisScope, isDiff, } from "../../types/RepositoryAnalysis.js";
3
3
  import { WorkspaceAuthType, getDefaultAuthHeader } from "../../utils/workspaceAuth.js";
4
4
  import { logger } from "../../utils/logger.js";
5
5
  import { buildArchitectPreamble, buildContextFetchingGuidance, buildReasoningProtocol, buildToolWorkflows, buildFewShotExamples, buildVerificationChecklist, getAuthSnippets, MAX_TESTS_TO_GENERATE, MAX_RECOMMENDATIONS, } from "./recommendationSections.js";
6
- import { CATEGORY_PRIORITY } from "../../types/TestRecommendation.js";
6
+ import { CATEGORY_PRIORITY, Novelty, PriorityTier } from "../../types/TestRecommendation.js";
7
7
  import { buildScopeAssessmentSection, isFrontendFile } from "./scopeAssessment.js";
8
8
  import { buildExecutionPlan, EXEC_STEP_CODE_REVIEW } from "./diffExecutionPlan.js";
9
9
  import { buildFullRepoRecommendations } from "./fullRepoCatalog.js";
@@ -36,21 +36,21 @@ const PRIORITY_ORDER = { CRITICAL: 4, HIGH: 3, MEDIUM: 2, LOW: 1 };
36
36
  const NOVELTY_ORDER = { new: 3, modified: 2, existing: 1 };
37
37
  function classifyNovelty(scenario, diffContext) {
38
38
  if (!diffContext)
39
- return "existing";
39
+ return Novelty.EXISTING;
40
40
  const paths = scenario.steps.map(s => s.path);
41
41
  const newPaths = new Set((diffContext.newEndpoints || []).map(ep => ep.path));
42
42
  const modPaths = new Set((diffContext.modifiedEndpoints || []).map(ep => ep.path));
43
43
  const removedPaths = new Set((diffContext.removedEndpoints || []).map(ep => ep.path));
44
44
  if (paths.some(p => newPaths.has(p)))
45
- return "new";
45
+ return Novelty.NEW;
46
46
  if (paths.some(p => modPaths.has(p) || removedPaths.has(p)))
47
- return "modified";
48
- return "existing";
47
+ return Novelty.MODIFIED;
48
+ return Novelty.EXISTING;
49
49
  }
50
50
  function prioritiseCandidate(scenario, diffContext) {
51
51
  const priority = isAttackSurfaceSecurityBoundary(scenario)
52
- ? "CRITICAL"
53
- : CATEGORY_PRIORITY[scenario.category] ?? "LOW";
52
+ ? PriorityTier.CRITICAL
53
+ : CATEGORY_PRIORITY[scenario.category] ?? PriorityTier.LOW;
54
54
  const novelty = classifyNovelty(scenario, diffContext);
55
55
  return { priority, novelty };
56
56
  }
@@ -58,15 +58,20 @@ function computeTiebreakerSeed(endpoints, diffFiles) {
58
58
  const canonical = [...endpoints].sort().join("|") + "::" + [...diffFiles].sort().join("|");
59
59
  return crypto.createHash("sha256").update(canonical).digest("hex").slice(0, 8);
60
60
  }
61
- // ── Execution Plan (replaces pre-ranked + scenarios + heuristic sections) ──
62
- export function buildRecommendationPrompt(analysis, analysisScope = AnalysisScope.FullRepo, topN = MAX_RECOMMENDATIONS, prContext, workspaceAuthHeader, workspaceAuthType, workspaceAuthScheme, maxGenerateOverride, sessionId) {
61
+ // Prevents bot-committed test files from being treated as application changes
62
+ // on subsequent testbot runs on the same PR.
63
+ const SKYRAMP_TEST_FILE_PATTERN = /(?:_test|_smoke|_contract|_fuzz|_integration|_load|_e2e|_ui)\.[^/]+$|scenario_[^/]+\.json$/;
64
+ /**
65
+ * Compute the pre-ranked scenario scoring, the UI/API classification flags,
66
+ * and the effective GENERATE cap — extracted from buildRecommendationPrompt
67
+ * so skyramp_analyze_changes (SKYR-3879 Path B pre-seed) can derive the exact
68
+ * same candidate set the Execution Plan prompt is built from, without
69
+ * duplicating (and risking drift from) this scoring logic.
70
+ */
71
+ export function computeScoredCandidates(analysis, analysisScope = AnalysisScope.FullRepo, topN = MAX_RECOMMENDATIONS, maxGenerateOverride) {
63
72
  const isDiffScope = isDiff(analysisScope);
64
73
  const diffContext = analysis.branchDiffContext;
65
- const openApiSpec = analysis.artifacts?.openApiSpecs?.[0];
66
74
  // ── Filter out bot-generated test files from changedFiles ──
67
- // Prevents bot-committed test files from being treated as application changes
68
- // on subsequent testbot runs on the same PR.
69
- const SKYRAMP_TEST_FILE_PATTERN = /(?:_test|_smoke|_contract|_fuzz|_integration|_load|_e2e|_ui)\.[^/]+$|scenario_[^/]+\.json$/;
70
75
  const filteredChangedFiles = diffContext
71
76
  ? diffContext.changedFiles.filter(f => !SKYRAMP_TEST_FILE_PATTERN.test(f))
72
77
  : [];
@@ -81,6 +86,56 @@ export function buildRecommendationPrompt(analysis, analysisScope = AnalysisScop
81
86
  ? (diffContext.newEndpoints.length > 0 || diffContext.modifiedEndpoints.length > 0 || (diffContext.removedEndpoints?.length ?? 0) > 0)
82
87
  : false;
83
88
  const isUIOnlyPR = hasFrontendChanges && !hasApiChanges;
89
+ // ── Scoring ──
90
+ const baseMaxGen = Math.min(Math.max(maxGenerateOverride ?? (isDiffScope ? MAX_TESTS_TO_GENERATE : topN), 0), topN);
91
+ const maxGen = isUIOnlyPR ? Math.max(baseMaxGen, 1) : baseMaxGen;
92
+ const scenarios = analysis.businessContext.draftedScenarios;
93
+ let scored = [];
94
+ let seed = "";
95
+ if (!isUIOnlyPR && scenarios.length > 0) {
96
+ const diffFiles = filteredChangedFiles; // use filtered list so bot-committed test files don't shift the seed
97
+ const endpointPaths = analysis.apiEndpoints.endpoints.map(ep => ep.path);
98
+ seed = computeTiebreakerSeed(endpointPaths, diffFiles);
99
+ scored = scenarios.map(s => {
100
+ const result = prioritiseCandidate(s, diffContext ?? undefined);
101
+ return { scenario: s, ...result };
102
+ });
103
+ scored.sort((a, b) => {
104
+ const pa = PRIORITY_ORDER[a.priority], pb = PRIORITY_ORDER[b.priority];
105
+ if (pb !== pa)
106
+ return pb - pa;
107
+ const na = NOVELTY_ORDER[a.novelty], nb = NOVELTY_ORDER[b.novelty];
108
+ if (nb !== na)
109
+ return nb - na;
110
+ const crossA = a.scenario.steps.length > 2 ? 1 : 0;
111
+ const crossB = b.scenario.steps.length > 2 ? 1 : 0;
112
+ if (crossB !== crossA)
113
+ return crossB - crossA;
114
+ if (b.scenario.steps.length !== a.scenario.steps.length)
115
+ return b.scenario.steps.length - a.scenario.steps.length;
116
+ const errorA = a.scenario.steps.some(s => s.interactionType === "error" || s.interactionType === "edge-case") ? 1 : 0;
117
+ const errorB = b.scenario.steps.some(s => s.interactionType === "error" || s.interactionType === "edge-case") ? 1 : 0;
118
+ if (errorB !== errorA)
119
+ return errorB - errorA;
120
+ // Use locale-independent comparison to avoid runtime-locale non-determinism
121
+ const nameA = a.scenario.scenarioName;
122
+ const nameB = b.scenario.scenarioName;
123
+ if (nameA < nameB)
124
+ return -1;
125
+ if (nameA > nameB)
126
+ return 1;
127
+ const hashA = parseInt(crypto.createHash("sha256").update(seed + a.scenario.scenarioName).digest("hex").slice(0, 8), 16);
128
+ const hashB = parseInt(crypto.createHash("sha256").update(seed + b.scenario.scenarioName).digest("hex").slice(0, 8), 16);
129
+ return hashA - hashB;
130
+ });
131
+ }
132
+ return { scored, filteredChangedFiles, isUIOnlyPR, hasFrontendChanges, hasApiChanges, maxGen, seed };
133
+ }
134
+ export function buildRecommendationPrompt(analysis, analysisScope = AnalysisScope.FullRepo, topN = MAX_RECOMMENDATIONS, prContext, workspaceAuthHeader, workspaceAuthType, workspaceAuthScheme, maxGenerateOverride, sessionId) {
135
+ const isDiffScope = isDiff(analysisScope);
136
+ const diffContext = analysis.branchDiffContext;
137
+ const openApiSpec = analysis.artifacts?.openApiSpecs?.[0];
138
+ const { scored, filteredChangedFiles, isUIOnlyPR, hasFrontendChanges, hasApiChanges, maxGen, seed, } = computeScoredCandidates(analysis, analysisScope, topN, maxGenerateOverride);
84
139
  const hasTraces = (analysis.artifacts?.traceFiles?.length ?? 0) > 0 ||
85
140
  (analysis.artifacts?.playwrightRecordings?.length ?? 0) > 0;
86
141
  // ── Mode preamble ──
@@ -277,50 +332,7 @@ ${isDiffScope ? "Changed endpoints only. " : ""}Use source code schemas (Zod/Pyd
277
332
  ${detailBlocks}
278
333
  `;
279
334
  }
280
- // ── Scoring ──
281
335
  const endpointCount = allEndpoints.reduce((acc, ep) => acc + (ep.methods ?? []).length, 0);
282
- const baseMaxGen = Math.min(Math.max(maxGenerateOverride ?? (isDiffScope ? MAX_TESTS_TO_GENERATE : topN), 0), topN);
283
- const maxGen = isUIOnlyPR ? Math.max(baseMaxGen, 1) : baseMaxGen;
284
- const scenarios = analysis.businessContext.draftedScenarios;
285
- let scored = [];
286
- let seed = "";
287
- if (!isUIOnlyPR && scenarios.length > 0) {
288
- const diffFiles = filteredChangedFiles; // use filtered list so bot-committed test files don't shift the seed
289
- const endpointPaths = allEndpoints.map(ep => ep.path);
290
- seed = computeTiebreakerSeed(endpointPaths, diffFiles);
291
- scored = scenarios.map(s => {
292
- const result = prioritiseCandidate(s, diffContext ?? undefined);
293
- return { scenario: s, ...result };
294
- });
295
- scored.sort((a, b) => {
296
- const pa = PRIORITY_ORDER[a.priority], pb = PRIORITY_ORDER[b.priority];
297
- if (pb !== pa)
298
- return pb - pa;
299
- const na = NOVELTY_ORDER[a.novelty], nb = NOVELTY_ORDER[b.novelty];
300
- if (nb !== na)
301
- return nb - na;
302
- const crossA = a.scenario.steps.length > 2 ? 1 : 0;
303
- const crossB = b.scenario.steps.length > 2 ? 1 : 0;
304
- if (crossB !== crossA)
305
- return crossB - crossA;
306
- if (b.scenario.steps.length !== a.scenario.steps.length)
307
- return b.scenario.steps.length - a.scenario.steps.length;
308
- const errorA = a.scenario.steps.some(s => s.interactionType === "error" || s.interactionType === "edge-case") ? 1 : 0;
309
- const errorB = b.scenario.steps.some(s => s.interactionType === "error" || s.interactionType === "edge-case") ? 1 : 0;
310
- if (errorB !== errorA)
311
- return errorB - errorA;
312
- // Use locale-independent comparison to avoid runtime-locale non-determinism
313
- const nameA = a.scenario.scenarioName;
314
- const nameB = b.scenario.scenarioName;
315
- if (nameA < nameB)
316
- return -1;
317
- if (nameA > nameB)
318
- return 1;
319
- const hashA = parseInt(crypto.createHash("sha256").update(seed + a.scenario.scenarioName).digest("hex").slice(0, 8), 16);
320
- const hashB = parseInt(crypto.createHash("sha256").update(seed + b.scenario.scenarioName).digest("hex").slice(0, 8), 16);
321
- return hashA - hashB;
322
- });
323
- }
324
336
  // ── Main section: execution plan, UI-only guidance, or draft-your-own ──
325
337
  let mainSection;
326
338
  if (!isDiffScope && scored.length > 0) {
@@ -1,10 +1,11 @@
1
1
  import { jest } from "@jest/globals";
2
+ import { Novelty, PriorityTier } from "../../types/TestRecommendation.js";
2
3
  import { TestType } from "../../types/TestTypes.js";
3
4
  jest.unstable_mockModule("@skyramp/skyramp", () => ({
4
5
  WorkspaceConfigManager: { create: jest.fn() },
5
6
  }));
6
7
  const { buildRecommendationPrompt, buildExternalCoverageSet, externalDedupKey } = await import("./test-recommendation-prompt.js");
7
- const { buildExecutionPlan } = await import("./diffExecutionPlan.js");
8
+ const { buildExecutionPlan, EXEC_STEP_REGISTER } = await import("./diffExecutionPlan.js");
8
9
  const { PATH_PARAM_UUID_GUIDANCE, MAX_TESTS_TO_GENERATE, buildTestQualityCriteria, buildArchitectPreamble, buildContextFetchingGuidance, buildReasoningProtocol, buildFewShotExamples, buildVerificationChecklist, } = await import("./recommendationSections.js");
9
10
  import { AnalysisScope } from "../../types/RepositoryAnalysis.js";
10
11
  // ---------------------------------------------------------------------------
@@ -746,6 +747,42 @@ describe("buildRecommendationPrompt — zero-classified diff (SKYR-3820)", () =>
746
747
  });
747
748
  });
748
749
  // ---------------------------------------------------------------------------
750
+ // Tests — REGISTER step (SKYR-3879 Path B checkpoint)
751
+ // ---------------------------------------------------------------------------
752
+ describe("buildRecommendationPrompt — REGISTER step (SKYR-3879 Path B)", () => {
753
+ it("diff-scope execution plan includes the REGISTER step and its skyramp_register_test_plan instruction", () => {
754
+ const prompt = buildRecommendationPrompt(minimalAnalysis(), AnalysisScope.CurrentBranchDiff, 5);
755
+ expect(prompt).toContain(`Step ${EXEC_STEP_REGISTER} — Register your test plan`);
756
+ expect(prompt).toContain("skyramp_register_test_plan");
757
+ });
758
+ it("diff-scope execution plan's GENERATE header references the REGISTER step", () => {
759
+ const prompt = buildRecommendationPrompt(minimalAnalysis(), AnalysisScope.CurrentBranchDiff, 5);
760
+ expect(prompt).toContain(`registering via Step ${EXEC_STEP_REGISTER}`);
761
+ });
762
+ it("full-repo mode (no execution plan) never mentions skyramp_register_test_plan", () => {
763
+ const scenariosForFullRepo = [
764
+ minimalScenario({
765
+ scenarioName: "scenario-0",
766
+ category: "crud",
767
+ priority: "low",
768
+ testType: TestType.CONTRACT,
769
+ steps: [{ order: 1, method: "GET", path: "/api/items", description: "Get items", interactionType: "success", expectedStatusCode: 200 }],
770
+ }),
771
+ ];
772
+ const analysis = minimalAnalysis({
773
+ businessContext: { mainPurpose: "Test API", userFlows: [], dataFlows: [], integrationPatterns: [], draftedScenarios: scenariosForFullRepo },
774
+ });
775
+ const prompt = buildRecommendationPrompt(analysis, AnalysisScope.FullRepo, 6);
776
+ expect(prompt).not.toContain("skyramp_register_test_plan");
777
+ });
778
+ it("diff-scope 'draft your own' fallback (no pre-drafted scenarios) still includes the REGISTER step — the execution plan is always used in diff scope", () => {
779
+ const prompt = buildRecommendationPrompt(minimalAnalysis(), AnalysisScope.CurrentBranchDiff, 5);
780
+ // minimalAnalysis() has no draftedScenarios and no branchDiffContext, so scored=[]
781
+ // but buildExecutionPlan still renders (see buildRecommendationPrompt's isDiffScope branch).
782
+ expect(prompt).toContain("skyramp_register_test_plan");
783
+ });
784
+ });
785
+ // ---------------------------------------------------------------------------
749
786
  // Tests — GENERATE slot allocation (UI vs backend slots)
750
787
  // ---------------------------------------------------------------------------
751
788
  describe("buildRecommendationPrompt — GENERATE slot allocation", () => {
@@ -920,13 +957,13 @@ describe("buildRecommendationPrompt — GENERATE slot allocation", () => {
920
957
  });
921
958
  const prompt = buildExecutionPlan([attackSurface, bugCaught, ordinary].map((scenario) => ({
922
959
  scenario,
923
- priority: scenario.category === "crud" ? "LOW" : "CRITICAL",
924
- novelty: "modified",
960
+ priority: scenario.category === "crud" ? PriorityTier.LOW : PriorityTier.CRITICAL,
961
+ novelty: Novelty.MODIFIED,
925
962
  })), 3, 3, "http://localhost:3000", "Authorization", ", authScheme: \"Bearer\"", "", "seed", 3, false, false, false, new Set(["POST::flows::contract", "POST::orders::contract", "POST::customers::contract"]));
926
963
  expect(prompt).toContain("POST /api/flows/bulk_delete → 401");
927
964
  expect(prompt).toContain("POST /api/orders → 201");
928
965
  expect(prompt).not.toContain("POST /api/customers → 201");
929
- expect(prompt).toContain("except for `bug_caught` and attack-surface `security_boundary` items");
966
+ expect(prompt).toContain("`bug_caught` and attack-surface `security_boundary` items get the strictest reading");
930
967
  });
931
968
  it("keeps symmetric attack-surface siblings together before ordinary auth boundaries", () => {
932
969
  const directWidgets = minimalScenario({
@@ -959,8 +996,8 @@ describe("buildRecommendationPrompt — GENERATE slot allocation", () => {
959
996
  });
960
997
  const prompt = buildExecutionPlan([directWidgets, directGadgets, widgetsBulk, gadgetsBulk].map((scenario) => ({
961
998
  scenario,
962
- priority: "CRITICAL",
963
- novelty: "modified",
999
+ priority: PriorityTier.CRITICAL,
1000
+ novelty: Novelty.MODIFIED,
964
1001
  })), 3, 4, "http://localhost:3000", "Authorization", ", authScheme: \"Bearer\"", "", "seed", 4, false);
965
1002
  const generatedBlock = prompt.slice(0, prompt.indexOf("#4 [ADDITIONAL]"));
966
1003
  const additionalBlock = prompt.slice(prompt.indexOf("#4 [ADDITIONAL]"));
@@ -975,7 +1012,7 @@ describe("buildExecutionPlan — even-with-spillover test-type distribution", ()
975
1012
  // Backend-only PR (hasFrontendChanges=false) so every GENERATE slot is a backend
976
1013
  // scenario drawn from `scored` — this is where contract-vs-integration distribution
977
1014
  // applies. baseUrl/auth args mirror the existing buildExecutionPlan tests above.
978
- const contract = (name, path, priority = "HIGH") => ({
1015
+ const contract = (name, path, priority = PriorityTier.HIGH) => ({
979
1016
  scenario: minimalScenario({
980
1017
  scenarioName: name,
981
1018
  category: "crud",
@@ -983,9 +1020,9 @@ describe("buildExecutionPlan — even-with-spillover test-type distribution", ()
983
1020
  steps: [{ order: 1, method: "POST", path, description: `POST ${path}`, interactionType: "success", expectedStatusCode: 201 }],
984
1021
  }),
985
1022
  priority: priority,
986
- novelty: "new",
1023
+ novelty: Novelty.NEW,
987
1024
  });
988
- const integration = (name, p1, p2, priority = "HIGH") => ({
1025
+ const integration = (name, p1, p2, priority = PriorityTier.HIGH) => ({
989
1026
  scenario: minimalScenario({
990
1027
  scenarioName: name,
991
1028
  category: "data_integrity",
@@ -996,7 +1033,7 @@ describe("buildExecutionPlan — even-with-spillover test-type distribution", ()
996
1033
  ],
997
1034
  }),
998
1035
  priority: priority,
999
- novelty: "new",
1036
+ novelty: Novelty.NEW,
1000
1037
  });
1001
1038
  const plan = (items, maxGen, topN) => buildExecutionPlan(items, maxGen, topN, "http://localhost:3000", "Authorization", ", authScheme: \"Bearer\"", "", "seed", items.length, false);
1002
1039
  it("distributes GENERATE slots across contract AND integration when budget < candidates", () => {
@@ -1025,7 +1062,7 @@ describe("buildExecutionPlan — even-with-spillover test-type distribution", ()
1025
1062
  // A CRITICAL integration sits after several HIGH contracts in rank order.
1026
1063
  const items = [
1027
1064
  contract("c1", "/api/a"), contract("c2", "/api/b"), contract("c3", "/api/c"),
1028
- integration("crit", "/api/orders", "/api/orders/{id}", "CRITICAL"),
1065
+ integration("crit", "/api/orders", "/api/orders/{id}", PriorityTier.CRITICAL),
1029
1066
  ];
1030
1067
  const prompt = plan(items, 3, 6);
1031
1068
  const generated = prompt.slice(0, prompt.indexOf("[ADDITIONAL]") >= 0 ? prompt.indexOf("[ADDITIONAL]") : prompt.length);
@@ -16,7 +16,7 @@ export interface RelatedRepository {
16
16
  */
17
17
  export declare function parseRelatedRepositories(raw: string | undefined): RelatedRepository[] | undefined;
18
18
  export declare function getTestbotPrompt(prTitle: string, prDescription: string, summaryOutputFile: string, repositoryPath: string, baseBranch?: string, maxRecommendations?: number, maxGenerate?: number, _maxCritical?: number, // Reserved — accepted for API compat but not yet wired into prompt
19
- prNumber?: number, userPrompt?: string, services?: Service[], uiCredentials?: string, testsRepoDir?: string, relatedRepositories?: RelatedRepository[], primaryRepo?: string): string;
19
+ prNumber?: number, userPrompt?: string, services?: Service[], uiCredentials?: string, testsRepoDir?: string, relatedRepositories?: RelatedRepository[], primaryRepo?: string, planOnly?: boolean): string;
20
20
  /**
21
21
  * Read services from .skyramp/workspace.yml. Returns empty array if
22
22
  * the workspace file doesn't exist or can't be parsed.