@skyramp/mcp 0.3.1 → 0.3.2-rc.pom-2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (177) hide show
  1. package/build/index.js +2 -1
  2. package/build/prompts/code-reuse.d.ts +7 -1
  3. package/build/prompts/code-reuse.js +8 -5
  4. package/build/prompts/code-reuse.test.d.ts +1 -0
  5. package/build/prompts/code-reuse.test.js +62 -0
  6. package/build/prompts/pom-aware-code-reuse.d.ts +6 -1
  7. package/build/prompts/pom-aware-code-reuse.js +100 -53
  8. package/build/prompts/pom-aware-code-reuse.test.d.ts +1 -0
  9. package/build/prompts/pom-aware-code-reuse.test.js +11 -0
  10. package/build/prompts/test-recommendation/diffExecutionPlan.d.ts +4 -2
  11. package/build/prompts/test-recommendation/diffExecutionPlan.js +11 -65
  12. package/build/prompts/test-recommendation/fullRepoCatalog.d.ts +2 -2
  13. package/build/prompts/test-recommendation/recommendationSections.js +5 -2
  14. package/build/prompts/test-recommendation/scopeAssessment.js +1 -1
  15. package/build/prompts/test-recommendation/test-recommendation-prompt.d.ts +26 -1
  16. package/build/prompts/test-recommendation/test-recommendation-prompt.js +68 -56
  17. package/build/prompts/test-recommendation/test-recommendation-prompt.test.js +48 -11
  18. package/build/prompts/testbot/testbot-prompts.d.ts +1 -1
  19. package/build/prompts/testbot/testbot-prompts.js +67 -21
  20. package/build/prompts/testbot/testbot-prompts.test.js +44 -0
  21. package/build/recommendation/budgeters/diversityBalancedBudgeter.d.ts +7 -0
  22. package/build/recommendation/budgeters/diversityBalancedBudgeter.js +71 -0
  23. package/build/recommendation/budgeters/diversityBalancedBudgeter.test.d.ts +1 -0
  24. package/build/recommendation/budgeters/diversityBalancedBudgeter.test.js +75 -0
  25. package/build/recommendation/budgeters/fixedNBudgeter.d.ts +7 -0
  26. package/build/recommendation/budgeters/fixedNBudgeter.js +11 -0
  27. package/build/recommendation/budgeters/fixedNBudgeter.test.d.ts +1 -0
  28. package/build/recommendation/budgeters/fixedNBudgeter.test.js +66 -0
  29. package/build/recommendation/budgeters/shared.d.ts +19 -0
  30. package/build/recommendation/budgeters/shared.js +66 -0
  31. package/build/recommendation/discriminators.d.ts +31 -0
  32. package/build/recommendation/discriminators.js +355 -0
  33. package/build/recommendation/discriminators.test.d.ts +1 -0
  34. package/build/recommendation/discriminators.test.js +324 -0
  35. package/build/recommendation/diversity.d.ts +47 -0
  36. package/build/recommendation/diversity.js +101 -0
  37. package/build/recommendation/diversity.test.d.ts +1 -0
  38. package/build/recommendation/diversity.test.js +77 -0
  39. package/build/recommendation/planRanker.d.ts +50 -0
  40. package/build/recommendation/planRanker.js +67 -0
  41. package/build/recommendation/planRanker.test.d.ts +1 -0
  42. package/build/recommendation/planRanker.test.js +110 -0
  43. package/build/recommendation/testFixtures.d.ts +25 -0
  44. package/build/recommendation/testFixtures.js +45 -0
  45. package/build/resources/testbotResource.js +4 -1
  46. package/build/services/ScenarioGenerationService.d.ts +5 -0
  47. package/build/services/ScenarioGenerationService.js +16 -1
  48. package/build/services/ScenarioGenerationService.test.js +44 -0
  49. package/build/services/TestExecutionService.d.ts +15 -1
  50. package/build/services/TestExecutionService.js +210 -55
  51. package/build/services/TestExecutionService.test.js +397 -0
  52. package/build/services/TestGenerationService.js +19 -1
  53. package/build/services/TestGenerationService.test.js +58 -0
  54. package/build/tool-phases.js +1 -0
  55. package/build/toolNames.d.ts +19 -0
  56. package/build/toolNames.js +19 -0
  57. package/build/tools/code-refactor/codeReuseTool.d.ts +7 -0
  58. package/build/tools/code-refactor/codeReuseTool.js +130 -4
  59. package/build/tools/code-refactor/codeReuseTool.test.d.ts +1 -0
  60. package/build/tools/code-refactor/codeReuseTool.test.js +290 -0
  61. package/build/tools/executeSkyrampTestTool.js +8 -2
  62. package/build/tools/generate-tests/generateBatchScenarioRestTool.d.ts +6 -1
  63. package/build/tools/generate-tests/generateBatchScenarioRestTool.js +110 -17
  64. package/build/tools/generate-tests/generateBatchScenarioRestTool.test.js +147 -0
  65. package/build/tools/generate-tests/generateContractRestTool.js +11 -1
  66. package/build/tools/generate-tests/generateIntegrationRestTool.js +24 -1
  67. package/build/tools/generate-tests/generateIntegrationRestTool.test.d.ts +1 -0
  68. package/build/tools/generate-tests/generateIntegrationRestTool.test.js +159 -0
  69. package/build/tools/generate-tests/planGuard.d.ts +13 -0
  70. package/build/tools/generate-tests/planGuard.js +78 -0
  71. package/build/tools/generate-tests/planGuard.test.d.ts +1 -0
  72. package/build/tools/generate-tests/planGuard.test.js +185 -0
  73. package/build/tools/generate-tests/scenarioFileIdentity.d.ts +10 -0
  74. package/build/tools/generate-tests/scenarioFileIdentity.js +46 -0
  75. package/build/tools/generate-tests/scenarioLint.d.ts +30 -0
  76. package/build/tools/generate-tests/scenarioLint.js +150 -0
  77. package/build/tools/generate-tests/scenarioLint.test.d.ts +1 -0
  78. package/build/tools/generate-tests/scenarioLint.test.js +100 -0
  79. package/build/tools/submitReportTool.js +78 -0
  80. package/build/tools/submitReportTool.test.js +255 -0
  81. package/build/tools/test-management/analyzeChangesTool.js +55 -2
  82. package/build/tools/test-management/analyzeChangesTool.test.js +12 -0
  83. package/build/tools/test-management/index.d.ts +1 -0
  84. package/build/tools/test-management/index.js +1 -0
  85. package/build/tools/test-management/registerTestPlanTool.d.ts +2 -0
  86. package/build/tools/test-management/registerTestPlanTool.js +329 -0
  87. package/build/tools/test-management/registerTestPlanTool.test.d.ts +1 -0
  88. package/build/tools/test-management/registerTestPlanTool.test.js +296 -0
  89. package/build/types/Recommendation.d.ts +97 -0
  90. package/build/types/Recommendation.js +48 -0
  91. package/build/types/RepositoryAnalysis.d.ts +14 -14
  92. package/build/types/TestExecution.d.ts +2 -0
  93. package/build/types/TestRecommendation.d.ts +12 -1
  94. package/build/types/TestRecommendation.js +26 -11
  95. package/build/types/TestTypes.js +1 -1
  96. package/build/utils/AnalysisStateManager.d.ts +47 -0
  97. package/build/utils/docker.test.js +1 -1
  98. package/build/utils/planMatchKeys.d.ts +61 -0
  99. package/build/utils/planMatchKeys.js +125 -0
  100. package/build/utils/pom-scope/import-expansion.d.ts +5 -0
  101. package/build/utils/pom-scope/import-expansion.js +32 -0
  102. package/build/utils/pom-scope/index.d.ts +39 -0
  103. package/build/utils/pom-scope/index.js +120 -0
  104. package/build/utils/pom-scope/index.test.d.ts +1 -0
  105. package/build/utils/pom-scope/index.test.js +239 -0
  106. package/build/utils/pom-scope/pom-files.d.ts +3 -0
  107. package/build/utils/pom-scope/pom-files.js +48 -0
  108. package/build/utils/pom-scope/pom-files.test.d.ts +1 -0
  109. package/build/utils/pom-scope/pom-files.test.js +29 -0
  110. package/build/utils/pom-scope/scoring.d.ts +20 -0
  111. package/build/utils/pom-scope/scoring.js +45 -0
  112. package/build/utils/pom-scope/scoring.test.d.ts +1 -0
  113. package/build/utils/pom-scope/scoring.test.js +39 -0
  114. package/build/utils/pom-scope/selector-extractor.d.ts +7 -0
  115. package/build/utils/pom-scope/selector-extractor.js +57 -0
  116. package/build/utils/pom-scope/selector-extractor.test.d.ts +1 -0
  117. package/build/utils/pom-scope/selector-extractor.test.js +67 -0
  118. package/build/utils/pom-verify/__fixtures__/af-style/asset-list.page.d.ts +5 -0
  119. package/build/utils/pom-verify/__fixtures__/af-style/asset-list.page.js +5 -0
  120. package/build/utils/pom-verify/__fixtures__/af-style/pageobjects/asset-list-page.d.ts +5 -0
  121. package/build/utils/pom-verify/__fixtures__/af-style/pageobjects/asset-list-page.js +9 -0
  122. package/build/utils/pom-verify/__fixtures__/af-style/report.iframe.page.d.ts +4 -0
  123. package/build/utils/pom-verify/__fixtures__/af-style/report.iframe.page.js +4 -0
  124. package/build/utils/pom-verify/__fixtures__/af-style/workflow-footer.page.d.ts +4 -0
  125. package/build/utils/pom-verify/__fixtures__/af-style/workflow-footer.page.js +4 -0
  126. package/build/utils/pom-verify/bindings.d.ts +19 -0
  127. package/build/utils/pom-verify/bindings.js +161 -0
  128. package/build/utils/pom-verify/bindings.test.d.ts +1 -0
  129. package/build/utils/pom-verify/bindings.test.js +164 -0
  130. package/build/utils/pom-verify/calls.d.ts +16 -0
  131. package/build/utils/pom-verify/calls.js +42 -0
  132. package/build/utils/pom-verify/calls.test.d.ts +1 -0
  133. package/build/utils/pom-verify/calls.test.js +61 -0
  134. package/build/utils/pom-verify/index.d.ts +4 -0
  135. package/build/utils/pom-verify/index.js +4 -0
  136. package/build/utils/pom-verify/resolve.d.ts +7 -0
  137. package/build/utils/pom-verify/resolve.js +27 -0
  138. package/build/utils/pom-verify/resolve.test.d.ts +1 -0
  139. package/build/utils/pom-verify/resolve.test.js +68 -0
  140. package/build/utils/pom-verify/strip.d.ts +9 -0
  141. package/build/utils/pom-verify/strip.js +89 -0
  142. package/build/utils/pom-verify/verify.d.ts +14 -0
  143. package/build/utils/pom-verify/verify.js +158 -0
  144. package/build/utils/pom-verify/verify.test.d.ts +1 -0
  145. package/build/utils/pom-verify/verify.test.js +325 -0
  146. package/build/utils/reportVerification.d.ts +61 -0
  147. package/build/utils/reportVerification.js +104 -0
  148. package/build/utils/reportVerification.test.d.ts +1 -0
  149. package/build/utils/reportVerification.test.js +185 -0
  150. package/build/utils/scenarioDrafting.js +5 -5
  151. package/build/utils/versions.d.ts +3 -3
  152. package/build/utils/versions.js +1 -1
  153. package/build/utils/workspaceAuth.d.ts +9 -1
  154. package/build/utils/workspaceAuth.js +25 -5
  155. package/build/utils/workspaceAuth.test.js +48 -0
  156. package/build/workspace/workspace.d.ts +20 -0
  157. package/build/workspace/workspace.js +4 -0
  158. package/build/workspace/workspace.test.js +10 -0
  159. package/node_modules/playwright/lib/mcp/skyramp/traceRecordingBackend.js +77 -8
  160. package/node_modules/playwright/node_modules/playwright-core/lib/generated/injectedScriptSource.js +1 -1
  161. package/node_modules/playwright/node_modules/playwright-core/lib/generated/pollingRecorderSource.js +1 -1
  162. package/node_modules/playwright/node_modules/playwright-core/lib/utils/isomorphic/volatileDate.js +101 -0
  163. package/node_modules/playwright/node_modules/playwright-core/lib/utils.js +2 -0
  164. package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/assets/{codeMirrorModule-aszq5EdG.js → codeMirrorModule-Bzd72-bG.js} +1 -1
  165. package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/assets/{defaultSettingsView-BxS7Jm4s.js → defaultSettingsView-DzxTioTK.js} +101 -101
  166. package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/{index.D4JTTy4R.js → index.BGc30U3S.js} +1 -1
  167. package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/index.html +2 -2
  168. package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/{uiMode.DaRMQKOI.js → uiMode.IaDrb29A.js} +1 -1
  169. package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/uiMode.html +2 -2
  170. package/node_modules/playwright/node_modules/playwright-core/package.json +1 -1
  171. package/node_modules/playwright/node_modules/playwright-core/src/generated/injectedScriptSource.ts +1 -1
  172. package/node_modules/playwright/node_modules/playwright-core/src/generated/pollingRecorderSource.ts +1 -1
  173. package/node_modules/playwright/node_modules/playwright-core/src/utils/isomorphic/volatileDate.ts +131 -0
  174. package/node_modules/playwright/node_modules/playwright-core/src/utils.ts +1 -0
  175. package/node_modules/playwright/package.json +1 -1
  176. package/package.json +3 -3
  177. package/node_modules/playwright/node_modules/playwright-core/.DS_Store +0 -0
@@ -47,7 +47,7 @@ export function parseRelatedRepositories(raw) {
47
47
  }
48
48
  }
49
49
  export function getTestbotPrompt(prTitle, prDescription, summaryOutputFile, repositoryPath, baseBranch, maxRecommendations = MAX_RECOMMENDATIONS, maxGenerate = MAX_TESTS_TO_GENERATE, _maxCritical = MAX_CRITICAL_TESTS, // Reserved — accepted for API compat but not yet wired into prompt
50
- prNumber, userPrompt, services, uiCredentials, testsRepoDir, relatedRepositories, primaryRepo) {
50
+ prNumber, userPrompt, services, uiCredentials, testsRepoDir, relatedRepositories, primaryRepo, planOnly = false) {
51
51
  maxGenerate = Math.min(Math.max(maxGenerate, 0), maxRecommendations);
52
52
  // TODO(SKYR-3636 follow-up): migrate Task 1 + Task 2 step bodies to PromptPlan
53
53
  // (src/prompts/test-recommendation/promptPlan.ts) so step numbers don't have
@@ -65,6 +65,17 @@ prNumber, userPrompt, services, uiCredentials, testsRepoDir, relatedRepositories
65
65
  // testDirectory, relative to the delivery root (the configured test repo if set,
66
66
  // otherwise the primary repo).
67
67
  const hasRelatedRepos = !!relatedRepositories?.length;
68
+ // Task 1 maintenance step 2c branches in plan-only eval runs (SKYR-3879
69
+ // plan-only lane): the SUT is not running, so the pre-edit baseline
70
+ // execution is replaced by static-analysis-only verdicts. Defined before
71
+ // task1Section, which interpolates it.
72
+ let maintenanceBeforeExecStep;
73
+ if (planOnly) {
74
+ maintenanceBeforeExecStep = ` c. Plan-only run: skip the pre-edit baseline execution — the application is not running. Record every maintenance verdict from static drift analysis alone; execution statuses simply remain unrecorded.`;
75
+ }
76
+ else {
77
+ maintenanceBeforeExecStep = ` c. Call \`skyramp_execute_test\` with \`phase: "before"\` and \`stateFile\` for every UPDATE/REGENERATE/DELETE test. Exclude tests marked \`[external]\`. Run them sequentially, not in parallel. This captures the pre-edit baseline — do not skip even if you expect the test to fail.`;
78
+ }
68
79
  // For follow-up requests: emit the @skyramp-testbot header + guardrails + retrieve-recommendations step.
69
80
  // For first-run prompts: emit the full Task 1 analysis + maintenance section.
70
81
  const task1Section = userPrompt
@@ -131,7 +142,7 @@ ${hasRelatedRepos ? `
131
142
 
132
143
  b. Write \`updateInstructions\` for each UPDATE or REGENERATE test before calling \`skyramp_actions\` — articulating the change first prevents file content from overriding diff-based reasoning.
133
144
 
134
- c. Call \`skyramp_execute_test\` with \`phase: "before"\` and \`stateFile\` for every UPDATE/REGENERATE/DELETE test. Exclude tests marked \`[external]\`. Run them sequentially, not in parallel. This captures the pre-edit baseline — do not skip even if you expect the test to fail.
145
+ ${maintenanceBeforeExecStep}
135
146
 
136
147
  d. Call \`skyramp_actions\` with \`stateFile\` (from \`skyramp_analyze_changes\` output) and apply the edits it returns.
137
148
 
@@ -343,19 +354,37 @@ ${hasRelatedRepos ? `
343
354
  const testDirInstruction = testsRepoDir
344
355
  ? `the \`<output_dir>\` from the \`<services>\` block, rooted under the test repository at \`${testsRepoDir}\` (i.e. \`${testsRepoDir}/<output_dir>\`). Write ALL test output files to paths under \`${testsRepoDir}\`, not under \`${repositoryPath}\`. Do NOT write any test files to the app repository.`
345
356
  : `${SERVICE_REFS.testDirRef}. Do NOT create a new \`tests/\` directory at the repo root — use that path. If no \`testDirectory\` is configured, default to the language-conventional location (e.g. \`src/test/java/...\` for Java, \`tests/\` for Python).`;
346
- const primaryRepoBlock = primaryRepo ? `<REPOSITORY>${primaryRepo}</REPOSITORY>\n` : '';
347
- return `<TITLE>${prTitle}</TITLE>
348
- <DESCRIPTION>${prDescription}</DESCRIPTION>
349
- ${primaryRepoBlock}<REPOSITORY PATH>${repositoryPath}</REPOSITORY PATH>
350
- ${relatedReposBlock}${testsRepoDirBlock}${serviceContext ? serviceContext + '\n' : ''}${uiCredentialsBlock ? uiCredentialsBlock + '\n' : ''}Use the Skyramp MCP server tools for all tasks below.
351
-
352
- ${task1Section}
353
-
354
- ## Task 2: Generate New Tests
357
+ // Task 3's non-zero-budget count rule branches in plan-only eval runs: the
358
+ // report DECLARES the final GENERATE selection instead of recording
359
+ // generated files. The zero-surface abstention rules above it are shared.
360
+ let task3CountRule;
361
+ if (planOnly) {
362
+ task3CountRule = `Otherwise (your Budget Plan is non-zero): this is a plan-only run — \`newTestsCreated\` DECLARES your final GENERATE selection instead of recording generated files: exactly one entry per item of your final GENERATE list (the list returned by \`skyramp_register_test_plan\` when that tool was available — it may contain fewer items than your committed budget — otherwise your Budget Plan selection, at most ${maxGenerate}), with \`fileName\` set to the file name you would have used. \`testResults\` must be \`[]\` — nothing was executed. The declaration itself is the deliverable; do not generate or backfill.`;
363
+ }
364
+ else {
365
+ task3CountRule = `Otherwise (your Budget Plan is non-zero): in \`newTestsCreated\`, you must have exactly as many budget-counting new tests as your committed Budget Plan's generate count (at most ${maxGenerate}). Only new files (ADD) created for the planned GENERATE items count toward this target — GENERATE items converted to UPDATE do not. You may also include at most one additional discovered-scenario file in \`newTestsCreated\` (the bug-catching test generated after all planned items); that extra test does **not** count against the budget. If you have fewer budget-counting new tests than your generate count, backfill from the remaining ADDITIONAL candidates before proceeding. Only proceed with fewer if all candidates failed after retry AND the fallback single-contract test also failed.`;
366
+ }
367
+ // Task 2 branches wholesale in plan-only eval runs (SKYR-3879 plan-only
368
+ // lane): the standard task mandates generation and execution, which a
369
+ // plan-only run must not do — branching the section (rather than appending
370
+ // an override) means the agent never sees conflicting instructions.
371
+ let task2Section;
372
+ if (planOnly) {
373
+ task2Section = `## Task 2: Commit the Test Plan (plan-only run)
374
+
375
+ This is a plan-only evaluation run: the application under test is NOT running, and this run evaluates test SELECTION only. Nothing is generated or executed in this task.
376
+
377
+ - Draft your complete candidate list exactly as the Execution Plan directs — every API test (contract / integration / batch-scenario) you would generate OR recommend for this PR, grounded in the analysis output and the diff. Favor tests that would FAIL if the changed logic were buggy, not just tests that exercise the new surface.
378
+ - If a tool named \`skyramp_register_test_plan\` is available, call it with the full candidate union — include a \`discriminator\` claim \`{kind, changedCodeAnchor}\` for every candidate that probes changed logic — and treat its returned GENERATE list as your final selection. If that tool is not available, commit to your Budget Plan's GENERATE selection (at most ${maxGenerate}).
379
+ - Skip UI and E2E candidates entirely — with no running app there are no blueprints to ground them, and this lane evaluates API test selection only.
380
+ - Take no other actions in this task: no test generation tools, no browser traces or blueprint captures, no test files written, no test executions. Proceed directly to ${taskRef(TASK_SUBMIT)}.`;
381
+ }
382
+ else {
383
+ task2Section = `## Task 2: Generate New Tests
355
384
 
356
385
  ${userPrompt ? "Generate only the tests that the user requested from the Additional Recommendations. The rules below still apply." : "Drift-based maintenance (Task 1) is complete. This step only processes the GENERATE list. Exception: if a GENERATE item targets a resource with an existing `[skyramp]` contract test, UPDATE that test file (see covered-resource handling below) — a new test case added to an existing file counts toward the budget and is reported in `newTestsCreated`."}
357
386
 
358
- - **MANDATORY — use the pre-ranked GENERATE list as-is**: The Execution Plan's GENERATE section governs ADD actions. You MUST generate exactly those scenarios in the exact order listed. Do NOT substitute, rename, or replace a GENERATE item. If parameter grounding uncovers a distinct bug-catching scenario not already in the GENERATE or ADDITIONAL list, generate it after all planned GENERATE items are complete and report it in \`newTestsCreated\` — this is an additional test driven by source-code analysis and does not count against the GENERATE budget.${hasRelatedRepos ? `\n - **Multi-repo exception:** this run has related repositories, so the per-repo GENERATE lists are NOT final — they are candidates re-selected by the cross-repo round-robin described in Task 1's "Cross-repo test generation". Follow that pooled, type-distributed selection instead of any single repo's GENERATE list. (In single-repo runs the GENERATE list IS final — generate it exactly as-is.)` : ""}
387
+ - **MANDATORY — use the plan returned by \`skyramp_register_test_plan\` as-is**: Before generating anything, call \`skyramp_register_test_plan\` (\`stateFile\` required) with your complete candidate list — every test you would generate OR recommend, including the Execution Plan's own pre-ranked GENERATE/ADDITIONAL items and any candidate you drafted yourself, with a \`discriminator\` claim \`{kind, changedCodeAnchor}\` for candidates probing changed logic. Its returned GENERATE list — not the Execution Plan's raw pre-ranked GENERATE section — governs ADD actions from this point on. You MUST generate exactly those scenarios in the exact order listed, keeping each item's \`scenarioName\` exactly as registered — the generation tools match on it and reject renamed or substituted scenarios. If parameter grounding uncovers a distinct bug-catching scenario not already registered, generate it after all planned GENERATE items are complete and report it in \`newTestsCreated\` — this is an additional test driven by source-code analysis and does not count against the GENERATE budget.${hasRelatedRepos ? `\n - **Multi-repo exception:** this run has related repositories, so the per-repo GENERATE lists are NOT final — they are candidates re-selected by the cross-repo round-robin described in Task 1's "Cross-repo test generation". Register the pooled, type-distributed selection instead of any single repo's GENERATE list. (In single-repo runs, register the GENERATE list exactly as-is.)` : ""}
359
388
  - **Do not fabricate tests outside the GENERATE list.** New test files cover NEW observable surface only — a new endpoint, or a newly-integrated component/route not already covered by an existing test. Changes that only modify, delete, or add fields to an EXISTING covered endpoint or component are maintenance: handle them in ${taskRef(TASK_ANALYZE_MAINTAIN)} by UPDATE/DELETE of the existing test, never by creating a new spec. If the GENERATE list is empty (deletion-only, cosmetic, or modification-of-existing PRs with no new surface), create zero new tests and proceed to ${taskRef(TASK_SUBMIT)} — do not invent a new spec to have something to report.
360
389
  - Scenario JSON files are always new files — always generate them for new methods. Every generated scenario JSON must have a corresponding new integration test generated from it via \`skyramp_integration_test_generation\`.
361
390
  - Covered-resource handling (aligns with Execution Plan Step 0): When a GENERATE item targets a resource that already has an existing test file covering the same endpoint:
@@ -416,7 +445,7 @@ ${userPrompt ? "Generate only the tests that the user requested from the Additio
416
445
  ${CONTRACT_MODE_GUIDANCE}
417
446
  - ${PATH_PARAM_UUID_GUIDANCE}
418
447
  - **UI**: First check for existing Playwright trace \`.zip\` files in the repo (Testbot scans recursively up to 5 directory levels — the per-service output directories, \`frontend/\`, \`public/\`, \`.skyramp/\`, or any subdirectory).
419
- If a relevant trace exists (covers the UI changes in this PR), use it directly with \`skyramp_ui_test_generation\` and \`modularizeCode: false\`.
448
+ If a relevant trace exists (covers the UI changes in this PR), use it directly with \`skyramp_ui_test_generation\`, \`modularizeCode: false\`, and \`codeReuse: true\` (when generating a TypeScript/JavaScript Playwright test — the default; leave \`codeReuse\` unset for other languages).
420
449
  If NO relevant trace exists, **you MUST write out your full trace plan as text BEFORE calling \`browser_navigate\`**. Do not touch the browser until the plan is written.
421
450
 
422
451
  **Browser authentication (check BEFORE navigating)**: If \`<ui-credentials>\` appears in your context above, the app requires login. Parse the credentials — one per line, two supported formats:
@@ -460,7 +489,7 @@ ${CONTRACT_MODE_GUIDANCE}
460
489
  Follow the **UI Recording Workflow** section at the end of this prompt. Additional CI constraints:
461
490
  - Navigate **directly** to the deepest relevant URL (e.g. \`/orders/1/edit\` instead of \`/\` then \`/orders\` then \`/orders/1\`) — minimize multi-hop navigation so the trace stays focused on the scenario under test.
462
491
  - \`skyramp_export_zip\` outputPath: \`${repositoryPath}/.skyramp/<test_name>_trace.zip\`
463
- - \`skyramp_ui_test_generation\`: set \`modularizeCode: false\`
492
+ - \`skyramp_ui_test_generation\`: set \`modularizeCode: false\` and \`codeReuse: true\` (TypeScript/JavaScript Playwright only — the default; leave \`codeReuse\` unset for other languages)
464
493
  - **\`browser_assert\` — MANDATORY**: at least one per page navigated. Call multiple assertions in the same tool call batch when checking independent elements. If you navigate to 2 pages, assert on both. Each assertion should verify a business outcome (state change, computed value, error condition) — not just that an element is visible.
465
494
  - **\`browser_visual_snapshot\` — for visual/appearance checks**: when the instruction asks to take a screenshot, capture a baseline, or verify how a page/element/region *looks* (not its text or value), call \`browser_visual_snapshot\` — it records a \`toHaveScreenshot()\` assertion so the generated test pixel-compares against a baseline on every run. Do NOT use \`browser_take_screenshot\` for this: it captures a throwaway image that is dropped at export and never appears in the generated test (use it only to view the page yourself).
466
495
  - **Wait for stable state before the second capture**: After performing an action that affects computed fields (filling a discount, submitting a form, adding an item), check the current page state before calling the second \`browser_blueprint\` (the capture after the action). If a computed field — total, price, count, derived text — still shows its initial empty or zero value (e.g. \`$0.00\`, \`0\`, \`Loading...\`, empty string), that means async data hasn't finished loading yet. Use \`browser_wait_for\` to wait up to 10 seconds for the field to update to a real value (for example, wait for the total to show a non-zero amount like \`$799.99\` instead of \`$0.00\`). Once the field shows a real value, THEN call the second \`browser_blueprint\` to capture stable state. If after 10 seconds the field still hasn't updated, skip the assertion on that field — don't capture and assert a value that hasn't loaded.
@@ -512,7 +541,7 @@ ${CONTRACT_MODE_GUIDANCE}
512
541
 
513
542
  **The Blueprint Citation Invariant applies during recording too.** Every assertion you emit cites element names — those names must come from blueprint captures, not invention. For N user-intent-level actions, the reference target is N+1 \`browser_blueprint\` calls (the first returns full, the rest return deltas). Traces that follow the pattern produce assertions grounded in observable state changes; traces that skip captures fall back to author-inferred assertions and risk citing names that don't exist in the rendered DOM.
514
543
 
515
- The rest of the UI workflow stays the same: trace plan, browser auth, navigation, export (\`skyramp_export_zip\`), generation (\`skyramp_ui_test_generation\`), \`skyramp_enhance_assertions\` post-call. Capture-act-capture adds blueprint captures alongside the existing steps; it doesn't replace anything.
544
+ The rest of the UI workflow stays the same: trace plan, browser auth, navigation, export (\`skyramp_export_zip\`), generation (\`skyramp_ui_test_generation\`), then \`skyramp_reuse_code\` (when \`codeReuse: true\`) and \`skyramp_enhance_assertions\` post-calls. Capture-act-capture adds blueprint captures alongside the existing steps; it doesn't replace anything.
516
545
  - **E2E**: Only if BOTH a backend trace \`.json\` AND a Playwright \`.zip\` already exist in the repo. Without both, move to \`additionalRecommendations\`.
517
546
  - Skip smoke tests entirely.
518
547
 
@@ -545,13 +574,24 @@ Do NOT use \`page.waitForTimeout()\` with fixed delays. Do NOT retry more than o
545
574
  **After generation, you MUST do exactly these steps — nothing more, nothing less:**
546
575
  1. **[MANDATORY] After \`skyramp_integration_test_generation\`**: Call \`skyramp_enhance_assertions\` with \`testFile\` set to the absolute path of the generated integration test file, \`testType: "integration"\`, and \`enhanceType: "generation"\`. Apply every instruction returned to that file.
547
576
  2. **[MANDATORY] After \`skyramp_contract_test_generation\` with \`providerMode\`**: Call \`skyramp_enhance_assertions\` with \`testFile\` set to the absolute path of the generated provider contract test file, \`testType: "contract"\`, and \`enhanceType: "generation"\`. Apply every instruction returned to that file.
548
- 3. **[MANDATORY] After \`skyramp_ui_test_generation\`**: Call \`skyramp_enhance_assertions\` with \`testFile\` set to the absolute path of the generated UI test file, \`testType: "ui"\`, and \`enhanceType: "generation"\`. Apply every instruction returned to that file. The HIGH-tier \`possibleAssertions\` from your second \`browser_blueprint\` captures (after each action) during trace recording are in your context — when the enhance instructions ask you to add assertions for state-changing actions, use those grounded candidates first (they contain exact computed values from the DOM delta, e.g. \`toHaveText('Total: $899.98')\`). Only fall back to deriving values from the test file or source code when no HIGH-tier candidate covers the action.
549
- 4. **Wait**: Do NOT proceed to test execution until steps 1–3 are complete and the verification checklist in the \`skyramp_enhance_assertions\` tool result has been validated for EVERY generated test file.
550
- Do not make any changes other than the assertion enhancements described above. For example: do not modify auth headers, cookies, tokens, env vars, or imports that the generation tool already set correctly — those are correct by construction and changing them breaks auth or execution.
577
+ 3. **[MANDATORY] After \`skyramp_ui_test_generation\` with \`codeReuse: true\`**: the generation result directs you to call \`skyramp_reuse_code\` — do this BEFORE \`skyramp_enhance_assertions\`. Two outcomes: (1) a response starting "No reusable POM layer detected" — this is a normal outcome, continue immediately (do NOT retry), and note "no POM layer — reuse skipped" in that test's \`testResults\` entry \`details\`; (2) a refactoring workflow — follow it to completion INCLUDING its verification loop (\`skyramp_reuse_code\` with \`verify: true\`), finish only when it reports PASSED, and include \`verification: passed\` in that test's \`testResults\` entry \`details\` in your final report. If a reused test later fails execution and the failure points at a substituted POM call, restore the saved \`<testFile>.raw.bak\` over the test file and re-run — do NOT hand-edit the customer's POM methods (this re-run counts toward the 2-attempt execution cap — prefer this restore over the generic timeout fix-up when the failing locator came from a POM substitution).
578
+ 4. **[MANDATORY] After \`skyramp_ui_test_generation\`**: Call \`skyramp_enhance_assertions\` with \`testFile\` set to the absolute path of the generated UI test file, \`testType: "ui"\`, and \`enhanceType: "generation"\`. Apply every instruction returned to that file. The HIGH-tier \`possibleAssertions\` from your second \`browser_blueprint\` captures (after each action) during trace recording are in your context — when the enhance instructions ask you to add assertions for state-changing actions, use those grounded candidates first (they contain exact computed values from the DOM delta, e.g. \`toHaveText('Total: $899.98')\`). Only fall back to deriving values from the test file or source code when no HIGH-tier candidate covers the action.
579
+ 5. **Wait**: Do NOT proceed to test execution until steps 1–4 are complete and the verification checklist in the \`skyramp_enhance_assertions\` tool result has been validated for EVERY generated test file.
580
+ Do not make any changes other than the code-reuse refactoring (step 3) and the assertion enhancements described above. For example: do not modify auth headers, cookies, tokens, env vars, or imports that the generation tool already set correctly — those are correct by construction and changing them breaks auth or execution.
551
581
 
552
582
  **Final execution (mandatory):** Do NOT call \`skyramp_execute_test\` until ALL maintenance edits AND ALL new test generation/enhancement are complete. Run these calls sequentially, not in parallel. Exclude tests marked \`[external]\`.
553
583
  - Only report test results for files you actually ran.
554
- **Auth**: If \`skyramp_analyze_changes\` reports an auth token or \`SKYRAMP_TEST_TOKEN\` is set, pass it in **every** \`skyramp_execute_test\` call from the first attempt — do NOT wait for a 401/403 to discover auth is needed.
584
+ **Auth**: If \`skyramp_analyze_changes\` reports an auth token or \`SKYRAMP_TEST_TOKEN\` is set, pass it in **every** \`skyramp_execute_test\` call from the first attempt — do NOT wait for a 401/403 to discover auth is needed.`;
585
+ }
586
+ const primaryRepoBlock = primaryRepo ? `<REPOSITORY>${primaryRepo}</REPOSITORY>\n` : '';
587
+ return `<TITLE>${prTitle}</TITLE>
588
+ <DESCRIPTION>${prDescription}</DESCRIPTION>
589
+ ${primaryRepoBlock}<REPOSITORY PATH>${repositoryPath}</REPOSITORY PATH>
590
+ ${relatedReposBlock}${testsRepoDirBlock}${serviceContext ? serviceContext + '\n' : ''}${uiCredentialsBlock ? uiCredentialsBlock + '\n' : ''}Use the Skyramp MCP server tools for all tasks below.
591
+
592
+ ${task1Section}
593
+
594
+ ${task2Section}
555
595
 
556
596
  ## Task 3: Submit Report
557
597
 
@@ -571,7 +611,7 @@ In these cases:
571
611
  - \`businessCaseAnalysis\` must be a one-sentence summary of what the PR actually does (do NOT leave it blank)
572
612
  - \`additionalRecommendations\` must be \`[]\` — do NOT recommend tests for a no-surface PR
573
613
 
574
- Otherwise (your Budget Plan is non-zero): in \`newTestsCreated\`, you must have exactly as many budget-counting new tests as your committed Budget Plan's generate count (at most ${maxGenerate}). Only new files (ADD) created for the planned GENERATE items count toward this target — GENERATE items converted to UPDATE do not. You may also include at most one additional discovered-scenario file in \`newTestsCreated\` (the bug-catching test generated after all planned items); that extra test does **not** count against the budget. If you have fewer budget-counting new tests than your generate count, backfill from the remaining ADDITIONAL candidates before proceeding. Only proceed with fewer if all candidates failed after retry AND the fallback single-contract test also failed.
614
+ ${task3CountRule}
575
615
 
576
616
  Call \`skyramp_submit_report\` with \`summaryOutputFile\`: "${summaryOutputFile}" and \`stateFile\` (from \`skyramp_analyze_changes\` output) — the stateFile is required for execution outcome tracking. Field names, types, and formats are defined in the tool's parameter schema — follow them exactly.
577
617
 
@@ -683,10 +723,16 @@ export function registerTestbotPrompt(server) {
683
723
  .string()
684
724
  .optional()
685
725
  .describe("The primary repository's owner/repo slug (where the testbot workflow runs). Used verbatim as the `repository` attribution for the primary repo's analysis, tests, and report items in multi-repo runs."),
726
+ planOnly: z
727
+ // Prompt/resource URL args arrive as strings — accept "true"/"1"
728
+ // alongside a real boolean; anything else (incl. undefined) is false.
729
+ .preprocess((v) => v === true || v === "true" || v === "1", z.boolean())
730
+ .optional()
731
+ .describe("Plan-only eval mode (eval harness only, SKYR-3879): run the full recommendation phase — analysis, maintenance verdicts, candidate drafting, plan registration — but generate and execute nothing; the report declares the final plan. Used by the eval pipeline to A/B selection changes without a running SUT."),
686
732
  },
687
733
  }, async (args) => {
688
734
  const services = await readWorkspaceServices(args.repositoryPath);
689
- let prompt = getTestbotPrompt(args.prTitle, args.prDescription, args.summaryOutputFile, args.repositoryPath, args.baseBranch, args.maxRecommendations, args.maxGenerate, args.maxCritical, args.prNumber, args.userPrompt, services.length ? services : undefined, args.uiCredentials, args.testsRepoDir, parseRelatedRepositories(args.relatedRepositories), args.primaryRepo);
735
+ let prompt = getTestbotPrompt(args.prTitle, args.prDescription, args.summaryOutputFile, args.repositoryPath, args.baseBranch, args.maxRecommendations, args.maxGenerate, args.maxCritical, args.prNumber, args.userPrompt, services.length ? services : undefined, args.uiCredentials, args.testsRepoDir, parseRelatedRepositories(args.relatedRepositories), args.primaryRepo, args.planOnly);
690
736
  if (args.workspaceValidationFailed) {
691
737
  prompt = buildWorkspaceRecoveryPrefix(args.repositoryPath) + prompt;
692
738
  }
@@ -378,3 +378,47 @@ describe("parseRelatedRepositories", () => {
378
378
  expect(parsed).toEqual([{ repo: "o/ok", repositoryPath: "/ok", baseBranch: undefined }]);
379
379
  });
380
380
  });
381
+ describe("plan-only eval mode (via getTestbotPrompt)", () => {
382
+ function callWithPlanOnly(planOnly) {
383
+ return getTestbotPrompt(baseArgs.prTitle, baseArgs.prDescription, baseArgs.summaryOutputFile, baseArgs.repositoryPath, undefined, // baseBranch
384
+ undefined, // maxRecommendations
385
+ undefined, // maxGenerate
386
+ undefined, // maxCritical
387
+ undefined, // prNumber
388
+ undefined, // userPrompt
389
+ undefined, // services
390
+ undefined, // uiCredentials
391
+ undefined, // testsRepoDir
392
+ undefined, // relatedRepositories
393
+ undefined, // primaryRepo
394
+ planOnly);
395
+ }
396
+ it("branches Task 2 to the plan-only variant when planOnly is true", () => {
397
+ const prompt = callWithPlanOnly(true);
398
+ expect(prompt).toContain("## Task 2: Commit the Test Plan (plan-only run)");
399
+ expect(prompt).not.toContain("## Task 2: Generate New Tests");
400
+ // The standard generation/execution mandates must be absent, not overridden.
401
+ expect(prompt).not.toContain("skyramp_integration_test_generation");
402
+ expect(prompt).not.toContain('phase: "before"');
403
+ // Task 3 declares the plan instead of counting generated files.
404
+ expect(prompt).toContain("DECLARES your final GENERATE selection");
405
+ });
406
+ it("renders the standard prompt untouched when planOnly is false or omitted", () => {
407
+ const prompt = callWithPlanOnly(false);
408
+ expect(prompt).toContain("## Task 2: Generate New Tests");
409
+ expect(prompt).not.toContain("plan-only");
410
+ expect(prompt).toBe(callWithServices([]));
411
+ });
412
+ });
413
+ describe("UI code reuse wiring", () => {
414
+ it("instructs codeReuse: true for TS/JS Playwright UI generation", () => {
415
+ const p = callWithServices([]);
416
+ expect(p).toContain("codeReuse: true");
417
+ });
418
+ it("teaches both reuse outcomes and the raw.bak insurance rule", () => {
419
+ const p = callWithServices([]);
420
+ expect(p).toContain("No reusable POM layer detected");
421
+ expect(p).toContain("verification: passed");
422
+ expect(p).toContain(".raw.bak");
423
+ });
424
+ });
@@ -0,0 +1,7 @@
1
+ import { Budgeter } from "../../types/Recommendation.js";
2
+ /**
3
+ * Budgeter that guarantees cross-test-type coverage in the GENERATE set.
4
+ * `maxGenerate` acts as an upper cap (not a hard N). Fixes the single-type skew
5
+ * that starves contract/error-path tests. Opt-in via strategy selection.
6
+ */
7
+ export declare const diversityBalancedBudgeter: Budgeter;
@@ -0,0 +1,71 @@
1
+ import { runBudget } from "./shared.js";
2
+ import { bucketByType, inferScenarioType, isProtectedCandidate, roundRobinFill, } from "../diversity.js";
3
+ const typeOf = (c) => inferScenarioType(c.scenario);
4
+ /**
5
+ * Pick `count` items guaranteeing each present test type at least one slot.
6
+ *
7
+ * Unlike roundRobinByType (which fills ALL protected items first and can
8
+ * starve other types at small N — the SKYR-3879 "0 contract executed" skew),
9
+ * this caps protected occupancy to `count - (numTypes - 1)` so at least one slot
10
+ * per remaining type is always reachable, then:
11
+ * Phase 1 — one item for each still-uncovered present type (by rank), then
12
+ * Phase 2 — round-robin the remainder across types until `count` is reached.
13
+ * Rank order is preserved within each type; degenerate cases match a rank slice.
14
+ */
15
+ function floorBalancedPick(items, count) {
16
+ if (count <= 0)
17
+ return [];
18
+ if (count >= items.length)
19
+ return items.slice(0, count);
20
+ // Present types in first-appearance (rank) order.
21
+ const typesOrder = [];
22
+ const seen = new Set();
23
+ for (const it of items) {
24
+ const t = typeOf(it);
25
+ if (!seen.has(t)) {
26
+ seen.add(t);
27
+ typesOrder.push(t);
28
+ }
29
+ }
30
+ const maxProtected = Math.max(0, count - (typesOrder.length - 1));
31
+ const selected = [];
32
+ const pool = [];
33
+ let protectedTaken = 0;
34
+ for (const it of items) {
35
+ if (isProtectedCandidate(it.priority, it.scenario) &&
36
+ protectedTaken < maxProtected &&
37
+ selected.length < count) {
38
+ selected.push(it);
39
+ protectedTaken++;
40
+ }
41
+ else {
42
+ pool.push(it);
43
+ }
44
+ }
45
+ const { buckets } = bucketByType(pool, typeOf);
46
+ const covered = new Set(selected.map(typeOf));
47
+ // Phase 1 — floor: one item per still-uncovered present type.
48
+ for (const t of typesOrder) {
49
+ if (selected.length >= count)
50
+ break;
51
+ if (covered.has(t))
52
+ continue;
53
+ const bucket = buckets.get(t);
54
+ if (bucket && bucket.length > 0) {
55
+ selected.push(bucket.shift());
56
+ covered.add(t);
57
+ }
58
+ }
59
+ // Phase 2 — round-robin the remainder across types until full.
60
+ roundRobinFill(selected, typesOrder, buckets, count);
61
+ return selected;
62
+ }
63
+ /**
64
+ * Budgeter that guarantees cross-test-type coverage in the GENERATE set.
65
+ * `maxGenerate` acts as an upper cap (not a hard N). Fixes the single-type skew
66
+ * that starves contract/error-path tests. Opt-in via strategy selection.
67
+ */
68
+ export const diversityBalancedBudgeter = {
69
+ name: "diversity-balanced",
70
+ select: (ranked, ctx) => runBudget(ranked, ctx, floorBalancedPick),
71
+ };
@@ -0,0 +1,75 @@
1
+ import { diversityBalancedBudgeter } from "./diversityBalancedBudgeter.js";
2
+ import { PriorityTier } from "../../types/TestRecommendation.js";
3
+ import { fixedNBudgeter } from "./fixedNBudgeter.js";
4
+ import { mkCandidate } from "../testFixtures.js";
5
+ function ctx(over = {}) {
6
+ return {
7
+ maxGenerate: 3,
8
+ maxTotal: 20,
9
+ isUIOnlyPR: false,
10
+ hasFrontendChanges: false,
11
+ externalCoverage: new Set(),
12
+ ...over,
13
+ };
14
+ }
15
+ const countType = (r, t) => r.generate.filter((c) => c.scenario.testType === t).length;
16
+ const typesIn = (r) => new Set(r.generate.map((c) => c.scenario.testType));
17
+ describe("diversityBalancedBudgeter", () => {
18
+ it("guarantees a contract slot where fixed-n starves it (the SKYR-3879 fix)", () => {
19
+ // Protected (CRITICAL) integration scenarios out-rank contract ones and, at N=3,
20
+ // fixed-n's protected-first fill consumes every slot with integration → 0 contract.
21
+ const ranked = [
22
+ mkCandidate("i1", { testType: "integration", priority: PriorityTier.CRITICAL }),
23
+ mkCandidate("i2", { testType: "integration", priority: PriorityTier.CRITICAL }),
24
+ mkCandidate("i3", { testType: "integration", priority: PriorityTier.CRITICAL }),
25
+ mkCandidate("c1", { testType: "contract" }),
26
+ mkCandidate("c2", { testType: "contract" }),
27
+ ];
28
+ const c = ctx({ maxGenerate: 3 });
29
+ // Baseline: fixed-n reproduces the skew.
30
+ expect(countType(fixedNBudgeter.select(ranked, c), "contract")).toBe(0);
31
+ // Fix: diversity-balanced reserves a per-type floor.
32
+ const div = diversityBalancedBudgeter.select(ranked, c);
33
+ expect(div.generate.length).toBe(3);
34
+ expect(countType(div, "contract")).toBeGreaterThanOrEqual(1);
35
+ expect(countType(div, "integration")).toBeGreaterThanOrEqual(1);
36
+ });
37
+ it("gives every present test type at least one slot before any type gets a second", () => {
38
+ const ranked = [
39
+ mkCandidate("i1", { testType: "integration" }),
40
+ mkCandidate("i2", { testType: "integration" }),
41
+ mkCandidate("i3", { testType: "integration" }),
42
+ mkCandidate("c1", { testType: "contract" }),
43
+ mkCandidate("c2", { testType: "contract" }),
44
+ mkCandidate("u1", { testType: "ui" }),
45
+ ];
46
+ const div = diversityBalancedBudgeter.select(ranked, ctx({ maxGenerate: 3 }));
47
+ expect(typesIn(div)).toEqual(new Set(["integration", "contract", "ui"]));
48
+ });
49
+ it("raising maxGenerate increases per-type coverage while keeping the floor", () => {
50
+ const ranked = [
51
+ mkCandidate("i1", { testType: "integration", priority: PriorityTier.CRITICAL }),
52
+ mkCandidate("i2", { testType: "integration", priority: PriorityTier.CRITICAL }),
53
+ mkCandidate("i3", { testType: "integration", priority: PriorityTier.CRITICAL }),
54
+ mkCandidate("i4", { testType: "integration", priority: PriorityTier.CRITICAL }),
55
+ mkCandidate("i5", { testType: "integration", priority: PriorityTier.CRITICAL }),
56
+ mkCandidate("c1", { testType: "contract" }),
57
+ mkCandidate("c2", { testType: "contract" }),
58
+ mkCandidate("u1", { testType: "ui" }),
59
+ ];
60
+ const atThree = diversityBalancedBudgeter.select(ranked, ctx({ maxGenerate: 3 }));
61
+ const atSix = diversityBalancedBudgeter.select(ranked, ctx({ maxGenerate: 6 }));
62
+ expect(countType(atThree, "contract")).toBeGreaterThanOrEqual(1);
63
+ expect(countType(atThree, "ui")).toBeGreaterThanOrEqual(1);
64
+ expect(countType(atSix, "contract")).toBeGreaterThanOrEqual(1);
65
+ expect(countType(atSix, "ui")).toBeGreaterThanOrEqual(1);
66
+ // more budget → more of the highest-ranked (integration) coverage
67
+ expect(countType(atSix, "integration")).toBeGreaterThan(countType(atThree, "integration"));
68
+ expect(atSix.generate.length).toBe(6);
69
+ });
70
+ it("returns everything (rank order) when maxGenerate >= candidate count", () => {
71
+ const ranked = [mkCandidate("i1", { testType: "integration" }), mkCandidate("c1", { testType: "contract" })];
72
+ const div = diversityBalancedBudgeter.select(ranked, ctx({ maxGenerate: 5 }));
73
+ expect(div.generate.map((c) => c.scenario.scenarioName)).toEqual(["i1", "c1"]);
74
+ });
75
+ });
@@ -0,0 +1,7 @@
1
+ import { Budgeter } from "../../types/Recommendation.js";
2
+ /**
3
+ * Default budgeter — reproduces the pre-refactor behavior exactly: fill up to
4
+ * `maxGenerate` GENERATE slots by round-robin across the present test types
5
+ * (protected items first), the rest become ADDITIONAL.
6
+ */
7
+ export declare const fixedNBudgeter: Budgeter;
@@ -0,0 +1,11 @@
1
+ import { roundRobinByType } from "../diversity.js";
2
+ import { runBudget } from "./shared.js";
3
+ /**
4
+ * Default budgeter — reproduces the pre-refactor behavior exactly: fill up to
5
+ * `maxGenerate` GENERATE slots by round-robin across the present test types
6
+ * (protected items first), the rest become ADDITIONAL.
7
+ */
8
+ export const fixedNBudgeter = {
9
+ name: "fixed-n",
10
+ select: (ranked, ctx) => runBudget(ranked, ctx, roundRobinByType),
11
+ };
@@ -0,0 +1,66 @@
1
+ import { fixedNBudgeter } from "./fixedNBudgeter.js";
2
+ import { mkCandidate } from "../testFixtures.js";
3
+ import { externalDedupKey } from "../../prompts/test-recommendation/recommendationShared.js";
4
+ function ctx(over = {}) {
5
+ return {
6
+ maxGenerate: 3,
7
+ maxTotal: 20,
8
+ isUIOnlyPR: false,
9
+ hasFrontendChanges: false,
10
+ externalCoverage: new Set(),
11
+ ...over,
12
+ };
13
+ }
14
+ const gen = (r) => r.generate.map((c) => c.scenario.scenarioName);
15
+ describe("fixedNBudgeter", () => {
16
+ it("backend-only PR fills maxGenerate slots via round-robin across test types", () => {
17
+ const ranked = [
18
+ mkCandidate("i1", { testType: "integration" }),
19
+ mkCandidate("i2", { testType: "integration" }),
20
+ mkCandidate("i3", { testType: "integration" }),
21
+ mkCandidate("c1", { testType: "contract" }),
22
+ mkCandidate("c2", { testType: "contract" }),
23
+ ];
24
+ const res = fixedNBudgeter.select(ranked, ctx());
25
+ expect(res.generate.length).toBe(3);
26
+ expect(res.reservedUISlots).toBe(0);
27
+ // buckets [integration, contract]; round 1 → i1, c1; round 2 → i2
28
+ expect(gen(res)).toEqual(["i1", "c1", "i2"]);
29
+ });
30
+ it("mixed (frontend) PR reserves one UI slot → backend count = maxGenerate - 1", () => {
31
+ const ranked = [
32
+ mkCandidate("i1", { testType: "integration" }),
33
+ mkCandidate("i2", { testType: "integration" }),
34
+ mkCandidate("c1", { testType: "contract" }),
35
+ ];
36
+ const res = fixedNBudgeter.select(ranked, ctx({ hasFrontendChanges: true }));
37
+ expect(res.generate.length).toBe(2);
38
+ expect(res.reservedUISlots).toBe(1);
39
+ });
40
+ it("UI-only PR selects no backend generate items", () => {
41
+ const res = fixedNBudgeter.select([mkCandidate("i1", { testType: "integration" })], ctx({ isUIOnlyPR: true }));
42
+ expect(res.generate.length).toBe(0);
43
+ });
44
+ it("additional = leftover candidates (rank order), excluding anything generated", () => {
45
+ const ranked = [
46
+ mkCandidate("i1", { testType: "integration" }),
47
+ mkCandidate("c1", { testType: "contract" }),
48
+ mkCandidate("i2", { testType: "integration" }),
49
+ mkCandidate("c2", { testType: "contract" }),
50
+ ];
51
+ const res = fixedNBudgeter.select(ranked, ctx());
52
+ const generated = new Set(gen(res));
53
+ for (const a of res.additional) {
54
+ expect(generated.has(a.scenario.scenarioName)).toBe(false);
55
+ }
56
+ expect(res.additional.length).toBe(1);
57
+ });
58
+ it("drops an ordinary candidate already covered by an external test", () => {
59
+ const covered = mkCandidate("covered", { testType: "contract" });
60
+ const fresh = mkCandidate("fresh", { testType: "integration" });
61
+ const res = fixedNBudgeter.select([covered, fresh], ctx({ maxGenerate: 5, externalCoverage: new Set([externalDedupKey(covered.scenario)]) }));
62
+ const all = [...res.generate, ...res.additional].map((c) => c.scenario.scenarioName);
63
+ expect(all).toContain("fresh");
64
+ expect(all).not.toContain("covered");
65
+ });
66
+ });
@@ -0,0 +1,19 @@
1
+ import { Candidate, BudgetContext, SelectionResult } from "../../types/Recommendation.js";
2
+ /**
3
+ * Backend GENERATE slot count:
4
+ * - UI-only PR: 0 (all slots are UI placeholders)
5
+ * - Mixed PR: maxGenerate - 1 (last slot reserved for a UI placeholder)
6
+ * - Backend-only PR: maxGenerate
7
+ */
8
+ export declare function backendGenerateCount(ctx: BudgetContext): number;
9
+ /** UI placeholder slots the render layer fills (UI-only → all; mixed → one). */
10
+ export declare function reservedUISlots(ctx: BudgetContext): number;
11
+ /**
12
+ * Shared budgeting pipeline. All Budgeters run the same external-dedup,
13
+ * attack-surface prioritization, and ADDITIONAL set-difference; they differ
14
+ * ONLY in how they pick the GENERATE items from the slot-ordered list (`pick`).
15
+ *
16
+ * With `pick = roundRobinByType` this reproduces the pre-refactor selection in
17
+ * diffExecutionPlan.ts exactly.
18
+ */
19
+ export declare function runBudget(ranked: Candidate[], ctx: BudgetContext, pick: (items: Candidate[], count: number) => Candidate[]): SelectionResult;
@@ -0,0 +1,66 @@
1
+ import { prioritizeAttackSurfaceBundles } from "../diversity.js";
2
+ import { externalDedupKey, scenarioCoverageKey, isAttackSurfaceSecurityBoundary, } from "../../prompts/test-recommendation/recommendationShared.js";
3
+ import { logger } from "../../utils/logger.js";
4
+ /**
5
+ * Backend GENERATE slot count:
6
+ * - UI-only PR: 0 (all slots are UI placeholders)
7
+ * - Mixed PR: maxGenerate - 1 (last slot reserved for a UI placeholder)
8
+ * - Backend-only PR: maxGenerate
9
+ */
10
+ export function backendGenerateCount(ctx) {
11
+ if (ctx.isUIOnlyPR)
12
+ return 0;
13
+ return ctx.hasFrontendChanges ? Math.max(0, ctx.maxGenerate - 1) : ctx.maxGenerate;
14
+ }
15
+ /** UI placeholder slots the render layer fills (UI-only → all; mixed → one). */
16
+ export function reservedUISlots(ctx) {
17
+ if (ctx.isUIOnlyPR)
18
+ return ctx.maxGenerate;
19
+ return ctx.hasFrontendChanges && ctx.maxGenerate > 0 ? 1 : 0;
20
+ }
21
+ /**
22
+ * Drop candidates whose method-aware resource+type is already covered by an
23
+ * external test — except protected `bug_caught` / attack-surface scenarios,
24
+ * which require semantic flaw coverage the external test may not provide.
25
+ */
26
+ function applyExternalDedup(ranked, externalCoverage) {
27
+ if (externalCoverage.size === 0)
28
+ return ranked;
29
+ return ranked.filter((item) => {
30
+ const key = externalDedupKey(item.scenario);
31
+ if (externalCoverage.has(key)) {
32
+ if (item.scenario.category === "bug_caught" || isAttackSurfaceSecurityBoundary(item.scenario)) {
33
+ logger.info(`External dedup: preserving "${item.scenario.scenarioName}" (${key}) — protected bug/attack-surface scenario requires semantic flaw coverage`);
34
+ return true;
35
+ }
36
+ logger.info(`External dedup: skipping "${item.scenario.scenarioName}" (${key}) — covered by external test`);
37
+ return false;
38
+ }
39
+ return true;
40
+ });
41
+ }
42
+ /**
43
+ * Shared budgeting pipeline. All Budgeters run the same external-dedup,
44
+ * attack-surface prioritization, and ADDITIONAL set-difference; they differ
45
+ * ONLY in how they pick the GENERATE items from the slot-ordered list (`pick`).
46
+ *
47
+ * With `pick = roundRobinByType` this reproduces the pre-refactor selection in
48
+ * diffExecutionPlan.ts exactly.
49
+ */
50
+ export function runBudget(ranked, ctx, pick) {
51
+ const backend = backendGenerateCount(ctx);
52
+ const deduped = applyExternalDedup(ranked, ctx.externalCoverage);
53
+ const slotOrdered = prioritizeAttackSurfaceBundles(deduped);
54
+ const generate = pick(slotOrdered, Math.min(backend, slotOrdered.length));
55
+ // ADDITIONAL = everything not chosen for GENERATE, in rank order, capped at the
56
+ // remaining budget, minus any whose resource+type is already covered by GENERATE.
57
+ const generateSet = new Set(generate);
58
+ const generatedCoverage = new Set(generate.map((item) => scenarioCoverageKey(item.scenario)));
59
+ const additional = slotOrdered
60
+ .filter((it) => !generateSet.has(it))
61
+ .slice(0, Math.max(0, ctx.maxTotal - backend))
62
+ .filter((item) => !generatedCoverage.has(scenarioCoverageKey(item.scenario)));
63
+ // Budgeting itself produces no demotions; the register-plan selection stage
64
+ // (planRanker.selectPlan) fills this channel from discriminator verification.
65
+ return { generate, additional, reservedUISlots: reservedUISlots(ctx), demotions: [] };
66
+ }
@@ -0,0 +1,31 @@
1
+ import { DraftedScenario } from "../types/RepositoryAnalysis.js";
2
+ import { DiscriminatorKind } from "../types/Recommendation.js";
3
+ /**
4
+ * A discriminator claim the agent (or server) attaches to a candidate: the
5
+ * structural bug-shape (`kind`) plus the verbatim diff snippet the test probes.
6
+ */
7
+ export interface DiscriminatorClaim {
8
+ kind: DiscriminatorKind;
9
+ changedCodeAnchor: string;
10
+ }
11
+ /** Result of verifying a claim against a scenario's steps and the PR diff. */
12
+ export interface DiscriminatorValidation {
13
+ verified: boolean;
14
+ /** One-sentence, human-readable reason present ONLY when `verified` is false;
15
+ * it is returned to the LLM to help it fix the claim. */
16
+ reason?: string;
17
+ }
18
+ /**
19
+ * Verify a declared discriminator against the candidate's `steps[]` and the PR
20
+ * diff. Mirrors the classification precedent (`recommendationShared.ts`): the
21
+ * claim is checked STRUCTURALLY, never by fuzzy keyword scans of prose. A failed
22
+ * check returns `verified: false` with a one-sentence reason (the claim is
23
+ * demoted, never used to reject the candidate). NEVER throws — malformed or
24
+ * partial input is treated as unverified.
25
+ *
26
+ * Two gates, both required to verify:
27
+ * 1. `changedCodeAnchor` must be a non-trivial string (>= 8 chars after trim)
28
+ * occurring verbatim in `diffText` — grounds the claim in the real change.
29
+ * 2. The `kind`-specific structural predicate must hold over `steps[]`.
30
+ */
31
+ export declare function validateDiscriminator(scenario: DraftedScenario, declared: DiscriminatorClaim, diffText: string): DiscriminatorValidation;