@skyramp/mcp 0.3.2-rc.pom-4 → 0.3.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (132) hide show
  1. package/build/adapters/jestAdapter.d.ts +14 -0
  2. package/build/adapters/jestAdapter.js +113 -0
  3. package/build/adapters/mochaAdapter.d.ts +13 -0
  4. package/build/adapters/mochaAdapter.js +87 -0
  5. package/build/adapters/playwrightAdapter.d.ts +17 -0
  6. package/build/adapters/playwrightAdapter.js +182 -0
  7. package/build/adapters/pytestAdapter.d.ts +15 -0
  8. package/build/adapters/pytestAdapter.js +108 -0
  9. package/build/commands/commandLibrary.d.ts +1 -0
  10. package/build/commands/commandLibrary.js +19 -13
  11. package/build/commands/localDevTestChangesCommand.d.ts +15 -0
  12. package/build/commands/localDevTestChangesCommand.js +201 -0
  13. package/build/index.js +82 -6
  14. package/build/prompts/code-reuse.js +3 -0
  15. package/build/prompts/enhance-assertions/sharedAssertionRules.d.ts +1 -0
  16. package/build/prompts/enhance-assertions/sharedAssertionRules.js +17 -0
  17. package/build/prompts/initialize-workspace/initializeWorkspacePrompt.js +11 -9
  18. package/build/prompts/local-dev/local-dev-plan.d.ts +35 -0
  19. package/build/prompts/local-dev/local-dev-plan.js +429 -0
  20. package/build/prompts/local-dev/local-dev-prompts.d.ts +4 -0
  21. package/build/prompts/local-dev/local-dev-prompts.js +190 -0
  22. package/build/prompts/prompt-utils.d.ts +8 -0
  23. package/build/prompts/prompt-utils.js +33 -0
  24. package/build/prompts/sut-setup/modes/adaptWorkflowPrompt.js +70 -7
  25. package/build/prompts/sut-setup/modes/dockerComposePrompt.js +1 -1
  26. package/build/prompts/sut-setup/shared.js +19 -17
  27. package/build/prompts/test-maintenance/drift-analysis-prompt.d.ts +10 -1
  28. package/build/prompts/test-maintenance/drift-analysis-prompt.js +54 -1
  29. package/build/prompts/test-recommendation/analysisOutputPrompt.js +21 -29
  30. package/build/prompts/test-recommendation/scopeAssessment.d.ts +5 -2
  31. package/build/prompts/test-recommendation/scopeAssessment.js +78 -6
  32. package/build/prompts/test-recommendation/test-recommendation-prompt.js +41 -4
  33. package/build/prompts/testbot/testbot-prompts.d.ts +0 -5
  34. package/build/prompts/testbot/testbot-prompts.js +39 -55
  35. package/build/recommendation/planRanker.d.ts +15 -2
  36. package/build/recommendation/planRanker.js +76 -5
  37. package/build/resources/testbotResource.js +2 -1
  38. package/build/services/AnalyticsService.d.ts +1 -1
  39. package/build/services/TestExecutionService.d.ts +2 -1
  40. package/build/services/TestExecutionService.js +8 -3
  41. package/build/services/TestGenerationService.d.ts +2 -2
  42. package/build/services/TestGenerationService.js +39 -21
  43. package/build/services/containerEnv.js +3 -1
  44. package/build/tool-phases.js +7 -0
  45. package/build/tools/code-refactor/codeReuseTool.js +43 -4
  46. package/build/tools/code-refactor/enhanceAssertionsTool.js +68 -18
  47. package/build/tools/code-refactor/reuse-outcome.d.ts +109 -0
  48. package/build/tools/code-refactor/reuse-outcome.js +158 -0
  49. package/build/tools/code-refactor/reuse-state.d.ts +45 -0
  50. package/build/tools/code-refactor/reuse-state.js +140 -0
  51. package/build/tools/enrichTestWithMocksTool.d.ts +28 -0
  52. package/build/tools/enrichTestWithMocksTool.js +726 -0
  53. package/build/tools/executeSkyrampTestTool.d.ts +11 -0
  54. package/build/tools/executeSkyrampTestTool.js +62 -21
  55. package/build/tools/generate-tests/batchMockGenerationTool.d.ts +106 -0
  56. package/build/tools/generate-tests/batchMockGenerationTool.js +545 -0
  57. package/build/tools/generate-tests/generateBatchScenarioRestTool.js +25 -23
  58. package/build/tools/generate-tests/generateContractRestTool.js +2 -2
  59. package/build/tools/generate-tests/generateE2ERestTool.d.ts +1 -1
  60. package/build/tools/generate-tests/generateE2ERestTool.js +1 -1
  61. package/build/tools/generate-tests/generateLoadRestTool.d.ts +1 -1
  62. package/build/tools/generate-tests/generateLoadRestTool.js +1 -1
  63. package/build/tools/generate-tests/generateMockRestTool.d.ts +179 -6
  64. package/build/tools/generate-tests/generateMockRestTool.js +391 -22
  65. package/build/tools/generate-tests/loadTestSchema.d.ts +1 -1
  66. package/build/tools/generate-tests/planGuard.js +2 -22
  67. package/build/tools/generateEnrichedIntegrationTestTool.d.ts +25 -0
  68. package/build/tools/generateEnrichedIntegrationTestTool.js +235 -0
  69. package/build/tools/localDevWorkerComposeTool.d.ts +24 -0
  70. package/build/tools/localDevWorkerComposeTool.js +264 -0
  71. package/build/tools/one-click/oneClickTool.d.ts +13 -0
  72. package/build/tools/one-click/oneClickTool.js +195 -24
  73. package/build/tools/preflightMockCheckTool.d.ts +2 -0
  74. package/build/tools/preflightMockCheckTool.js +96 -0
  75. package/build/tools/queryProxyMocksTool.d.ts +70 -0
  76. package/build/tools/queryProxyMocksTool.js +522 -0
  77. package/build/tools/runExistingTestsTool.d.ts +138 -0
  78. package/build/tools/runExistingTestsTool.js +644 -0
  79. package/build/tools/submitReportTool.js +30 -1
  80. package/build/tools/test-management/analyzeChangesTool.d.ts +2 -1
  81. package/build/tools/test-management/analyzeChangesTool.js +63 -40
  82. package/build/tools/test-management/analyzeTestHealthTool.d.ts +11 -0
  83. package/build/tools/test-management/analyzeTestHealthTool.js +63 -1
  84. package/build/tools/test-management/registerTestPlanTool.js +55 -7
  85. package/build/tools/trace/startTraceCollectionTool.js +3 -3
  86. package/build/types/ExternalTestExecution.d.ts +67 -0
  87. package/build/types/ExternalTestExecution.js +8 -0
  88. package/build/types/OneClickCommands.d.ts +1 -1
  89. package/build/types/Recommendation.d.ts +20 -0
  90. package/build/types/Recommendation.js +32 -6
  91. package/build/types/RepositoryAnalysis.d.ts +131 -14
  92. package/build/types/RepositoryAnalysis.js +16 -2
  93. package/build/types/ReuseOutcome.d.ts +63 -0
  94. package/build/types/ReuseOutcome.js +33 -0
  95. package/build/types/TestExecution.d.ts +1 -0
  96. package/build/types/TestTypes.d.ts +25 -7
  97. package/build/types/TestTypes.js +25 -7
  98. package/build/types/TestbotReport.d.ts +7 -0
  99. package/build/types/index.d.ts +2 -0
  100. package/build/types/index.js +1 -0
  101. package/build/utils/AnalysisStateManager.d.ts +33 -0
  102. package/build/utils/AnalysisStateManager.js +36 -2
  103. package/build/utils/analyze-openapi.js +18 -1
  104. package/build/utils/branchDiff.d.ts +17 -1
  105. package/build/utils/branchDiff.js +99 -14
  106. package/build/utils/featureFlags.d.ts +31 -0
  107. package/build/utils/featureFlags.js +37 -0
  108. package/build/utils/grpcMockValidation.d.ts +1 -0
  109. package/build/utils/grpcMockValidation.js +49 -0
  110. package/build/utils/httpMethodValidation.d.ts +4 -0
  111. package/build/utils/httpMethodValidation.js +15 -0
  112. package/build/utils/logger.js +1 -1
  113. package/build/utils/mockCompatibility.d.ts +49 -0
  114. package/build/utils/mockCompatibility.js +82 -0
  115. package/build/utils/pom-verify/verify.d.ts +5 -0
  116. package/build/utils/pom-verify/verify.js +1 -0
  117. package/build/utils/progress.js +10 -5
  118. package/build/utils/proxy-terminal.js +3 -3
  119. package/build/utils/routeParsers.d.ts +3 -9
  120. package/build/utils/routeParsers.js +79 -4
  121. package/build/utils/utils.js +2 -2
  122. package/build/utils/versions.d.ts +4 -3
  123. package/build/utils/versions.js +3 -1
  124. package/build/utils/workspaceAuth.d.ts +46 -0
  125. package/build/utils/workspaceAuth.js +156 -1
  126. package/build/workspace/frameworks.d.ts +11 -0
  127. package/build/workspace/frameworks.js +22 -0
  128. package/build/workspace/testSuites.d.ts +20 -0
  129. package/build/workspace/testSuites.js +17 -0
  130. package/build/workspace/workspace.d.ts +206 -24
  131. package/build/workspace/workspace.js +52 -2
  132. package/package.json +5 -2
@@ -2,6 +2,53 @@ import { buildActionDecisionTree, buildCheckAdditiveFields, buildCheckEndpointEx
2
2
  import { buildUiActionDecisionTree, buildUiCheckRouteExistence, buildUiCheckSelectors, buildUiCheckPageObjects, buildUiCheckBehavioralChanges, buildUiCheckAssignAction, buildUiDriftOutputChecklist, } from "./uiDriftAnalysisSections.js";
3
3
  import { RECOMMENDATIONS_INSTRUCTION, buildSymbolDiscoveryStep } from "./driftAnalysisShared.js";
4
4
  import { PromptPlan } from "../test-recommendation/promptPlan.js";
5
+ /**
6
+ * Render CONFIRMED failing tests from prior external-exec runs, plus the §4.5
7
+ * intent rules that gate whether a confirmed failure becomes an edit. Only
8
+ * environment-healthy CONFIRM runs count — a wall of red from a broken env is
9
+ * NOT PR signal. Returns "" when there is nothing confirmed (so the drift
10
+ * prompt is unchanged when external execution didn't run or found no failures).
11
+ */
12
+ export function buildConfirmedFailuresSection(externalTestResults) {
13
+ if (!externalTestResults?.length)
14
+ return "";
15
+ const confirmed = externalTestResults
16
+ .filter((r) => r.mode === "confirm" && r.environmentHealthy && !r.skipped)
17
+ .flatMap((r) => r.results.filter((t) => t.status === "fail" || t.status === "error"));
18
+ if (confirmed.length === 0)
19
+ return "";
20
+ const seen = new Set();
21
+ const unique = confirmed.filter((t) => {
22
+ if (seen.has(t.testId))
23
+ return false;
24
+ seen.add(t.testId);
25
+ return true;
26
+ });
27
+ // Escape angle brackets / ampersands so run output (message/file/testId) can't
28
+ // break out of the <confirmed_failures> section or inject pseudo-tags.
29
+ const esc = (s) => s.replace(/[<>&]/g, (c) => ({ "<": "&lt;", ">": "&gt;", "&": "&amp;" })[c]);
30
+ const rows = unique
31
+ .map((t) => {
32
+ const msg = t.message
33
+ ? `: ${esc(t.message.slice(0, 200).replace(/\s+/g, " ").trim())}`
34
+ : "";
35
+ return `- \`${esc(t.file)}\` — ${esc(t.testId)} [${t.status}]${msg}`;
36
+ })
37
+ .join("\n");
38
+ return `<confirmed_failures>
39
+ These tests were RUN and CONFIRMED failing under this change (facts, not diff guesses) — treat them as ground truth for WHICH tests broke. \`error\` = failed at fixture/collection (never reached its assertions); \`fail\` = assertion failure.
40
+
41
+ ${rows}
42
+ </confirmed_failures>
43
+
44
+ <confirmed_failure_intent_rules>
45
+ A confirmed failure is NOT automatically a test to UPDATE. Before editing ANY confirmed-failing test, classify the breakage:
46
+ - Intended behavior change (PR title/description/linked issue say so and the change is coherent across the codebase) → UPDATE the test, preserving what it ASSERTS (never gut assertions to force green).
47
+ - Unintended or incomplete (the PR broke behavior users depend on, or a consumer wasn't updated) → do NOT edit the test; record the confirmed failure in \`issuesFound\` with the evidence. Editing it green would launder a real regression into a passing suite — the worst outcome.
48
+ - Ambiguous → fix conservatively and flag the intent question in the report.
49
+ This classification is a precondition for any edit to a confirmed-failing test.
50
+ </confirmed_failure_intent_rules>`;
51
+ }
5
52
  const _apiPlan = new PromptPlan()
6
53
  .addPhase("maintenance", "Test Maintenance Assessment", {
7
54
  headerLevel: "##",
@@ -35,7 +82,7 @@ const _uiPlan = new PromptPlan()
35
82
  * - ui: when provided, appends a UI drift section for component/browser tests.
36
83
  * Omit when no frontend files changed or no UI tests exist.
37
84
  */
38
- export function buildDriftAnalysisPrompt(stateFile, apiTests, ui, repoPaths) {
85
+ export function buildDriftAnalysisPrompt(stateFile, apiTests, ui, repoPaths, externalTestResults) {
39
86
  const parts = [];
40
87
  // Emit symbol discovery once at the top regardless of how many sections follow.
41
88
  // Placing it inside each plan's checklist caused duplication on mixed diffs.
@@ -43,6 +90,12 @@ export function buildDriftAnalysisPrompt(stateFile, apiTests, ui, repoPaths) {
43
90
  const discovery = buildSymbolDiscoveryStep(hasAnyTests, repoPaths);
44
91
  if (discovery)
45
92
  parts.push(discovery);
93
+ // Fold in RUN-CONFIRMED failures (external execution) so the agent reasons over
94
+ // facts, not diff guesses — with the §4.5 intent gate. No-op when external
95
+ // execution didn't run or found nothing (keeps the static path unchanged).
96
+ const confirmedFailures = buildConfirmedFailuresSection(externalTestResults);
97
+ if (confirmedFailures)
98
+ parts.push(confirmedFailures);
46
99
  // Include API drift when there are API tests, or when UI drift is not running
47
100
  // (ensures skyramp_actions is always reachable even if apiTests is empty).
48
101
  if (apiTests.length > 0 || !ui) {
@@ -288,11 +288,11 @@ The ranked test recommendation catalog is pre-built and shown below (after the s
288
288
  **If** Steps 1–2 revealed additional scenarios the catalog does not cover (e.g. a computed formula or foreign-key relationship that was missed), you may optionally call \`skyramp_recommend_tests\` with \`stateFile: "${p.stateFile ?? p.sessionId}"\` and \`enrichedScenarios\` to regenerate a more complete catalog — but only after presenting the current one.`;
289
289
  const hasJavaFiles = p.candidateRouteFiles?.some((f) => /\.(java|kt)$/.test(f)) ?? false;
290
290
  const routeFilesSection = p.candidateRouteFiles && p.candidateRouteFiles.length > 0
291
- ? `\nRoute/controller files found by static scan (read these to discover endpoints — the regex-based catalog below may be incomplete for your framework):\n${p.candidateRouteFiles.map((f) => `- ${f}`).join("\n")}\n`
291
+ ? `\nCandidate route/controller files for LLM inspection (read these to discover endpoints; static parser output is only a hint):\n${p.candidateRouteFiles.map((f) => `- ${f}`).join("\n")}\n`
292
292
  : "";
293
293
  const resolvePathsNote = p.routerMountContext.length
294
- ? `**Resolve nested paths** using your Step ${ANALYSIS_STEP_RESOLVE_PATHS} table — a router in the table with prefix \`/api/v1/products/{product_id}/reviews\` means every endpoint in that file lives under that full path.`
295
- : `**Resolve full paths** using the prefixes you identified in Step ${ANALYSIS_STEP_READ_FILES} (e.g. Java Spring class-level \`@RequestMapping\` prefix + method-level path).`;
294
+ ? `**Resolve nested paths** using your Step ${ANALYSIS_STEP_RESOLVE_PATHS} table. The table you build from source/router context is authoritative over static parser hints.`
295
+ : `**Resolve full paths** using the prefixes, mounts, annotations, decorators, or routing DSL you identify in Step ${ANALYSIS_STEP_READ_FILES}.`;
296
296
  return `## Your Task — Fill in and Present the Catalog (full repo)
297
297
 
298
298
  ### Step ${ANALYSIS_STEP_READ_FILES}: Read key files
@@ -335,36 +335,28 @@ ${p.routerFileContents?.length
335
335
  p.routerMountContext.map((f) => `- \`${f}\``).join("\n")}`
336
336
  : "";
337
337
  const enrichment = buildEnrichmentInstructions(p);
338
- // LLM fallback: when heuristic scanning may have missed backend route files.
339
- const scannerFallbackHasCatalog = p.scannedEndpoints.length > 0;
340
- const scannerFallbackReason = scannerFallbackHasCatalog
341
- ? "The heuristic endpoint scanner may have missed route/controller files in this diff, even though endpoint data is present from another source"
342
- : "The heuristic endpoint scanner found **0 endpoints**";
343
- const scannerFallbackInstruction = scannerFallbackHasCatalog
344
- ? "Treat this as a supplemental gap check: merge only HTTP endpoints that are visibly missing from the existing catalog/spec data, and keep the existing catalog as the primary endpoint source."
345
- : "Build a table of discovered endpoints (method, full path, source file) and use it as the authoritative endpoint list for all subsequent steps.";
346
- const llmFallbackSection = (isDiffScope || p.scannedEndpoints.length === 0) &&
347
- p.candidateRouteFiles &&
348
- p.candidateRouteFiles.length > 0
338
+ const staticHintCount = p.scannedEndpoints.length;
339
+ const routeDiscoverySection = isDiffScope ||
340
+ staticHintCount > 0 ||
341
+ (p.candidateRouteFiles?.length ?? 0) > 0
349
342
  ? `
350
- ## Scanner Fallback — Manual Endpoint Discovery Required
343
+ ## LLM Route Discovery Inputs
351
344
 
352
- ${scannerFallbackReason}, and this repository contains ${p.candidateRouteFiles.length} file(s) that appear to define HTTP routes. The scanner likely failed due to multi-line definitions, framework-specific patterns, indirection the regex does not cover, or endpoint data being supplied by an OpenAPI spec instead of source scanning.
345
+ Static endpoint scan results are **best-effort hints only**. Do not assume the scanner supports this repository's language, framework, routing DSL, or helper abstractions.
353
346
 
354
- **Read the following controller/router files and identify all HTTP route registrations (method + path).** Include these discovered endpoints when executing the steps above.
347
+ ${staticHintCount > 0 ? `Static hints available: ${staticHintCount}. Verify every hinted method/path against source before using it.` : "Static hints available: 0. Build the endpoint list from source, router context, spec, and diff."}
355
348
 
356
- ${p.candidateRouteFiles.slice(0, 15).map((f) => `- \`${f}\``).join("\n")}
357
- ${p.candidateRouteFiles.length > 15 ? `\n_(${p.candidateRouteFiles.length - 15} more files not shown)_` : ""}
349
+ ${p.candidateRouteFiles && p.candidateRouteFiles.length > 0
350
+ ? `Candidate files to inspect:\n${p.candidateRouteFiles
351
+ .slice(0, 15)
352
+ .map((f) => `- \`${f}\``)
353
+ .join("\n")}${p.candidateRouteFiles.length > 15 ? `\n_(${p.candidateRouteFiles.length - 15} more files not shown)_` : ""}`
354
+ : staticHintCount > 0
355
+ ? "Candidate files to inspect: none identified by static scanning. Verify the static hints against their source files, changed files, router context, and specs above."
356
+ : "Candidate files to inspect: use the changed files and routing entry-point files above."}
358
357
 
359
- For each file, look for:
360
- - Express/Fastify/Koa/Hapi: \`router.<method>('/path', ...)\` or \`app.<method>('/path', ...)\`
361
- - FastAPI/Flask: \`@router.<method>('/path')\` or \`@app.route('/path', ...)\`
362
- - Spring/NestJS: \`@GetMapping\`, \`@PostMapping\`, \`@Controller\`, etc.
363
- - Go (Gin/Echo/Chi): \`r.GET("/path", ...)\`, \`r.Group("/prefix")\`
364
- - GraphQL: schema/resolver artifacts are unsupported for REST test generation; do not invent REST endpoints from them
365
- - Any other framework-specific route registration pattern
366
-
367
- ${scannerFallbackInstruction}`
358
+ For each candidate file, infer route registrations from the source's actual framework or DSL. Record method, full path, and source file. Resolve mount prefixes through the routing entry-point files. If a file is GraphQL-only, do not invent REST endpoints from it.
359
+ `
368
360
  : "";
369
361
  return `# Repository Analysis
370
362
 
@@ -373,7 +365,7 @@ ${scannerFallbackInstruction}`
373
365
  **Analysis Scope**: \`${p.analysisScope}\`
374
366
  ${isDiffScope ? `**Diff endpoints**: ${(p.parsedDiff?.newEndpoints.length ?? 0) + (p.parsedDiff?.modifiedEndpoints.length ?? 0) + (p.parsedDiff?.removedEndpoints?.length ?? 0)}` : `**Pre-scanned endpoints**: ${p.scannedEndpoints.length}`}
375
367
  ${routerSection}
376
- ${llmFallbackSection}
368
+ ${routeDiscoverySection}
377
369
  ${enrichment}
378
370
 
379
371
  **CRITICAL**: No .json/.md file creation. Prioritize cross-resource workflows.`;
@@ -54,8 +54,11 @@ export declare function isTestFile(filePath: string): boolean;
54
54
  /**
55
55
  * Builds the PR scope assessment section.
56
56
  *
57
- * When `precomputedUIPct` is provided (0 = backend-only, 100 = UI-only) the server
58
- * has already determined the split unambiguously — skip Steps A–C and emit one line.
57
+ * When `precomputedUIPct` is provided (0 = backend-only, 100 = UI-only) the server has
58
+ * already determined the split unambiguously, so Steps A–C are skipped. Backend-only
59
+ * (0) renders a single Budget Plan line; UI-only (100) renders that line plus the
60
+ * zero-new-surface override (SKYR-4099), because the budget is a default there rather
61
+ * than a mandate and a diff that adds no new surface must be able to abstain.
59
62
  *
60
63
  * For mixed PRs (`precomputedUIPct` is undefined, `hasFrontendChanges` is true) skip
61
64
  * Steps A–C but keep Step D so the LLM can apply judgment to determine the UI%.
@@ -188,11 +188,77 @@ export function isTestFile(filePath) {
188
188
  /(?:^|\/)__tests__\//.test(filePath));
189
189
  }
190
190
  // ── LLM scope assessment ──────────────────────────────────────────────────────
191
+ /**
192
+ * The zero-new-surface abstention rule, shared by every branch that can see a frontend
193
+ * diff (SKYR-4099).
194
+ *
195
+ * It previously existed only on the mixed-PR branch, so a frontend-ONLY diff — which
196
+ * takes the precomputed branch — had no sanctioned path to zero tests and the agent
197
+ * generated unnecessary UI tests while stating in its own reasoning that the change was
198
+ * cosmetic. Two copies then meant two definitions, and the mixed-PR one carved out
199
+ * "changes that alter visibility, layout, or state", which classifies a spacing-token
200
+ * change as non-cosmetic and made the override inert for exactly the diffs it should
201
+ * catch. One definition, both branches.
202
+ *
203
+ * Scoped to match `testbot-prompts.ts`'s "Do not fabricate tests outside the GENERATE
204
+ * list", which names three zero-test cases — deletion-only, cosmetic, and
205
+ * modification-of-existing-with-no-new-surface — under one principle: a new spec covers
206
+ * NEW observable surface only. An earlier revision of this section implemented cosmetic
207
+ * alone, and its keep-the-budget list contradicted the other two (an element being
208
+ * removed, or a `data-testid` being renamed, both forced the budget to stand). The test
209
+ * is coverage, not the kind of edit: does an existing test already reach this surface?
210
+ *
211
+ * `skipClause` is appended to the opening paragraph: the mixed-PR branch has a UI%
212
+ * step that becomes irrelevant once the budget is 0, the precomputed branch does not.
213
+ */
214
+ /**
215
+ * The zero-new-surface abstention rule, shared by every branch that can see a frontend
216
+ * diff (SKYR-4099).
217
+ *
218
+ * It previously existed only on the mixed-PR branch, so a frontend-ONLY diff — which
219
+ * takes the precomputed branch — had no sanctioned path to zero tests and the agent
220
+ * generated unnecessary UI tests while stating in its own reasoning that the change was
221
+ * cosmetic. Two copies then meant two definitions, and the mixed-PR one carved out
222
+ * "changes that alter visibility, layout, or state", which classifies a spacing-token
223
+ * change as non-cosmetic and made the override inert for exactly the diffs it should
224
+ * catch. One definition, both branches.
225
+ *
226
+ * Scoped to match `testbot-prompts.ts`'s "Do not fabricate tests outside the GENERATE
227
+ * list", which names three zero-test cases — deletion-only, cosmetic, and
228
+ * modification-of-existing-with-no-new-surface — under one principle: a new spec covers
229
+ * NEW observable surface only. The test is coverage, not the kind of edit: does an
230
+ * existing test already reach this surface?
231
+ *
232
+ * `skipClause` is appended to the opening paragraph: the mixed-PR branch has a UI% step
233
+ * that becomes irrelevant once the budget is 0, the precomputed branch does not.
234
+ */
235
+ function zeroSurfaceSection(skipClause = "") {
236
+ return `**Zero-new-surface override:** The total above is a default, not a mandate. A new spec exists to cover **new observable surface** — a component, route, page or flow that no existing test reaches. If your code review finds the diff adds none, set your Budget Plan to **0 total** and abstain — recommend and generate zero tests${skipClause}.
237
+
238
+ Abstain — the diff adds no new surface:
239
+ - **Cosmetic.** A styling-only value change (a spacing, size, color or font token, or a utility class swap such as \`size-4\`→\`size-5\`), or a \`.css\`/\`.scss\` reformat (property reordering, comment or whitespace edits, \`0px\`→\`0\`).
240
+ - **Deletion-only.** A component, route, element or feature was removed. The work is DELETING the tests that covered it — a removed surface cannot be the subject of a new spec.
241
+ - **Modification of an already-covered surface.** A renamed or moved selector, \`data-testid\`, \`aria-*\` or role; changed copy; an added field; a reordered or conditionally hidden element — where an existing test already reaches it. The work is UPDATING that test in place.
242
+
243
+ Keep the budget only for surface no existing test covers:
244
+ - A newly added component, route, page or flow.
245
+ - A component that was previously unintegrated and now has an integration point.
246
+ - New interactive behavior, state or validation on a surface no existing test reaches.
247
+
248
+ Two things that are NOT evidence of new surface: a frontend file appearing in the diff, and a large diff. Judge by whether an existing test already reaches the changed surface.
249
+
250
+ With a 0-total Budget Plan the work this diff needs is maintenance of the tests that already cover the affected surface — update the ones whose selectors or copy moved, and delete the ones that covered something this diff removed. Do NOT add a spec asserting that a removed feature is absent: the tests that covered it are the ones to delete, and an "is not present" assertion breaks the next time an unrelated sibling element changes.
251
+
252
+ If nothing currently covers the changed surface, there is no maintenance to do — and still no new spec to write, because a test added now would assert behavior this diff did not change. The missing coverage is a pre-existing gap, not something this PR introduced. Record it in \`additionalRecommendations\` in recommendatory voice ("would verify …") so the gap is visible without claiming a test was written.`;
253
+ }
191
254
  /**
192
255
  * Builds the PR scope assessment section.
193
256
  *
194
- * When `precomputedUIPct` is provided (0 = backend-only, 100 = UI-only) the server
195
- * has already determined the split unambiguously — skip Steps A–C and emit one line.
257
+ * When `precomputedUIPct` is provided (0 = backend-only, 100 = UI-only) the server has
258
+ * already determined the split unambiguously, so Steps A–C are skipped. Backend-only
259
+ * (0) renders a single Budget Plan line; UI-only (100) renders that line plus the
260
+ * zero-new-surface override (SKYR-4099), because the budget is a default there rather
261
+ * than a mandate and a diff that adds no new surface must be able to abstain.
196
262
  *
197
263
  * For mixed PRs (`precomputedUIPct` is undefined, `hasFrontendChanges` is true) skip
198
264
  * Steps A–C but keep Step D so the LLM can apply judgment to determine the UI%.
@@ -229,12 +295,18 @@ With a 0-total Budget Plan: generate zero tests, recommend zero tests, and follo
229
295
 
230
296
  **Exception — claim the ceiling only with evidence:** if your code review of the changed files shows an observable API behavior change the classifier missed (e.g. a DTO/serializer/service change that alters a response shape, a shared library/default-value or business-rule constant change that alters the behavior of an existing, unchanged endpoint (e.g. a default schedule, threshold, or config constant imported by a route handler elsewhere in the codebase), a deployment/config change that newly exposes or removes endpoints, or a schema-defined API contract change — a CRD type/kubebuilder validation marker, GraphQL schema, or gRPC proto edit that adds, removes, or re-validates what the server accepts or returns), raise your Budget Plan to cover exactly those affected endpoints, up to ${maxTotal} total (${effectiveGenerate} generate + ${additional} additional), 0% UI/E2E. Note: repositories whose entire API surface is schema-defined (e.g. a Kubernetes operator serving CRDs through the kube-apiserver) ALWAYS classify zero endpoints — for these, a schema change in the diff IS the endpoint change; evaluate this exception against the schema files instead of concluding there is nothing to test. Similarly, a changed file with zero classified endpoints is not by itself evidence of "no testable surface" — trace what imports the changed export (grep for its name) to check whether it feeds an existing endpoint's behavior before concluding the diff has no test value. Every test must name the changed file that justifies it. State your raised plan now in the canonical format — \`Budget Plan: <total> total (<generate> generate + <additional> additional), 0% UI/E2E\` — and use those exact numbers throughout the rest of the prompt; the raised generate count is your committed generate count.`;
231
297
  }
232
- // Unambiguous backend-only or UI-only: emit a single Budget Plan line — no LLM counting needed.
298
+ // Unambiguous backend-only or UI-only: no LLM counting needed. Backend-only emits just
299
+ // the Budget Plan line; UI-only appends the zero-new-surface override (see below).
233
300
  if (precomputedUIPct !== undefined) {
301
+ const uiSuffix = precomputedUIPct > 0 ? `, ${precomputedUIPct}% UI/E2E` : "";
302
+ // Ordered ahead of the "use these exact numbers" line — that line reads as final, so
303
+ // an override printed after it cannot fire. Backend-only (precomputedUIPct === 0) has
304
+ // no frontend file to call cosmetic, so it gets no override.
305
+ const cosmeticOverride = precomputedUIPct > 0 ? `${zeroSurfaceSection()}\n\n` : "";
234
306
  return `### PR Scope Assessment
235
- Budget Plan: ${maxTotal} total (${effectiveGenerate} generate + ${additional} additional), ${precomputedUIPct}% UI/E2E
307
+ Budget Plan: ${maxTotal} total (${effectiveGenerate} generate + ${additional} additional)${uiSuffix}
236
308
 
237
- Use these exact numbers throughout the rest of the prompt.`;
309
+ ${cosmeticOverride}Use these exact numbers throughout the rest of the prompt.`;
238
310
  }
239
311
  // Mixed PR: server can pre-compute the total but not the UI/E2E split — keep Step D.
240
312
  if (hasFrontendChanges) {
@@ -242,7 +314,7 @@ Use these exact numbers throughout the rest of the prompt.`;
242
314
 
243
315
  Budget Plan (total already determined): **${maxTotal} total (${effectiveGenerate} generate + ${additional} additional)**
244
316
 
245
- **Cosmetic-only override:** If, after code review, the entire diff is cosmetic with no observable rendering or interaction change (e.g. a \`.css\`/\`.scss\` reformat — property reordering, comment/whitespace edits, \`0px\`→\`0\`), the total above does NOT apply — set your Budget Plan to **0 total**, abstain (recommend and generate zero tests), and skip Step D below. A frontend file appearing in the diff is not, by itself, a behavior change. (Style changes that alter visibility, layout, or state are NOT cosmetic — keep the budget for those.)
317
+ ${zeroSurfaceSection(", and skip Step D below")}
246
318
 
247
319
  **Step D — Determine UI vs backend split for the budget above:**
248
320
  - Non-UI slots are backend tests; start from file-count ratio for UI%, then apply judgment:
@@ -178,6 +178,9 @@ Write UI recommendation \`reasoning\` fields in **natural prose** that names ele
178
178
  }
179
179
  }
180
180
  const fmtEndpoint = (m, ep) => ` ${m.method} ${ep.path}${m.authRequired ? " [auth]" : ""} (${(m.interactions ?? []).length} interactions)`;
181
+ // In diff scope, cap the reference endpoint list to prevent context overflow.
182
+ // Changed endpoints are always shown in full; only the "other" list is capped.
183
+ const DIFF_SCOPE_OTHER_ENDPOINT_CAP = 20;
181
184
  let endpointLines;
182
185
  if (isDiffScope && changedEndpointKeys.size > 0) {
183
186
  const changedLines = [];
@@ -201,7 +204,25 @@ Write UI recommendation \`reasoning\` fields in **natural prose** that names ele
201
204
  changedLines.push(` ${m.method} ${ep.path} [removed]`);
202
205
  }
203
206
  }
204
- endpointLines = `**Likely changed in this PR (from static file→endpoint mapping — verify against diff in Step ${ANALYSIS_STEP_EXTRACT}):**\n${changedLines.join("\n") || " none"}\n\n**Other endpoints (reference only):**\n${otherLines.join("\n") || " none"}`;
207
+ const cappedOther = otherLines.slice(0, DIFF_SCOPE_OTHER_ENDPOINT_CAP);
208
+ const hiddenOtherCount = otherLines.length - cappedOther.length;
209
+ const otherSuffix = hiddenOtherCount > 0
210
+ ? `\n ... and ${hiddenOtherCount} more (full endpoint list in state file)`
211
+ : "";
212
+ const otherLabel = hiddenOtherCount > 0
213
+ ? `Other endpoints (reference only, ${cappedOther.length} of ${otherLines.length} shown)`
214
+ : "Other endpoints (reference only)";
215
+ endpointLines = `**Likely changed in this PR (from static file→endpoint mapping — verify against diff in Step ${ANALYSIS_STEP_EXTRACT}):**\n${changedLines.join("\n") || " none"}\n\n**${otherLabel}:**\n${cappedOther.join("\n") || " none"}${otherSuffix}`;
216
+ }
217
+ else if (isDiffScope) {
218
+ // Diff scope but no changed endpoints detected — cap to avoid dumping the full catalog.
219
+ const allMethodLines = allEndpoints.flatMap((ep) => (ep.methods ?? []).map((m) => fmtEndpoint(m, ep)));
220
+ const cappedLines = allMethodLines.slice(0, DIFF_SCOPE_OTHER_ENDPOINT_CAP);
221
+ const hiddenCount = allMethodLines.length - cappedLines.length;
222
+ const suffix = hiddenCount > 0
223
+ ? `\n ... and ${hiddenCount} more (full endpoint list in state file — trace changed files directly from the diff to find affected endpoints)`
224
+ : "";
225
+ endpointLines = `${cappedLines.join("\n") || " none"}${suffix}`;
205
226
  }
206
227
  else {
207
228
  endpointLines = allEndpoints
@@ -230,9 +251,22 @@ Write UI recommendation \`reasoning\` fields in **natural prose** that names ele
230
251
  }
231
252
  }
232
253
  const { authSchemeSnippet } = getAuthSnippets(authHeaderValue, authTypeValue, workspaceAuthScheme);
254
+ const routeDiscovery = analysis.routeDiscovery;
255
+ const routeDiscoverySection = routeDiscovery
256
+ ? `
257
+ ## LLM Route Discovery Inputs
258
+ Static endpoint data below is a best-effort hint, not a complete parser for every framework.
259
+ Authoritative endpoint extraction must come from reading the changed source files, router/module context, OpenAPI paths when available, and the diff.
260
+ Candidate files to inspect: ${routeDiscovery.candidateFiles.length > 0 ? routeDiscovery.candidateFiles.join(", ") : "none"}
261
+ Router/module context files: ${routeDiscovery.routerMountContext.length > 0 ? routeDiscovery.routerMountContext.join(", ") : "none"}
262
+ OpenAPI paths available: ${routeDiscovery.openApiPaths.length}
263
+ Static hints available: ${routeDiscovery.staticHints.length}
264
+ ${routeDiscovery.diffFilePath ? `Diff file: ${routeDiscovery.diffFilePath}` : ""}
265
+ `.trim()
266
+ : "";
233
267
  const sourcePriority = `
234
268
  ## Source Priority
235
- When information conflicts, prefer: **Traces** (actual behavior) > **Code** (implemented behavior) > **Spec/Docs** (documented behavior).
269
+ When information conflicts, prefer: **Traces** (actual behavior) > **Source code read by the LLM** (implemented behavior) > **OpenAPI spec/docs** (documented behavior) > **Static parser hints** (best-effort, may be incomplete or framework-blind).
236
270
  `;
237
271
  // Compact fingerprint of tests already covering endpoints in this repo (Skyramp + external).
238
272
  // Re-derived fresh each run from test files on disk — no separate persistence needed.
@@ -274,8 +308,11 @@ Framework: ${analysis.projectClassification.primaryFramework} (${analysis.projec
274
308
  Project type: ${analysis.projectClassification.projectType}
275
309
  Auth: ${authMethod} (header: ${authHeaderValue}${authTypeValue ? `, type: ${authTypeValue}` : ""})
276
310
  Base URL: ${analysis.apiEndpoints.baseUrl}
277
- Candidate endpoints from static scan — unverified, confirm paths against spec or source before use (${analysis.apiEndpoints.totalCount}):
311
+ Candidate endpoint hints from static scan — unverified and non-exhaustive; confirm paths by reading source/router context before use (${analysis.apiEndpoints.totalCount}):
278
312
  ${endpointLines}${testFingerprint}
313
+ ${routeDiscoverySection ? `
314
+
315
+ ${routeDiscoverySection}` : ""}
279
316
  `.trim();
280
317
  // ── Branch diff ──
281
318
  let diffSection = "";
@@ -295,7 +332,7 @@ Affected services: ${diffContext.affectedServices.join(", ") || "N/A"}
295
332
 
296
333
  Focus on tests that validate these changes and how they interact with existing resources.
297
334
  For removed endpoints: verify they now return 404 or the appropriate deprecation status code.
298
- Allocate your test budget to endpoints listed under "Likely changed in this PR". Use other endpoints only as setup steps (e.g. creating a resource before testing its deletion).
335
+ Treat the endpoint lists above as static hints. If source/diff inspection finds a different changed endpoint set, prefer the source-grounded set and use other endpoints only as setup steps.
299
336
  `;
300
337
  }
301
338
  // ── Interactions ──
@@ -17,10 +17,5 @@ export interface RelatedRepository {
17
17
  export declare function parseRelatedRepositories(raw: string | undefined): RelatedRepository[] | undefined;
18
18
  export declare function getTestbotPrompt(prTitle: string, prDescription: string, summaryOutputFile: string, repositoryPath: string, baseBranch?: string, maxRecommendations?: number, maxGenerate?: number, _maxCritical?: number, // Reserved — accepted for API compat but not yet wired into prompt
19
19
  prNumber?: number, userPrompt?: string, services?: Service[], uiCredentials?: string, testsRepoDir?: string, relatedRepositories?: RelatedRepository[], primaryRepo?: string, planOnly?: boolean, language?: string): string;
20
- /**
21
- * Read services from .skyramp/workspace.yml. Returns empty array if
22
- * the workspace file doesn't exist or can't be parsed.
23
- */
24
- export declare function readWorkspaceServices(repositoryPath: string): Promise<Service[]>;
25
20
  export declare function buildWorkspaceRecoveryPrefix(repositoryPath: string): string;
26
21
  export declare function registerTestbotPrompt(server: McpServer): void;