@skyramp/mcp 0.3.5 → 0.3.6-rc.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/build/playwright/registerPlaywrightTools.js +92 -30
- package/build/playwright/traceRecordingPrompt.d.ts +6 -0
- package/build/playwright/traceRecordingPrompt.js +6 -2
- package/build/prompts/code-reuse.d.ts +1 -2
- package/build/prompts/code-reuse.js +182 -77
- package/build/prompts/modularization/integration-test-modularization.d.ts +2 -0
- package/build/prompts/modularization/integration-test-modularization.js +83 -41
- package/build/prompts/modularization/render.d.ts +18 -0
- package/build/prompts/modularization/render.js +12 -0
- package/build/prompts/modularization/ui-test-modularization.d.ts +3 -1
- package/build/prompts/modularization/ui-test-modularization.js +89 -47
- package/build/prompts/pom-aware-code-reuse.js +3 -1
- package/build/prompts/shared-helper-policy.d.ts +57 -0
- package/build/prompts/shared-helper-policy.js +135 -0
- package/build/prompts/test-recommendation/diffExecutionPlan.js +62 -56
- package/build/prompts/test-recommendation/fullRepoCatalog.js +19 -8
- package/build/prompts/test-recommendation/recommendationShared.d.ts +28 -6
- package/build/prompts/test-recommendation/recommendationShared.js +90 -16
- package/build/prompts/test-recommendation/registerRecommendTestsPrompt.js +22 -0
- package/build/prompts/test-recommendation/test-recommendation-prompt.d.ts +2 -2
- package/build/prompts/test-recommendation/test-recommendation-prompt.js +3 -3
- package/build/prompts/testbot/testbot-prompts.js +80 -34
- package/build/recommendation/budgeters/shared.js +105 -27
- package/build/recommendation/discriminators.js +13 -2
- package/build/recommendation/planRanker.d.ts +6 -6
- package/build/recommendation/planRanker.js +6 -61
- package/build/services/AnalyticsService.d.ts +7 -0
- package/build/services/AnalyticsService.js +7 -1
- package/build/services/ModularizationService.js +1 -3
- package/build/services/TestDiscoveryService.d.ts +0 -2
- package/build/services/TestDiscoveryService.js +2 -37
- package/build/services/TestGenerationService.d.ts +16 -0
- package/build/services/TestGenerationService.js +86 -10
- package/build/services/containerEnv.js +13 -12
- package/build/tools/code-refactor/codeReuseTool.js +279 -93
- package/build/tools/code-refactor/enhance-state.d.ts +49 -0
- package/build/tools/code-refactor/enhance-state.js +109 -0
- package/build/tools/code-refactor/enhanceAssertionsTool.js +34 -1
- package/build/tools/code-refactor/modularizationTool.js +9 -2
- package/build/tools/code-refactor/reuse-outcome.d.ts +23 -1
- package/build/tools/code-refactor/reuse-outcome.js +14 -4
- package/build/tools/code-refactor/reuse-state.d.ts +127 -5
- package/build/tools/code-refactor/reuse-state.js +628 -16
- package/build/tools/code-refactor/utils-verify-gates.d.ts +26 -0
- package/build/tools/code-refactor/utils-verify-gates.js +100 -0
- package/build/tools/code-refactor/verify-gates.d.ts +2 -1
- package/build/tools/code-refactor/verify-gates.js +90 -25
- package/build/tools/executeSkyrampTestTool.d.ts +19 -0
- package/build/tools/executeSkyrampTestTool.js +158 -8
- package/build/tools/generate-tests/generateBatchScenarioRestTool.js +2 -2
- package/build/tools/generate-tests/generateE2ERestTool.js +16 -0
- package/build/tools/generate-tests/generateUIRestTool.d.ts +1 -0
- package/build/tools/generate-tests/generateUIRestTool.js +22 -0
- package/build/tools/generate-tests/scenarioLint.d.ts +2 -0
- package/build/tools/generate-tests/scenarioLint.js +127 -19
- package/build/tools/generate-tests/trace-reuse-guard.d.ts +20 -0
- package/build/tools/generate-tests/trace-reuse-guard.js +93 -0
- package/build/tools/submitReportTool.d.ts +38 -38
- package/build/tools/submitReportTool.js +411 -114
- package/build/tools/test-management/analyzeChangesTool.d.ts +24 -1
- package/build/tools/test-management/analyzeChangesTool.js +71 -10
- package/build/tools/test-management/analyzeTestHealthTool.js +7 -7
- package/build/tools/test-management/registerTestPlanTool.d.ts +203 -0
- package/build/tools/test-management/registerTestPlanTool.js +70 -12
- package/build/types/Recommendation.d.ts +34 -5
- package/build/types/RepositoryAnalysis.d.ts +133 -114
- package/build/types/RepositoryAnalysis.js +1 -1
- package/build/types/ReuseOutcome.d.ts +102 -6
- package/build/types/ReuseOutcome.js +16 -2
- package/build/types/TestRecommendation.js +21 -3
- package/build/types/TestTypes.js +14 -8
- package/build/types/TestbotReport.d.ts +10 -1
- package/build/types/index.d.ts +2 -2
- package/build/types/index.js +1 -1
- package/build/utils/AnalysisStateManager.d.ts +57 -1
- package/build/utils/AnalysisStateManager.js +54 -5
- package/build/utils/branchDiff.d.ts +10 -0
- package/build/utils/branchDiff.js +28 -0
- package/build/utils/changedRoutes.d.ts +29 -0
- package/build/utils/changedRoutes.js +87 -0
- package/build/utils/featureFlags.d.ts +21 -0
- package/build/utils/featureFlags.js +23 -0
- package/build/utils/frontendIntegration.js +34 -4
- package/build/utils/importerHop.d.ts +2 -8
- package/build/utils/importerHop.js +15 -53
- package/build/utils/pathMatching.d.ts +38 -0
- package/build/utils/pathMatching.js +71 -0
- package/build/utils/pathSignatures.d.ts +22 -0
- package/build/utils/pathSignatures.js +57 -0
- package/build/utils/planMatchKeys.d.ts +16 -3
- package/build/utils/planMatchKeys.js +26 -10
- package/build/utils/pluralization.d.ts +10 -0
- package/build/utils/pluralization.js +18 -0
- package/build/utils/pom-catalog-parse.d.ts +52 -0
- package/build/utils/pom-catalog-parse.js +141 -0
- package/build/utils/pom-scope/selector-extractor.d.ts +12 -0
- package/build/utils/pom-scope/selector-extractor.js +34 -8
- package/build/utils/pom-verify/verify.d.ts +6 -5
- package/build/utils/pom-verify/verify.js +8 -6
- package/build/utils/reportVerification.d.ts +64 -4
- package/build/utils/reportVerification.js +228 -3
- package/build/utils/reuseRouting.d.ts +3 -0
- package/build/utils/reuseRouting.js +50 -0
- package/build/utils/routeParsers.d.ts +2 -0
- package/build/utils/routeParsers.js +65 -8
- package/build/utils/scenarioDrafting.d.ts +1 -1
- package/build/utils/scenarioDrafting.js +57 -45
- package/build/utils/subjectEndpoints.d.ts +19 -0
- package/build/utils/subjectEndpoints.js +98 -0
- package/build/utils/testFileClassification.d.ts +11 -0
- package/build/utils/testFileClassification.js +47 -0
- package/build/utils/uiPageEnumerator.d.ts +45 -19
- package/build/utils/uiPageEnumerator.js +95 -51
- package/build/utils/utils-verify/allow.d.ts +16 -0
- package/build/utils/utils-verify/allow.js +68 -0
- package/build/utils/utils-verify/call-sites.d.ts +34 -0
- package/build/utils/utils-verify/call-sites.js +154 -0
- package/build/utils/utils-verify/index.d.ts +7 -0
- package/build/utils/utils-verify/index.js +7 -0
- package/build/utils/utils-verify/language-spec.d.ts +91 -0
- package/build/utils/utils-verify/language-spec.js +210 -0
- package/build/utils/utils-verify/locate.d.ts +39 -0
- package/build/utils/utils-verify/locate.js +199 -0
- package/build/utils/utils-verify/parse.d.ts +34 -0
- package/build/utils/utils-verify/parse.js +177 -0
- package/build/utils/utils-verify/stage.d.ts +24 -0
- package/build/utils/utils-verify/stage.js +107 -0
- package/build/utils/utils-verify/verify.d.ts +63 -0
- package/build/utils/utils-verify/verify.js +168 -0
- package/build/utils/utils.d.ts +3 -1
- package/build/utils/utils.js +3 -1
- package/build/workspace/workspace.d.ts +32 -32
- package/node_modules/playwright/lib/mcp/skyramp/assertTool.js +9 -5
- package/node_modules/playwright/lib/mcp/skyramp/loadTraceTool.js +16 -0
- package/node_modules/playwright/lib/mcp/skyramp/skyRampImport.js +2 -0
- package/node_modules/playwright/lib/mcp/skyramp/traceRecordingBackend.js +115 -14
- package/node_modules/playwright/lib/mcp/test/skyRampExport.js +13 -1
- package/node_modules/playwright/node_modules/playwright-core/.DS_Store +0 -0
- package/node_modules/playwright/node_modules/playwright-core/lib/server/codegen/skyramp/jsonlReader.js +2 -0
- package/node_modules/playwright/node_modules/playwright-core/lib/vite/htmlReport/index.html +27 -253
- package/node_modules/playwright/node_modules/playwright-core/lib/vite/recorder/assets/{codeMirrorModule-DtudTj_v.js → codeMirrorModule-DJMC4zNo.js} +1 -1
- package/node_modules/playwright/node_modules/playwright-core/lib/vite/recorder/assets/index-BW82eAUI.js +196 -0
- package/node_modules/playwright/node_modules/playwright-core/lib/vite/recorder/index.html +1 -1
- package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/assets/{codeMirrorModule-FNMuBzX1.js → codeMirrorModule-CZfp96qZ.js} +1 -1
- package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/assets/defaultSettingsView-gpLo02E0.js +809 -0
- package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/index.Bq1r1URj.js +2 -0
- package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/index.html +2 -2
- package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/uiMode.VEfqi1qN.js +5 -0
- package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/uiMode.html +2 -2
- package/node_modules/playwright/node_modules/playwright-core/package.json +1 -1
- package/node_modules/playwright/node_modules/playwright-core/src/server/codegen/skyramp/jsonlReader.ts +1 -1
- package/node_modules/playwright/package.json +1 -1
- package/package.json +2 -2
- package/node_modules/playwright/node_modules/playwright-core/lib/vite/recorder/assets/index-BpDwp16L.js +0 -422
- package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/assets/defaultSettingsView-Co9upU5h.js +0 -1035
- package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/index.DXNIQ_dx.js +0 -2
- package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/uiMode.CIKB3XSv.js +0 -5
|
@@ -4,6 +4,7 @@ import { CallToolResult } from "@modelcontextprotocol/sdk/types.js";
|
|
|
4
4
|
import { CandidateUiPage } from "../../utils/uiPageEnumerator.js";
|
|
5
5
|
import type { FrontendFileIntegration } from "../../types/FrontendIntegration.js";
|
|
6
6
|
import { TraceFile } from "../../types/RepositoryAnalysis.js";
|
|
7
|
+
import { BranchDiffData } from "../../utils/branchDiff.js";
|
|
7
8
|
import { ScannedEndpoint } from "../../utils/repoScanner.js";
|
|
8
9
|
import { TraceParseResult } from "../../utils/trace-parser.js";
|
|
9
10
|
/** Exported for testing: maps a parsed trace result to a TraceFile. */
|
|
@@ -25,7 +26,7 @@ export declare const analyzeChangesInputSchema: {
|
|
|
25
26
|
includeUncommitted: z.ZodDefault<z.ZodOptional<z.ZodBoolean>>;
|
|
26
27
|
};
|
|
27
28
|
export declare const NO_UI_INSTRUCTIONS = "No UI changes detected \u2014 no blueprint capture needed.";
|
|
28
|
-
export declare const NO_RESOLVABLE_URLS_INSTRUCTIONS = "Frontend changes detected but no candidate URLs could be resolved (
|
|
29
|
+
export declare const NO_RESOLVABLE_URLS_INSTRUCTIONS = "Frontend changes detected but no candidate URLs could be resolved (no route files matched the changed files or their importers, and no frontend baseUrl to fall back to). UI recommendations will be source-grounded only.";
|
|
29
30
|
export declare function buildCaptureInstructions(pages: CandidateUiPage[]): string;
|
|
30
31
|
/**
|
|
31
32
|
* Instruction block for changed frontend files the server determined have no
|
|
@@ -45,4 +46,26 @@ export declare function buildAnalyzeChangesResult(parts: {
|
|
|
45
46
|
outputText: string;
|
|
46
47
|
recommendationPrompt: string;
|
|
47
48
|
}): CallToolResult;
|
|
49
|
+
/**
|
|
50
|
+
* Coverage is discovered from the working tree, so a test file the PR itself adds looks
|
|
51
|
+
* identical to pre-existing coverage. This is what the budgeter reads to decide whether
|
|
52
|
+
* an empty plan is a dedup artifact or the author's own coverage.
|
|
53
|
+
*
|
|
54
|
+
* No diff means no answer, so it reports `true` and the reserve stays shut. `diffData` is
|
|
55
|
+
* absent in full-repo scope — where fullRepoCatalog deliberately drops externally covered
|
|
56
|
+
* scenarios, and promoting them back would contradict the prompt the agent reads — and
|
|
57
|
+
* after a branch-diff failure, where nothing is known about the changed files at all.
|
|
58
|
+
*
|
|
59
|
+
* Classification comes from discovery, not from `isTestFile`: the flag has to agree with
|
|
60
|
+
* whatever built the coverage keys it guards.
|
|
61
|
+
*
|
|
62
|
+
* `userChangedFiles` is the bot-filtered list. The bot commits its generated tests onto
|
|
63
|
+
* the branch it analyzed, and those filenames are exactly what the discovery patterns
|
|
64
|
+
* exist to match — so reading the raw base..HEAD list shut the reserve on every re-run
|
|
65
|
+
* of a branch: run 1 generates the tests, run 2 sees them. When the filter cannot answer
|
|
66
|
+
* (no bot commit yet, or a git failure) the raw list stands. That is safe in the same
|
|
67
|
+
* direction as the `!diffData` case: the raw list can only hold MORE files, so the
|
|
68
|
+
* fallback shuts the reserve rather than opening it.
|
|
69
|
+
*/
|
|
70
|
+
export declare function computeDiffChangesTestFiles(diffData: BranchDiffData | undefined, userChangedFiles?: string[] | null): boolean;
|
|
48
71
|
export declare function registerAnalyzeChangesTool(server: McpServer): void;
|
|
@@ -14,13 +14,16 @@ import { StateManager, registerSession, storeSessionData, rememberTestsRepoDir,
|
|
|
14
14
|
import { buildRecommendationPrompt, computeScoredCandidates } from "../../prompts/test-recommendation/test-recommendation-prompt.js";
|
|
15
15
|
import { hasFlutterSdkDep, isFrontendFile, isTestFile } from "../../prompts/test-recommendation/scopeAssessment.js";
|
|
16
16
|
import { buildExternalCoverageSet } from "../../prompts/test-recommendation/recommendationShared.js";
|
|
17
|
+
import { resolveSubjectEndpoints } from "../../utils/subjectEndpoints.js";
|
|
18
|
+
import { collectChangedRouteLines } from "../../utils/changedRoutes.js";
|
|
17
19
|
import { CandidateSource, computeCandidateId } from "../../types/Recommendation.js";
|
|
18
20
|
import { selectPlan } from "../../recommendation/planRanker.js";
|
|
19
21
|
import { buildApprovedPlanItem } from "../../utils/planMatchKeys.js";
|
|
20
|
-
import { enumerateCandidateUiPages } from "../../utils/uiPageEnumerator.js";
|
|
22
|
+
import { enumerateCandidateUiPages, MAX_CANDIDATE_PAGES } from "../../utils/uiPageEnumerator.js";
|
|
21
23
|
import { checkFrontendFileIntegration } from "../../utils/frontendIntegration.js";
|
|
22
24
|
import { MAX_RECOMMENDATIONS, MAX_TESTS_TO_GENERATE } from "../../prompts/test-recommendation/recommendationSections.js";
|
|
23
25
|
import { TestDiscoveryService } from "../../services/TestDiscoveryService.js";
|
|
26
|
+
import { isDiscoveredTestFile } from "../../utils/testFileClassification.js";
|
|
24
27
|
import { ScenarioSource, AnalysisScope } from "../../types/RepositoryAnalysis.js";
|
|
25
28
|
import { computeBranchDiff } from "../../utils/branchDiff.js";
|
|
26
29
|
import { classifyEndpointsByChangedFiles, selectRemovalCandidateFiles, recoverRemovedEndpointsFromBase, } from "../../utils/routeParsers.js";
|
|
@@ -324,23 +327,35 @@ export const analyzeChangesInputSchema = {
|
|
|
324
327
|
// for UI recommendation reasoning (enforced at submit time by the Blueprint
|
|
325
328
|
// Citation Invariant in the testbot prompt).
|
|
326
329
|
export const NO_UI_INSTRUCTIONS = `No UI changes detected — no blueprint capture needed.`;
|
|
327
|
-
export const NO_RESOLVABLE_URLS_INSTRUCTIONS = `Frontend changes detected but no candidate URLs could be resolved (
|
|
330
|
+
export const NO_RESOLVABLE_URLS_INSTRUCTIONS = `Frontend changes detected but no candidate URLs could be resolved (no route files matched the changed files or their importers, and no frontend baseUrl to fall back to). UI recommendations will be source-grounded only.`;
|
|
328
331
|
export function buildCaptureInstructions(pages) {
|
|
332
|
+
const pathOnly = pages.some((p) => p.baseUrlResolved === false);
|
|
329
333
|
const pagesYaml = pages
|
|
330
|
-
.map((p, i) => ` ${i + 1}. ${p.url} (sourcedFrom: ${p.sourcedFrom.join(", ") || "(none)"}, strategy: ${p.strategy})`)
|
|
334
|
+
.map((p, i) => ` ${i + 1}. ${p.url} (sourcedFrom: ${p.sourcedFrom.join(", ") || "(none)"}${p.via ? `, via: ${p.via.join(", ")}` : ""}, strategy: ${p.strategy})`)
|
|
331
335
|
.join("\n");
|
|
336
|
+
const capNote = pages.length >= MAX_CANDIDATE_PAGES
|
|
337
|
+
? `\n(The list is capped at ${MAX_CANDIDATE_PAGES} pages — direct route matches first, then the pages that render the most changed files.)`
|
|
338
|
+
: "";
|
|
339
|
+
// The server could not resolve a frontend baseUrl (no frontend service with
|
|
340
|
+
// `api.baseUrl`, no SKYRAMP_TEST_BASE_URL). The paths are still exact; the
|
|
341
|
+
// agent supplies the host.
|
|
342
|
+
const baseUrlStep = pathOnly
|
|
343
|
+
? `
|
|
344
|
+
**The entries above are URL paths, not full URLs** — the workspace declares no frontend service with \`api.baseUrl\`. Determine the frontend base URL once before capturing — the testbot workflow's \`targetReadyCheckCommand\` or the dev-server port in \`package.json\` scripts usually names it — then prefix every path with it. Log an \`issuesFound\` info entry recommending that \`api.baseUrl\` be set on the frontend service in \`.skyramp/workspace.yml\`.
|
|
345
|
+
`
|
|
346
|
+
: "";
|
|
332
347
|
return `Frontend changes detected. **Before writing any UI recommendation \`reasoning\`, capture blueprints on the candidate UI pages below.** Those captures stay in your tool-result history and serve as element vocabulary — the recommendation catalog further down gives you the authoring rules; you bring the observed elements.
|
|
333
348
|
|
|
334
|
-
**Candidate URLs:**
|
|
335
|
-
${pagesYaml}
|
|
336
|
-
|
|
349
|
+
**Candidate ${pathOnly ? "URL paths" : "URLs"}:**
|
|
350
|
+
${pagesYaml}${capNote}
|
|
351
|
+
${baseUrlStep}
|
|
337
352
|
**For each candidate URL:**
|
|
338
353
|
- \`browser_navigate\` to the URL
|
|
339
354
|
- \`browser_blueprint\` to capture the page
|
|
340
355
|
|
|
341
356
|
You don't need to thread the blueprints back into a tool call — they're in your context once captured.
|
|
342
357
|
|
|
343
|
-
If a candidate URL 404s or redirects unexpectedly, navigate from the
|
|
358
|
+
If a candidate URL 404s or redirects unexpectedly, navigate from the frontend base URL and explore (admin apps mount routes under base prefixes the source extraction can't see). If the page rendered but lacks the changed feature (gated UI: modal, dropdown, accordion), do NOT iterate further during this step — UI recs will fall back to source-grounded prose for those, and the agent's later trace recording (Task 2) will navigate into the gate via capture-act-capture.
|
|
344
359
|
|
|
345
360
|
If \`browser_blueprint\` fails on every candidate URL (app unreachable, all 404s), proceed and log an \`issuesFound\` info entry. Recommendations will be source-grounded; non-UI work is unaffected.`;
|
|
346
361
|
}
|
|
@@ -376,6 +391,32 @@ export function buildAnalyzeChangesResult(parts) {
|
|
|
376
391
|
const executionPlan = `\`\`\`json\n${parts.structuredSummary}\n\`\`\`\n\n## UI Blueprint Capture — do this BEFORE writing UI recommendation reasoning\n${parts.uiInstructions}\n\n${parts.outputText}\n\n---\n\n## Pre-built Test Catalog — Fill in placeholders from source code, then display verbatim\n⚠️ Do NOT reformat, rename sections, or generate a new catalog. Replace \`<…from source>\` values, then show this output exactly as-is, grouped by test type.\n\n${parts.recommendationPrompt}`;
|
|
377
392
|
return dualChannelResult({ executionPlan });
|
|
378
393
|
}
|
|
394
|
+
/**
|
|
395
|
+
* Coverage is discovered from the working tree, so a test file the PR itself adds looks
|
|
396
|
+
* identical to pre-existing coverage. This is what the budgeter reads to decide whether
|
|
397
|
+
* an empty plan is a dedup artifact or the author's own coverage.
|
|
398
|
+
*
|
|
399
|
+
* No diff means no answer, so it reports `true` and the reserve stays shut. `diffData` is
|
|
400
|
+
* absent in full-repo scope — where fullRepoCatalog deliberately drops externally covered
|
|
401
|
+
* scenarios, and promoting them back would contradict the prompt the agent reads — and
|
|
402
|
+
* after a branch-diff failure, where nothing is known about the changed files at all.
|
|
403
|
+
*
|
|
404
|
+
* Classification comes from discovery, not from `isTestFile`: the flag has to agree with
|
|
405
|
+
* whatever built the coverage keys it guards.
|
|
406
|
+
*
|
|
407
|
+
* `userChangedFiles` is the bot-filtered list. The bot commits its generated tests onto
|
|
408
|
+
* the branch it analyzed, and those filenames are exactly what the discovery patterns
|
|
409
|
+
* exist to match — so reading the raw base..HEAD list shut the reserve on every re-run
|
|
410
|
+
* of a branch: run 1 generates the tests, run 2 sees them. When the filter cannot answer
|
|
411
|
+
* (no bot commit yet, or a git failure) the raw list stands. That is safe in the same
|
|
412
|
+
* direction as the `!diffData` case: the raw list can only hold MORE files, so the
|
|
413
|
+
* fallback shuts the reserve rather than opening it.
|
|
414
|
+
*/
|
|
415
|
+
export function computeDiffChangesTestFiles(diffData, userChangedFiles) {
|
|
416
|
+
if (!diffData)
|
|
417
|
+
return true;
|
|
418
|
+
return (userChangedFiles ?? diffData.changedFiles).some(isDiscoveredTestFile);
|
|
419
|
+
}
|
|
379
420
|
export function registerAnalyzeChangesTool(server) {
|
|
380
421
|
server.registerTool(TOOL_NAME, {
|
|
381
422
|
annotations: {
|
|
@@ -1323,19 +1364,21 @@ export function registerAnalyzeChangesTool(server) {
|
|
|
1323
1364
|
// same classification without re-deriving it. Absent on backend-only PRs.
|
|
1324
1365
|
//
|
|
1325
1366
|
// candidateUiPages is enumerated programmatically via the strategy
|
|
1326
|
-
// ladder in uiPageEnumerator (framework route grep,
|
|
1327
|
-
// routes, root fallback). The agent uses these to capture
|
|
1367
|
+
// ladder in uiPageEnumerator (framework route grep + import graph,
|
|
1368
|
+
// source-grounded routes, root fallback). The agent uses these to capture
|
|
1328
1369
|
// browser_blueprints — see uiInstructions below, which this tool returns
|
|
1329
1370
|
// so the agent captures element vocabulary for UI rec reasoning.
|
|
1330
1371
|
const uiContext = await (async () => {
|
|
1331
1372
|
// changedFrontendFiles computed above (before discoverTests) — reuse here.
|
|
1332
1373
|
if (changedFrontendFiles.length === 0)
|
|
1333
1374
|
return undefined;
|
|
1334
|
-
const candidateUiPages = await enumerateCandidateUiPages(params.repositoryPath, changedFrontendFiles);
|
|
1335
1375
|
// SKYR-3855: deterministic production-importer check, computed here so
|
|
1336
1376
|
// downstream consumers (testbot prompt) can skip UI generation for
|
|
1337
1377
|
// unintegrated components on a server fact instead of a mid-run grep.
|
|
1378
|
+
// Computed first: the enumerator resolves changed non-route files to
|
|
1379
|
+
// the pages that import them.
|
|
1338
1380
|
const frontendFileIntegration = checkFrontendFileIntegration(params.repositoryPath, changedFrontendFiles);
|
|
1381
|
+
const candidateUiPages = await enumerateCandidateUiPages(params.repositoryPath, changedFrontendFiles, frontendFileIntegration);
|
|
1339
1382
|
return {
|
|
1340
1383
|
changedFrontendFiles,
|
|
1341
1384
|
candidateUiPages,
|
|
@@ -1372,13 +1415,29 @@ export function registerAnalyzeChangesTool(server) {
|
|
|
1372
1415
|
// computeScoredCandidates is the single source of truth for both.
|
|
1373
1416
|
const topN = params.topN ?? MAX_RECOMMENDATIONS;
|
|
1374
1417
|
const scoredResult = computeScoredCandidates(fullAnalysis, analysisScope, topN, params.maxGenerate);
|
|
1418
|
+
// Resolve each scenario's subject endpoints once, here — this is the only
|
|
1419
|
+
// point that has the scenarios, the diff, and runs before BOTH consumers
|
|
1420
|
+
// (selectPlan below, and the two prompt renderers). `scored.scenario` is
|
|
1421
|
+
// the SAME object as the entry in `allDraftedScenarios` (assigned by
|
|
1422
|
+
// reference into businessContext.draftedScenarios, then into
|
|
1423
|
+
// repositoryAnalysis.scenarios below), so this mutation also fills what
|
|
1424
|
+
// gets persisted. Filling later would leave one consumer keying off a
|
|
1425
|
+
// different subject (SKYR-4214).
|
|
1426
|
+
const changedRoutesForSubjects = collectChangedRouteLines(diffText ?? "");
|
|
1427
|
+
for (const scored of scoredResult.scored) {
|
|
1428
|
+
scored.scenario.subjectEndpoints = resolveSubjectEndpoints(scored.scenario, {
|
|
1429
|
+
changedRoutes: changedRoutesForSubjects,
|
|
1430
|
+
});
|
|
1431
|
+
}
|
|
1375
1432
|
const externalCoverage = buildExternalCoverageSet(testLocationsByType);
|
|
1433
|
+
const diffChangesTestFiles = computeDiffChangesTestFiles(diffData, await getUserChangedFiles(params.repositoryPath));
|
|
1376
1434
|
const planBudgetContext = {
|
|
1377
1435
|
maxGenerate: scoredResult.maxGen,
|
|
1378
1436
|
maxTotal: topN,
|
|
1379
1437
|
isUIOnlyPR: scoredResult.isUIOnlyPR,
|
|
1380
1438
|
hasFrontendChanges: scoredResult.hasFrontendChanges,
|
|
1381
1439
|
externalCoverageKeys: [...externalCoverage],
|
|
1440
|
+
diffChangesTestFiles,
|
|
1382
1441
|
};
|
|
1383
1442
|
let approvedPlan;
|
|
1384
1443
|
if (scoredResult.scored.length > 0) {
|
|
@@ -1388,6 +1447,7 @@ export function registerAnalyzeChangesTool(server) {
|
|
|
1388
1447
|
isUIOnlyPR: planBudgetContext.isUIOnlyPR,
|
|
1389
1448
|
hasFrontendChanges: planBudgetContext.hasFrontendChanges,
|
|
1390
1449
|
externalCoverage,
|
|
1450
|
+
diffChangesTestFiles,
|
|
1391
1451
|
};
|
|
1392
1452
|
const serverCandidates = scoredResult.scored.map(({ scenario, priority, novelty }) => ({
|
|
1393
1453
|
scenario,
|
|
@@ -1403,6 +1463,7 @@ export function registerAnalyzeChangesTool(server) {
|
|
|
1403
1463
|
generate: plan.generate.map(buildApprovedPlanItem),
|
|
1404
1464
|
additional: plan.additional.map(buildApprovedPlanItem),
|
|
1405
1465
|
demotions: [],
|
|
1466
|
+
dropped: plan.dropped,
|
|
1406
1467
|
};
|
|
1407
1468
|
}
|
|
1408
1469
|
const unifiedState = {
|
|
@@ -175,13 +175,13 @@ export function registerAnalyzeTestHealthTool(server) {
|
|
|
175
175
|
// every repo in a multi-repo run (endpoints in Repo A, tests in Repo B).
|
|
176
176
|
// TODO(multi-repo): existingTests is scoped to the current repo only — related
|
|
177
177
|
// repo tests are not pre-loaded, discoverable only via the grep instruction.
|
|
178
|
-
//
|
|
179
|
-
|
|
180
|
-
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
|
|
178
|
+
// Every root in the RUN, not the requested repo plus the related ones: on a
|
|
179
|
+
// related-repo call the latter omits the primary and repeats the requested
|
|
180
|
+
// repo (SKYR-4246). Dedupe by root because a legacy untagged primary can also
|
|
181
|
+
// appear as a related section holding the same path.
|
|
182
|
+
const allRepoPaths = [
|
|
183
|
+
...new Set((await stateManager.listRepoCheckouts()).map(c => c.root)),
|
|
184
|
+
];
|
|
185
185
|
const promptText = buildDriftAnalysisPrompt(stateManager.getStatePath(), apiTests.map((t) => ({ testFile: t.testFile, source: t.source })), uiDriftParams, allRepoPaths, stateData?.externalTestResults);
|
|
186
186
|
return {
|
|
187
187
|
structuredContent: { prompt: promptText },
|
|
@@ -1,2 +1,205 @@
|
|
|
1
|
+
import { z } from "zod";
|
|
1
2
|
import { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js";
|
|
3
|
+
import { HttpMethod, TestType } from "../../types/TestTypes.js";
|
|
4
|
+
import { Candidate, DiscriminatorKind } from "../../types/Recommendation.js";
|
|
5
|
+
import { ChangedRoute } from "../../utils/changedRoutes.js";
|
|
6
|
+
declare const registerCandidateSchema: z.ZodObject<{
|
|
7
|
+
scenarioName: z.ZodString;
|
|
8
|
+
description: z.ZodString;
|
|
9
|
+
category: z.ZodEnum<["new_endpoint", "bug_caught", "business_rule", "security_boundary", "data_integrity", "breaking_change", "auth", "error_handling", "workflow", "data_validation", "crud"]>;
|
|
10
|
+
priority: z.ZodEnum<["high", "medium", "low"]>;
|
|
11
|
+
testType: z.ZodEffects<z.ZodNativeEnum<typeof TestType>, TestType, TestType>;
|
|
12
|
+
steps: z.ZodArray<z.ZodObject<{
|
|
13
|
+
order: z.ZodNumber;
|
|
14
|
+
method: z.ZodNativeEnum<typeof HttpMethod>;
|
|
15
|
+
path: z.ZodString;
|
|
16
|
+
description: z.ZodString;
|
|
17
|
+
interactionType: z.ZodEnum<["success", "error", "edge-case"]>;
|
|
18
|
+
requestBody: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodAny>>;
|
|
19
|
+
queryParams: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodAny>>;
|
|
20
|
+
responseBody: z.ZodOptional<z.ZodUnion<[z.ZodRecord<z.ZodString, z.ZodAny>, z.ZodArray<z.ZodAny, "many">]>>;
|
|
21
|
+
expectedStatusCode: z.ZodNumber;
|
|
22
|
+
expectedResponseFields: z.ZodOptional<z.ZodArray<z.ZodString, "many">>;
|
|
23
|
+
bodyMustInclude: z.ZodOptional<z.ZodArray<z.ZodString, "many">>;
|
|
24
|
+
chainsFrom: z.ZodOptional<z.ZodUnion<[z.ZodObject<{
|
|
25
|
+
sourceStep: z.ZodNumber;
|
|
26
|
+
sourceField: z.ZodString;
|
|
27
|
+
sourceLocation: z.ZodEnum<["body", "header", "cookie"]>;
|
|
28
|
+
targetParam: z.ZodString;
|
|
29
|
+
targetLocation: z.ZodEnum<["path", "body", "query", "header", "cookie"]>;
|
|
30
|
+
}, "strip", z.ZodTypeAny, {
|
|
31
|
+
sourceStep: number;
|
|
32
|
+
sourceField: string;
|
|
33
|
+
sourceLocation: "body" | "header" | "cookie";
|
|
34
|
+
targetParam: string;
|
|
35
|
+
targetLocation: "path" | "body" | "header" | "cookie" | "query";
|
|
36
|
+
}, {
|
|
37
|
+
sourceStep: number;
|
|
38
|
+
sourceField: string;
|
|
39
|
+
sourceLocation: "body" | "header" | "cookie";
|
|
40
|
+
targetParam: string;
|
|
41
|
+
targetLocation: "path" | "body" | "header" | "cookie" | "query";
|
|
42
|
+
}>, z.ZodArray<z.ZodObject<{
|
|
43
|
+
sourceStep: z.ZodNumber;
|
|
44
|
+
sourceField: z.ZodString;
|
|
45
|
+
sourceLocation: z.ZodEnum<["body", "header", "cookie"]>;
|
|
46
|
+
targetParam: z.ZodString;
|
|
47
|
+
targetLocation: z.ZodEnum<["path", "body", "query", "header", "cookie"]>;
|
|
48
|
+
}, "strip", z.ZodTypeAny, {
|
|
49
|
+
sourceStep: number;
|
|
50
|
+
sourceField: string;
|
|
51
|
+
sourceLocation: "body" | "header" | "cookie";
|
|
52
|
+
targetParam: string;
|
|
53
|
+
targetLocation: "path" | "body" | "header" | "cookie" | "query";
|
|
54
|
+
}, {
|
|
55
|
+
sourceStep: number;
|
|
56
|
+
sourceField: string;
|
|
57
|
+
sourceLocation: "body" | "header" | "cookie";
|
|
58
|
+
targetParam: string;
|
|
59
|
+
targetLocation: "path" | "body" | "header" | "cookie" | "query";
|
|
60
|
+
}>, "many">]>>;
|
|
61
|
+
}, "strip", z.ZodTypeAny, {
|
|
62
|
+
path: string;
|
|
63
|
+
method: HttpMethod;
|
|
64
|
+
description: string;
|
|
65
|
+
order: number;
|
|
66
|
+
interactionType: "error" | "success" | "edge-case";
|
|
67
|
+
expectedStatusCode: number;
|
|
68
|
+
queryParams?: Record<string, any> | undefined;
|
|
69
|
+
requestBody?: Record<string, any> | undefined;
|
|
70
|
+
responseBody?: any[] | Record<string, any> | undefined;
|
|
71
|
+
expectedResponseFields?: string[] | undefined;
|
|
72
|
+
chainsFrom?: {
|
|
73
|
+
sourceStep: number;
|
|
74
|
+
sourceField: string;
|
|
75
|
+
sourceLocation: "body" | "header" | "cookie";
|
|
76
|
+
targetParam: string;
|
|
77
|
+
targetLocation: "path" | "body" | "header" | "cookie" | "query";
|
|
78
|
+
} | {
|
|
79
|
+
sourceStep: number;
|
|
80
|
+
sourceField: string;
|
|
81
|
+
sourceLocation: "body" | "header" | "cookie";
|
|
82
|
+
targetParam: string;
|
|
83
|
+
targetLocation: "path" | "body" | "header" | "cookie" | "query";
|
|
84
|
+
}[] | undefined;
|
|
85
|
+
bodyMustInclude?: string[] | undefined;
|
|
86
|
+
}, {
|
|
87
|
+
path: string;
|
|
88
|
+
method: HttpMethod;
|
|
89
|
+
description: string;
|
|
90
|
+
order: number;
|
|
91
|
+
interactionType: "error" | "success" | "edge-case";
|
|
92
|
+
expectedStatusCode: number;
|
|
93
|
+
queryParams?: Record<string, any> | undefined;
|
|
94
|
+
requestBody?: Record<string, any> | undefined;
|
|
95
|
+
responseBody?: any[] | Record<string, any> | undefined;
|
|
96
|
+
expectedResponseFields?: string[] | undefined;
|
|
97
|
+
chainsFrom?: {
|
|
98
|
+
sourceStep: number;
|
|
99
|
+
sourceField: string;
|
|
100
|
+
sourceLocation: "body" | "header" | "cookie";
|
|
101
|
+
targetParam: string;
|
|
102
|
+
targetLocation: "path" | "body" | "header" | "cookie" | "query";
|
|
103
|
+
} | {
|
|
104
|
+
sourceStep: number;
|
|
105
|
+
sourceField: string;
|
|
106
|
+
sourceLocation: "body" | "header" | "cookie";
|
|
107
|
+
targetParam: string;
|
|
108
|
+
targetLocation: "path" | "body" | "header" | "cookie" | "query";
|
|
109
|
+
}[] | undefined;
|
|
110
|
+
bodyMustInclude?: string[] | undefined;
|
|
111
|
+
}>, "many">;
|
|
112
|
+
discriminator: z.ZodOptional<z.ZodObject<{
|
|
113
|
+
kind: z.ZodNativeEnum<typeof DiscriminatorKind>;
|
|
114
|
+
changedCodeAnchor: z.ZodString;
|
|
115
|
+
}, "strip", z.ZodTypeAny, {
|
|
116
|
+
kind: DiscriminatorKind;
|
|
117
|
+
changedCodeAnchor: string;
|
|
118
|
+
}, {
|
|
119
|
+
kind: DiscriminatorKind;
|
|
120
|
+
changedCodeAnchor: string;
|
|
121
|
+
}>>;
|
|
122
|
+
}, "strip", z.ZodTypeAny, {
|
|
123
|
+
description: string;
|
|
124
|
+
priority: "high" | "medium" | "low";
|
|
125
|
+
testType: TestType;
|
|
126
|
+
scenarioName: string;
|
|
127
|
+
category: "new_endpoint" | "bug_caught" | "business_rule" | "security_boundary" | "data_integrity" | "breaking_change" | "auth" | "error_handling" | "workflow" | "data_validation" | "crud";
|
|
128
|
+
steps: {
|
|
129
|
+
path: string;
|
|
130
|
+
method: HttpMethod;
|
|
131
|
+
description: string;
|
|
132
|
+
order: number;
|
|
133
|
+
interactionType: "error" | "success" | "edge-case";
|
|
134
|
+
expectedStatusCode: number;
|
|
135
|
+
queryParams?: Record<string, any> | undefined;
|
|
136
|
+
requestBody?: Record<string, any> | undefined;
|
|
137
|
+
responseBody?: any[] | Record<string, any> | undefined;
|
|
138
|
+
expectedResponseFields?: string[] | undefined;
|
|
139
|
+
chainsFrom?: {
|
|
140
|
+
sourceStep: number;
|
|
141
|
+
sourceField: string;
|
|
142
|
+
sourceLocation: "body" | "header" | "cookie";
|
|
143
|
+
targetParam: string;
|
|
144
|
+
targetLocation: "path" | "body" | "header" | "cookie" | "query";
|
|
145
|
+
} | {
|
|
146
|
+
sourceStep: number;
|
|
147
|
+
sourceField: string;
|
|
148
|
+
sourceLocation: "body" | "header" | "cookie";
|
|
149
|
+
targetParam: string;
|
|
150
|
+
targetLocation: "path" | "body" | "header" | "cookie" | "query";
|
|
151
|
+
}[] | undefined;
|
|
152
|
+
bodyMustInclude?: string[] | undefined;
|
|
153
|
+
}[];
|
|
154
|
+
discriminator?: {
|
|
155
|
+
kind: DiscriminatorKind;
|
|
156
|
+
changedCodeAnchor: string;
|
|
157
|
+
} | undefined;
|
|
158
|
+
}, {
|
|
159
|
+
description: string;
|
|
160
|
+
priority: "high" | "medium" | "low";
|
|
161
|
+
testType: TestType;
|
|
162
|
+
scenarioName: string;
|
|
163
|
+
category: "new_endpoint" | "bug_caught" | "business_rule" | "security_boundary" | "data_integrity" | "breaking_change" | "auth" | "error_handling" | "workflow" | "data_validation" | "crud";
|
|
164
|
+
steps: {
|
|
165
|
+
path: string;
|
|
166
|
+
method: HttpMethod;
|
|
167
|
+
description: string;
|
|
168
|
+
order: number;
|
|
169
|
+
interactionType: "error" | "success" | "edge-case";
|
|
170
|
+
expectedStatusCode: number;
|
|
171
|
+
queryParams?: Record<string, any> | undefined;
|
|
172
|
+
requestBody?: Record<string, any> | undefined;
|
|
173
|
+
responseBody?: any[] | Record<string, any> | undefined;
|
|
174
|
+
expectedResponseFields?: string[] | undefined;
|
|
175
|
+
chainsFrom?: {
|
|
176
|
+
sourceStep: number;
|
|
177
|
+
sourceField: string;
|
|
178
|
+
sourceLocation: "body" | "header" | "cookie";
|
|
179
|
+
targetParam: string;
|
|
180
|
+
targetLocation: "path" | "body" | "header" | "cookie" | "query";
|
|
181
|
+
} | {
|
|
182
|
+
sourceStep: number;
|
|
183
|
+
sourceField: string;
|
|
184
|
+
sourceLocation: "body" | "header" | "cookie";
|
|
185
|
+
targetParam: string;
|
|
186
|
+
targetLocation: "path" | "body" | "header" | "cookie" | "query";
|
|
187
|
+
}[] | undefined;
|
|
188
|
+
bodyMustInclude?: string[] | undefined;
|
|
189
|
+
}[];
|
|
190
|
+
discriminator?: {
|
|
191
|
+
kind: DiscriminatorKind;
|
|
192
|
+
changedCodeAnchor: string;
|
|
193
|
+
} | undefined;
|
|
194
|
+
}>;
|
|
195
|
+
type RegisterCandidateInput = z.infer<typeof registerCandidateSchema>;
|
|
196
|
+
/** Build the agent-submitted candidates. Discriminator claims are verified
|
|
197
|
+
* later, by {@link applyDiscriminatorClaims}, once the merge has settled which
|
|
198
|
+
* scenario each name resolves to. Exported for the direct unit test in
|
|
199
|
+
* registerTestPlanTool.test.ts — a candidate that does not survive selection
|
|
200
|
+
* never reaches the persisted plan, so its subjectEndpoints fill is not
|
|
201
|
+
* observable through the tool's output, and testing here avoids coupling the
|
|
202
|
+
* fill to what selection happens to keep. */
|
|
203
|
+
export declare function buildAgentCandidates(candidates: RegisterCandidateInput[], changedRoutes: ChangedRoute[]): Candidate[];
|
|
2
204
|
export declare function registerRegisterTestPlanTool(server: McpServer): void;
|
|
205
|
+
export {};
|
|
@@ -10,6 +10,8 @@ import { SCENARIO_CATEGORIES, CATEGORY_PRIORITY, Novelty, PriorityTier } from ".
|
|
|
10
10
|
import { HttpMethod, TestType } from "../../types/TestTypes.js";
|
|
11
11
|
import { CandidateSource, computeCandidateId, scenarioMergeKey, DiscriminatorKind } from "../../types/Recommendation.js";
|
|
12
12
|
import { selectPlan } from "../../recommendation/planRanker.js";
|
|
13
|
+
import { resolveSubjectEndpoints } from "../../utils/subjectEndpoints.js";
|
|
14
|
+
import { collectChangedRouteLines } from "../../utils/changedRoutes.js";
|
|
13
15
|
import { reservedUISlots } from "../../recommendation/budgeters/shared.js";
|
|
14
16
|
import { inferScenarioType } from "../../recommendation/diversity.js";
|
|
15
17
|
import { validateDiscriminator } from "../../recommendation/discriminators.js";
|
|
@@ -49,7 +51,10 @@ const registerStepSchema = z.object({
|
|
|
49
51
|
.record(z.any())
|
|
50
52
|
.optional()
|
|
51
53
|
.describe("Query parameters with concrete values. Also scanned for boundary_equality and negative_match verification."),
|
|
52
|
-
responseBody: z
|
|
54
|
+
responseBody: z
|
|
55
|
+
.union([z.record(z.any()), z.array(z.any())])
|
|
56
|
+
.optional()
|
|
57
|
+
.describe("Expected response body shape/values, when asserted. An array for a collection/list endpoint that returns a JSON array."),
|
|
53
58
|
expectedStatusCode: z.number().int().describe("HTTP status code this step asserts (e.g. 201, 400, 404)."),
|
|
54
59
|
expectedResponseFields: z
|
|
55
60
|
.array(z.string())
|
|
@@ -119,8 +124,8 @@ function toScenarioStep(step) {
|
|
|
119
124
|
chainsFrom: step.chainsFrom,
|
|
120
125
|
};
|
|
121
126
|
}
|
|
122
|
-
function toDraftedScenario(input) {
|
|
123
|
-
|
|
127
|
+
function toDraftedScenario(input, changedRoutes) {
|
|
128
|
+
const scenario = {
|
|
124
129
|
scenarioName: input.scenarioName,
|
|
125
130
|
description: input.description,
|
|
126
131
|
category: input.category,
|
|
@@ -131,13 +136,20 @@ function toDraftedScenario(input) {
|
|
|
131
136
|
estimatedComplexity: "moderate",
|
|
132
137
|
testType: input.testType,
|
|
133
138
|
};
|
|
139
|
+
// Resolved here, at the one boundary that has both the steps and the diff, so
|
|
140
|
+
// every later dedup stage reads one recorded value (SKYR-4214).
|
|
141
|
+
return { ...scenario, subjectEndpoints: resolveSubjectEndpoints(scenario, { changedRoutes }) };
|
|
134
142
|
}
|
|
135
143
|
/** Build the agent-submitted candidates. Discriminator claims are verified
|
|
136
144
|
* later, by {@link applyDiscriminatorClaims}, once the merge has settled which
|
|
137
|
-
* scenario each name resolves to.
|
|
138
|
-
|
|
145
|
+
* scenario each name resolves to. Exported for the direct unit test in
|
|
146
|
+
* registerTestPlanTool.test.ts — a candidate that does not survive selection
|
|
147
|
+
* never reaches the persisted plan, so its subjectEndpoints fill is not
|
|
148
|
+
* observable through the tool's output, and testing here avoids coupling the
|
|
149
|
+
* fill to what selection happens to keep. */
|
|
150
|
+
export function buildAgentCandidates(candidates, changedRoutes) {
|
|
139
151
|
return candidates.map((input) => {
|
|
140
|
-
const scenario = toDraftedScenario(input);
|
|
152
|
+
const scenario = toDraftedScenario(input, changedRoutes);
|
|
141
153
|
return {
|
|
142
154
|
scenario,
|
|
143
155
|
priority: derivePriorityTier(scenario),
|
|
@@ -230,6 +242,13 @@ function resolveBudgetContext(stateData, fullState) {
|
|
|
230
242
|
isUIOnlyPR: pbc?.isUIOnlyPR ?? false,
|
|
231
243
|
hasFrontendChanges: pbc?.hasFrontendChanges ?? false,
|
|
232
244
|
externalCoverage: new Set(pbc?.externalCoverageKeys ?? []),
|
|
245
|
+
// `true` for the same reason `computeDiffChangesTestFiles` answers `true` with
|
|
246
|
+
// no diff: an unknown must shut the reserve, not open it. State written by an
|
|
247
|
+
// older MCP build carries no such field, and if that PR did add its own tests
|
|
248
|
+
// the reserve would recommend duplicates of them — the failure this flag
|
|
249
|
+
// exists to stop, reached by version skew alone. It is also the pre-reserve
|
|
250
|
+
// behaviour: covered candidates were dropped.
|
|
251
|
+
diffChangesTestFiles: pbc?.diffChangesTestFiles ?? true,
|
|
233
252
|
};
|
|
234
253
|
const relatedSections = Object.values(fullState?.relatedRepos ?? {}).map((section) => section.data);
|
|
235
254
|
if (relatedSections.length === 0)
|
|
@@ -260,6 +279,12 @@ function resolveBudgetContext(stateData, fullState) {
|
|
|
260
279
|
isUIOnlyPR: hasFrontendChanges && !hasApiChanges,
|
|
261
280
|
hasFrontendChanges,
|
|
262
281
|
externalCoverage,
|
|
282
|
+
// Unioned across sections like the flags above: a pull request that adds
|
|
283
|
+
// its own tests in ANY repo of the run must shut the reserve, or that repo's
|
|
284
|
+
// own coverage becomes grounds to recommend duplicates of it. A section with
|
|
285
|
+
// no budget context is an unknown, and an unknown shuts the reserve — the
|
|
286
|
+
// same direction `computeDiffChangesTestFiles` takes with no diff.
|
|
287
|
+
diffChangesTestFiles: sections.some((s) => s?.planBudgetContext?.diffChangesTestFiles ?? true),
|
|
263
288
|
};
|
|
264
289
|
}
|
|
265
290
|
function describeGenerationCall(item) {
|
|
@@ -315,7 +340,7 @@ function renderGenerationDirective(plan) {
|
|
|
315
340
|
(nonUICount > 0 ? ` Generate the ${tests(nonUICount)} of other types as well.` : ""),
|
|
316
341
|
];
|
|
317
342
|
}
|
|
318
|
-
function renderPlanText(plan) {
|
|
343
|
+
function renderPlanText(plan, dropped) {
|
|
319
344
|
const lines = [];
|
|
320
345
|
lines.push(`## Approved Test Plan (${plan.planId})`);
|
|
321
346
|
lines.push("");
|
|
@@ -338,10 +363,41 @@ function renderPlanText(plan) {
|
|
|
338
363
|
plan.additional.forEach((item, i) => lines.push(renderPlanItem(item, plan.generate.length + i + 1)));
|
|
339
364
|
}
|
|
340
365
|
if (plan.demotions.length > 0) {
|
|
366
|
+
// A demotion (failed discriminator claim) and a drop (removed by the
|
|
367
|
+
// budgeter for an unrelated reason — coverage or budget) are independent:
|
|
368
|
+
// a candidate can suffer both. Split so the report never claims a
|
|
369
|
+
// dropped candidate is "still registered".
|
|
370
|
+
const dropReasonByCandidateId = new Map(dropped.map((d) => [d.candidateId, d.reason]));
|
|
371
|
+
const registeredDemotions = plan.demotions.filter((d) => !dropReasonByCandidateId.has(d.candidateId));
|
|
372
|
+
const removedDemotions = plan.demotions.filter((d) => dropReasonByCandidateId.has(d.candidateId));
|
|
373
|
+
if (registeredDemotions.length > 0) {
|
|
374
|
+
lines.push("");
|
|
375
|
+
lines.push(`### Demotions (${registeredDemotions.length}) — discriminator claims that did NOT verify (candidate still registered, claim dropped)`);
|
|
376
|
+
for (const demotion of registeredDemotions) {
|
|
377
|
+
lines.push(`- ${demotion.candidateId}: ${demotion.reason}`);
|
|
378
|
+
}
|
|
379
|
+
}
|
|
380
|
+
if (removedDemotions.length > 0) {
|
|
381
|
+
lines.push("");
|
|
382
|
+
lines.push(`### Demoted AND removed (${removedDemotions.length}) — the claim failed to verify AND the candidate left the plan. It is NOT registered.`);
|
|
383
|
+
for (const demotion of removedDemotions) {
|
|
384
|
+
lines.push(`- ${demotion.candidateId}: claim failed — ${demotion.reason}; removed — ${dropReasonByCandidateId.get(demotion.candidateId)}`);
|
|
385
|
+
}
|
|
386
|
+
}
|
|
387
|
+
}
|
|
388
|
+
// Every other removal — budget or coverage, with no failed claim attached.
|
|
389
|
+
// This section is independent of `demotions`: the two channels are unrelated,
|
|
390
|
+
// and the motivating case has drops and NO demotions. Nesting it inside the
|
|
391
|
+
// demotion block returned empty GENERATE/ADDITIONAL lists with no candidate id
|
|
392
|
+
// and no reason, which is the state item 5 exists to make visible. Measured on
|
|
393
|
+
// eval run 32586610615, fixture repo15-P14: 10 drops, 0 demotions.
|
|
394
|
+
const demotedIds = new Set(plan.demotions.map((d) => d.candidateId));
|
|
395
|
+
const removedWithoutDemotion = dropped.filter((d) => !demotedIds.has(d.candidateId));
|
|
396
|
+
if (removedWithoutDemotion.length > 0) {
|
|
341
397
|
lines.push("");
|
|
342
|
-
lines.push(`###
|
|
343
|
-
for (const
|
|
344
|
-
lines.push(`- ${
|
|
398
|
+
lines.push(`### Removed (${removedWithoutDemotion.length}) — candidates the selection stage dropped. They are NOT registered.`);
|
|
399
|
+
for (const drop of removedWithoutDemotion) {
|
|
400
|
+
lines.push(`- ${drop.candidateId}: ${drop.reason}`);
|
|
345
401
|
}
|
|
346
402
|
}
|
|
347
403
|
lines.push("");
|
|
@@ -409,7 +465,8 @@ export function registerRegisterTestPlanTool(server) {
|
|
|
409
465
|
...(fullState?.repositoryAnalysis?.scenarios ?? []),
|
|
410
466
|
...Object.values(fullState?.relatedRepos ?? {}).flatMap((section) => section.data?.repositoryAnalysis?.scenarios ?? []),
|
|
411
467
|
];
|
|
412
|
-
const
|
|
468
|
+
const changedRoutes = collectChangedRouteLines(diffText);
|
|
469
|
+
const agentCandidates = buildAgentCandidates(params.candidates ?? [], changedRoutes);
|
|
413
470
|
// The approved plan is run-wide and persists at the ROOT (see the
|
|
414
471
|
// persist step below), so prior-plan recovery reads the root first;
|
|
415
472
|
// the section fallback covers state written by older builds.
|
|
@@ -501,6 +558,7 @@ export function registerRegisterTestPlanTool(server) {
|
|
|
501
558
|
generate: result.generate.map(buildApprovedPlanItem),
|
|
502
559
|
additional: result.additional.map(buildApprovedPlanItem),
|
|
503
560
|
demotions: result.demotions,
|
|
561
|
+
dropped: result.dropped,
|
|
504
562
|
};
|
|
505
563
|
// The approved plan is the ONE run-wide authority (the prompt mandates
|
|
506
564
|
// a single pooled registration on multi-repo runs), so it persists at
|
|
@@ -518,7 +576,7 @@ export function registerRegisterTestPlanTool(server) {
|
|
|
518
576
|
return errorResult;
|
|
519
577
|
}
|
|
520
578
|
return {
|
|
521
|
-
content: [{ type: "text", text: renderPlanText(approvedPlan) }],
|
|
579
|
+
content: [{ type: "text", text: renderPlanText(approvedPlan, result.dropped) }],
|
|
522
580
|
};
|
|
523
581
|
}
|
|
524
582
|
catch (error) {
|