@skyramp/mcp 0.3.1 → 0.3.2-rc.pom-2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/build/index.js +2 -1
- package/build/prompts/code-reuse.d.ts +7 -1
- package/build/prompts/code-reuse.js +8 -5
- package/build/prompts/code-reuse.test.d.ts +1 -0
- package/build/prompts/code-reuse.test.js +62 -0
- package/build/prompts/pom-aware-code-reuse.d.ts +6 -1
- package/build/prompts/pom-aware-code-reuse.js +100 -53
- package/build/prompts/pom-aware-code-reuse.test.d.ts +1 -0
- package/build/prompts/pom-aware-code-reuse.test.js +11 -0
- package/build/prompts/test-recommendation/diffExecutionPlan.d.ts +4 -2
- package/build/prompts/test-recommendation/diffExecutionPlan.js +11 -65
- package/build/prompts/test-recommendation/fullRepoCatalog.d.ts +2 -2
- package/build/prompts/test-recommendation/recommendationSections.js +5 -2
- package/build/prompts/test-recommendation/scopeAssessment.js +1 -1
- package/build/prompts/test-recommendation/test-recommendation-prompt.d.ts +26 -1
- package/build/prompts/test-recommendation/test-recommendation-prompt.js +68 -56
- package/build/prompts/test-recommendation/test-recommendation-prompt.test.js +48 -11
- package/build/prompts/testbot/testbot-prompts.d.ts +1 -1
- package/build/prompts/testbot/testbot-prompts.js +67 -21
- package/build/prompts/testbot/testbot-prompts.test.js +44 -0
- package/build/recommendation/budgeters/diversityBalancedBudgeter.d.ts +7 -0
- package/build/recommendation/budgeters/diversityBalancedBudgeter.js +71 -0
- package/build/recommendation/budgeters/diversityBalancedBudgeter.test.d.ts +1 -0
- package/build/recommendation/budgeters/diversityBalancedBudgeter.test.js +75 -0
- package/build/recommendation/budgeters/fixedNBudgeter.d.ts +7 -0
- package/build/recommendation/budgeters/fixedNBudgeter.js +11 -0
- package/build/recommendation/budgeters/fixedNBudgeter.test.d.ts +1 -0
- package/build/recommendation/budgeters/fixedNBudgeter.test.js +66 -0
- package/build/recommendation/budgeters/shared.d.ts +19 -0
- package/build/recommendation/budgeters/shared.js +66 -0
- package/build/recommendation/discriminators.d.ts +31 -0
- package/build/recommendation/discriminators.js +355 -0
- package/build/recommendation/discriminators.test.d.ts +1 -0
- package/build/recommendation/discriminators.test.js +324 -0
- package/build/recommendation/diversity.d.ts +47 -0
- package/build/recommendation/diversity.js +101 -0
- package/build/recommendation/diversity.test.d.ts +1 -0
- package/build/recommendation/diversity.test.js +77 -0
- package/build/recommendation/planRanker.d.ts +50 -0
- package/build/recommendation/planRanker.js +67 -0
- package/build/recommendation/planRanker.test.d.ts +1 -0
- package/build/recommendation/planRanker.test.js +110 -0
- package/build/recommendation/testFixtures.d.ts +25 -0
- package/build/recommendation/testFixtures.js +45 -0
- package/build/resources/testbotResource.js +4 -1
- package/build/services/ScenarioGenerationService.d.ts +5 -0
- package/build/services/ScenarioGenerationService.js +16 -1
- package/build/services/ScenarioGenerationService.test.js +44 -0
- package/build/services/TestExecutionService.d.ts +15 -1
- package/build/services/TestExecutionService.js +210 -55
- package/build/services/TestExecutionService.test.js +397 -0
- package/build/services/TestGenerationService.js +19 -1
- package/build/services/TestGenerationService.test.js +58 -0
- package/build/tool-phases.js +1 -0
- package/build/toolNames.d.ts +19 -0
- package/build/toolNames.js +19 -0
- package/build/tools/code-refactor/codeReuseTool.d.ts +7 -0
- package/build/tools/code-refactor/codeReuseTool.js +130 -4
- package/build/tools/code-refactor/codeReuseTool.test.d.ts +1 -0
- package/build/tools/code-refactor/codeReuseTool.test.js +290 -0
- package/build/tools/executeSkyrampTestTool.js +8 -2
- package/build/tools/generate-tests/generateBatchScenarioRestTool.d.ts +6 -1
- package/build/tools/generate-tests/generateBatchScenarioRestTool.js +110 -17
- package/build/tools/generate-tests/generateBatchScenarioRestTool.test.js +147 -0
- package/build/tools/generate-tests/generateContractRestTool.js +11 -1
- package/build/tools/generate-tests/generateIntegrationRestTool.js +24 -1
- package/build/tools/generate-tests/generateIntegrationRestTool.test.d.ts +1 -0
- package/build/tools/generate-tests/generateIntegrationRestTool.test.js +159 -0
- package/build/tools/generate-tests/planGuard.d.ts +13 -0
- package/build/tools/generate-tests/planGuard.js +78 -0
- package/build/tools/generate-tests/planGuard.test.d.ts +1 -0
- package/build/tools/generate-tests/planGuard.test.js +185 -0
- package/build/tools/generate-tests/scenarioFileIdentity.d.ts +10 -0
- package/build/tools/generate-tests/scenarioFileIdentity.js +46 -0
- package/build/tools/generate-tests/scenarioLint.d.ts +30 -0
- package/build/tools/generate-tests/scenarioLint.js +150 -0
- package/build/tools/generate-tests/scenarioLint.test.d.ts +1 -0
- package/build/tools/generate-tests/scenarioLint.test.js +100 -0
- package/build/tools/submitReportTool.js +78 -0
- package/build/tools/submitReportTool.test.js +255 -0
- package/build/tools/test-management/analyzeChangesTool.js +55 -2
- package/build/tools/test-management/analyzeChangesTool.test.js +12 -0
- package/build/tools/test-management/index.d.ts +1 -0
- package/build/tools/test-management/index.js +1 -0
- package/build/tools/test-management/registerTestPlanTool.d.ts +2 -0
- package/build/tools/test-management/registerTestPlanTool.js +329 -0
- package/build/tools/test-management/registerTestPlanTool.test.d.ts +1 -0
- package/build/tools/test-management/registerTestPlanTool.test.js +296 -0
- package/build/types/Recommendation.d.ts +97 -0
- package/build/types/Recommendation.js +48 -0
- package/build/types/RepositoryAnalysis.d.ts +14 -14
- package/build/types/TestExecution.d.ts +2 -0
- package/build/types/TestRecommendation.d.ts +12 -1
- package/build/types/TestRecommendation.js +26 -11
- package/build/types/TestTypes.js +1 -1
- package/build/utils/AnalysisStateManager.d.ts +47 -0
- package/build/utils/docker.test.js +1 -1
- package/build/utils/planMatchKeys.d.ts +61 -0
- package/build/utils/planMatchKeys.js +125 -0
- package/build/utils/pom-scope/import-expansion.d.ts +5 -0
- package/build/utils/pom-scope/import-expansion.js +32 -0
- package/build/utils/pom-scope/index.d.ts +39 -0
- package/build/utils/pom-scope/index.js +120 -0
- package/build/utils/pom-scope/index.test.d.ts +1 -0
- package/build/utils/pom-scope/index.test.js +239 -0
- package/build/utils/pom-scope/pom-files.d.ts +3 -0
- package/build/utils/pom-scope/pom-files.js +48 -0
- package/build/utils/pom-scope/pom-files.test.d.ts +1 -0
- package/build/utils/pom-scope/pom-files.test.js +29 -0
- package/build/utils/pom-scope/scoring.d.ts +20 -0
- package/build/utils/pom-scope/scoring.js +45 -0
- package/build/utils/pom-scope/scoring.test.d.ts +1 -0
- package/build/utils/pom-scope/scoring.test.js +39 -0
- package/build/utils/pom-scope/selector-extractor.d.ts +7 -0
- package/build/utils/pom-scope/selector-extractor.js +57 -0
- package/build/utils/pom-scope/selector-extractor.test.d.ts +1 -0
- package/build/utils/pom-scope/selector-extractor.test.js +67 -0
- package/build/utils/pom-verify/__fixtures__/af-style/asset-list.page.d.ts +5 -0
- package/build/utils/pom-verify/__fixtures__/af-style/asset-list.page.js +5 -0
- package/build/utils/pom-verify/__fixtures__/af-style/pageobjects/asset-list-page.d.ts +5 -0
- package/build/utils/pom-verify/__fixtures__/af-style/pageobjects/asset-list-page.js +9 -0
- package/build/utils/pom-verify/__fixtures__/af-style/report.iframe.page.d.ts +4 -0
- package/build/utils/pom-verify/__fixtures__/af-style/report.iframe.page.js +4 -0
- package/build/utils/pom-verify/__fixtures__/af-style/workflow-footer.page.d.ts +4 -0
- package/build/utils/pom-verify/__fixtures__/af-style/workflow-footer.page.js +4 -0
- package/build/utils/pom-verify/bindings.d.ts +19 -0
- package/build/utils/pom-verify/bindings.js +161 -0
- package/build/utils/pom-verify/bindings.test.d.ts +1 -0
- package/build/utils/pom-verify/bindings.test.js +164 -0
- package/build/utils/pom-verify/calls.d.ts +16 -0
- package/build/utils/pom-verify/calls.js +42 -0
- package/build/utils/pom-verify/calls.test.d.ts +1 -0
- package/build/utils/pom-verify/calls.test.js +61 -0
- package/build/utils/pom-verify/index.d.ts +4 -0
- package/build/utils/pom-verify/index.js +4 -0
- package/build/utils/pom-verify/resolve.d.ts +7 -0
- package/build/utils/pom-verify/resolve.js +27 -0
- package/build/utils/pom-verify/resolve.test.d.ts +1 -0
- package/build/utils/pom-verify/resolve.test.js +68 -0
- package/build/utils/pom-verify/strip.d.ts +9 -0
- package/build/utils/pom-verify/strip.js +89 -0
- package/build/utils/pom-verify/verify.d.ts +14 -0
- package/build/utils/pom-verify/verify.js +158 -0
- package/build/utils/pom-verify/verify.test.d.ts +1 -0
- package/build/utils/pom-verify/verify.test.js +325 -0
- package/build/utils/reportVerification.d.ts +61 -0
- package/build/utils/reportVerification.js +104 -0
- package/build/utils/reportVerification.test.d.ts +1 -0
- package/build/utils/reportVerification.test.js +185 -0
- package/build/utils/scenarioDrafting.js +5 -5
- package/build/utils/versions.d.ts +3 -3
- package/build/utils/versions.js +1 -1
- package/build/utils/workspaceAuth.d.ts +9 -1
- package/build/utils/workspaceAuth.js +25 -5
- package/build/utils/workspaceAuth.test.js +48 -0
- package/build/workspace/workspace.d.ts +20 -0
- package/build/workspace/workspace.js +4 -0
- package/build/workspace/workspace.test.js +10 -0
- package/node_modules/playwright/lib/mcp/skyramp/traceRecordingBackend.js +77 -8
- package/node_modules/playwright/node_modules/playwright-core/lib/generated/injectedScriptSource.js +1 -1
- package/node_modules/playwright/node_modules/playwright-core/lib/generated/pollingRecorderSource.js +1 -1
- package/node_modules/playwright/node_modules/playwright-core/lib/utils/isomorphic/volatileDate.js +101 -0
- package/node_modules/playwright/node_modules/playwright-core/lib/utils.js +2 -0
- package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/assets/{codeMirrorModule-aszq5EdG.js → codeMirrorModule-Bzd72-bG.js} +1 -1
- package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/assets/{defaultSettingsView-BxS7Jm4s.js → defaultSettingsView-DzxTioTK.js} +101 -101
- package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/{index.D4JTTy4R.js → index.BGc30U3S.js} +1 -1
- package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/index.html +2 -2
- package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/{uiMode.DaRMQKOI.js → uiMode.IaDrb29A.js} +1 -1
- package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/uiMode.html +2 -2
- package/node_modules/playwright/node_modules/playwright-core/package.json +1 -1
- package/node_modules/playwright/node_modules/playwright-core/src/generated/injectedScriptSource.ts +1 -1
- package/node_modules/playwright/node_modules/playwright-core/src/generated/pollingRecorderSource.ts +1 -1
- package/node_modules/playwright/node_modules/playwright-core/src/utils/isomorphic/volatileDate.ts +131 -0
- package/node_modules/playwright/node_modules/playwright-core/src/utils.ts +1 -0
- package/node_modules/playwright/package.json +1 -1
- package/package.json +3 -3
- package/node_modules/playwright/node_modules/playwright-core/.DS_Store +0 -0
|
@@ -47,7 +47,7 @@ export function parseRelatedRepositories(raw) {
|
|
|
47
47
|
}
|
|
48
48
|
}
|
|
49
49
|
export function getTestbotPrompt(prTitle, prDescription, summaryOutputFile, repositoryPath, baseBranch, maxRecommendations = MAX_RECOMMENDATIONS, maxGenerate = MAX_TESTS_TO_GENERATE, _maxCritical = MAX_CRITICAL_TESTS, // Reserved — accepted for API compat but not yet wired into prompt
|
|
50
|
-
prNumber, userPrompt, services, uiCredentials, testsRepoDir, relatedRepositories, primaryRepo) {
|
|
50
|
+
prNumber, userPrompt, services, uiCredentials, testsRepoDir, relatedRepositories, primaryRepo, planOnly = false) {
|
|
51
51
|
maxGenerate = Math.min(Math.max(maxGenerate, 0), maxRecommendations);
|
|
52
52
|
// TODO(SKYR-3636 follow-up): migrate Task 1 + Task 2 step bodies to PromptPlan
|
|
53
53
|
// (src/prompts/test-recommendation/promptPlan.ts) so step numbers don't have
|
|
@@ -65,6 +65,17 @@ prNumber, userPrompt, services, uiCredentials, testsRepoDir, relatedRepositories
|
|
|
65
65
|
// testDirectory, relative to the delivery root (the configured test repo if set,
|
|
66
66
|
// otherwise the primary repo).
|
|
67
67
|
const hasRelatedRepos = !!relatedRepositories?.length;
|
|
68
|
+
// Task 1 maintenance step 2c branches in plan-only eval runs (SKYR-3879
|
|
69
|
+
// plan-only lane): the SUT is not running, so the pre-edit baseline
|
|
70
|
+
// execution is replaced by static-analysis-only verdicts. Defined before
|
|
71
|
+
// task1Section, which interpolates it.
|
|
72
|
+
let maintenanceBeforeExecStep;
|
|
73
|
+
if (planOnly) {
|
|
74
|
+
maintenanceBeforeExecStep = ` c. Plan-only run: skip the pre-edit baseline execution — the application is not running. Record every maintenance verdict from static drift analysis alone; execution statuses simply remain unrecorded.`;
|
|
75
|
+
}
|
|
76
|
+
else {
|
|
77
|
+
maintenanceBeforeExecStep = ` c. Call \`skyramp_execute_test\` with \`phase: "before"\` and \`stateFile\` for every UPDATE/REGENERATE/DELETE test. Exclude tests marked \`[external]\`. Run them sequentially, not in parallel. This captures the pre-edit baseline — do not skip even if you expect the test to fail.`;
|
|
78
|
+
}
|
|
68
79
|
// For follow-up requests: emit the @skyramp-testbot header + guardrails + retrieve-recommendations step.
|
|
69
80
|
// For first-run prompts: emit the full Task 1 analysis + maintenance section.
|
|
70
81
|
const task1Section = userPrompt
|
|
@@ -131,7 +142,7 @@ ${hasRelatedRepos ? `
|
|
|
131
142
|
|
|
132
143
|
b. Write \`updateInstructions\` for each UPDATE or REGENERATE test before calling \`skyramp_actions\` — articulating the change first prevents file content from overriding diff-based reasoning.
|
|
133
144
|
|
|
134
|
-
|
|
145
|
+
${maintenanceBeforeExecStep}
|
|
135
146
|
|
|
136
147
|
d. Call \`skyramp_actions\` with \`stateFile\` (from \`skyramp_analyze_changes\` output) and apply the edits it returns.
|
|
137
148
|
|
|
@@ -343,19 +354,37 @@ ${hasRelatedRepos ? `
|
|
|
343
354
|
const testDirInstruction = testsRepoDir
|
|
344
355
|
? `the \`<output_dir>\` from the \`<services>\` block, rooted under the test repository at \`${testsRepoDir}\` (i.e. \`${testsRepoDir}/<output_dir>\`). Write ALL test output files to paths under \`${testsRepoDir}\`, not under \`${repositoryPath}\`. Do NOT write any test files to the app repository.`
|
|
345
356
|
: `${SERVICE_REFS.testDirRef}. Do NOT create a new \`tests/\` directory at the repo root — use that path. If no \`testDirectory\` is configured, default to the language-conventional location (e.g. \`src/test/java/...\` for Java, \`tests/\` for Python).`;
|
|
346
|
-
|
|
347
|
-
|
|
348
|
-
|
|
349
|
-
|
|
350
|
-
|
|
351
|
-
|
|
352
|
-
|
|
353
|
-
|
|
354
|
-
|
|
357
|
+
// Task 3's non-zero-budget count rule branches in plan-only eval runs: the
|
|
358
|
+
// report DECLARES the final GENERATE selection instead of recording
|
|
359
|
+
// generated files. The zero-surface abstention rules above it are shared.
|
|
360
|
+
let task3CountRule;
|
|
361
|
+
if (planOnly) {
|
|
362
|
+
task3CountRule = `Otherwise (your Budget Plan is non-zero): this is a plan-only run — \`newTestsCreated\` DECLARES your final GENERATE selection instead of recording generated files: exactly one entry per item of your final GENERATE list (the list returned by \`skyramp_register_test_plan\` when that tool was available — it may contain fewer items than your committed budget — otherwise your Budget Plan selection, at most ${maxGenerate}), with \`fileName\` set to the file name you would have used. \`testResults\` must be \`[]\` — nothing was executed. The declaration itself is the deliverable; do not generate or backfill.`;
|
|
363
|
+
}
|
|
364
|
+
else {
|
|
365
|
+
task3CountRule = `Otherwise (your Budget Plan is non-zero): in \`newTestsCreated\`, you must have exactly as many budget-counting new tests as your committed Budget Plan's generate count (at most ${maxGenerate}). Only new files (ADD) created for the planned GENERATE items count toward this target — GENERATE items converted to UPDATE do not. You may also include at most one additional discovered-scenario file in \`newTestsCreated\` (the bug-catching test generated after all planned items); that extra test does **not** count against the budget. If you have fewer budget-counting new tests than your generate count, backfill from the remaining ADDITIONAL candidates before proceeding. Only proceed with fewer if all candidates failed after retry AND the fallback single-contract test also failed.`;
|
|
366
|
+
}
|
|
367
|
+
// Task 2 branches wholesale in plan-only eval runs (SKYR-3879 plan-only
|
|
368
|
+
// lane): the standard task mandates generation and execution, which a
|
|
369
|
+
// plan-only run must not do — branching the section (rather than appending
|
|
370
|
+
// an override) means the agent never sees conflicting instructions.
|
|
371
|
+
let task2Section;
|
|
372
|
+
if (planOnly) {
|
|
373
|
+
task2Section = `## Task 2: Commit the Test Plan (plan-only run)
|
|
374
|
+
|
|
375
|
+
This is a plan-only evaluation run: the application under test is NOT running, and this run evaluates test SELECTION only. Nothing is generated or executed in this task.
|
|
376
|
+
|
|
377
|
+
- Draft your complete candidate list exactly as the Execution Plan directs — every API test (contract / integration / batch-scenario) you would generate OR recommend for this PR, grounded in the analysis output and the diff. Favor tests that would FAIL if the changed logic were buggy, not just tests that exercise the new surface.
|
|
378
|
+
- If a tool named \`skyramp_register_test_plan\` is available, call it with the full candidate union — include a \`discriminator\` claim \`{kind, changedCodeAnchor}\` for every candidate that probes changed logic — and treat its returned GENERATE list as your final selection. If that tool is not available, commit to your Budget Plan's GENERATE selection (at most ${maxGenerate}).
|
|
379
|
+
- Skip UI and E2E candidates entirely — with no running app there are no blueprints to ground them, and this lane evaluates API test selection only.
|
|
380
|
+
- Take no other actions in this task: no test generation tools, no browser traces or blueprint captures, no test files written, no test executions. Proceed directly to ${taskRef(TASK_SUBMIT)}.`;
|
|
381
|
+
}
|
|
382
|
+
else {
|
|
383
|
+
task2Section = `## Task 2: Generate New Tests
|
|
355
384
|
|
|
356
385
|
${userPrompt ? "Generate only the tests that the user requested from the Additional Recommendations. The rules below still apply." : "Drift-based maintenance (Task 1) is complete. This step only processes the GENERATE list. Exception: if a GENERATE item targets a resource with an existing `[skyramp]` contract test, UPDATE that test file (see covered-resource handling below) — a new test case added to an existing file counts toward the budget and is reported in `newTestsCreated`."}
|
|
357
386
|
|
|
358
|
-
- **MANDATORY — use the pre-ranked GENERATE list
|
|
387
|
+
- **MANDATORY — use the plan returned by \`skyramp_register_test_plan\` as-is**: Before generating anything, call \`skyramp_register_test_plan\` (\`stateFile\` required) with your complete candidate list — every test you would generate OR recommend, including the Execution Plan's own pre-ranked GENERATE/ADDITIONAL items and any candidate you drafted yourself, with a \`discriminator\` claim \`{kind, changedCodeAnchor}\` for candidates probing changed logic. Its returned GENERATE list — not the Execution Plan's raw pre-ranked GENERATE section — governs ADD actions from this point on. You MUST generate exactly those scenarios in the exact order listed, keeping each item's \`scenarioName\` exactly as registered — the generation tools match on it and reject renamed or substituted scenarios. If parameter grounding uncovers a distinct bug-catching scenario not already registered, generate it after all planned GENERATE items are complete and report it in \`newTestsCreated\` — this is an additional test driven by source-code analysis and does not count against the GENERATE budget.${hasRelatedRepos ? `\n - **Multi-repo exception:** this run has related repositories, so the per-repo GENERATE lists are NOT final — they are candidates re-selected by the cross-repo round-robin described in Task 1's "Cross-repo test generation". Register the pooled, type-distributed selection instead of any single repo's GENERATE list. (In single-repo runs, register the GENERATE list exactly as-is.)` : ""}
|
|
359
388
|
- **Do not fabricate tests outside the GENERATE list.** New test files cover NEW observable surface only — a new endpoint, or a newly-integrated component/route not already covered by an existing test. Changes that only modify, delete, or add fields to an EXISTING covered endpoint or component are maintenance: handle them in ${taskRef(TASK_ANALYZE_MAINTAIN)} by UPDATE/DELETE of the existing test, never by creating a new spec. If the GENERATE list is empty (deletion-only, cosmetic, or modification-of-existing PRs with no new surface), create zero new tests and proceed to ${taskRef(TASK_SUBMIT)} — do not invent a new spec to have something to report.
|
|
360
389
|
- Scenario JSON files are always new files — always generate them for new methods. Every generated scenario JSON must have a corresponding new integration test generated from it via \`skyramp_integration_test_generation\`.
|
|
361
390
|
- Covered-resource handling (aligns with Execution Plan Step 0): When a GENERATE item targets a resource that already has an existing test file covering the same endpoint:
|
|
@@ -416,7 +445,7 @@ ${userPrompt ? "Generate only the tests that the user requested from the Additio
|
|
|
416
445
|
${CONTRACT_MODE_GUIDANCE}
|
|
417
446
|
- ${PATH_PARAM_UUID_GUIDANCE}
|
|
418
447
|
- **UI**: First check for existing Playwright trace \`.zip\` files in the repo (Testbot scans recursively up to 5 directory levels — the per-service output directories, \`frontend/\`, \`public/\`, \`.skyramp/\`, or any subdirectory).
|
|
419
|
-
If a relevant trace exists (covers the UI changes in this PR), use it directly with \`skyramp_ui_test_generation\` and \`
|
|
448
|
+
If a relevant trace exists (covers the UI changes in this PR), use it directly with \`skyramp_ui_test_generation\`, \`modularizeCode: false\`, and \`codeReuse: true\` (when generating a TypeScript/JavaScript Playwright test — the default; leave \`codeReuse\` unset for other languages).
|
|
420
449
|
If NO relevant trace exists, **you MUST write out your full trace plan as text BEFORE calling \`browser_navigate\`**. Do not touch the browser until the plan is written.
|
|
421
450
|
|
|
422
451
|
**Browser authentication (check BEFORE navigating)**: If \`<ui-credentials>\` appears in your context above, the app requires login. Parse the credentials — one per line, two supported formats:
|
|
@@ -460,7 +489,7 @@ ${CONTRACT_MODE_GUIDANCE}
|
|
|
460
489
|
Follow the **UI Recording Workflow** section at the end of this prompt. Additional CI constraints:
|
|
461
490
|
- Navigate **directly** to the deepest relevant URL (e.g. \`/orders/1/edit\` instead of \`/\` then \`/orders\` then \`/orders/1\`) — minimize multi-hop navigation so the trace stays focused on the scenario under test.
|
|
462
491
|
- \`skyramp_export_zip\` outputPath: \`${repositoryPath}/.skyramp/<test_name>_trace.zip\`
|
|
463
|
-
- \`skyramp_ui_test_generation\`: set \`modularizeCode: false\`
|
|
492
|
+
- \`skyramp_ui_test_generation\`: set \`modularizeCode: false\` and \`codeReuse: true\` (TypeScript/JavaScript Playwright only — the default; leave \`codeReuse\` unset for other languages)
|
|
464
493
|
- **\`browser_assert\` — MANDATORY**: at least one per page navigated. Call multiple assertions in the same tool call batch when checking independent elements. If you navigate to 2 pages, assert on both. Each assertion should verify a business outcome (state change, computed value, error condition) — not just that an element is visible.
|
|
465
494
|
- **\`browser_visual_snapshot\` — for visual/appearance checks**: when the instruction asks to take a screenshot, capture a baseline, or verify how a page/element/region *looks* (not its text or value), call \`browser_visual_snapshot\` — it records a \`toHaveScreenshot()\` assertion so the generated test pixel-compares against a baseline on every run. Do NOT use \`browser_take_screenshot\` for this: it captures a throwaway image that is dropped at export and never appears in the generated test (use it only to view the page yourself).
|
|
466
495
|
- **Wait for stable state before the second capture**: After performing an action that affects computed fields (filling a discount, submitting a form, adding an item), check the current page state before calling the second \`browser_blueprint\` (the capture after the action). If a computed field — total, price, count, derived text — still shows its initial empty or zero value (e.g. \`$0.00\`, \`0\`, \`Loading...\`, empty string), that means async data hasn't finished loading yet. Use \`browser_wait_for\` to wait up to 10 seconds for the field to update to a real value (for example, wait for the total to show a non-zero amount like \`$799.99\` instead of \`$0.00\`). Once the field shows a real value, THEN call the second \`browser_blueprint\` to capture stable state. If after 10 seconds the field still hasn't updated, skip the assertion on that field — don't capture and assert a value that hasn't loaded.
|
|
@@ -512,7 +541,7 @@ ${CONTRACT_MODE_GUIDANCE}
|
|
|
512
541
|
|
|
513
542
|
**The Blueprint Citation Invariant applies during recording too.** Every assertion you emit cites element names — those names must come from blueprint captures, not invention. For N user-intent-level actions, the reference target is N+1 \`browser_blueprint\` calls (the first returns full, the rest return deltas). Traces that follow the pattern produce assertions grounded in observable state changes; traces that skip captures fall back to author-inferred assertions and risk citing names that don't exist in the rendered DOM.
|
|
514
543
|
|
|
515
|
-
The rest of the UI workflow stays the same: trace plan, browser auth, navigation, export (\`skyramp_export_zip\`), generation (\`skyramp_ui_test_generation\`), \`skyramp_enhance_assertions\` post-
|
|
544
|
+
The rest of the UI workflow stays the same: trace plan, browser auth, navigation, export (\`skyramp_export_zip\`), generation (\`skyramp_ui_test_generation\`), then \`skyramp_reuse_code\` (when \`codeReuse: true\`) and \`skyramp_enhance_assertions\` post-calls. Capture-act-capture adds blueprint captures alongside the existing steps; it doesn't replace anything.
|
|
516
545
|
- **E2E**: Only if BOTH a backend trace \`.json\` AND a Playwright \`.zip\` already exist in the repo. Without both, move to \`additionalRecommendations\`.
|
|
517
546
|
- Skip smoke tests entirely.
|
|
518
547
|
|
|
@@ -545,13 +574,24 @@ Do NOT use \`page.waitForTimeout()\` with fixed delays. Do NOT retry more than o
|
|
|
545
574
|
**After generation, you MUST do exactly these steps — nothing more, nothing less:**
|
|
546
575
|
1. **[MANDATORY] After \`skyramp_integration_test_generation\`**: Call \`skyramp_enhance_assertions\` with \`testFile\` set to the absolute path of the generated integration test file, \`testType: "integration"\`, and \`enhanceType: "generation"\`. Apply every instruction returned to that file.
|
|
547
576
|
2. **[MANDATORY] After \`skyramp_contract_test_generation\` with \`providerMode\`**: Call \`skyramp_enhance_assertions\` with \`testFile\` set to the absolute path of the generated provider contract test file, \`testType: "contract"\`, and \`enhanceType: "generation"\`. Apply every instruction returned to that file.
|
|
548
|
-
3. **[MANDATORY] After \`skyramp_ui_test_generation
|
|
549
|
-
4. **
|
|
550
|
-
|
|
577
|
+
3. **[MANDATORY] After \`skyramp_ui_test_generation\` with \`codeReuse: true\`**: the generation result directs you to call \`skyramp_reuse_code\` — do this BEFORE \`skyramp_enhance_assertions\`. Two outcomes: (1) a response starting "No reusable POM layer detected" — this is a normal outcome, continue immediately (do NOT retry), and note "no POM layer — reuse skipped" in that test's \`testResults\` entry \`details\`; (2) a refactoring workflow — follow it to completion INCLUDING its verification loop (\`skyramp_reuse_code\` with \`verify: true\`), finish only when it reports PASSED, and include \`verification: passed\` in that test's \`testResults\` entry \`details\` in your final report. If a reused test later fails execution and the failure points at a substituted POM call, restore the saved \`<testFile>.raw.bak\` over the test file and re-run — do NOT hand-edit the customer's POM methods (this re-run counts toward the 2-attempt execution cap — prefer this restore over the generic timeout fix-up when the failing locator came from a POM substitution).
|
|
578
|
+
4. **[MANDATORY] After \`skyramp_ui_test_generation\`**: Call \`skyramp_enhance_assertions\` with \`testFile\` set to the absolute path of the generated UI test file, \`testType: "ui"\`, and \`enhanceType: "generation"\`. Apply every instruction returned to that file. The HIGH-tier \`possibleAssertions\` from your second \`browser_blueprint\` captures (after each action) during trace recording are in your context — when the enhance instructions ask you to add assertions for state-changing actions, use those grounded candidates first (they contain exact computed values from the DOM delta, e.g. \`toHaveText('Total: $899.98')\`). Only fall back to deriving values from the test file or source code when no HIGH-tier candidate covers the action.
|
|
579
|
+
5. **Wait**: Do NOT proceed to test execution until steps 1–4 are complete and the verification checklist in the \`skyramp_enhance_assertions\` tool result has been validated for EVERY generated test file.
|
|
580
|
+
Do not make any changes other than the code-reuse refactoring (step 3) and the assertion enhancements described above. For example: do not modify auth headers, cookies, tokens, env vars, or imports that the generation tool already set correctly — those are correct by construction and changing them breaks auth or execution.
|
|
551
581
|
|
|
552
582
|
**Final execution (mandatory):** Do NOT call \`skyramp_execute_test\` until ALL maintenance edits AND ALL new test generation/enhancement are complete. Run these calls sequentially, not in parallel. Exclude tests marked \`[external]\`.
|
|
553
583
|
- Only report test results for files you actually ran.
|
|
554
|
-
**Auth**: If \`skyramp_analyze_changes\` reports an auth token or \`SKYRAMP_TEST_TOKEN\` is set, pass it in **every** \`skyramp_execute_test\` call from the first attempt — do NOT wait for a 401/403 to discover auth is needed
|
|
584
|
+
**Auth**: If \`skyramp_analyze_changes\` reports an auth token or \`SKYRAMP_TEST_TOKEN\` is set, pass it in **every** \`skyramp_execute_test\` call from the first attempt — do NOT wait for a 401/403 to discover auth is needed.`;
|
|
585
|
+
}
|
|
586
|
+
const primaryRepoBlock = primaryRepo ? `<REPOSITORY>${primaryRepo}</REPOSITORY>\n` : '';
|
|
587
|
+
return `<TITLE>${prTitle}</TITLE>
|
|
588
|
+
<DESCRIPTION>${prDescription}</DESCRIPTION>
|
|
589
|
+
${primaryRepoBlock}<REPOSITORY PATH>${repositoryPath}</REPOSITORY PATH>
|
|
590
|
+
${relatedReposBlock}${testsRepoDirBlock}${serviceContext ? serviceContext + '\n' : ''}${uiCredentialsBlock ? uiCredentialsBlock + '\n' : ''}Use the Skyramp MCP server tools for all tasks below.
|
|
591
|
+
|
|
592
|
+
${task1Section}
|
|
593
|
+
|
|
594
|
+
${task2Section}
|
|
555
595
|
|
|
556
596
|
## Task 3: Submit Report
|
|
557
597
|
|
|
@@ -571,7 +611,7 @@ In these cases:
|
|
|
571
611
|
- \`businessCaseAnalysis\` must be a one-sentence summary of what the PR actually does (do NOT leave it blank)
|
|
572
612
|
- \`additionalRecommendations\` must be \`[]\` — do NOT recommend tests for a no-surface PR
|
|
573
613
|
|
|
574
|
-
|
|
614
|
+
${task3CountRule}
|
|
575
615
|
|
|
576
616
|
Call \`skyramp_submit_report\` with \`summaryOutputFile\`: "${summaryOutputFile}" and \`stateFile\` (from \`skyramp_analyze_changes\` output) — the stateFile is required for execution outcome tracking. Field names, types, and formats are defined in the tool's parameter schema — follow them exactly.
|
|
577
617
|
|
|
@@ -683,10 +723,16 @@ export function registerTestbotPrompt(server) {
|
|
|
683
723
|
.string()
|
|
684
724
|
.optional()
|
|
685
725
|
.describe("The primary repository's owner/repo slug (where the testbot workflow runs). Used verbatim as the `repository` attribution for the primary repo's analysis, tests, and report items in multi-repo runs."),
|
|
726
|
+
planOnly: z
|
|
727
|
+
// Prompt/resource URL args arrive as strings — accept "true"/"1"
|
|
728
|
+
// alongside a real boolean; anything else (incl. undefined) is false.
|
|
729
|
+
.preprocess((v) => v === true || v === "true" || v === "1", z.boolean())
|
|
730
|
+
.optional()
|
|
731
|
+
.describe("Plan-only eval mode (eval harness only, SKYR-3879): run the full recommendation phase — analysis, maintenance verdicts, candidate drafting, plan registration — but generate and execute nothing; the report declares the final plan. Used by the eval pipeline to A/B selection changes without a running SUT."),
|
|
686
732
|
},
|
|
687
733
|
}, async (args) => {
|
|
688
734
|
const services = await readWorkspaceServices(args.repositoryPath);
|
|
689
|
-
let prompt = getTestbotPrompt(args.prTitle, args.prDescription, args.summaryOutputFile, args.repositoryPath, args.baseBranch, args.maxRecommendations, args.maxGenerate, args.maxCritical, args.prNumber, args.userPrompt, services.length ? services : undefined, args.uiCredentials, args.testsRepoDir, parseRelatedRepositories(args.relatedRepositories), args.primaryRepo);
|
|
735
|
+
let prompt = getTestbotPrompt(args.prTitle, args.prDescription, args.summaryOutputFile, args.repositoryPath, args.baseBranch, args.maxRecommendations, args.maxGenerate, args.maxCritical, args.prNumber, args.userPrompt, services.length ? services : undefined, args.uiCredentials, args.testsRepoDir, parseRelatedRepositories(args.relatedRepositories), args.primaryRepo, args.planOnly);
|
|
690
736
|
if (args.workspaceValidationFailed) {
|
|
691
737
|
prompt = buildWorkspaceRecoveryPrefix(args.repositoryPath) + prompt;
|
|
692
738
|
}
|
|
@@ -378,3 +378,47 @@ describe("parseRelatedRepositories", () => {
|
|
|
378
378
|
expect(parsed).toEqual([{ repo: "o/ok", repositoryPath: "/ok", baseBranch: undefined }]);
|
|
379
379
|
});
|
|
380
380
|
});
|
|
381
|
+
describe("plan-only eval mode (via getTestbotPrompt)", () => {
|
|
382
|
+
function callWithPlanOnly(planOnly) {
|
|
383
|
+
return getTestbotPrompt(baseArgs.prTitle, baseArgs.prDescription, baseArgs.summaryOutputFile, baseArgs.repositoryPath, undefined, // baseBranch
|
|
384
|
+
undefined, // maxRecommendations
|
|
385
|
+
undefined, // maxGenerate
|
|
386
|
+
undefined, // maxCritical
|
|
387
|
+
undefined, // prNumber
|
|
388
|
+
undefined, // userPrompt
|
|
389
|
+
undefined, // services
|
|
390
|
+
undefined, // uiCredentials
|
|
391
|
+
undefined, // testsRepoDir
|
|
392
|
+
undefined, // relatedRepositories
|
|
393
|
+
undefined, // primaryRepo
|
|
394
|
+
planOnly);
|
|
395
|
+
}
|
|
396
|
+
it("branches Task 2 to the plan-only variant when planOnly is true", () => {
|
|
397
|
+
const prompt = callWithPlanOnly(true);
|
|
398
|
+
expect(prompt).toContain("## Task 2: Commit the Test Plan (plan-only run)");
|
|
399
|
+
expect(prompt).not.toContain("## Task 2: Generate New Tests");
|
|
400
|
+
// The standard generation/execution mandates must be absent, not overridden.
|
|
401
|
+
expect(prompt).not.toContain("skyramp_integration_test_generation");
|
|
402
|
+
expect(prompt).not.toContain('phase: "before"');
|
|
403
|
+
// Task 3 declares the plan instead of counting generated files.
|
|
404
|
+
expect(prompt).toContain("DECLARES your final GENERATE selection");
|
|
405
|
+
});
|
|
406
|
+
it("renders the standard prompt untouched when planOnly is false or omitted", () => {
|
|
407
|
+
const prompt = callWithPlanOnly(false);
|
|
408
|
+
expect(prompt).toContain("## Task 2: Generate New Tests");
|
|
409
|
+
expect(prompt).not.toContain("plan-only");
|
|
410
|
+
expect(prompt).toBe(callWithServices([]));
|
|
411
|
+
});
|
|
412
|
+
});
|
|
413
|
+
describe("UI code reuse wiring", () => {
|
|
414
|
+
it("instructs codeReuse: true for TS/JS Playwright UI generation", () => {
|
|
415
|
+
const p = callWithServices([]);
|
|
416
|
+
expect(p).toContain("codeReuse: true");
|
|
417
|
+
});
|
|
418
|
+
it("teaches both reuse outcomes and the raw.bak insurance rule", () => {
|
|
419
|
+
const p = callWithServices([]);
|
|
420
|
+
expect(p).toContain("No reusable POM layer detected");
|
|
421
|
+
expect(p).toContain("verification: passed");
|
|
422
|
+
expect(p).toContain(".raw.bak");
|
|
423
|
+
});
|
|
424
|
+
});
|
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
import { Budgeter } from "../../types/Recommendation.js";
|
|
2
|
+
/**
|
|
3
|
+
* Budgeter that guarantees cross-test-type coverage in the GENERATE set.
|
|
4
|
+
* `maxGenerate` acts as an upper cap (not a hard N). Fixes the single-type skew
|
|
5
|
+
* that starves contract/error-path tests. Opt-in via strategy selection.
|
|
6
|
+
*/
|
|
7
|
+
export declare const diversityBalancedBudgeter: Budgeter;
|
|
@@ -0,0 +1,71 @@
|
|
|
1
|
+
import { runBudget } from "./shared.js";
|
|
2
|
+
import { bucketByType, inferScenarioType, isProtectedCandidate, roundRobinFill, } from "../diversity.js";
|
|
3
|
+
const typeOf = (c) => inferScenarioType(c.scenario);
|
|
4
|
+
/**
|
|
5
|
+
* Pick `count` items guaranteeing each present test type at least one slot.
|
|
6
|
+
*
|
|
7
|
+
* Unlike roundRobinByType (which fills ALL protected items first and can
|
|
8
|
+
* starve other types at small N — the SKYR-3879 "0 contract executed" skew),
|
|
9
|
+
* this caps protected occupancy to `count - (numTypes - 1)` so at least one slot
|
|
10
|
+
* per remaining type is always reachable, then:
|
|
11
|
+
* Phase 1 — one item for each still-uncovered present type (by rank), then
|
|
12
|
+
* Phase 2 — round-robin the remainder across types until `count` is reached.
|
|
13
|
+
* Rank order is preserved within each type; degenerate cases match a rank slice.
|
|
14
|
+
*/
|
|
15
|
+
function floorBalancedPick(items, count) {
|
|
16
|
+
if (count <= 0)
|
|
17
|
+
return [];
|
|
18
|
+
if (count >= items.length)
|
|
19
|
+
return items.slice(0, count);
|
|
20
|
+
// Present types in first-appearance (rank) order.
|
|
21
|
+
const typesOrder = [];
|
|
22
|
+
const seen = new Set();
|
|
23
|
+
for (const it of items) {
|
|
24
|
+
const t = typeOf(it);
|
|
25
|
+
if (!seen.has(t)) {
|
|
26
|
+
seen.add(t);
|
|
27
|
+
typesOrder.push(t);
|
|
28
|
+
}
|
|
29
|
+
}
|
|
30
|
+
const maxProtected = Math.max(0, count - (typesOrder.length - 1));
|
|
31
|
+
const selected = [];
|
|
32
|
+
const pool = [];
|
|
33
|
+
let protectedTaken = 0;
|
|
34
|
+
for (const it of items) {
|
|
35
|
+
if (isProtectedCandidate(it.priority, it.scenario) &&
|
|
36
|
+
protectedTaken < maxProtected &&
|
|
37
|
+
selected.length < count) {
|
|
38
|
+
selected.push(it);
|
|
39
|
+
protectedTaken++;
|
|
40
|
+
}
|
|
41
|
+
else {
|
|
42
|
+
pool.push(it);
|
|
43
|
+
}
|
|
44
|
+
}
|
|
45
|
+
const { buckets } = bucketByType(pool, typeOf);
|
|
46
|
+
const covered = new Set(selected.map(typeOf));
|
|
47
|
+
// Phase 1 — floor: one item per still-uncovered present type.
|
|
48
|
+
for (const t of typesOrder) {
|
|
49
|
+
if (selected.length >= count)
|
|
50
|
+
break;
|
|
51
|
+
if (covered.has(t))
|
|
52
|
+
continue;
|
|
53
|
+
const bucket = buckets.get(t);
|
|
54
|
+
if (bucket && bucket.length > 0) {
|
|
55
|
+
selected.push(bucket.shift());
|
|
56
|
+
covered.add(t);
|
|
57
|
+
}
|
|
58
|
+
}
|
|
59
|
+
// Phase 2 — round-robin the remainder across types until full.
|
|
60
|
+
roundRobinFill(selected, typesOrder, buckets, count);
|
|
61
|
+
return selected;
|
|
62
|
+
}
|
|
63
|
+
/**
|
|
64
|
+
* Budgeter that guarantees cross-test-type coverage in the GENERATE set.
|
|
65
|
+
* `maxGenerate` acts as an upper cap (not a hard N). Fixes the single-type skew
|
|
66
|
+
* that starves contract/error-path tests. Opt-in via strategy selection.
|
|
67
|
+
*/
|
|
68
|
+
export const diversityBalancedBudgeter = {
|
|
69
|
+
name: "diversity-balanced",
|
|
70
|
+
select: (ranked, ctx) => runBudget(ranked, ctx, floorBalancedPick),
|
|
71
|
+
};
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export {};
|
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
import { diversityBalancedBudgeter } from "./diversityBalancedBudgeter.js";
|
|
2
|
+
import { PriorityTier } from "../../types/TestRecommendation.js";
|
|
3
|
+
import { fixedNBudgeter } from "./fixedNBudgeter.js";
|
|
4
|
+
import { mkCandidate } from "../testFixtures.js";
|
|
5
|
+
function ctx(over = {}) {
|
|
6
|
+
return {
|
|
7
|
+
maxGenerate: 3,
|
|
8
|
+
maxTotal: 20,
|
|
9
|
+
isUIOnlyPR: false,
|
|
10
|
+
hasFrontendChanges: false,
|
|
11
|
+
externalCoverage: new Set(),
|
|
12
|
+
...over,
|
|
13
|
+
};
|
|
14
|
+
}
|
|
15
|
+
const countType = (r, t) => r.generate.filter((c) => c.scenario.testType === t).length;
|
|
16
|
+
const typesIn = (r) => new Set(r.generate.map((c) => c.scenario.testType));
|
|
17
|
+
describe("diversityBalancedBudgeter", () => {
|
|
18
|
+
it("guarantees a contract slot where fixed-n starves it (the SKYR-3879 fix)", () => {
|
|
19
|
+
// Protected (CRITICAL) integration scenarios out-rank contract ones and, at N=3,
|
|
20
|
+
// fixed-n's protected-first fill consumes every slot with integration → 0 contract.
|
|
21
|
+
const ranked = [
|
|
22
|
+
mkCandidate("i1", { testType: "integration", priority: PriorityTier.CRITICAL }),
|
|
23
|
+
mkCandidate("i2", { testType: "integration", priority: PriorityTier.CRITICAL }),
|
|
24
|
+
mkCandidate("i3", { testType: "integration", priority: PriorityTier.CRITICAL }),
|
|
25
|
+
mkCandidate("c1", { testType: "contract" }),
|
|
26
|
+
mkCandidate("c2", { testType: "contract" }),
|
|
27
|
+
];
|
|
28
|
+
const c = ctx({ maxGenerate: 3 });
|
|
29
|
+
// Baseline: fixed-n reproduces the skew.
|
|
30
|
+
expect(countType(fixedNBudgeter.select(ranked, c), "contract")).toBe(0);
|
|
31
|
+
// Fix: diversity-balanced reserves a per-type floor.
|
|
32
|
+
const div = diversityBalancedBudgeter.select(ranked, c);
|
|
33
|
+
expect(div.generate.length).toBe(3);
|
|
34
|
+
expect(countType(div, "contract")).toBeGreaterThanOrEqual(1);
|
|
35
|
+
expect(countType(div, "integration")).toBeGreaterThanOrEqual(1);
|
|
36
|
+
});
|
|
37
|
+
it("gives every present test type at least one slot before any type gets a second", () => {
|
|
38
|
+
const ranked = [
|
|
39
|
+
mkCandidate("i1", { testType: "integration" }),
|
|
40
|
+
mkCandidate("i2", { testType: "integration" }),
|
|
41
|
+
mkCandidate("i3", { testType: "integration" }),
|
|
42
|
+
mkCandidate("c1", { testType: "contract" }),
|
|
43
|
+
mkCandidate("c2", { testType: "contract" }),
|
|
44
|
+
mkCandidate("u1", { testType: "ui" }),
|
|
45
|
+
];
|
|
46
|
+
const div = diversityBalancedBudgeter.select(ranked, ctx({ maxGenerate: 3 }));
|
|
47
|
+
expect(typesIn(div)).toEqual(new Set(["integration", "contract", "ui"]));
|
|
48
|
+
});
|
|
49
|
+
it("raising maxGenerate increases per-type coverage while keeping the floor", () => {
|
|
50
|
+
const ranked = [
|
|
51
|
+
mkCandidate("i1", { testType: "integration", priority: PriorityTier.CRITICAL }),
|
|
52
|
+
mkCandidate("i2", { testType: "integration", priority: PriorityTier.CRITICAL }),
|
|
53
|
+
mkCandidate("i3", { testType: "integration", priority: PriorityTier.CRITICAL }),
|
|
54
|
+
mkCandidate("i4", { testType: "integration", priority: PriorityTier.CRITICAL }),
|
|
55
|
+
mkCandidate("i5", { testType: "integration", priority: PriorityTier.CRITICAL }),
|
|
56
|
+
mkCandidate("c1", { testType: "contract" }),
|
|
57
|
+
mkCandidate("c2", { testType: "contract" }),
|
|
58
|
+
mkCandidate("u1", { testType: "ui" }),
|
|
59
|
+
];
|
|
60
|
+
const atThree = diversityBalancedBudgeter.select(ranked, ctx({ maxGenerate: 3 }));
|
|
61
|
+
const atSix = diversityBalancedBudgeter.select(ranked, ctx({ maxGenerate: 6 }));
|
|
62
|
+
expect(countType(atThree, "contract")).toBeGreaterThanOrEqual(1);
|
|
63
|
+
expect(countType(atThree, "ui")).toBeGreaterThanOrEqual(1);
|
|
64
|
+
expect(countType(atSix, "contract")).toBeGreaterThanOrEqual(1);
|
|
65
|
+
expect(countType(atSix, "ui")).toBeGreaterThanOrEqual(1);
|
|
66
|
+
// more budget → more of the highest-ranked (integration) coverage
|
|
67
|
+
expect(countType(atSix, "integration")).toBeGreaterThan(countType(atThree, "integration"));
|
|
68
|
+
expect(atSix.generate.length).toBe(6);
|
|
69
|
+
});
|
|
70
|
+
it("returns everything (rank order) when maxGenerate >= candidate count", () => {
|
|
71
|
+
const ranked = [mkCandidate("i1", { testType: "integration" }), mkCandidate("c1", { testType: "contract" })];
|
|
72
|
+
const div = diversityBalancedBudgeter.select(ranked, ctx({ maxGenerate: 5 }));
|
|
73
|
+
expect(div.generate.map((c) => c.scenario.scenarioName)).toEqual(["i1", "c1"]);
|
|
74
|
+
});
|
|
75
|
+
});
|
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
import { Budgeter } from "../../types/Recommendation.js";
|
|
2
|
+
/**
|
|
3
|
+
* Default budgeter — reproduces the pre-refactor behavior exactly: fill up to
|
|
4
|
+
* `maxGenerate` GENERATE slots by round-robin across the present test types
|
|
5
|
+
* (protected items first), the rest become ADDITIONAL.
|
|
6
|
+
*/
|
|
7
|
+
export declare const fixedNBudgeter: Budgeter;
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
import { roundRobinByType } from "../diversity.js";
|
|
2
|
+
import { runBudget } from "./shared.js";
|
|
3
|
+
/**
|
|
4
|
+
* Default budgeter — reproduces the pre-refactor behavior exactly: fill up to
|
|
5
|
+
* `maxGenerate` GENERATE slots by round-robin across the present test types
|
|
6
|
+
* (protected items first), the rest become ADDITIONAL.
|
|
7
|
+
*/
|
|
8
|
+
export const fixedNBudgeter = {
|
|
9
|
+
name: "fixed-n",
|
|
10
|
+
select: (ranked, ctx) => runBudget(ranked, ctx, roundRobinByType),
|
|
11
|
+
};
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export {};
|
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
import { fixedNBudgeter } from "./fixedNBudgeter.js";
|
|
2
|
+
import { mkCandidate } from "../testFixtures.js";
|
|
3
|
+
import { externalDedupKey } from "../../prompts/test-recommendation/recommendationShared.js";
|
|
4
|
+
function ctx(over = {}) {
|
|
5
|
+
return {
|
|
6
|
+
maxGenerate: 3,
|
|
7
|
+
maxTotal: 20,
|
|
8
|
+
isUIOnlyPR: false,
|
|
9
|
+
hasFrontendChanges: false,
|
|
10
|
+
externalCoverage: new Set(),
|
|
11
|
+
...over,
|
|
12
|
+
};
|
|
13
|
+
}
|
|
14
|
+
const gen = (r) => r.generate.map((c) => c.scenario.scenarioName);
|
|
15
|
+
describe("fixedNBudgeter", () => {
|
|
16
|
+
it("backend-only PR fills maxGenerate slots via round-robin across test types", () => {
|
|
17
|
+
const ranked = [
|
|
18
|
+
mkCandidate("i1", { testType: "integration" }),
|
|
19
|
+
mkCandidate("i2", { testType: "integration" }),
|
|
20
|
+
mkCandidate("i3", { testType: "integration" }),
|
|
21
|
+
mkCandidate("c1", { testType: "contract" }),
|
|
22
|
+
mkCandidate("c2", { testType: "contract" }),
|
|
23
|
+
];
|
|
24
|
+
const res = fixedNBudgeter.select(ranked, ctx());
|
|
25
|
+
expect(res.generate.length).toBe(3);
|
|
26
|
+
expect(res.reservedUISlots).toBe(0);
|
|
27
|
+
// buckets [integration, contract]; round 1 → i1, c1; round 2 → i2
|
|
28
|
+
expect(gen(res)).toEqual(["i1", "c1", "i2"]);
|
|
29
|
+
});
|
|
30
|
+
it("mixed (frontend) PR reserves one UI slot → backend count = maxGenerate - 1", () => {
|
|
31
|
+
const ranked = [
|
|
32
|
+
mkCandidate("i1", { testType: "integration" }),
|
|
33
|
+
mkCandidate("i2", { testType: "integration" }),
|
|
34
|
+
mkCandidate("c1", { testType: "contract" }),
|
|
35
|
+
];
|
|
36
|
+
const res = fixedNBudgeter.select(ranked, ctx({ hasFrontendChanges: true }));
|
|
37
|
+
expect(res.generate.length).toBe(2);
|
|
38
|
+
expect(res.reservedUISlots).toBe(1);
|
|
39
|
+
});
|
|
40
|
+
it("UI-only PR selects no backend generate items", () => {
|
|
41
|
+
const res = fixedNBudgeter.select([mkCandidate("i1", { testType: "integration" })], ctx({ isUIOnlyPR: true }));
|
|
42
|
+
expect(res.generate.length).toBe(0);
|
|
43
|
+
});
|
|
44
|
+
it("additional = leftover candidates (rank order), excluding anything generated", () => {
|
|
45
|
+
const ranked = [
|
|
46
|
+
mkCandidate("i1", { testType: "integration" }),
|
|
47
|
+
mkCandidate("c1", { testType: "contract" }),
|
|
48
|
+
mkCandidate("i2", { testType: "integration" }),
|
|
49
|
+
mkCandidate("c2", { testType: "contract" }),
|
|
50
|
+
];
|
|
51
|
+
const res = fixedNBudgeter.select(ranked, ctx());
|
|
52
|
+
const generated = new Set(gen(res));
|
|
53
|
+
for (const a of res.additional) {
|
|
54
|
+
expect(generated.has(a.scenario.scenarioName)).toBe(false);
|
|
55
|
+
}
|
|
56
|
+
expect(res.additional.length).toBe(1);
|
|
57
|
+
});
|
|
58
|
+
it("drops an ordinary candidate already covered by an external test", () => {
|
|
59
|
+
const covered = mkCandidate("covered", { testType: "contract" });
|
|
60
|
+
const fresh = mkCandidate("fresh", { testType: "integration" });
|
|
61
|
+
const res = fixedNBudgeter.select([covered, fresh], ctx({ maxGenerate: 5, externalCoverage: new Set([externalDedupKey(covered.scenario)]) }));
|
|
62
|
+
const all = [...res.generate, ...res.additional].map((c) => c.scenario.scenarioName);
|
|
63
|
+
expect(all).toContain("fresh");
|
|
64
|
+
expect(all).not.toContain("covered");
|
|
65
|
+
});
|
|
66
|
+
});
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
import { Candidate, BudgetContext, SelectionResult } from "../../types/Recommendation.js";
|
|
2
|
+
/**
|
|
3
|
+
* Backend GENERATE slot count:
|
|
4
|
+
* - UI-only PR: 0 (all slots are UI placeholders)
|
|
5
|
+
* - Mixed PR: maxGenerate - 1 (last slot reserved for a UI placeholder)
|
|
6
|
+
* - Backend-only PR: maxGenerate
|
|
7
|
+
*/
|
|
8
|
+
export declare function backendGenerateCount(ctx: BudgetContext): number;
|
|
9
|
+
/** UI placeholder slots the render layer fills (UI-only → all; mixed → one). */
|
|
10
|
+
export declare function reservedUISlots(ctx: BudgetContext): number;
|
|
11
|
+
/**
|
|
12
|
+
* Shared budgeting pipeline. All Budgeters run the same external-dedup,
|
|
13
|
+
* attack-surface prioritization, and ADDITIONAL set-difference; they differ
|
|
14
|
+
* ONLY in how they pick the GENERATE items from the slot-ordered list (`pick`).
|
|
15
|
+
*
|
|
16
|
+
* With `pick = roundRobinByType` this reproduces the pre-refactor selection in
|
|
17
|
+
* diffExecutionPlan.ts exactly.
|
|
18
|
+
*/
|
|
19
|
+
export declare function runBudget(ranked: Candidate[], ctx: BudgetContext, pick: (items: Candidate[], count: number) => Candidate[]): SelectionResult;
|
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
import { prioritizeAttackSurfaceBundles } from "../diversity.js";
|
|
2
|
+
import { externalDedupKey, scenarioCoverageKey, isAttackSurfaceSecurityBoundary, } from "../../prompts/test-recommendation/recommendationShared.js";
|
|
3
|
+
import { logger } from "../../utils/logger.js";
|
|
4
|
+
/**
|
|
5
|
+
* Backend GENERATE slot count:
|
|
6
|
+
* - UI-only PR: 0 (all slots are UI placeholders)
|
|
7
|
+
* - Mixed PR: maxGenerate - 1 (last slot reserved for a UI placeholder)
|
|
8
|
+
* - Backend-only PR: maxGenerate
|
|
9
|
+
*/
|
|
10
|
+
export function backendGenerateCount(ctx) {
|
|
11
|
+
if (ctx.isUIOnlyPR)
|
|
12
|
+
return 0;
|
|
13
|
+
return ctx.hasFrontendChanges ? Math.max(0, ctx.maxGenerate - 1) : ctx.maxGenerate;
|
|
14
|
+
}
|
|
15
|
+
/** UI placeholder slots the render layer fills (UI-only → all; mixed → one). */
|
|
16
|
+
export function reservedUISlots(ctx) {
|
|
17
|
+
if (ctx.isUIOnlyPR)
|
|
18
|
+
return ctx.maxGenerate;
|
|
19
|
+
return ctx.hasFrontendChanges && ctx.maxGenerate > 0 ? 1 : 0;
|
|
20
|
+
}
|
|
21
|
+
/**
|
|
22
|
+
* Drop candidates whose method-aware resource+type is already covered by an
|
|
23
|
+
* external test — except protected `bug_caught` / attack-surface scenarios,
|
|
24
|
+
* which require semantic flaw coverage the external test may not provide.
|
|
25
|
+
*/
|
|
26
|
+
function applyExternalDedup(ranked, externalCoverage) {
|
|
27
|
+
if (externalCoverage.size === 0)
|
|
28
|
+
return ranked;
|
|
29
|
+
return ranked.filter((item) => {
|
|
30
|
+
const key = externalDedupKey(item.scenario);
|
|
31
|
+
if (externalCoverage.has(key)) {
|
|
32
|
+
if (item.scenario.category === "bug_caught" || isAttackSurfaceSecurityBoundary(item.scenario)) {
|
|
33
|
+
logger.info(`External dedup: preserving "${item.scenario.scenarioName}" (${key}) — protected bug/attack-surface scenario requires semantic flaw coverage`);
|
|
34
|
+
return true;
|
|
35
|
+
}
|
|
36
|
+
logger.info(`External dedup: skipping "${item.scenario.scenarioName}" (${key}) — covered by external test`);
|
|
37
|
+
return false;
|
|
38
|
+
}
|
|
39
|
+
return true;
|
|
40
|
+
});
|
|
41
|
+
}
|
|
42
|
+
/**
|
|
43
|
+
* Shared budgeting pipeline. All Budgeters run the same external-dedup,
|
|
44
|
+
* attack-surface prioritization, and ADDITIONAL set-difference; they differ
|
|
45
|
+
* ONLY in how they pick the GENERATE items from the slot-ordered list (`pick`).
|
|
46
|
+
*
|
|
47
|
+
* With `pick = roundRobinByType` this reproduces the pre-refactor selection in
|
|
48
|
+
* diffExecutionPlan.ts exactly.
|
|
49
|
+
*/
|
|
50
|
+
export function runBudget(ranked, ctx, pick) {
|
|
51
|
+
const backend = backendGenerateCount(ctx);
|
|
52
|
+
const deduped = applyExternalDedup(ranked, ctx.externalCoverage);
|
|
53
|
+
const slotOrdered = prioritizeAttackSurfaceBundles(deduped);
|
|
54
|
+
const generate = pick(slotOrdered, Math.min(backend, slotOrdered.length));
|
|
55
|
+
// ADDITIONAL = everything not chosen for GENERATE, in rank order, capped at the
|
|
56
|
+
// remaining budget, minus any whose resource+type is already covered by GENERATE.
|
|
57
|
+
const generateSet = new Set(generate);
|
|
58
|
+
const generatedCoverage = new Set(generate.map((item) => scenarioCoverageKey(item.scenario)));
|
|
59
|
+
const additional = slotOrdered
|
|
60
|
+
.filter((it) => !generateSet.has(it))
|
|
61
|
+
.slice(0, Math.max(0, ctx.maxTotal - backend))
|
|
62
|
+
.filter((item) => !generatedCoverage.has(scenarioCoverageKey(item.scenario)));
|
|
63
|
+
// Budgeting itself produces no demotions; the register-plan selection stage
|
|
64
|
+
// (planRanker.selectPlan) fills this channel from discriminator verification.
|
|
65
|
+
return { generate, additional, reservedUISlots: reservedUISlots(ctx), demotions: [] };
|
|
66
|
+
}
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
import { DraftedScenario } from "../types/RepositoryAnalysis.js";
|
|
2
|
+
import { DiscriminatorKind } from "../types/Recommendation.js";
|
|
3
|
+
/**
|
|
4
|
+
* A discriminator claim the agent (or server) attaches to a candidate: the
|
|
5
|
+
* structural bug-shape (`kind`) plus the verbatim diff snippet the test probes.
|
|
6
|
+
*/
|
|
7
|
+
export interface DiscriminatorClaim {
|
|
8
|
+
kind: DiscriminatorKind;
|
|
9
|
+
changedCodeAnchor: string;
|
|
10
|
+
}
|
|
11
|
+
/** Result of verifying a claim against a scenario's steps and the PR diff. */
|
|
12
|
+
export interface DiscriminatorValidation {
|
|
13
|
+
verified: boolean;
|
|
14
|
+
/** One-sentence, human-readable reason present ONLY when `verified` is false;
|
|
15
|
+
* it is returned to the LLM to help it fix the claim. */
|
|
16
|
+
reason?: string;
|
|
17
|
+
}
|
|
18
|
+
/**
|
|
19
|
+
* Verify a declared discriminator against the candidate's `steps[]` and the PR
|
|
20
|
+
* diff. Mirrors the classification precedent (`recommendationShared.ts`): the
|
|
21
|
+
* claim is checked STRUCTURALLY, never by fuzzy keyword scans of prose. A failed
|
|
22
|
+
* check returns `verified: false` with a one-sentence reason (the claim is
|
|
23
|
+
* demoted, never used to reject the candidate). NEVER throws — malformed or
|
|
24
|
+
* partial input is treated as unverified.
|
|
25
|
+
*
|
|
26
|
+
* Two gates, both required to verify:
|
|
27
|
+
* 1. `changedCodeAnchor` must be a non-trivial string (>= 8 chars after trim)
|
|
28
|
+
* occurring verbatim in `diffText` — grounds the claim in the real change.
|
|
29
|
+
* 2. The `kind`-specific structural predicate must hold over `steps[]`.
|
|
30
|
+
*/
|
|
31
|
+
export declare function validateDiscriminator(scenario: DraftedScenario, declared: DiscriminatorClaim, diffText: string): DiscriminatorValidation;
|