@skyramp/mcp 0.3.1 → 0.3.2-rc.pom-2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/build/index.js +2 -1
- package/build/prompts/code-reuse.d.ts +7 -1
- package/build/prompts/code-reuse.js +8 -5
- package/build/prompts/code-reuse.test.d.ts +1 -0
- package/build/prompts/code-reuse.test.js +62 -0
- package/build/prompts/pom-aware-code-reuse.d.ts +6 -1
- package/build/prompts/pom-aware-code-reuse.js +100 -53
- package/build/prompts/pom-aware-code-reuse.test.d.ts +1 -0
- package/build/prompts/pom-aware-code-reuse.test.js +11 -0
- package/build/prompts/test-recommendation/diffExecutionPlan.d.ts +4 -2
- package/build/prompts/test-recommendation/diffExecutionPlan.js +11 -65
- package/build/prompts/test-recommendation/fullRepoCatalog.d.ts +2 -2
- package/build/prompts/test-recommendation/recommendationSections.js +5 -2
- package/build/prompts/test-recommendation/scopeAssessment.js +1 -1
- package/build/prompts/test-recommendation/test-recommendation-prompt.d.ts +26 -1
- package/build/prompts/test-recommendation/test-recommendation-prompt.js +68 -56
- package/build/prompts/test-recommendation/test-recommendation-prompt.test.js +48 -11
- package/build/prompts/testbot/testbot-prompts.d.ts +1 -1
- package/build/prompts/testbot/testbot-prompts.js +67 -21
- package/build/prompts/testbot/testbot-prompts.test.js +44 -0
- package/build/recommendation/budgeters/diversityBalancedBudgeter.d.ts +7 -0
- package/build/recommendation/budgeters/diversityBalancedBudgeter.js +71 -0
- package/build/recommendation/budgeters/diversityBalancedBudgeter.test.d.ts +1 -0
- package/build/recommendation/budgeters/diversityBalancedBudgeter.test.js +75 -0
- package/build/recommendation/budgeters/fixedNBudgeter.d.ts +7 -0
- package/build/recommendation/budgeters/fixedNBudgeter.js +11 -0
- package/build/recommendation/budgeters/fixedNBudgeter.test.d.ts +1 -0
- package/build/recommendation/budgeters/fixedNBudgeter.test.js +66 -0
- package/build/recommendation/budgeters/shared.d.ts +19 -0
- package/build/recommendation/budgeters/shared.js +66 -0
- package/build/recommendation/discriminators.d.ts +31 -0
- package/build/recommendation/discriminators.js +355 -0
- package/build/recommendation/discriminators.test.d.ts +1 -0
- package/build/recommendation/discriminators.test.js +324 -0
- package/build/recommendation/diversity.d.ts +47 -0
- package/build/recommendation/diversity.js +101 -0
- package/build/recommendation/diversity.test.d.ts +1 -0
- package/build/recommendation/diversity.test.js +77 -0
- package/build/recommendation/planRanker.d.ts +50 -0
- package/build/recommendation/planRanker.js +67 -0
- package/build/recommendation/planRanker.test.d.ts +1 -0
- package/build/recommendation/planRanker.test.js +110 -0
- package/build/recommendation/testFixtures.d.ts +25 -0
- package/build/recommendation/testFixtures.js +45 -0
- package/build/resources/testbotResource.js +4 -1
- package/build/services/ScenarioGenerationService.d.ts +5 -0
- package/build/services/ScenarioGenerationService.js +16 -1
- package/build/services/ScenarioGenerationService.test.js +44 -0
- package/build/services/TestExecutionService.d.ts +15 -1
- package/build/services/TestExecutionService.js +210 -55
- package/build/services/TestExecutionService.test.js +397 -0
- package/build/services/TestGenerationService.js +19 -1
- package/build/services/TestGenerationService.test.js +58 -0
- package/build/tool-phases.js +1 -0
- package/build/toolNames.d.ts +19 -0
- package/build/toolNames.js +19 -0
- package/build/tools/code-refactor/codeReuseTool.d.ts +7 -0
- package/build/tools/code-refactor/codeReuseTool.js +130 -4
- package/build/tools/code-refactor/codeReuseTool.test.d.ts +1 -0
- package/build/tools/code-refactor/codeReuseTool.test.js +290 -0
- package/build/tools/executeSkyrampTestTool.js +8 -2
- package/build/tools/generate-tests/generateBatchScenarioRestTool.d.ts +6 -1
- package/build/tools/generate-tests/generateBatchScenarioRestTool.js +110 -17
- package/build/tools/generate-tests/generateBatchScenarioRestTool.test.js +147 -0
- package/build/tools/generate-tests/generateContractRestTool.js +11 -1
- package/build/tools/generate-tests/generateIntegrationRestTool.js +24 -1
- package/build/tools/generate-tests/generateIntegrationRestTool.test.d.ts +1 -0
- package/build/tools/generate-tests/generateIntegrationRestTool.test.js +159 -0
- package/build/tools/generate-tests/planGuard.d.ts +13 -0
- package/build/tools/generate-tests/planGuard.js +78 -0
- package/build/tools/generate-tests/planGuard.test.d.ts +1 -0
- package/build/tools/generate-tests/planGuard.test.js +185 -0
- package/build/tools/generate-tests/scenarioFileIdentity.d.ts +10 -0
- package/build/tools/generate-tests/scenarioFileIdentity.js +46 -0
- package/build/tools/generate-tests/scenarioLint.d.ts +30 -0
- package/build/tools/generate-tests/scenarioLint.js +150 -0
- package/build/tools/generate-tests/scenarioLint.test.d.ts +1 -0
- package/build/tools/generate-tests/scenarioLint.test.js +100 -0
- package/build/tools/submitReportTool.js +78 -0
- package/build/tools/submitReportTool.test.js +255 -0
- package/build/tools/test-management/analyzeChangesTool.js +55 -2
- package/build/tools/test-management/analyzeChangesTool.test.js +12 -0
- package/build/tools/test-management/index.d.ts +1 -0
- package/build/tools/test-management/index.js +1 -0
- package/build/tools/test-management/registerTestPlanTool.d.ts +2 -0
- package/build/tools/test-management/registerTestPlanTool.js +329 -0
- package/build/tools/test-management/registerTestPlanTool.test.d.ts +1 -0
- package/build/tools/test-management/registerTestPlanTool.test.js +296 -0
- package/build/types/Recommendation.d.ts +97 -0
- package/build/types/Recommendation.js +48 -0
- package/build/types/RepositoryAnalysis.d.ts +14 -14
- package/build/types/TestExecution.d.ts +2 -0
- package/build/types/TestRecommendation.d.ts +12 -1
- package/build/types/TestRecommendation.js +26 -11
- package/build/types/TestTypes.js +1 -1
- package/build/utils/AnalysisStateManager.d.ts +47 -0
- package/build/utils/docker.test.js +1 -1
- package/build/utils/planMatchKeys.d.ts +61 -0
- package/build/utils/planMatchKeys.js +125 -0
- package/build/utils/pom-scope/import-expansion.d.ts +5 -0
- package/build/utils/pom-scope/import-expansion.js +32 -0
- package/build/utils/pom-scope/index.d.ts +39 -0
- package/build/utils/pom-scope/index.js +120 -0
- package/build/utils/pom-scope/index.test.d.ts +1 -0
- package/build/utils/pom-scope/index.test.js +239 -0
- package/build/utils/pom-scope/pom-files.d.ts +3 -0
- package/build/utils/pom-scope/pom-files.js +48 -0
- package/build/utils/pom-scope/pom-files.test.d.ts +1 -0
- package/build/utils/pom-scope/pom-files.test.js +29 -0
- package/build/utils/pom-scope/scoring.d.ts +20 -0
- package/build/utils/pom-scope/scoring.js +45 -0
- package/build/utils/pom-scope/scoring.test.d.ts +1 -0
- package/build/utils/pom-scope/scoring.test.js +39 -0
- package/build/utils/pom-scope/selector-extractor.d.ts +7 -0
- package/build/utils/pom-scope/selector-extractor.js +57 -0
- package/build/utils/pom-scope/selector-extractor.test.d.ts +1 -0
- package/build/utils/pom-scope/selector-extractor.test.js +67 -0
- package/build/utils/pom-verify/__fixtures__/af-style/asset-list.page.d.ts +5 -0
- package/build/utils/pom-verify/__fixtures__/af-style/asset-list.page.js +5 -0
- package/build/utils/pom-verify/__fixtures__/af-style/pageobjects/asset-list-page.d.ts +5 -0
- package/build/utils/pom-verify/__fixtures__/af-style/pageobjects/asset-list-page.js +9 -0
- package/build/utils/pom-verify/__fixtures__/af-style/report.iframe.page.d.ts +4 -0
- package/build/utils/pom-verify/__fixtures__/af-style/report.iframe.page.js +4 -0
- package/build/utils/pom-verify/__fixtures__/af-style/workflow-footer.page.d.ts +4 -0
- package/build/utils/pom-verify/__fixtures__/af-style/workflow-footer.page.js +4 -0
- package/build/utils/pom-verify/bindings.d.ts +19 -0
- package/build/utils/pom-verify/bindings.js +161 -0
- package/build/utils/pom-verify/bindings.test.d.ts +1 -0
- package/build/utils/pom-verify/bindings.test.js +164 -0
- package/build/utils/pom-verify/calls.d.ts +16 -0
- package/build/utils/pom-verify/calls.js +42 -0
- package/build/utils/pom-verify/calls.test.d.ts +1 -0
- package/build/utils/pom-verify/calls.test.js +61 -0
- package/build/utils/pom-verify/index.d.ts +4 -0
- package/build/utils/pom-verify/index.js +4 -0
- package/build/utils/pom-verify/resolve.d.ts +7 -0
- package/build/utils/pom-verify/resolve.js +27 -0
- package/build/utils/pom-verify/resolve.test.d.ts +1 -0
- package/build/utils/pom-verify/resolve.test.js +68 -0
- package/build/utils/pom-verify/strip.d.ts +9 -0
- package/build/utils/pom-verify/strip.js +89 -0
- package/build/utils/pom-verify/verify.d.ts +14 -0
- package/build/utils/pom-verify/verify.js +158 -0
- package/build/utils/pom-verify/verify.test.d.ts +1 -0
- package/build/utils/pom-verify/verify.test.js +325 -0
- package/build/utils/reportVerification.d.ts +61 -0
- package/build/utils/reportVerification.js +104 -0
- package/build/utils/reportVerification.test.d.ts +1 -0
- package/build/utils/reportVerification.test.js +185 -0
- package/build/utils/scenarioDrafting.js +5 -5
- package/build/utils/versions.d.ts +3 -3
- package/build/utils/versions.js +1 -1
- package/build/utils/workspaceAuth.d.ts +9 -1
- package/build/utils/workspaceAuth.js +25 -5
- package/build/utils/workspaceAuth.test.js +48 -0
- package/build/workspace/workspace.d.ts +20 -0
- package/build/workspace/workspace.js +4 -0
- package/build/workspace/workspace.test.js +10 -0
- package/node_modules/playwright/lib/mcp/skyramp/traceRecordingBackend.js +77 -8
- package/node_modules/playwright/node_modules/playwright-core/lib/generated/injectedScriptSource.js +1 -1
- package/node_modules/playwright/node_modules/playwright-core/lib/generated/pollingRecorderSource.js +1 -1
- package/node_modules/playwright/node_modules/playwright-core/lib/utils/isomorphic/volatileDate.js +101 -0
- package/node_modules/playwright/node_modules/playwright-core/lib/utils.js +2 -0
- package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/assets/{codeMirrorModule-aszq5EdG.js → codeMirrorModule-Bzd72-bG.js} +1 -1
- package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/assets/{defaultSettingsView-BxS7Jm4s.js → defaultSettingsView-DzxTioTK.js} +101 -101
- package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/{index.D4JTTy4R.js → index.BGc30U3S.js} +1 -1
- package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/index.html +2 -2
- package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/{uiMode.DaRMQKOI.js → uiMode.IaDrb29A.js} +1 -1
- package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/uiMode.html +2 -2
- package/node_modules/playwright/node_modules/playwright-core/package.json +1 -1
- package/node_modules/playwright/node_modules/playwright-core/src/generated/injectedScriptSource.ts +1 -1
- package/node_modules/playwright/node_modules/playwright-core/src/generated/pollingRecorderSource.ts +1 -1
- package/node_modules/playwright/node_modules/playwright-core/src/utils/isomorphic/volatileDate.ts +131 -0
- package/node_modules/playwright/node_modules/playwright-core/src/utils.ts +1 -0
- package/node_modules/playwright/package.json +1 -1
- package/package.json +3 -3
- package/node_modules/playwright/node_modules/playwright-core/.DS_Store +0 -0
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { roundRobinByType } from "../../recommendation/diversity.js";
|
|
1
2
|
import { AUTH_MIDDLEWARE_PATTERNS_STR } from "../../utils/workspaceAuth.js";
|
|
2
3
|
import { resolveServiceDetailsRef } from "../../utils/utils.js";
|
|
3
4
|
import { logger } from "../../utils/logger.js";
|
|
@@ -21,7 +22,7 @@ The highest-severity \`<bug_found>\` block from this code review triggers a mand
|
|
|
21
22
|
}
|
|
22
23
|
function _execCoverageBody(ctx) {
|
|
23
24
|
return `${ctx.externalTestFilesList}For every GENERATE item below, check its endpoint path and test type against the Existing Tests list (further down in the prompt).
|
|
24
|
-
- **\`[external]\` tests**: If the endpoint is already covered by an \`[external]\` test of the same type → skip the resource entirely (do NOT create or update),
|
|
25
|
+
- **\`[external]\` tests**: If the endpoint is already covered by an \`[external]\` test of the same type **that exercises the behavior this PR changes** → skip the resource entirely (do NOT create or update). Resource-level overlap alone is NOT coverage: when the PR tightens or adds a constraint (e.g. a max-items/max-length/enum bound), an external test that never crosses the new bound does not cover it — generate the boundary test. This matters most for single-resource APIs (e.g. one CRD served by the kube-apiserver), where resource-level dedup would permanently block all generation. \`bug_caught\` and attack-surface \`security_boundary\` items get the strictest reading: they may be skipped only if the external test directly asserts the same flaw/bypass and would fail/pass based on that same bug. Backfill from ADDITIONAL using the priority order below:
|
|
25
26
|
1. **BUG-CATCHING TESTS FIRST (CRITICAL)**: If source code analysis revealed a bug, logic error, or incorrect formula (e.g. discount math adding instead of subtracting, off-by-one errors, missing validation), CREATE A TEST THAT EXPOSES IT. The test SHOULD FAIL — that's the point. Document the bug. Example: if discount formula is wrong, test with discount=20% and assert correct math. If no bug found, skip to #2.
|
|
26
27
|
2. **PR-endpoint edge cases**: Look for integration test candidates covering error paths, boundary values, or alternative scenarios for the SAME endpoints changed in the PR diff. If no suitable candidate exists in ADDITIONAL, derive one from your source-code enrichment findings.
|
|
27
28
|
3. **Same-resource other scenarios**: Other HTTP methods or flows on the same resource group touched by the PR.
|
|
@@ -94,6 +95,11 @@ For each pair of GENERATE items, ask: same HTTP method + path + step sequence +
|
|
|
94
95
|
|
|
95
96
|
Same step sequence with only payload differences (e.g. 10% vs 5% discount both returning 200) = same code path = duplicate. Different scenario names do not make duplicate tests distinct.`;
|
|
96
97
|
}
|
|
98
|
+
function _execRegisterBody(_ctx) {
|
|
99
|
+
return `Register your complete candidate list — every test you would generate OR recommend — via \`skyramp_register_test_plan\` (\`stateFile\` required). Include a discriminator claim (\`discriminator\` field — valid kinds and anchor rules are in the tool schema) for candidates probing the changed logic identified in Step ${EXEC_STEP_CODE_REVIEW}/Step ${EXEC_STEP_ENRICH}.
|
|
100
|
+
|
|
101
|
+
The returned GENERATE list is mandatory and final — generation tools reject unregistered scenarios. If the tool demotes a discriminator claim (returned in \`demotions\` with a reason), either strengthen the claim — a step that actually exercises the declared \`kind\`, or a verbatim anchor that occurs in the diff — or drop it; the candidate itself stays in the plan either way.`;
|
|
102
|
+
}
|
|
97
103
|
function _execExecuteBody(ctx) {
|
|
98
104
|
return `Replace any scenario that pairs unrelated resources with one reflecting actual foreign-key relationships in the codebase.
|
|
99
105
|
Use the field names and values from the \`<source_evidence>\` blocks you quoted in Step ${EXEC_STEP_ENRICH} to fill all tool call parameters. Prefer reusing Step ${EXEC_STEP_ENRICH} evidence when it already resolves a placeholder, but if a placeholder cannot be replaced with concrete values from files already read, you may read the specific schema, model, or handler file needed to resolve it. Assert response field values, not just status codes.
|
|
@@ -127,6 +133,7 @@ const _execPlan = new PromptPlan({ startFrom: 0 })
|
|
|
127
133
|
.step("ENRICH", "Parameter Grounding & Priority Assignment", _execEnrichBody)
|
|
128
134
|
.step("DIVERSITY", (ctx) => `Diversity check (using enriched knowledge from Step ${ctx.enrichStepLabel})`, _execDiversityBody)
|
|
129
135
|
.step("EXECUTE", "Execute merged plan in rank order", _execExecuteBody)
|
|
136
|
+
.step("REGISTER", "Register your test plan", _execRegisterBody)
|
|
130
137
|
.done();
|
|
131
138
|
// ── Exported step label constants ─────────────────────────────────────────────
|
|
132
139
|
/** "0" — Code Review: correctness analysis */
|
|
@@ -139,6 +146,8 @@ export const EXEC_STEP_ENRICH = _execPlan.labels.ENRICH; // "2"
|
|
|
139
146
|
export const EXEC_STEP_DIVERSITY = _execPlan.labels.DIVERSITY; // "3"
|
|
140
147
|
/** "4" — Execute merged plan */
|
|
141
148
|
export const EXEC_STEP_EXECUTE = _execPlan.labels.EXECUTE; // "4"
|
|
149
|
+
/** "5" — Register test plan (SKYR-3879 Path B checkpoint) */
|
|
150
|
+
export const EXEC_STEP_REGISTER = _execPlan.labels.REGISTER; // "5"
|
|
142
151
|
const SERVICE_REFS = resolveServiceDetailsRef();
|
|
143
152
|
function prioritizeAttackSurfaceBundles(items) {
|
|
144
153
|
const reordered = [];
|
|
@@ -154,69 +163,6 @@ function prioritizeAttackSurfaceBundles(items) {
|
|
|
154
163
|
}
|
|
155
164
|
return reordered;
|
|
156
165
|
}
|
|
157
|
-
/**
|
|
158
|
-
* Select `count` items from a rank-ordered list, distributing GENERATE slots
|
|
159
|
-
* EVENLY across the test types present, with spillover.
|
|
160
|
-
*
|
|
161
|
-
* Policy (kept identical to the multi-repo prose in testbot-prompts.ts's
|
|
162
|
-
* "Cross-repo test generation" block — change both together):
|
|
163
|
-
* - Protected items first: CRITICAL-priority and attack-surface security_boundary
|
|
164
|
-
* scenarios always take a slot before round-robin (they must stay in GENERATE).
|
|
165
|
-
* - Bucket the rest by inferred test type (contract vs integration — the same
|
|
166
|
-
* inference used when rendering: `testType ?? (steps===1 ? contract : integration)`).
|
|
167
|
-
* - Round-robin one item per non-empty bucket per round, in the buckets' order of
|
|
168
|
-
* first appearance in the rank-ordered list (so the highest-ranked type wins
|
|
169
|
-
* round 1), preserving rank order within each bucket.
|
|
170
|
-
* - Spillover: an exhausted bucket is skipped on later rounds, so its freed slots
|
|
171
|
-
* go to the next type's next-highest item.
|
|
172
|
-
*
|
|
173
|
-
* Degenerate cases match the previous pure rank-order slice exactly: a single type
|
|
174
|
-
* present, or `count >= items.length`, returns the same items in the same order —
|
|
175
|
-
* so backend-only / single-type runs are unchanged (no regression).
|
|
176
|
-
*/
|
|
177
|
-
function roundRobinByType(rankOrdered, count) {
|
|
178
|
-
if (count <= 0)
|
|
179
|
-
return [];
|
|
180
|
-
// Everything fits → no need to bucket; identical to the old slice.
|
|
181
|
-
if (count >= rankOrdered.length)
|
|
182
|
-
return rankOrdered.slice(0, count);
|
|
183
|
-
const inferType = (s) => s.testType ?? (s.steps.length === 1 ? "contract" : "integration");
|
|
184
|
-
// Protected items occupy GENERATE slots first, in rank order.
|
|
185
|
-
const selected = [];
|
|
186
|
-
const remaining = [];
|
|
187
|
-
for (const item of rankOrdered) {
|
|
188
|
-
if (selected.length < count &&
|
|
189
|
-
(item.priority === "CRITICAL" || isAttackSurfaceSecurityBoundary(item.scenario))) {
|
|
190
|
-
selected.push(item);
|
|
191
|
-
}
|
|
192
|
-
else {
|
|
193
|
-
remaining.push(item);
|
|
194
|
-
}
|
|
195
|
-
}
|
|
196
|
-
// Bucket the remainder by inferred type, preserving rank order and first-appearance
|
|
197
|
-
// bucket order.
|
|
198
|
-
const order = [];
|
|
199
|
-
const buckets = new Map();
|
|
200
|
-
for (const item of remaining) {
|
|
201
|
-
const t = inferType(item.scenario);
|
|
202
|
-
if (!buckets.has(t)) {
|
|
203
|
-
buckets.set(t, []);
|
|
204
|
-
order.push(t);
|
|
205
|
-
}
|
|
206
|
-
buckets.get(t).push(item);
|
|
207
|
-
}
|
|
208
|
-
// Round-robin one per non-empty bucket per round until full.
|
|
209
|
-
while (selected.length < count && order.some((t) => buckets.get(t).length > 0)) {
|
|
210
|
-
for (const t of order) {
|
|
211
|
-
if (selected.length >= count)
|
|
212
|
-
break;
|
|
213
|
-
const bucket = buckets.get(t);
|
|
214
|
-
if (bucket.length > 0)
|
|
215
|
-
selected.push(bucket.shift());
|
|
216
|
-
}
|
|
217
|
-
}
|
|
218
|
-
return selected;
|
|
219
|
-
}
|
|
220
166
|
export function buildExecutionPlan(scored, maxGen, topN, baseUrl, authHeaderValue, authSchemeSnippet, authTypeValue, seed, endpointCount, isUIOnlyPR, hasFrontendChanges = false, hasTraces = false, externalCoverage = new Set(), relevantExternalTestPaths = [],
|
|
221
167
|
/**
|
|
222
168
|
* Whether the diff classified at least one new/modified/removed endpoint.
|
|
@@ -473,7 +419,7 @@ ${buildScopeAssessmentSection(topN, maxGen, isUIOnlyPR, isUIOnlyPR ? 100 : hasFr
|
|
|
473
419
|
|
|
474
420
|
${_execPlan.render(_ctx)}
|
|
475
421
|
|
|
476
|
-
### GENERATE (after completing Steps ${EXEC_STEP_CODE_REVIEW}–${EXEC_STEP_EXECUTE} above) —
|
|
422
|
+
### GENERATE (after completing Steps ${EXEC_STEP_CODE_REVIEW}–${EXEC_STEP_EXECUTE} above and registering via Step ${EXEC_STEP_REGISTER}) — the list below is a starting point; \`skyramp_register_test_plan\`'s returned GENERATE list is the final, mandatory one. Generate exactly those items in order; add variations to ADDITIONAL instead. If Step ${EXEC_STEP_COVERAGE} converts an item to UPDATE, backfill from ADDITIONAL (priority order in Step ${EXEC_STEP_COVERAGE})
|
|
477
423
|
|
|
478
424
|
${isUIOnlyPR
|
|
479
425
|
? uiGenerateBlocks ||
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import { DraftedScenario } from "../../types/RepositoryAnalysis.js";
|
|
2
|
-
import { PriorityTier } from "../../types/TestRecommendation.js";
|
|
2
|
+
import { Novelty, PriorityTier } from "../../types/TestRecommendation.js";
|
|
3
3
|
export declare function buildFullRepoRecommendations(scored: Array<{
|
|
4
4
|
scenario: DraftedScenario;
|
|
5
5
|
priority: PriorityTier;
|
|
6
|
-
novelty:
|
|
6
|
+
novelty: Novelty;
|
|
7
7
|
}>, topN: number, baseUrl: string, authHeaderValue: string, authSchemeSnippet: string, authTypeValue: string, isFrontendProject?: boolean, isFrontendOnlyProject?: boolean, externalCoverage?: Set<string>): string;
|
|
@@ -51,7 +51,7 @@ Before each GENERATE tool call, confirm WHERE each key value comes from:
|
|
|
51
51
|
- **endpointURL** → workspace \`baseUrl\` + endpoint path (both required — never path alone)
|
|
52
52
|
- **authHeader / authScheme** → workspace config or OpenAPI \`securitySchemes\`
|
|
53
53
|
- **Foreign-key path params** → chained from a prior step's response — never invented or hardcoded. Common field names: \`id\`, \`uuid\`, \`_id\`, \`*_id\`; use whatever identifier field the server returns for this resource. The chaining source can be a response body (POST or GET), a response header (e.g. \`Location\`), or a cookie.
|
|
54
|
-
- **Names / string values** → realistic
|
|
54
|
+
- **Names / string values** → realistic. Do NOT hardcode a timestamp/uuid suffix. Instead, for create fields that carry a UNIQUE constraint (e.g. \`name\`, \`slug\`, \`email\` — confirm from the source schema: \`unique=True\`, SQL \`UNIQUE\`, DTO), list their field paths in the step's \`uniqueFields\` (gjson notation) — the generator injects a run-unique value so re-runs don't 409.
|
|
55
55
|
|
|
56
56
|
## Ranking Rule
|
|
57
57
|
For each GENERATE item, include one sentence in your output (before the tool calls) stating the specific bug or failure it targets — derived from \`bugCatchingTarget\` or your source-code reading. Example: "Targets: order total miscalculation — total_amount = sum(item.price × item.quantity) should recompute when items array changes."
|
|
@@ -332,7 +332,10 @@ ${authGuidance}
|
|
|
332
332
|
collection array (e.g. \`"items": [{"product_id": <chained from prior POST>, "quantity": 2}]\`).
|
|
333
333
|
Never send a PATCH that only modifies metadata (discount, status) without also including the
|
|
334
334
|
items/products collection — such a test will not catch collection-level or total-recalculation bugs.
|
|
335
|
-
|
|
335
|
+
For UNIQUE-constrained create fields, set the step's \`uniqueFields\` (gjson paths) rather than
|
|
336
|
+
hardcoding timestamp suffixes — the generator makes them run-unique so re-runs don't 409.
|
|
337
|
+
For every resource-creating step (POST/PUT), also add a matching DELETE step for the created
|
|
338
|
+
resource (path ID chained from the create response) so the scenario cleans up after itself.
|
|
336
339
|
For GET/PUT/DELETE with path IDs, use a placeholder — chaining resolves the real ID.
|
|
337
340
|
2. Produces a \`scenario_<name>.json\` in the same \`outputDir\` as the test files (not \`.skyramp/\`).
|
|
338
341
|
3. Call \`skyramp_integration_test_generation\` with \`scenarioFile\`: ${integrationAuthNote}
|
|
@@ -200,7 +200,7 @@ Budget Plan: 0 total — no new, modified, or removed endpoints were classified
|
|
|
200
200
|
|
|
201
201
|
With a 0-total Budget Plan: generate zero tests, recommend zero tests, and follow the zero-test report path. Do NOT draft baseline or generic tests for unchanged endpoints to fill a budget — an empty diff surface is a valid, expected outcome.
|
|
202
202
|
|
|
203
|
-
**Exception — claim the ceiling only with evidence:** if your code review of the changed files shows an observable API behavior change the classifier missed (e.g. a DTO/serializer/service change that alters a response shape,
|
|
203
|
+
**Exception — claim the ceiling only with evidence:** if your code review of the changed files shows an observable API behavior change the classifier missed (e.g. a DTO/serializer/service change that alters a response shape, a deployment/config change that newly exposes or removes endpoints, or a schema-defined API contract change — a CRD type/kubebuilder validation marker, GraphQL schema, or gRPC proto edit that adds, removes, or re-validates what the server accepts or returns), raise your Budget Plan to cover exactly those affected endpoints, up to ${maxTotal} total (${effectiveGenerate} generate + ${additional} additional), 0% UI/E2E. Note: repositories whose entire API surface is schema-defined (e.g. a Kubernetes operator serving CRDs through the kube-apiserver) ALWAYS classify zero endpoints — for these, a schema change in the diff IS the endpoint change; evaluate this exception against the schema files instead of concluding there is nothing to test. Every test must name the changed file that justifies it. State your raised plan now in the canonical format — \`Budget Plan: <total> total (<generate> generate + <additional> additional), 0% UI/E2E\` — and use those exact numbers throughout the rest of the prompt; the raised generate count is your committed generate count.`;
|
|
204
204
|
}
|
|
205
205
|
// Unambiguous backend-only or UI-only: emit a single Budget Plan line — no LLM counting needed.
|
|
206
206
|
if (precomputedUIPct !== undefined) {
|
|
@@ -1,6 +1,31 @@
|
|
|
1
|
-
import { RepositoryAnalysis, AnalysisScope } from "../../types/RepositoryAnalysis.js";
|
|
1
|
+
import { RepositoryAnalysis, AnalysisScope, DraftedScenario } from "../../types/RepositoryAnalysis.js";
|
|
2
2
|
import { WorkspaceAuthType } from "../../utils/workspaceAuth.js";
|
|
3
3
|
import { PRTestContext } from "../../utils/pr-comment-parser.js";
|
|
4
|
+
import { Novelty, PriorityTier } from "../../types/TestRecommendation.js";
|
|
4
5
|
import { buildExternalCoverageSet, externalDedupKey } from "./recommendationShared.js";
|
|
5
6
|
export { buildExternalCoverageSet, externalDedupKey };
|
|
7
|
+
/** Result of {@link computeScoredCandidates} — the scoring/classification
|
|
8
|
+
* inputs shared between prompt rendering and the SKYR-3879 register-plan
|
|
9
|
+
* pre-seed (analyzeChangesTool.ts), so both derive the exact same numbers. */
|
|
10
|
+
export interface ScoredCandidatesResult {
|
|
11
|
+
scored: Array<{
|
|
12
|
+
scenario: DraftedScenario;
|
|
13
|
+
priority: PriorityTier;
|
|
14
|
+
novelty: Novelty;
|
|
15
|
+
}>;
|
|
16
|
+
filteredChangedFiles: string[];
|
|
17
|
+
isUIOnlyPR: boolean;
|
|
18
|
+
hasFrontendChanges: boolean;
|
|
19
|
+
hasApiChanges: boolean;
|
|
20
|
+
maxGen: number;
|
|
21
|
+
seed: string;
|
|
22
|
+
}
|
|
23
|
+
/**
|
|
24
|
+
* Compute the pre-ranked scenario scoring, the UI/API classification flags,
|
|
25
|
+
* and the effective GENERATE cap — extracted from buildRecommendationPrompt
|
|
26
|
+
* so skyramp_analyze_changes (SKYR-3879 Path B pre-seed) can derive the exact
|
|
27
|
+
* same candidate set the Execution Plan prompt is built from, without
|
|
28
|
+
* duplicating (and risking drift from) this scoring logic.
|
|
29
|
+
*/
|
|
30
|
+
export declare function computeScoredCandidates(analysis: RepositoryAnalysis, analysisScope?: AnalysisScope, topN?: number, maxGenerateOverride?: number): ScoredCandidatesResult;
|
|
6
31
|
export declare function buildRecommendationPrompt(analysis: RepositoryAnalysis, analysisScope?: AnalysisScope, topN?: number, prContext?: PRTestContext, workspaceAuthHeader?: string, workspaceAuthType?: WorkspaceAuthType, workspaceAuthScheme?: string, maxGenerateOverride?: number, sessionId?: string): string;
|
|
@@ -3,7 +3,7 @@ import { AnalysisScope, isDiff, } from "../../types/RepositoryAnalysis.js";
|
|
|
3
3
|
import { WorkspaceAuthType, getDefaultAuthHeader } from "../../utils/workspaceAuth.js";
|
|
4
4
|
import { logger } from "../../utils/logger.js";
|
|
5
5
|
import { buildArchitectPreamble, buildContextFetchingGuidance, buildReasoningProtocol, buildToolWorkflows, buildFewShotExamples, buildVerificationChecklist, getAuthSnippets, MAX_TESTS_TO_GENERATE, MAX_RECOMMENDATIONS, } from "./recommendationSections.js";
|
|
6
|
-
import { CATEGORY_PRIORITY } from "../../types/TestRecommendation.js";
|
|
6
|
+
import { CATEGORY_PRIORITY, Novelty, PriorityTier } from "../../types/TestRecommendation.js";
|
|
7
7
|
import { buildScopeAssessmentSection, isFrontendFile } from "./scopeAssessment.js";
|
|
8
8
|
import { buildExecutionPlan, EXEC_STEP_CODE_REVIEW } from "./diffExecutionPlan.js";
|
|
9
9
|
import { buildFullRepoRecommendations } from "./fullRepoCatalog.js";
|
|
@@ -36,21 +36,21 @@ const PRIORITY_ORDER = { CRITICAL: 4, HIGH: 3, MEDIUM: 2, LOW: 1 };
|
|
|
36
36
|
const NOVELTY_ORDER = { new: 3, modified: 2, existing: 1 };
|
|
37
37
|
function classifyNovelty(scenario, diffContext) {
|
|
38
38
|
if (!diffContext)
|
|
39
|
-
return
|
|
39
|
+
return Novelty.EXISTING;
|
|
40
40
|
const paths = scenario.steps.map(s => s.path);
|
|
41
41
|
const newPaths = new Set((diffContext.newEndpoints || []).map(ep => ep.path));
|
|
42
42
|
const modPaths = new Set((diffContext.modifiedEndpoints || []).map(ep => ep.path));
|
|
43
43
|
const removedPaths = new Set((diffContext.removedEndpoints || []).map(ep => ep.path));
|
|
44
44
|
if (paths.some(p => newPaths.has(p)))
|
|
45
|
-
return
|
|
45
|
+
return Novelty.NEW;
|
|
46
46
|
if (paths.some(p => modPaths.has(p) || removedPaths.has(p)))
|
|
47
|
-
return
|
|
48
|
-
return
|
|
47
|
+
return Novelty.MODIFIED;
|
|
48
|
+
return Novelty.EXISTING;
|
|
49
49
|
}
|
|
50
50
|
function prioritiseCandidate(scenario, diffContext) {
|
|
51
51
|
const priority = isAttackSurfaceSecurityBoundary(scenario)
|
|
52
|
-
?
|
|
53
|
-
: CATEGORY_PRIORITY[scenario.category] ??
|
|
52
|
+
? PriorityTier.CRITICAL
|
|
53
|
+
: CATEGORY_PRIORITY[scenario.category] ?? PriorityTier.LOW;
|
|
54
54
|
const novelty = classifyNovelty(scenario, diffContext);
|
|
55
55
|
return { priority, novelty };
|
|
56
56
|
}
|
|
@@ -58,15 +58,20 @@ function computeTiebreakerSeed(endpoints, diffFiles) {
|
|
|
58
58
|
const canonical = [...endpoints].sort().join("|") + "::" + [...diffFiles].sort().join("|");
|
|
59
59
|
return crypto.createHash("sha256").update(canonical).digest("hex").slice(0, 8);
|
|
60
60
|
}
|
|
61
|
-
//
|
|
62
|
-
|
|
61
|
+
// Prevents bot-committed test files from being treated as application changes
|
|
62
|
+
// on subsequent testbot runs on the same PR.
|
|
63
|
+
const SKYRAMP_TEST_FILE_PATTERN = /(?:_test|_smoke|_contract|_fuzz|_integration|_load|_e2e|_ui)\.[^/]+$|scenario_[^/]+\.json$/;
|
|
64
|
+
/**
|
|
65
|
+
* Compute the pre-ranked scenario scoring, the UI/API classification flags,
|
|
66
|
+
* and the effective GENERATE cap — extracted from buildRecommendationPrompt
|
|
67
|
+
* so skyramp_analyze_changes (SKYR-3879 Path B pre-seed) can derive the exact
|
|
68
|
+
* same candidate set the Execution Plan prompt is built from, without
|
|
69
|
+
* duplicating (and risking drift from) this scoring logic.
|
|
70
|
+
*/
|
|
71
|
+
export function computeScoredCandidates(analysis, analysisScope = AnalysisScope.FullRepo, topN = MAX_RECOMMENDATIONS, maxGenerateOverride) {
|
|
63
72
|
const isDiffScope = isDiff(analysisScope);
|
|
64
73
|
const diffContext = analysis.branchDiffContext;
|
|
65
|
-
const openApiSpec = analysis.artifacts?.openApiSpecs?.[0];
|
|
66
74
|
// ── Filter out bot-generated test files from changedFiles ──
|
|
67
|
-
// Prevents bot-committed test files from being treated as application changes
|
|
68
|
-
// on subsequent testbot runs on the same PR.
|
|
69
|
-
const SKYRAMP_TEST_FILE_PATTERN = /(?:_test|_smoke|_contract|_fuzz|_integration|_load|_e2e|_ui)\.[^/]+$|scenario_[^/]+\.json$/;
|
|
70
75
|
const filteredChangedFiles = diffContext
|
|
71
76
|
? diffContext.changedFiles.filter(f => !SKYRAMP_TEST_FILE_PATTERN.test(f))
|
|
72
77
|
: [];
|
|
@@ -81,6 +86,56 @@ export function buildRecommendationPrompt(analysis, analysisScope = AnalysisScop
|
|
|
81
86
|
? (diffContext.newEndpoints.length > 0 || diffContext.modifiedEndpoints.length > 0 || (diffContext.removedEndpoints?.length ?? 0) > 0)
|
|
82
87
|
: false;
|
|
83
88
|
const isUIOnlyPR = hasFrontendChanges && !hasApiChanges;
|
|
89
|
+
// ── Scoring ──
|
|
90
|
+
const baseMaxGen = Math.min(Math.max(maxGenerateOverride ?? (isDiffScope ? MAX_TESTS_TO_GENERATE : topN), 0), topN);
|
|
91
|
+
const maxGen = isUIOnlyPR ? Math.max(baseMaxGen, 1) : baseMaxGen;
|
|
92
|
+
const scenarios = analysis.businessContext.draftedScenarios;
|
|
93
|
+
let scored = [];
|
|
94
|
+
let seed = "";
|
|
95
|
+
if (!isUIOnlyPR && scenarios.length > 0) {
|
|
96
|
+
const diffFiles = filteredChangedFiles; // use filtered list so bot-committed test files don't shift the seed
|
|
97
|
+
const endpointPaths = analysis.apiEndpoints.endpoints.map(ep => ep.path);
|
|
98
|
+
seed = computeTiebreakerSeed(endpointPaths, diffFiles);
|
|
99
|
+
scored = scenarios.map(s => {
|
|
100
|
+
const result = prioritiseCandidate(s, diffContext ?? undefined);
|
|
101
|
+
return { scenario: s, ...result };
|
|
102
|
+
});
|
|
103
|
+
scored.sort((a, b) => {
|
|
104
|
+
const pa = PRIORITY_ORDER[a.priority], pb = PRIORITY_ORDER[b.priority];
|
|
105
|
+
if (pb !== pa)
|
|
106
|
+
return pb - pa;
|
|
107
|
+
const na = NOVELTY_ORDER[a.novelty], nb = NOVELTY_ORDER[b.novelty];
|
|
108
|
+
if (nb !== na)
|
|
109
|
+
return nb - na;
|
|
110
|
+
const crossA = a.scenario.steps.length > 2 ? 1 : 0;
|
|
111
|
+
const crossB = b.scenario.steps.length > 2 ? 1 : 0;
|
|
112
|
+
if (crossB !== crossA)
|
|
113
|
+
return crossB - crossA;
|
|
114
|
+
if (b.scenario.steps.length !== a.scenario.steps.length)
|
|
115
|
+
return b.scenario.steps.length - a.scenario.steps.length;
|
|
116
|
+
const errorA = a.scenario.steps.some(s => s.interactionType === "error" || s.interactionType === "edge-case") ? 1 : 0;
|
|
117
|
+
const errorB = b.scenario.steps.some(s => s.interactionType === "error" || s.interactionType === "edge-case") ? 1 : 0;
|
|
118
|
+
if (errorB !== errorA)
|
|
119
|
+
return errorB - errorA;
|
|
120
|
+
// Use locale-independent comparison to avoid runtime-locale non-determinism
|
|
121
|
+
const nameA = a.scenario.scenarioName;
|
|
122
|
+
const nameB = b.scenario.scenarioName;
|
|
123
|
+
if (nameA < nameB)
|
|
124
|
+
return -1;
|
|
125
|
+
if (nameA > nameB)
|
|
126
|
+
return 1;
|
|
127
|
+
const hashA = parseInt(crypto.createHash("sha256").update(seed + a.scenario.scenarioName).digest("hex").slice(0, 8), 16);
|
|
128
|
+
const hashB = parseInt(crypto.createHash("sha256").update(seed + b.scenario.scenarioName).digest("hex").slice(0, 8), 16);
|
|
129
|
+
return hashA - hashB;
|
|
130
|
+
});
|
|
131
|
+
}
|
|
132
|
+
return { scored, filteredChangedFiles, isUIOnlyPR, hasFrontendChanges, hasApiChanges, maxGen, seed };
|
|
133
|
+
}
|
|
134
|
+
export function buildRecommendationPrompt(analysis, analysisScope = AnalysisScope.FullRepo, topN = MAX_RECOMMENDATIONS, prContext, workspaceAuthHeader, workspaceAuthType, workspaceAuthScheme, maxGenerateOverride, sessionId) {
|
|
135
|
+
const isDiffScope = isDiff(analysisScope);
|
|
136
|
+
const diffContext = analysis.branchDiffContext;
|
|
137
|
+
const openApiSpec = analysis.artifacts?.openApiSpecs?.[0];
|
|
138
|
+
const { scored, filteredChangedFiles, isUIOnlyPR, hasFrontendChanges, hasApiChanges, maxGen, seed, } = computeScoredCandidates(analysis, analysisScope, topN, maxGenerateOverride);
|
|
84
139
|
const hasTraces = (analysis.artifacts?.traceFiles?.length ?? 0) > 0 ||
|
|
85
140
|
(analysis.artifacts?.playwrightRecordings?.length ?? 0) > 0;
|
|
86
141
|
// ── Mode preamble ──
|
|
@@ -277,50 +332,7 @@ ${isDiffScope ? "Changed endpoints only. " : ""}Use source code schemas (Zod/Pyd
|
|
|
277
332
|
${detailBlocks}
|
|
278
333
|
`;
|
|
279
334
|
}
|
|
280
|
-
// ── Scoring ──
|
|
281
335
|
const endpointCount = allEndpoints.reduce((acc, ep) => acc + (ep.methods ?? []).length, 0);
|
|
282
|
-
const baseMaxGen = Math.min(Math.max(maxGenerateOverride ?? (isDiffScope ? MAX_TESTS_TO_GENERATE : topN), 0), topN);
|
|
283
|
-
const maxGen = isUIOnlyPR ? Math.max(baseMaxGen, 1) : baseMaxGen;
|
|
284
|
-
const scenarios = analysis.businessContext.draftedScenarios;
|
|
285
|
-
let scored = [];
|
|
286
|
-
let seed = "";
|
|
287
|
-
if (!isUIOnlyPR && scenarios.length > 0) {
|
|
288
|
-
const diffFiles = filteredChangedFiles; // use filtered list so bot-committed test files don't shift the seed
|
|
289
|
-
const endpointPaths = allEndpoints.map(ep => ep.path);
|
|
290
|
-
seed = computeTiebreakerSeed(endpointPaths, diffFiles);
|
|
291
|
-
scored = scenarios.map(s => {
|
|
292
|
-
const result = prioritiseCandidate(s, diffContext ?? undefined);
|
|
293
|
-
return { scenario: s, ...result };
|
|
294
|
-
});
|
|
295
|
-
scored.sort((a, b) => {
|
|
296
|
-
const pa = PRIORITY_ORDER[a.priority], pb = PRIORITY_ORDER[b.priority];
|
|
297
|
-
if (pb !== pa)
|
|
298
|
-
return pb - pa;
|
|
299
|
-
const na = NOVELTY_ORDER[a.novelty], nb = NOVELTY_ORDER[b.novelty];
|
|
300
|
-
if (nb !== na)
|
|
301
|
-
return nb - na;
|
|
302
|
-
const crossA = a.scenario.steps.length > 2 ? 1 : 0;
|
|
303
|
-
const crossB = b.scenario.steps.length > 2 ? 1 : 0;
|
|
304
|
-
if (crossB !== crossA)
|
|
305
|
-
return crossB - crossA;
|
|
306
|
-
if (b.scenario.steps.length !== a.scenario.steps.length)
|
|
307
|
-
return b.scenario.steps.length - a.scenario.steps.length;
|
|
308
|
-
const errorA = a.scenario.steps.some(s => s.interactionType === "error" || s.interactionType === "edge-case") ? 1 : 0;
|
|
309
|
-
const errorB = b.scenario.steps.some(s => s.interactionType === "error" || s.interactionType === "edge-case") ? 1 : 0;
|
|
310
|
-
if (errorB !== errorA)
|
|
311
|
-
return errorB - errorA;
|
|
312
|
-
// Use locale-independent comparison to avoid runtime-locale non-determinism
|
|
313
|
-
const nameA = a.scenario.scenarioName;
|
|
314
|
-
const nameB = b.scenario.scenarioName;
|
|
315
|
-
if (nameA < nameB)
|
|
316
|
-
return -1;
|
|
317
|
-
if (nameA > nameB)
|
|
318
|
-
return 1;
|
|
319
|
-
const hashA = parseInt(crypto.createHash("sha256").update(seed + a.scenario.scenarioName).digest("hex").slice(0, 8), 16);
|
|
320
|
-
const hashB = parseInt(crypto.createHash("sha256").update(seed + b.scenario.scenarioName).digest("hex").slice(0, 8), 16);
|
|
321
|
-
return hashA - hashB;
|
|
322
|
-
});
|
|
323
|
-
}
|
|
324
336
|
// ── Main section: execution plan, UI-only guidance, or draft-your-own ──
|
|
325
337
|
let mainSection;
|
|
326
338
|
if (!isDiffScope && scored.length > 0) {
|
|
@@ -1,10 +1,11 @@
|
|
|
1
1
|
import { jest } from "@jest/globals";
|
|
2
|
+
import { Novelty, PriorityTier } from "../../types/TestRecommendation.js";
|
|
2
3
|
import { TestType } from "../../types/TestTypes.js";
|
|
3
4
|
jest.unstable_mockModule("@skyramp/skyramp", () => ({
|
|
4
5
|
WorkspaceConfigManager: { create: jest.fn() },
|
|
5
6
|
}));
|
|
6
7
|
const { buildRecommendationPrompt, buildExternalCoverageSet, externalDedupKey } = await import("./test-recommendation-prompt.js");
|
|
7
|
-
const { buildExecutionPlan } = await import("./diffExecutionPlan.js");
|
|
8
|
+
const { buildExecutionPlan, EXEC_STEP_REGISTER } = await import("./diffExecutionPlan.js");
|
|
8
9
|
const { PATH_PARAM_UUID_GUIDANCE, MAX_TESTS_TO_GENERATE, buildTestQualityCriteria, buildArchitectPreamble, buildContextFetchingGuidance, buildReasoningProtocol, buildFewShotExamples, buildVerificationChecklist, } = await import("./recommendationSections.js");
|
|
9
10
|
import { AnalysisScope } from "../../types/RepositoryAnalysis.js";
|
|
10
11
|
// ---------------------------------------------------------------------------
|
|
@@ -746,6 +747,42 @@ describe("buildRecommendationPrompt — zero-classified diff (SKYR-3820)", () =>
|
|
|
746
747
|
});
|
|
747
748
|
});
|
|
748
749
|
// ---------------------------------------------------------------------------
|
|
750
|
+
// Tests — REGISTER step (SKYR-3879 Path B checkpoint)
|
|
751
|
+
// ---------------------------------------------------------------------------
|
|
752
|
+
describe("buildRecommendationPrompt — REGISTER step (SKYR-3879 Path B)", () => {
|
|
753
|
+
it("diff-scope execution plan includes the REGISTER step and its skyramp_register_test_plan instruction", () => {
|
|
754
|
+
const prompt = buildRecommendationPrompt(minimalAnalysis(), AnalysisScope.CurrentBranchDiff, 5);
|
|
755
|
+
expect(prompt).toContain(`Step ${EXEC_STEP_REGISTER} — Register your test plan`);
|
|
756
|
+
expect(prompt).toContain("skyramp_register_test_plan");
|
|
757
|
+
});
|
|
758
|
+
it("diff-scope execution plan's GENERATE header references the REGISTER step", () => {
|
|
759
|
+
const prompt = buildRecommendationPrompt(minimalAnalysis(), AnalysisScope.CurrentBranchDiff, 5);
|
|
760
|
+
expect(prompt).toContain(`registering via Step ${EXEC_STEP_REGISTER}`);
|
|
761
|
+
});
|
|
762
|
+
it("full-repo mode (no execution plan) never mentions skyramp_register_test_plan", () => {
|
|
763
|
+
const scenariosForFullRepo = [
|
|
764
|
+
minimalScenario({
|
|
765
|
+
scenarioName: "scenario-0",
|
|
766
|
+
category: "crud",
|
|
767
|
+
priority: "low",
|
|
768
|
+
testType: TestType.CONTRACT,
|
|
769
|
+
steps: [{ order: 1, method: "GET", path: "/api/items", description: "Get items", interactionType: "success", expectedStatusCode: 200 }],
|
|
770
|
+
}),
|
|
771
|
+
];
|
|
772
|
+
const analysis = minimalAnalysis({
|
|
773
|
+
businessContext: { mainPurpose: "Test API", userFlows: [], dataFlows: [], integrationPatterns: [], draftedScenarios: scenariosForFullRepo },
|
|
774
|
+
});
|
|
775
|
+
const prompt = buildRecommendationPrompt(analysis, AnalysisScope.FullRepo, 6);
|
|
776
|
+
expect(prompt).not.toContain("skyramp_register_test_plan");
|
|
777
|
+
});
|
|
778
|
+
it("diff-scope 'draft your own' fallback (no pre-drafted scenarios) still includes the REGISTER step — the execution plan is always used in diff scope", () => {
|
|
779
|
+
const prompt = buildRecommendationPrompt(minimalAnalysis(), AnalysisScope.CurrentBranchDiff, 5);
|
|
780
|
+
// minimalAnalysis() has no draftedScenarios and no branchDiffContext, so scored=[]
|
|
781
|
+
// but buildExecutionPlan still renders (see buildRecommendationPrompt's isDiffScope branch).
|
|
782
|
+
expect(prompt).toContain("skyramp_register_test_plan");
|
|
783
|
+
});
|
|
784
|
+
});
|
|
785
|
+
// ---------------------------------------------------------------------------
|
|
749
786
|
// Tests — GENERATE slot allocation (UI vs backend slots)
|
|
750
787
|
// ---------------------------------------------------------------------------
|
|
751
788
|
describe("buildRecommendationPrompt — GENERATE slot allocation", () => {
|
|
@@ -920,13 +957,13 @@ describe("buildRecommendationPrompt — GENERATE slot allocation", () => {
|
|
|
920
957
|
});
|
|
921
958
|
const prompt = buildExecutionPlan([attackSurface, bugCaught, ordinary].map((scenario) => ({
|
|
922
959
|
scenario,
|
|
923
|
-
priority: scenario.category === "crud" ?
|
|
924
|
-
novelty:
|
|
960
|
+
priority: scenario.category === "crud" ? PriorityTier.LOW : PriorityTier.CRITICAL,
|
|
961
|
+
novelty: Novelty.MODIFIED,
|
|
925
962
|
})), 3, 3, "http://localhost:3000", "Authorization", ", authScheme: \"Bearer\"", "", "seed", 3, false, false, false, new Set(["POST::flows::contract", "POST::orders::contract", "POST::customers::contract"]));
|
|
926
963
|
expect(prompt).toContain("POST /api/flows/bulk_delete → 401");
|
|
927
964
|
expect(prompt).toContain("POST /api/orders → 201");
|
|
928
965
|
expect(prompt).not.toContain("POST /api/customers → 201");
|
|
929
|
-
expect(prompt).toContain("
|
|
966
|
+
expect(prompt).toContain("`bug_caught` and attack-surface `security_boundary` items get the strictest reading");
|
|
930
967
|
});
|
|
931
968
|
it("keeps symmetric attack-surface siblings together before ordinary auth boundaries", () => {
|
|
932
969
|
const directWidgets = minimalScenario({
|
|
@@ -959,8 +996,8 @@ describe("buildRecommendationPrompt — GENERATE slot allocation", () => {
|
|
|
959
996
|
});
|
|
960
997
|
const prompt = buildExecutionPlan([directWidgets, directGadgets, widgetsBulk, gadgetsBulk].map((scenario) => ({
|
|
961
998
|
scenario,
|
|
962
|
-
priority:
|
|
963
|
-
novelty:
|
|
999
|
+
priority: PriorityTier.CRITICAL,
|
|
1000
|
+
novelty: Novelty.MODIFIED,
|
|
964
1001
|
})), 3, 4, "http://localhost:3000", "Authorization", ", authScheme: \"Bearer\"", "", "seed", 4, false);
|
|
965
1002
|
const generatedBlock = prompt.slice(0, prompt.indexOf("#4 [ADDITIONAL]"));
|
|
966
1003
|
const additionalBlock = prompt.slice(prompt.indexOf("#4 [ADDITIONAL]"));
|
|
@@ -975,7 +1012,7 @@ describe("buildExecutionPlan — even-with-spillover test-type distribution", ()
|
|
|
975
1012
|
// Backend-only PR (hasFrontendChanges=false) so every GENERATE slot is a backend
|
|
976
1013
|
// scenario drawn from `scored` — this is where contract-vs-integration distribution
|
|
977
1014
|
// applies. baseUrl/auth args mirror the existing buildExecutionPlan tests above.
|
|
978
|
-
const contract = (name, path, priority =
|
|
1015
|
+
const contract = (name, path, priority = PriorityTier.HIGH) => ({
|
|
979
1016
|
scenario: minimalScenario({
|
|
980
1017
|
scenarioName: name,
|
|
981
1018
|
category: "crud",
|
|
@@ -983,9 +1020,9 @@ describe("buildExecutionPlan — even-with-spillover test-type distribution", ()
|
|
|
983
1020
|
steps: [{ order: 1, method: "POST", path, description: `POST ${path}`, interactionType: "success", expectedStatusCode: 201 }],
|
|
984
1021
|
}),
|
|
985
1022
|
priority: priority,
|
|
986
|
-
novelty:
|
|
1023
|
+
novelty: Novelty.NEW,
|
|
987
1024
|
});
|
|
988
|
-
const integration = (name, p1, p2, priority =
|
|
1025
|
+
const integration = (name, p1, p2, priority = PriorityTier.HIGH) => ({
|
|
989
1026
|
scenario: minimalScenario({
|
|
990
1027
|
scenarioName: name,
|
|
991
1028
|
category: "data_integrity",
|
|
@@ -996,7 +1033,7 @@ describe("buildExecutionPlan — even-with-spillover test-type distribution", ()
|
|
|
996
1033
|
],
|
|
997
1034
|
}),
|
|
998
1035
|
priority: priority,
|
|
999
|
-
novelty:
|
|
1036
|
+
novelty: Novelty.NEW,
|
|
1000
1037
|
});
|
|
1001
1038
|
const plan = (items, maxGen, topN) => buildExecutionPlan(items, maxGen, topN, "http://localhost:3000", "Authorization", ", authScheme: \"Bearer\"", "", "seed", items.length, false);
|
|
1002
1039
|
it("distributes GENERATE slots across contract AND integration when budget < candidates", () => {
|
|
@@ -1025,7 +1062,7 @@ describe("buildExecutionPlan — even-with-spillover test-type distribution", ()
|
|
|
1025
1062
|
// A CRITICAL integration sits after several HIGH contracts in rank order.
|
|
1026
1063
|
const items = [
|
|
1027
1064
|
contract("c1", "/api/a"), contract("c2", "/api/b"), contract("c3", "/api/c"),
|
|
1028
|
-
integration("crit", "/api/orders", "/api/orders/{id}",
|
|
1065
|
+
integration("crit", "/api/orders", "/api/orders/{id}", PriorityTier.CRITICAL),
|
|
1029
1066
|
];
|
|
1030
1067
|
const prompt = plan(items, 3, 6);
|
|
1031
1068
|
const generated = prompt.slice(0, prompt.indexOf("[ADDITIONAL]") >= 0 ? prompt.indexOf("[ADDITIONAL]") : prompt.length);
|
|
@@ -16,7 +16,7 @@ export interface RelatedRepository {
|
|
|
16
16
|
*/
|
|
17
17
|
export declare function parseRelatedRepositories(raw: string | undefined): RelatedRepository[] | undefined;
|
|
18
18
|
export declare function getTestbotPrompt(prTitle: string, prDescription: string, summaryOutputFile: string, repositoryPath: string, baseBranch?: string, maxRecommendations?: number, maxGenerate?: number, _maxCritical?: number, // Reserved — accepted for API compat but not yet wired into prompt
|
|
19
|
-
prNumber?: number, userPrompt?: string, services?: Service[], uiCredentials?: string, testsRepoDir?: string, relatedRepositories?: RelatedRepository[], primaryRepo?: string): string;
|
|
19
|
+
prNumber?: number, userPrompt?: string, services?: Service[], uiCredentials?: string, testsRepoDir?: string, relatedRepositories?: RelatedRepository[], primaryRepo?: string, planOnly?: boolean): string;
|
|
20
20
|
/**
|
|
21
21
|
* Read services from .skyramp/workspace.yml. Returns empty array if
|
|
22
22
|
* the workspace file doesn't exist or can't be parsed.
|