@skyramp/mcp 0.3.5 → 0.3.6-rc.2.ac20
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/build/adapters/jestAdapter.js +3 -0
- package/build/adapters/mochaAdapter.js +2 -0
- package/build/adapters/playwrightAdapter.js +3 -0
- package/build/adapters/pytestAdapter.js +12 -0
- package/build/playwright/registerPlaywrightTools.js +92 -30
- package/build/playwright/traceRecordingPrompt.d.ts +6 -0
- package/build/playwright/traceRecordingPrompt.js +6 -2
- package/build/prompts/code-reuse.d.ts +1 -2
- package/build/prompts/code-reuse.js +182 -77
- package/build/prompts/modularization/integration-test-modularization.d.ts +2 -0
- package/build/prompts/modularization/integration-test-modularization.js +83 -41
- package/build/prompts/modularization/render.d.ts +18 -0
- package/build/prompts/modularization/render.js +12 -0
- package/build/prompts/modularization/ui-test-modularization.d.ts +3 -1
- package/build/prompts/modularization/ui-test-modularization.js +89 -47
- package/build/prompts/pom-aware-code-reuse.js +3 -1
- package/build/prompts/shared-helper-policy.d.ts +57 -0
- package/build/prompts/shared-helper-policy.js +135 -0
- package/build/prompts/test-recommendation/diffExecutionPlan.js +62 -56
- package/build/prompts/test-recommendation/fullRepoCatalog.js +19 -8
- package/build/prompts/test-recommendation/recommendationShared.d.ts +28 -6
- package/build/prompts/test-recommendation/recommendationShared.js +90 -16
- package/build/prompts/test-recommendation/registerRecommendTestsPrompt.js +22 -0
- package/build/prompts/test-recommendation/test-recommendation-prompt.d.ts +2 -2
- package/build/prompts/test-recommendation/test-recommendation-prompt.js +3 -3
- package/build/prompts/testbot/testbot-prompts.js +81 -35
- package/build/recommendation/budgeters/shared.js +105 -27
- package/build/recommendation/discriminators.js +13 -2
- package/build/recommendation/planRanker.d.ts +6 -6
- package/build/recommendation/planRanker.js +6 -61
- package/build/services/AnalyticsService.d.ts +7 -0
- package/build/services/AnalyticsService.js +7 -1
- package/build/services/ModularizationService.js +1 -3
- package/build/services/TestDiscoveryService.d.ts +0 -2
- package/build/services/TestDiscoveryService.js +2 -37
- package/build/services/TestGenerationService.d.ts +16 -0
- package/build/services/TestGenerationService.js +86 -10
- package/build/services/containerEnv.js +13 -12
- package/build/tools/code-refactor/codeReuseTool.js +279 -93
- package/build/tools/code-refactor/enhance-state.d.ts +49 -0
- package/build/tools/code-refactor/enhance-state.js +109 -0
- package/build/tools/code-refactor/enhanceAssertionsTool.js +34 -1
- package/build/tools/code-refactor/modularizationTool.js +9 -2
- package/build/tools/code-refactor/reuse-outcome.d.ts +23 -1
- package/build/tools/code-refactor/reuse-outcome.js +14 -4
- package/build/tools/code-refactor/reuse-state.d.ts +127 -5
- package/build/tools/code-refactor/reuse-state.js +628 -16
- package/build/tools/code-refactor/utils-verify-gates.d.ts +26 -0
- package/build/tools/code-refactor/utils-verify-gates.js +100 -0
- package/build/tools/code-refactor/verify-gates.d.ts +2 -1
- package/build/tools/code-refactor/verify-gates.js +90 -25
- package/build/tools/executeSkyrampTestTool.d.ts +19 -0
- package/build/tools/executeSkyrampTestTool.js +158 -8
- package/build/tools/generate-tests/generateBatchScenarioRestTool.js +2 -2
- package/build/tools/generate-tests/generateE2ERestTool.js +16 -0
- package/build/tools/generate-tests/generateUIRestTool.d.ts +1 -0
- package/build/tools/generate-tests/generateUIRestTool.js +22 -0
- package/build/tools/generate-tests/scenarioLint.d.ts +2 -0
- package/build/tools/generate-tests/scenarioLint.js +127 -19
- package/build/tools/generate-tests/trace-reuse-guard.d.ts +20 -0
- package/build/tools/generate-tests/trace-reuse-guard.js +93 -0
- package/build/tools/runExistingTestsTool.d.ts +34 -2
- package/build/tools/runExistingTestsTool.js +104 -4
- package/build/tools/submitReportTool.d.ts +38 -38
- package/build/tools/submitReportTool.js +537 -120
- package/build/tools/test-management/analyzeChangesTool.d.ts +24 -1
- package/build/tools/test-management/analyzeChangesTool.js +71 -10
- package/build/tools/test-management/analyzeTestHealthTool.js +7 -7
- package/build/tools/test-management/registerTestPlanTool.d.ts +203 -0
- package/build/tools/test-management/registerTestPlanTool.js +70 -12
- package/build/types/ExternalTestExecution.d.ts +67 -1
- package/build/types/Recommendation.d.ts +34 -5
- package/build/types/RepositoryAnalysis.d.ts +133 -114
- package/build/types/RepositoryAnalysis.js +1 -1
- package/build/types/ReuseOutcome.d.ts +102 -6
- package/build/types/ReuseOutcome.js +16 -2
- package/build/types/TestRecommendation.js +21 -3
- package/build/types/TestTypes.js +14 -8
- package/build/types/TestbotReport.d.ts +10 -1
- package/build/types/index.d.ts +2 -2
- package/build/types/index.js +1 -1
- package/build/utils/AnalysisStateManager.d.ts +57 -1
- package/build/utils/AnalysisStateManager.js +54 -5
- package/build/utils/branchDiff.d.ts +10 -0
- package/build/utils/branchDiff.js +28 -0
- package/build/utils/changedRoutes.d.ts +29 -0
- package/build/utils/changedRoutes.js +87 -0
- package/build/utils/featureFlags.d.ts +21 -0
- package/build/utils/featureFlags.js +23 -0
- package/build/utils/frontendIntegration.js +34 -4
- package/build/utils/importerHop.d.ts +2 -8
- package/build/utils/importerHop.js +15 -53
- package/build/utils/pathMatching.d.ts +38 -0
- package/build/utils/pathMatching.js +71 -0
- package/build/utils/pathSignatures.d.ts +22 -0
- package/build/utils/pathSignatures.js +57 -0
- package/build/utils/planMatchKeys.d.ts +16 -3
- package/build/utils/planMatchKeys.js +26 -10
- package/build/utils/pluralization.d.ts +10 -0
- package/build/utils/pluralization.js +18 -0
- package/build/utils/pom-catalog-parse.d.ts +52 -0
- package/build/utils/pom-catalog-parse.js +141 -0
- package/build/utils/pom-scope/selector-extractor.d.ts +12 -0
- package/build/utils/pom-scope/selector-extractor.js +34 -8
- package/build/utils/pom-verify/verify.d.ts +6 -5
- package/build/utils/pom-verify/verify.js +8 -6
- package/build/utils/reportVerification.d.ts +64 -4
- package/build/utils/reportVerification.js +228 -3
- package/build/utils/reuseRouting.d.ts +3 -0
- package/build/utils/reuseRouting.js +50 -0
- package/build/utils/routeParsers.d.ts +2 -0
- package/build/utils/routeParsers.js +65 -8
- package/build/utils/scenarioDrafting.d.ts +1 -1
- package/build/utils/scenarioDrafting.js +57 -45
- package/build/utils/subjectEndpoints.d.ts +19 -0
- package/build/utils/subjectEndpoints.js +98 -0
- package/build/utils/testFileClassification.d.ts +11 -0
- package/build/utils/testFileClassification.js +47 -0
- package/build/utils/uiPageEnumerator.d.ts +45 -19
- package/build/utils/uiPageEnumerator.js +95 -51
- package/build/utils/utils-verify/allow.d.ts +16 -0
- package/build/utils/utils-verify/allow.js +68 -0
- package/build/utils/utils-verify/call-sites.d.ts +34 -0
- package/build/utils/utils-verify/call-sites.js +154 -0
- package/build/utils/utils-verify/index.d.ts +7 -0
- package/build/utils/utils-verify/index.js +7 -0
- package/build/utils/utils-verify/language-spec.d.ts +91 -0
- package/build/utils/utils-verify/language-spec.js +210 -0
- package/build/utils/utils-verify/locate.d.ts +39 -0
- package/build/utils/utils-verify/locate.js +199 -0
- package/build/utils/utils-verify/parse.d.ts +34 -0
- package/build/utils/utils-verify/parse.js +177 -0
- package/build/utils/utils-verify/stage.d.ts +24 -0
- package/build/utils/utils-verify/stage.js +107 -0
- package/build/utils/utils-verify/verify.d.ts +63 -0
- package/build/utils/utils-verify/verify.js +168 -0
- package/build/utils/utils.d.ts +3 -1
- package/build/utils/utils.js +3 -1
- package/build/workspace/workspace.d.ts +32 -32
- package/node_modules/playwright/lib/mcp/skyramp/assertTool.js +9 -5
- package/node_modules/playwright/lib/mcp/skyramp/loadTraceTool.js +16 -0
- package/node_modules/playwright/lib/mcp/skyramp/skyRampImport.js +2 -0
- package/node_modules/playwright/lib/mcp/skyramp/traceRecordingBackend.js +115 -14
- package/node_modules/playwright/lib/mcp/test/skyRampExport.js +13 -1
- package/node_modules/playwright/node_modules/playwright-core/.DS_Store +0 -0
- package/node_modules/playwright/node_modules/playwright-core/lib/server/codegen/skyramp/jsonlReader.js +2 -0
- package/node_modules/playwright/node_modules/playwright-core/lib/vite/htmlReport/index.html +27 -253
- package/node_modules/playwright/node_modules/playwright-core/lib/vite/recorder/assets/{codeMirrorModule-DtudTj_v.js → codeMirrorModule-DJMC4zNo.js} +1 -1
- package/node_modules/playwright/node_modules/playwright-core/lib/vite/recorder/assets/index-BW82eAUI.js +196 -0
- package/node_modules/playwright/node_modules/playwright-core/lib/vite/recorder/index.html +1 -1
- package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/assets/{codeMirrorModule-FNMuBzX1.js → codeMirrorModule-CZfp96qZ.js} +1 -1
- package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/assets/defaultSettingsView-gpLo02E0.js +809 -0
- package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/index.Bq1r1URj.js +2 -0
- package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/index.html +2 -2
- package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/uiMode.VEfqi1qN.js +5 -0
- package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/uiMode.html +2 -2
- package/node_modules/playwright/node_modules/playwright-core/package.json +1 -1
- package/node_modules/playwright/node_modules/playwright-core/src/server/codegen/skyramp/jsonlReader.ts +1 -1
- package/node_modules/playwright/package.json +1 -1
- package/package.json +2 -2
- package/node_modules/playwright/node_modules/playwright-core/lib/vite/recorder/assets/index-BpDwp16L.js +0 -422
- package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/assets/defaultSettingsView-Co9upU5h.js +0 -1035
- package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/index.DXNIQ_dx.js +0 -2
- package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/uiMode.CIKB3XSv.js +0 -5
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { logger } from "../../utils/logger.js";
|
|
2
2
|
import { buildTestQualityCriteria } from "./recommendationSections.js";
|
|
3
|
-
import {
|
|
3
|
+
import { externalDedupKeys, isAttackSurfaceSecurityBoundary } from "./recommendationShared.js";
|
|
4
4
|
export function buildFullRepoRecommendations(scored, topN, baseUrl, authHeaderValue, authSchemeSnippet, authTypeValue, isFrontendProject = false, isFrontendOnlyProject = false, externalCoverage = new Set()) {
|
|
5
5
|
// Full-repo mode only — percentage-based UI/E2E slot targets (15% each, floor 1).
|
|
6
6
|
const rawE2E = isFrontendProject ? Math.max(1, Math.round(topN * 0.15)) : 0;
|
|
@@ -17,22 +17,29 @@ export function buildFullRepoRecommendations(scored, topN, baseUrl, authHeaderVa
|
|
|
17
17
|
: authHeaderValue
|
|
18
18
|
? `, authHeader: "${authHeaderValue}"`
|
|
19
19
|
: `, authHeader: <check OpenAPI securitySchemes or auth middleware; "" if confirmed unauthenticated>`;
|
|
20
|
-
// Supplement count for full-repo mode
|
|
21
|
-
const supplementCount = topN - Math.min(scored.length, topN);
|
|
22
20
|
const toTitle = (name) => name.replace(/-/g, " ").replace(/\b\w/g, c => c.toUpperCase());
|
|
23
21
|
const TYPE_ORDER = ["e2e", "ui", "integration", "contract"];
|
|
24
22
|
const TYPE_LABEL = {
|
|
25
23
|
e2e: "E2E", ui: "UI", integration: "Integration", contract: "Contract",
|
|
26
24
|
};
|
|
27
25
|
// Filter out scenarios already covered by external tests before slicing.
|
|
26
|
+
// The bug/attack-surface exemption matches applyExternalDedup in
|
|
27
|
+
// budgeters/shared.ts: those scenarios cover a semantic flaw that a
|
|
28
|
+
// same-endpoint external test does not, so an external match must not
|
|
29
|
+
// remove them here either.
|
|
28
30
|
const scoredFiltered = externalCoverage.size > 0
|
|
29
31
|
? scored.filter(item => {
|
|
30
|
-
const
|
|
31
|
-
if (
|
|
32
|
-
|
|
33
|
-
|
|
32
|
+
const keys = externalDedupKeys(item.scenario);
|
|
33
|
+
if (keys.length === 0)
|
|
34
|
+
return true;
|
|
35
|
+
if (!keys.every((key) => externalCoverage.has(key)))
|
|
36
|
+
return true;
|
|
37
|
+
if (item.scenario.category === "bug_caught" || isAttackSurfaceSecurityBoundary(item.scenario)) {
|
|
38
|
+
logger.info(`External dedup (full-repo): preserving "${item.scenario.scenarioName}" (${keys.join(", ")}) — protected bug/attack-surface scenario requires semantic flaw coverage`);
|
|
39
|
+
return true;
|
|
34
40
|
}
|
|
35
|
-
|
|
41
|
+
logger.info(`External dedup (full-repo): skipping "${item.scenario.scenarioName}" (${keys.join(", ")})`);
|
|
42
|
+
return false;
|
|
36
43
|
})
|
|
37
44
|
: scored;
|
|
38
45
|
// For full-stack repos, carve out E2E and UI slots before filling with backend tests.
|
|
@@ -40,6 +47,10 @@ export function buildFullRepoRecommendations(scored, topN, baseUrl, authHeaderVa
|
|
|
40
47
|
? Math.max(0, topN - minE2ESlots - minUISlots)
|
|
41
48
|
: topN;
|
|
42
49
|
const allItems = scoredFiltered.slice(0, backendSlotCount);
|
|
50
|
+
// Count the supplements against what this prompt actually lists. Reading the
|
|
51
|
+
// pre-filter `scored.length` claimed more pre-ranked items than the sections
|
|
52
|
+
// below render, so a pool thinned by external dedup asked for no supplements.
|
|
53
|
+
const supplementCount = topN - Math.min(allItems.length, topN);
|
|
43
54
|
const byType = new Map();
|
|
44
55
|
for (const t of TYPE_ORDER)
|
|
45
56
|
byType.set(t, []);
|
|
@@ -3,14 +3,36 @@
|
|
|
3
3
|
* full-repo mode (fullRepoCatalog.ts). Extracted here to avoid circular imports.
|
|
4
4
|
*/
|
|
5
5
|
import { DraftedScenario } from "../../types/RepositoryAnalysis.js";
|
|
6
|
-
export declare function scenarioCoverageKey(scenario: DraftedScenario): string;
|
|
7
6
|
/**
|
|
8
|
-
*
|
|
9
|
-
*
|
|
10
|
-
*
|
|
11
|
-
*
|
|
7
|
+
* The one spelling of a coverage key, `METHOD::resource::testType`. Both
|
|
8
|
+
* operands of the dedup comparison call it: a proposed test through
|
|
9
|
+
* `externalDedupKeys`, and an existing test through `buildExternalCoverageSet`.
|
|
10
|
+
* They used to build the same string in two places from two different inputs —
|
|
11
|
+
* a typed step list on one side, regex-scraped prose on the other — so a fix to
|
|
12
|
+
* one side left the other producing keys that could never match (SKYR-4214).
|
|
12
13
|
*/
|
|
13
|
-
export declare function
|
|
14
|
+
export declare function coverageKey(input: {
|
|
15
|
+
method: string;
|
|
16
|
+
path: string;
|
|
17
|
+
testType: string;
|
|
18
|
+
}): string;
|
|
19
|
+
/**
|
|
20
|
+
* Method-aware coverage keys for external test dedup — one per recorded subject
|
|
21
|
+
* endpoint. Method-aware so that an external test covering "GET /orders" does
|
|
22
|
+
* not block a test for "PUT /orders", a different operation on the same
|
|
23
|
+
* resource.
|
|
24
|
+
*
|
|
25
|
+
* A caller must remove a candidate ONLY when the coverage set holds EVERY key
|
|
26
|
+
* returned here. Removing it on one match would discard the coverage of its
|
|
27
|
+
* other endpoints. An empty list must never remove anything.
|
|
28
|
+
*/
|
|
29
|
+
export declare function externalDedupKeys(scenario: DraftedScenario): string[];
|
|
30
|
+
/**
|
|
31
|
+
* Resource+type keys (no method) for the GENERATE/ADDITIONAL overlap filter —
|
|
32
|
+
* one per recorded subject endpoint. Same rule as `externalDedupKeys`: remove
|
|
33
|
+
* only on ALL keys, never on an empty list.
|
|
34
|
+
*/
|
|
35
|
+
export declare function scenarioCoverageKeys(scenario: DraftedScenario): string[];
|
|
14
36
|
export declare function isAttackSurfaceSecurityBoundary(scenario: DraftedScenario): boolean;
|
|
15
37
|
export declare function isOrdinaryDirectAuthBoundary(scenario: DraftedScenario): boolean;
|
|
16
38
|
/**
|
|
@@ -10,22 +10,96 @@ function resolvePrimaryStep(scenario) {
|
|
|
10
10
|
const primaryStep = mutatingSteps[mutatingSteps.length - 1] ?? scenario.steps[scenario.steps.length - 1];
|
|
11
11
|
return { primaryStep, testType };
|
|
12
12
|
}
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
13
|
+
/**
|
|
14
|
+
* The one spelling of a coverage key, `METHOD::resource::testType`. Both
|
|
15
|
+
* operands of the dedup comparison call it: a proposed test through
|
|
16
|
+
* `externalDedupKeys`, and an existing test through `buildExternalCoverageSet`.
|
|
17
|
+
* They used to build the same string in two places from two different inputs —
|
|
18
|
+
* a typed step list on one side, regex-scraped prose on the other — so a fix to
|
|
19
|
+
* one side left the other producing keys that could never match (SKYR-4214).
|
|
20
|
+
*/
|
|
21
|
+
export function coverageKey(input) {
|
|
22
|
+
const method = (input.method ?? "GET").toUpperCase();
|
|
23
|
+
return `${method}::${extractResourceFromPath(input.path ?? "")}::${input.testType}`;
|
|
24
|
+
}
|
|
25
|
+
/** Test type of a scenario: its own label, else one step means a contract test. */
|
|
26
|
+
function resolveTestType(scenario) {
|
|
27
|
+
return scenario.testType ?? (scenario.steps.length === 1 ? "contract" : "integration");
|
|
28
|
+
}
|
|
29
|
+
/**
|
|
30
|
+
* Method-aware coverage keys for external test dedup — one per recorded subject
|
|
31
|
+
* endpoint. Method-aware so that an external test covering "GET /orders" does
|
|
32
|
+
* not block a test for "PUT /orders", a different operation on the same
|
|
33
|
+
* resource.
|
|
34
|
+
*
|
|
35
|
+
* A caller must remove a candidate ONLY when the coverage set holds EVERY key
|
|
36
|
+
* returned here. Removing it on one match would discard the coverage of its
|
|
37
|
+
* other endpoints. An empty list must never remove anything.
|
|
38
|
+
*/
|
|
39
|
+
export function externalDedupKeys(scenario) {
|
|
40
|
+
const testType = resolveTestType(scenario);
|
|
41
|
+
const subjects = scenario.subjectEndpoints;
|
|
42
|
+
if (subjects && subjects.length > 0) {
|
|
43
|
+
return subjects.map((s) => coverageKey({ ...s, testType }));
|
|
44
|
+
}
|
|
45
|
+
if (subjects)
|
|
46
|
+
return [];
|
|
47
|
+
const { primaryStep } = resolvePrimaryStep(scenario);
|
|
48
|
+
if (!primaryStep)
|
|
49
|
+
return [];
|
|
50
|
+
return [coverageKey({ method: primaryStep.method, path: primaryStep.path, testType })];
|
|
51
|
+
}
|
|
52
|
+
/**
|
|
53
|
+
* Resource+type keys (no method) for the GENERATE/ADDITIONAL overlap filter —
|
|
54
|
+
* one per recorded subject endpoint. Same rule as `externalDedupKeys`: remove
|
|
55
|
+
* only on ALL keys, never on an empty list.
|
|
56
|
+
*/
|
|
57
|
+
export function scenarioCoverageKeys(scenario) {
|
|
58
|
+
const testType = resolveTestType(scenario);
|
|
59
|
+
const steps = scenario.steps ?? [];
|
|
60
|
+
const subjects = scenario.subjectEndpoints;
|
|
61
|
+
// No recorded subject: fall back to the primary step, as externalDedupKeys does.
|
|
62
|
+
if (!subjects) {
|
|
63
|
+
const { primaryStep } = resolvePrimaryStep(scenario);
|
|
64
|
+
if (!primaryStep)
|
|
65
|
+
return [];
|
|
66
|
+
return [
|
|
67
|
+
overlapKey(primaryStep.method, extractResourceFromPath(primaryStep.path ?? ""), testType, primaryStep),
|
|
68
|
+
];
|
|
69
|
+
}
|
|
70
|
+
return subjects.map((subject) => {
|
|
71
|
+
const step = steps.find((s) => s.method === subject.method && s.path === subject.path);
|
|
72
|
+
return overlapKey(subject.method, extractResourceFromPath(subject.path ?? ""), testType, step);
|
|
73
|
+
});
|
|
17
74
|
}
|
|
18
75
|
/**
|
|
19
|
-
*
|
|
20
|
-
*
|
|
21
|
-
*
|
|
22
|
-
*
|
|
76
|
+
* The overlap key names the endpoint AND what the test asserts about it.
|
|
77
|
+
*
|
|
78
|
+
* It deliberately does NOT reuse `coverageKey`. That key's other operand comes
|
|
79
|
+
* from `buildExternalCoverageSet`, which scrapes prose and can only ever know a
|
|
80
|
+
* method and a path — adding a status there would stop every external key from
|
|
81
|
+
* matching. Both operands of the OVERLAP comparison are drafted candidates with
|
|
82
|
+
* a step list, so both sides know the method, the interaction type and the
|
|
83
|
+
* status.
|
|
84
|
+
*
|
|
85
|
+
* The method is part of the key because `extractResourceFromPath` maps
|
|
86
|
+
* `/api/orders` and `/api/orders/{id}` to the same resource. Without it a
|
|
87
|
+
* GENERATE `PATCH /api/orders/{id}` (success, 200) removed an ADDITIONAL
|
|
88
|
+
* `GET /api/orders/{id}` (success, 200) as a duplicate — a read test and an
|
|
89
|
+
* update test on one resource are not duplicates.
|
|
90
|
+
*
|
|
91
|
+
* Measured on 839 nightly fixtures: 781 of them change exactly ONE route, so
|
|
92
|
+
* every candidate resolves to the same subject and `resource::testType` yields
|
|
93
|
+
* 2 keys for the whole pool. In 11 fixtures that removed EVERY ADDITIONAL
|
|
94
|
+
* candidate — `01-site-stats-endpoint` lost all 18. Adding the interaction type
|
|
95
|
+
* and the status roughly doubles the distinct keys on those pools. It does not
|
|
96
|
+
* make the key complete: two tests that assert different things about the same
|
|
97
|
+
* 200 response still collide, and no key built from a route can separate them.
|
|
23
98
|
*/
|
|
24
|
-
|
|
25
|
-
const
|
|
26
|
-
const
|
|
27
|
-
|
|
28
|
-
return `${method}::${resource}::${testType}`;
|
|
99
|
+
function overlapKey(method, resource, testType, step) {
|
|
100
|
+
const interaction = step?.interactionType ?? "any";
|
|
101
|
+
const status = step?.expectedStatusCode ?? 0;
|
|
102
|
+
return `${(method ?? "ANY").toUpperCase()}::${resource}::${testType}::${interaction}::${status}`;
|
|
29
103
|
}
|
|
30
104
|
export function isAttackSurfaceSecurityBoundary(scenario) {
|
|
31
105
|
return scenario.category === "security_boundary" &&
|
|
@@ -60,11 +134,11 @@ export function buildExternalCoverageSet(testLocations) {
|
|
|
60
134
|
const resource = extractResourceFromPath(epPath);
|
|
61
135
|
if (resource !== "unknown") {
|
|
62
136
|
if (testType === "unknown") {
|
|
63
|
-
coverage.add(
|
|
64
|
-
coverage.add(
|
|
137
|
+
coverage.add(coverageKey({ method, path: epPath, testType: "integration" }));
|
|
138
|
+
coverage.add(coverageKey({ method, path: epPath, testType: "contract" }));
|
|
65
139
|
}
|
|
66
140
|
else {
|
|
67
|
-
coverage.add(
|
|
141
|
+
coverage.add(coverageKey({ method, path: epPath, testType }));
|
|
68
142
|
}
|
|
69
143
|
}
|
|
70
144
|
}
|
|
@@ -5,6 +5,18 @@ import { buildRecommendationPrompt } from "./test-recommendation-prompt.js";
|
|
|
5
5
|
import { ScenarioSource, AnalysisScope } from "../../types/RepositoryAnalysis.js";
|
|
6
6
|
import { SCENARIO_CATEGORIES } from "../../types/TestRecommendation.js";
|
|
7
7
|
import { inferExpectedStatus } from "../../utils/httpDefaults.js";
|
|
8
|
+
/**
|
|
9
|
+
* True when two step lists name the same method+path sequence, ignoring
|
|
10
|
+
* everything else (body, description, ...). Used to decide whether a
|
|
11
|
+
* previously-resolved `subjectEndpoints` still applies to an enriched
|
|
12
|
+
* replacement — this function has no PR diff in scope, so it cannot call
|
|
13
|
+
* `resolveSubjectEndpoints` itself (SKYR-4214).
|
|
14
|
+
*/
|
|
15
|
+
function sameStepRouteSequence(a, b) {
|
|
16
|
+
if (a.length !== b.length)
|
|
17
|
+
return false;
|
|
18
|
+
return a.every((step, i) => (step.method ?? "").toUpperCase() === (b[i].method ?? "").toUpperCase() && step.path === b[i].path);
|
|
19
|
+
}
|
|
8
20
|
export function mergeEnrichedScenarios(serverScenarios, raw) {
|
|
9
21
|
const rejectionNotes = [];
|
|
10
22
|
let parsed;
|
|
@@ -72,6 +84,16 @@ export function mergeEnrichedScenarios(serverScenarios, raw) {
|
|
|
72
84
|
}
|
|
73
85
|
const merged = new Map(serverScenarios.map(s => [s.scenarioName, s]));
|
|
74
86
|
for (const s of agentScenarios) {
|
|
87
|
+
const replaced = merged.get(s.scenarioName);
|
|
88
|
+
// The agent-submitted copy never carries subjectEndpoints — it is built
|
|
89
|
+
// fresh from the enriched JSON, not through resolveSubjectEndpoints. When
|
|
90
|
+
// the replaced server scenario had one and the steps still name the same
|
|
91
|
+
// endpoints, carry it forward so this candidate keeps the one recorded
|
|
92
|
+
// subject instead of falling back to a fresh last-mutating-step guess
|
|
93
|
+
// that could disagree with its siblings (SKYR-4214).
|
|
94
|
+
if (replaced?.subjectEndpoints && sameStepRouteSequence(replaced.steps, s.steps)) {
|
|
95
|
+
s.subjectEndpoints = replaced.subjectEndpoints;
|
|
96
|
+
}
|
|
75
97
|
merged.set(s.scenarioName, s);
|
|
76
98
|
}
|
|
77
99
|
logger.info("Merged agent-enriched scenarios", {
|
|
@@ -2,8 +2,8 @@ import { RepositoryAnalysis, AnalysisScope, DraftedScenario } from "../../types/
|
|
|
2
2
|
import { WorkspaceAuthType } from "../../utils/workspaceAuth.js";
|
|
3
3
|
import { PRTestContext } from "../../utils/pr-comment-parser.js";
|
|
4
4
|
import { Novelty, PriorityTier } from "../../types/TestRecommendation.js";
|
|
5
|
-
import { buildExternalCoverageSet,
|
|
6
|
-
export { buildExternalCoverageSet,
|
|
5
|
+
import { buildExternalCoverageSet, externalDedupKeys } from "./recommendationShared.js";
|
|
6
|
+
export { buildExternalCoverageSet, externalDedupKeys };
|
|
7
7
|
/** Result of {@link computeScoredCandidates} — the scoring/classification
|
|
8
8
|
* inputs shared between prompt rendering and the SKYR-3879 register-plan
|
|
9
9
|
* pre-seed (analyzeChangesTool.ts), so both derive the exact same numbers. */
|
|
@@ -8,9 +8,9 @@ import { buildScopeAssessmentSection, isFrontendFile } from "./scopeAssessment.j
|
|
|
8
8
|
import { buildExecutionPlan, EXEC_STEP_CODE_REVIEW, EXEC_STEP_ENRICH } from "./diffExecutionPlan.js";
|
|
9
9
|
import { buildFullRepoRecommendations } from "./fullRepoCatalog.js";
|
|
10
10
|
import { ANALYSIS_STEP_EXTRACT } from "./analysisOutputPrompt.js";
|
|
11
|
-
import { TASK_GENERATE, buildExternalCoverageSet,
|
|
11
|
+
import { TASK_GENERATE, buildExternalCoverageSet, externalDedupKeys, isAttackSurfaceSecurityBoundary, taskRef, } from "./recommendationShared.js";
|
|
12
12
|
// Re-export for backward compatibility (tests and external callers import these from this module)
|
|
13
|
-
export { buildExternalCoverageSet,
|
|
13
|
+
export { buildExternalCoverageSet, externalDedupKeys };
|
|
14
14
|
function formatTestLocations(locs) {
|
|
15
15
|
const entries = Object.entries(locs || {});
|
|
16
16
|
if (entries.length === 0)
|
|
@@ -23,7 +23,7 @@ function formatTestLocations(locs) {
|
|
|
23
23
|
"**Deduplication rule (apply this table before generating anything):**\n" +
|
|
24
24
|
"- `[external]` tests: if a resource is covered by an `[external]` test, do NOT create a new parallel test for the same HTTP method + resource + test type. These tests still break when the API changes — Task 1 maintenance applies to them the same as Skyramp tests (in-place UPDATE only; do not regenerate or delete).\n" +
|
|
25
25
|
"- `[skyramp]` contract test: if the HTTP method + path already appears in a `[skyramp]` `covers:` entry of type `contract` → UPDATE that file, do NOT create a new one.\n" +
|
|
26
|
-
"- `[skyramp]` integration test:
|
|
26
|
+
"- `[skyramp]` integration test: use the endpoints this PR changed. A setup step or a cleanup step is not an endpoint under test. A scenario can test more than one changed endpoint. UPDATE an existing `[skyramp]` `covers:` entry of type `integration` only when it already covers EVERY changed endpoint the scenario tests. If it covers some but not all of them, the scenario is not a duplicate — create it.\n" +
|
|
27
27
|
"- UI/E2E test: always create a new file — traces are distinct recordings.\n" +
|
|
28
28
|
"For `[skyramp]` contract and integration tests: if in doubt, prefer UPDATE over creating a duplicate.");
|
|
29
29
|
}
|
|
@@ -2,12 +2,12 @@ import { z } from "zod";
|
|
|
2
2
|
import { logger } from "../../utils/logger.js";
|
|
3
3
|
import { AnalyticsService } from "../../services/AnalyticsService.js";
|
|
4
4
|
import { MAX_TESTS_TO_GENERATE, MAX_RECOMMENDATIONS, MAX_CRITICAL_TESTS, PATH_PARAM_UUID_GUIDANCE, AUTH_CONFLICT_ERROR_MSG, } from "../test-recommendation/recommendationSections.js";
|
|
5
|
-
import { TASK_ANALYZE_MAINTAIN, TASK_GENERATE, TASK_SUBMIT, taskRef } from "../test-recommendation/recommendationShared.js";
|
|
6
|
-
import { getTraceRecordingPromptText } from "../../playwright/traceRecordingPrompt.js";
|
|
7
|
-
import { isContractConsumerModeEnabled, isPomReuseEnabled } from "../../utils/featureFlags.js";
|
|
8
5
|
import { setReportLanguage } from "../../utils/reportLanguage.js";
|
|
6
|
+
import { TASK_ANALYZE_MAINTAIN, TASK_GENERATE, TASK_SUBMIT, taskRef, } from "../test-recommendation/recommendationShared.js";
|
|
7
|
+
import { getTraceRecordingPromptText } from "../../playwright/traceRecordingPrompt.js";
|
|
8
|
+
import { isContractConsumerModeEnabled, isPomReuseEnabled, isUtilsReuseEnabled, } from "../../utils/featureFlags.js";
|
|
9
9
|
import { resolveServiceDetailsRef } from "../../utils/utils.js";
|
|
10
|
-
import { buildServiceContext, readWorkspaceServices
|
|
10
|
+
import { buildServiceContext, readWorkspaceServices } from "../prompt-utils.js";
|
|
11
11
|
// Cached at module-load — flags are process-wide and cannot change per call.
|
|
12
12
|
const CONSUMER_MODE_ENABLED = isContractConsumerModeEnabled();
|
|
13
13
|
const SERVICE_REFS = resolveServiceDetailsRef();
|
|
@@ -20,13 +20,46 @@ const CONTRACT_MODE_GUIDANCE = CONSUMER_MODE_ENABLED
|
|
|
20
20
|
Both modes (\`providerMode: true, consumerMode: true\`): For diff that contains BOTH provider signals (such as new/modified endpoint handlers, route changes this service owns) AND consumer signals (outbound HTTP client calls to another service, no new endpoint handlers).`
|
|
21
21
|
: ` Always add \`providerMode: true\` — the tool generates provider-side contract tests only.`;
|
|
22
22
|
const POM_REUSE_ENABLED = isPomReuseEnabled();
|
|
23
|
+
// Utils reuse (SkyrampUtils consolidation for integration tests, and for UI tests
|
|
24
|
+
// while the POM path is off) is customer-gated,
|
|
25
|
+
// default OFF — like the POM flag, its value is baked into the prompt at module load,
|
|
26
|
+
// so the MCP server must be restarted after flipping it.
|
|
27
|
+
const UTILS_REUSE_ENABLED = isUtilsReuseEnabled();
|
|
28
|
+
// The UI flow seeds a shared utils file (modularize-first) only with utils reuse
|
|
29
|
+
// on and the POM path off — the same condition isModularizeFirstTarget applies
|
|
30
|
+
// server-side. Every UI-chain statement in this prompt keys on this constant so
|
|
31
|
+
// no line can ban the modularization call the generation result instructs.
|
|
32
|
+
const UI_UTILS_REUSE = UTILS_REUSE_ENABLED && !POM_REUSE_ENABLED;
|
|
33
|
+
// Step 4 of the post-generation list. On the modularize-first UI flow the
|
|
34
|
+
// enhance call is the first sub-step of the generation result's chain — but
|
|
35
|
+
// that chain exists only for a TS/JS UI generation with enhanceAssertions on,
|
|
36
|
+
// which this prompt cannot see per call. So the step keeps the call and lets
|
|
37
|
+
// the agent skip it only when it already made it for that file.
|
|
38
|
+
const UI_ENHANCE_STEP = UI_UTILS_REUSE
|
|
39
|
+
? `Call \`skyramp_enhance_assertions\` with \`testFile\` set to the absolute path of the generated UI test file, \`testType: "ui"\`, and \`enhanceType: "generation"\`, and apply every instruction returned to that file — UNLESS you already called it for this file as the first sub-step of step 3's chain, in which case skip it (never enhance the same file twice; never leave a generated UI test un-enhanced). The guidance below on which assertions to add applies to whichever call enhances the file.`
|
|
40
|
+
: `Call \`skyramp_enhance_assertions\` with \`testFile\` set to the absolute path of the generated UI test file, \`testType: "ui"\`, and \`enhanceType: "generation"\`. Apply every instruction returned to that file.`;
|
|
23
41
|
// Post-generation code-reuse step for UI tests. The POM catalog, the `verify: true`
|
|
24
42
|
// loop and the `.raw.bak` restore only exist on the POM-aware path — when that path
|
|
25
43
|
// is flagged off, `skyramp_reuse_code` returns the SkyrampUtils workflow instead, so
|
|
26
44
|
// the agent must not be sent looking for artifacts nothing produces.
|
|
27
45
|
const UI_CODE_REUSE_STEP = POM_REUSE_ENABLED
|
|
28
46
|
? `3. **[MANDATORY] After \`skyramp_ui_test_generation\` with \`codeReuse: true\`**: the generation result directs you to call \`skyramp_reuse_code\` — do this BEFORE \`skyramp_enhance_assertions\`. Two outcomes: (1) a response starting "No reusable POM layer detected" — this is a normal outcome, continue immediately (do NOT retry); (2) a refactoring workflow — follow it to completion INCLUDING its verification loop (\`skyramp_reuse_code\` with \`verify: true\`), finish only when it reports PASSED. If a reused test later fails execution and the failure points at a substituted POM call, restore the saved \`<testFile>.raw.bak\` over the test file and re-run — do NOT hand-edit the customer's POM methods (this re-run counts toward the 2-attempt execution cap — prefer this restore over the generic timeout fix-up when the failing locator came from a POM substitution). **Cleanup before reporting:** \`.raw.bak\` files are internal scratch — after ALL test executions are complete (pass or fail) and before calling \`skyramp_submit_report\`, delete every \`*.raw.bak\` you created so they are not committed to the customer-facing branch. The \`skyramp-pom-catalog.md\` is NOT scratch — leave it in place (later runs reuse it).`
|
|
29
|
-
:
|
|
47
|
+
: UI_UTILS_REUSE
|
|
48
|
+
? `3. **[MANDATORY] After \`skyramp_ui_test_generation\` with \`codeReuse: true\`**: the generation result carries three CRITICAL NEXT STEPS — \`skyramp_enhance_assertions\` (on the freshly generated file, while its selectors are still inline), then \`skyramp_modularization\`, then \`skyramp_reuse_code\` — with the exact arguments for each. Do all three, in that order and to completion, for EVERY generated UI test file — the enhance sub-step is that file's step 4 below (do not call \`skyramp_enhance_assertions\` a second time for the same file, and do not skip it for later files). Sibling tests offering nothing to reuse is a normal outcome, but the reuse steps still move this test's own helpers into the shared utils file and import them back, so the test file IS expected to change. Then continue (do NOT retry).`
|
|
49
|
+
: `3. **[MANDATORY] After \`skyramp_ui_test_generation\` with \`codeReuse: true\`**: the generation result directs you to call \`skyramp_reuse_code\` — do this BEFORE \`skyramp_enhance_assertions\`. Follow the returned steps exactly. If it finds no existing helper functions to reuse, that is a normal outcome — leave the test file unchanged and continue immediately (do NOT retry).`;
|
|
50
|
+
// Generation-call clause for the integration pipeline: only ask for codeReuse when
|
|
51
|
+
// utils reuse is enabled — with the flag off, integration generation behaves exactly
|
|
52
|
+
// as it did before the feature existed (no file-rewriting post-steps).
|
|
53
|
+
const INTEGRATION_CODE_REUSE_GEN_CLAUSE = UTILS_REUSE_ENABLED
|
|
54
|
+
? `, setting \`modularizeCode: false\` and \`codeReuse: true\` (TypeScript/JavaScript/Python — leave \`codeReuse\` unset for Java: cross-file helpers do not compile in the test executor)`
|
|
55
|
+
: ``;
|
|
56
|
+
// The post-generation modularize → reuse steps are NOT listed here: the
|
|
57
|
+
// generation tool's own result emits them (see TestGenerationService) when called
|
|
58
|
+
// with codeReuse: true on a modularize-first flow (routing: isModularizeFirstTarget).
|
|
59
|
+
// The UI clauses below keep `modularizeCode: false` deliberately: the
|
|
60
|
+
// modularization happens via that hand-off, not the flag, so the instructions
|
|
61
|
+
// exist only on runs that actually generated that test type. This prompt's job
|
|
62
|
+
// is only the generation-call clauses that ask for codeReuse.
|
|
30
63
|
/**
|
|
31
64
|
* Parse the JSON-encoded `relatedRepositories` argument passed via the testbot
|
|
32
65
|
* prompt/resource. Returns undefined for missing/blank input or malformed JSON so the
|
|
@@ -85,10 +118,11 @@ export function getTestbotPrompt(opts) {
|
|
|
85
118
|
// Set on EVERY render, not just non-English ones: last render wins, so an
|
|
86
119
|
// en/argless render disarms a language captured earlier in a long-lived
|
|
87
120
|
// server process instead of falsely rejecting an English report.
|
|
88
|
-
setReportLanguage(language && language !==
|
|
89
|
-
let reportLanguageBlock =
|
|
90
|
-
if (language && language !==
|
|
91
|
-
const reportLanguageName = new Intl.DisplayNames([
|
|
121
|
+
setReportLanguage(language && language !== "en" ? language : undefined);
|
|
122
|
+
let reportLanguageBlock = "";
|
|
123
|
+
if (language && language !== "en") {
|
|
124
|
+
const reportLanguageName = new Intl.DisplayNames(["en"], { type: "language" }).of(language) ??
|
|
125
|
+
language;
|
|
92
126
|
reportLanguageBlock = `**Report language: ${reportLanguageName}.** Write ALL user-facing free-text report fields in ${reportLanguageName}: \`businessCaseAnalysis\`, every \`description\` and \`reasoning\`, \`testResults[].details\`, \`issuesFound[].description\`, \`nextSteps[]\`, test-maintenance \`beforeDetails\`/\`afterDetails\`, and \`commitMessage\`. Do NOT translate: code identifiers, endpoint paths, file names, test IDs, \`scenarioName\`, enum values (\`Pass\`/\`Fail\`/\`Skipped\`, severity values, \`testType\`), or anything inside backticks.
|
|
93
127
|
|
|
94
128
|
`;
|
|
@@ -143,7 +177,8 @@ Use those recommendations as your baseline. Only add or remove tests that the us
|
|
|
143
177
|
|
|
144
178
|
**If \`skyramp_analyze_changes\` returns an error:** retry once only if the error is transient (timeout, network blip, temporary unavailability) — do NOT retry for permanent errors (invalid repository path, missing required parameter, authentication failure). If it fails again, call \`skyramp_submit_report\` with a minimal valid payload: leave all test arrays empty and add the error to \`issuesFound\`. Refer to the \`skyramp_submit_report\` schema for required fields. Do NOT attempt Task 2 without a valid stateFile.
|
|
145
179
|
**If all changed files are non-application** (CI/CD, docs, lock files, config) → skip to Task 3 (Submit Report) with empty arrays. Put the one-paragraph summary in \`businessCaseAnalysis\` (always populated; that's where end-state narration belongs); leave \`issuesFound\` empty — a non-application diff is not an issue. Example narration for a Testbot onboarding PR (\`.github/workflows/skyramp-testbot.yml\` and/or files under \`.skyramp/\`): "This PR adds Skyramp Testbot GitHub Actions workflow configuration to enable automated test generation on every pull request. It also adds System Under Test (SUT) setup files under \`.skyramp/sut/\` required for the testbot workflow, to bring up services for testing. It contains no application code changes and has no testable behavioral surface."
|
|
146
|
-
${hasRelatedRepos
|
|
180
|
+
${hasRelatedRepos
|
|
181
|
+
? `
|
|
147
182
|
**MULTI-REPO CONTEXT (MANDATORY).** This run includes ${relatedRepositories.length} related ${relatedRepositories.length === 1 ? "repository" : "repositories"} listed in the \`<related_repositories>\` block below, each with an explicit \`repository\` (\`owner/repo\`), \`path\`, and \`base_branch\`. Use the \`repository\` value verbatim — do NOT infer it from git remotes or paths. You MUST analyze EACH related repository — exactly one \`skyramp_analyze_changes\` call per listed repo (${relatedRepositories.length} ${relatedRepositories.length === 1 ? "call" : "calls"}), in addition to the primary call in step 2.
|
|
148
183
|
|
|
149
184
|
**Run the primary call (step 2) FIRST, then the related repos in listed order — not in parallel.** All calls in this run automatically share ONE run-scoped state file — you do NOT pass a state-file path; setting \`repository\` is enough. The primary writes its root section; each related repo's call upserts its own section into that same file. Concurrent calls would race on the shared file, so they must be sequential. For each related repo:
|
|
@@ -162,7 +197,8 @@ ${hasRelatedRepos ? `
|
|
|
162
197
|
Within a type, higher score wins regardless of which repo it came from. Everything not selected becomes an ADDITIONAL recommendation. This guarantees a frontend-only primary repo cannot starve a related backend repo's contract/integration tests of GENERATE slots (and vice versa). When you generate a test for a related repo's endpoint:
|
|
163
198
|
- **Execute it only if that repo's service is already running and reachable.** The workflow's setup may have started multiple services; before generating an API test for a related repo, confirm its \`base_url\` (from that repo's workspace/Execution Plan) responds. If the service is unreachable, still GENERATE the test but mark its \`testResults\` status as \`Skipped\` with details "service not running in this run" — do NOT count an unreachable service as a failure.
|
|
164
199
|
- **Write the test file into that service's own \`testDirectory\`** — the one declared for the service in the unified workspace.yml (the related repo's services were registered there in step 1(a), each with its \`repository\`). The \`testDirectory\` is interpreted relative to the **single delivery root** (the configured test repo if set, otherwise the primary repo), so all generated tests are delivered together by the existing single-target delivery. Do NOT invent a per-source-repo subdirectory, and do NOT write into the related repo's own checkout — it is read-only context. (If two repos happen to declare the same \`testDirectory\`, their files coexist there; the \`repository\` field on each report item — below — is what attributes ownership, not the path.)
|
|
165
|
-
- **Set the \`repository\` field** (\`owner/repo\`) on every such \`newTestsCreated\` / \`testResults\` item so the report attributes it to the originating repo (see Report Guidelines).`
|
|
200
|
+
- **Set the \`repository\` field** (\`owner/repo\`) on every such \`newTestsCreated\` / \`testResults\` item so the report attributes it to the originating repo (see Report Guidelines).`
|
|
201
|
+
: ""}
|
|
166
202
|
|
|
167
203
|
2. **Maintain existing tests:**
|
|
168
204
|
|
|
@@ -176,9 +212,9 @@ ${maintenanceBeforeExecStep}
|
|
|
176
212
|
|
|
177
213
|
e. Call \`skyramp_actions\` with \`stateFile\` (from \`skyramp_analyze_changes\` output) and apply the edits it returns.
|
|
178
214
|
|
|
179
|
-
f. Verify external-test fixes. **This step is not optional and it is the easiest one to forget — you have just edited files in step 2(e), so come back here before you move on to anything else.** It applies whenever step 2(a) reported a real pass/fail result for a file you then edited. It does NOT apply when step 2(a) returned \`skipped: true\` or \`ran: 0\` for every suite — there is no baseline to compare against, so say so in your report instead of re-running. When it applies: re-run those \`[external]\` files with \`skyramp_run_existing_tests\` (\`mode: "verify"\`, \`stateFile\`)
|
|
215
|
+
f. Verify external-test fixes. **This step is not optional and it is the easiest one to forget — you have just edited files in step 2(e), so come back here before you move on to anything else.** It applies whenever step 2(a) reported a real pass/fail result for a file you then edited. It does NOT apply when step 2(a) returned \`skipped: true\` or \`ran: 0\` for every suite — there is no baseline to compare against, so say so in your report instead of re-running. When it applies: re-run those \`[external]\` files with \`skyramp_run_existing_tests\` (\`mode: "verify"\`, \`stateFile\`) — the server reads each file's result back as its \`afterStatus\`. Editing an \`[external]\` file that step 2(a) confirmed failing and NOT re-running it leaves your own fix unverified — you would be reporting a repair you never saw work. A still-failing verify is surfaced in the report — do not loop.
|
|
180
216
|
|
|
181
|
-
3. **Code review:**
|
|
217
|
+
3. **Code review:** Find the logic bugs in the code that this change touches. Read the implementation of each changed endpoint: the route handler, and the functions that it calls to read or write data. For a changed screen, read the component and the functions that it calls. Read these files even when the diff does not contain them — a defect often sits in the code that the change depends on. Report each finding in \`issuesFound\` with a severity, and say which file and line holds it. Common patterns to flag:
|
|
182
218
|
- Computed fields not recalculated after mutation (e.g. \`total_amount\` unchanged after items are added/removed)
|
|
183
219
|
- Incomplete CRUD: create without cleanup, update that adds new records without removing old ones
|
|
184
220
|
- Missing input validation on new endpoints
|
|
@@ -195,8 +231,6 @@ ${maintenanceBeforeExecStep}
|
|
|
195
231
|
"role": "button",
|
|
196
232
|
"accessibleName": "Save changes",
|
|
197
233
|
"testId": "save-changes-btn",
|
|
198
|
-
"stableId": null,
|
|
199
|
-
"contextText": null,
|
|
200
234
|
"mutability": "mutable",
|
|
201
235
|
"widgetType": "native"
|
|
202
236
|
}
|
|
@@ -204,10 +238,10 @@ ${maintenanceBeforeExecStep}
|
|
|
204
238
|
\`\`\`
|
|
205
239
|
- **One element when the test has a single dominant target** (a click, a type, a single visibility check). Most tests fall here — use a length-1 array.
|
|
206
240
|
- **Multiple elements when the test verifies several elements together** — render-state tests (heading + input + button on a form), workflow tests (click button A, assert state appears in element B), or form-fill tests (multiple inputs + submit button). Each element is its own array entry.
|
|
207
|
-
- Each element's fields
|
|
241
|
+
- Each element's fields come from a captured blueprint element. Copy the values that are present, and never invent one. \`role\` and \`accessibleName\` are always in the capture; the capture omits \`testId\`/\`stableId\`/\`contextText\` when the element has no such value, so omit them too.
|
|
208
242
|
- \`mutability\` — copy from \`blueprint.element.mutability\`. \`'mutable'\` = behavioral-test target; \`'immutable'\` = smoke target.
|
|
209
243
|
- \`widgetType\` — copy from \`blueprint.element.widgetType\`. \`'custom'\` = JavaScript-composite control (Radix, MUI, etc.) requiring click-to-open interaction; \`'native'\` = standard HTML element.
|
|
210
|
-
- \`contextText\` — only for elements inside repeating sections (table rows, list items). Lift from \`repeatingElement.items[].contextText\`.
|
|
244
|
+
- \`contextText\` — only for elements inside repeating sections (table rows, list items). Lift from \`repeatingElement.items[].contextText\`. Omit it otherwise.
|
|
211
245
|
|
|
212
246
|
**Field 2 — \`pageContext\`** (where the test runs):
|
|
213
247
|
\`\`\`json
|
|
@@ -359,19 +393,21 @@ ${maintenanceBeforeExecStep}
|
|
|
359
393
|
|
|
360
394
|
**No upstream captures available?** Set \`targetElements\` to \`null\`, omit \`pageContext\`, and prefix \`description\` and \`reasoning\` with \`[no-blueprint-data]\`. Use page/feature-level prose; don't cite specific element names without grounding. Apply the marker per entry, not per PR — affected recs only. Log capture failures in \`issuesFound\` (one info-severity entry per failure mode, naming counts). Don't pre-emptively fall back without attempting capture first. Non-UI work is unaffected.
|
|
361
395
|
`;
|
|
362
|
-
const serviceContext = services?.length ? buildServiceContext(services) :
|
|
396
|
+
const serviceContext = services?.length ? buildServiceContext(services) : "";
|
|
363
397
|
// The <ui-credentials> tags are framing for the agent's prompt context —
|
|
364
398
|
// not real XML — so credentials pass through verbatim and the agent can
|
|
365
399
|
// type them directly into login fields. Reject the only string that would
|
|
366
400
|
// break the framing: a credential containing the closing tag itself.
|
|
367
401
|
const trimmedCredentials = uiCredentials?.trim();
|
|
368
|
-
if (trimmedCredentials && trimmedCredentials.includes(
|
|
402
|
+
if (trimmedCredentials && trimmedCredentials.includes("</ui-credentials>")) {
|
|
369
403
|
throw new Error("uiCredentials must not contain '</ui-credentials>'");
|
|
370
404
|
}
|
|
371
405
|
const uiCredentialsBlock = trimmedCredentials
|
|
372
406
|
? `<ui-credentials>\n${trimmedCredentials}\n</ui-credentials>`
|
|
373
|
-
:
|
|
374
|
-
const testsRepoDirBlock = testsRepoDir
|
|
407
|
+
: "";
|
|
408
|
+
const testsRepoDirBlock = testsRepoDir
|
|
409
|
+
? `<TESTS REPO DIR>${testsRepoDir}</TESTS REPO DIR>\n`
|
|
410
|
+
: "";
|
|
375
411
|
// Multi-repo context block. Each entry's `repositoryPath` is checked out at its
|
|
376
412
|
// FEATURE ref; `baseBranch` is that repo's default branch (the diff base), so
|
|
377
413
|
// skyramp_analyze_changes computes a real default…feature diff for the related repo.
|
|
@@ -382,7 +418,7 @@ ${maintenanceBeforeExecStep}
|
|
|
382
418
|
.map((r) => ` <repository repository="${r.repo}" path="${r.repositoryPath}" base_branch="${r.baseBranch || "auto-detect"}" />`)
|
|
383
419
|
.join("\n") +
|
|
384
420
|
`\n</related_repositories>\n`
|
|
385
|
-
:
|
|
421
|
+
: "";
|
|
386
422
|
const testDirInstruction = testsRepoDir
|
|
387
423
|
? `the \`<output_dir>\` from the \`<services>\` block, rooted under the test repository at \`${testsRepoDir}\` (i.e. \`${testsRepoDir}/<output_dir>\`). Write ALL test output files to paths under \`${testsRepoDir}\`, not under \`${repositoryPath}\`. Do NOT write any test files to the app repository.`
|
|
388
424
|
: `${SERVICE_REFS.testDirRef}. Do NOT create a new \`tests/\` directory at the repo root — use that path. If no \`testDirectory\` is configured, default to the language-conventional location (e.g. \`src/test/java/...\` for Java, \`tests/\` for Python).`;
|
|
@@ -469,7 +505,7 @@ ${userPrompt ? "Generate only the tests that the user requested from the Additio
|
|
|
469
505
|
4. Only pass \`authHeader: ""\` if you can confirm the endpoint is truly unauthenticated.
|
|
470
506
|
|
|
471
507
|
**How to generate each type (for ADD):**
|
|
472
|
-
- **Integration**: call \`skyramp_batch_scenario_test_generation\` with ALL steps in a single call (pass the \`steps\` array with method, path, requestBody, statusCode for each step). Then call \`skyramp_integration_test_generation\` with the returned scenario file.
|
|
508
|
+
- **Integration**: call \`skyramp_batch_scenario_test_generation\` with ALL steps in a single call (pass the \`steps\` array with method, path, requestBody, statusCode for each step). Then call \`skyramp_integration_test_generation\` with the returned scenario file${INTEGRATION_CODE_REUSE_GEN_CLAUSE}.
|
|
473
509
|
**Use the pre-built scenario JSON from the Execution Plan** — pass the steps array directly. Do NOT read source code models to construct request bodies if the plan already provides them.
|
|
474
510
|
Scenario JSON and test files go in ${testDirInstruction}
|
|
475
511
|
**Pipeline for speed**: Call ALL \`skyramp_batch_scenario_test_generation\` calls in one batch. When they return, call ALL \`skyramp_integration_test_generation\` calls in the next batch. Do NOT serialize per-scenario (batch→integration→batch→integration) — batch ALL scenarios first, then generate ALL integration tests.
|
|
@@ -487,7 +523,7 @@ ${CONTRACT_MODE_GUIDANCE}
|
|
|
487
523
|
- Legacy format: \`username:password\` — the first \`:\` splits username from password.
|
|
488
524
|
These are format hints, not a strict grammar — apply judgment on ambiguous input (e.g. a \`=\` or \`;\` inside a value of the key=value form: split on the \`;\` that precedes a plausible login-field key).
|
|
489
525
|
|
|
490
|
-
**Credential selection**:
|
|
526
|
+
**Credential selection**: When several credentials are provided, reason carefully about which one each test case needs BEFORE logging in. Multiple credentials exist so tests can exercise the app as different identities — an app with authorization levels behaves differently per account, and a test only has value when it runs as the identity it is about: an admin workflow needs the admin account, a permission-boundary test needs the restricted one, a plain user flow needs an ordinary user. Read what each credential says about itself — a labeling field (\`role\`, or whatever the customer named it: \`accessLevel\`, \`permissionLevel\`, a team/tenant name, …), the username itself, any extra fields — and match that against the test case's intent. When nothing about a test case calls for a specific identity, use the first credential. If the identity a test case needs is not among the credentials, use the closest match and add a note to \`issuesFound\` naming the identity that was missing. NEVER mix fields across credential lines — type the username, password, and every extra field from the SAME line. The exact values you type identify which credential the generated test will read from the environment at replay time, so a mixed or altered value breaks that binding.
|
|
491
527
|
|
|
492
528
|
Type all values verbatim. Before navigating to ANY feature URL:
|
|
493
529
|
1. \`browser_navigate\` to the login URL (e.g. \`{baseUrl}/login\`, \`/user/login\`, \`/signin\` — infer from the app's base URL and framework)
|
|
@@ -576,7 +612,7 @@ ${CONTRACT_MODE_GUIDANCE}
|
|
|
576
612
|
|
|
577
613
|
**The Blueprint Citation Invariant applies during recording too.** Every assertion you emit cites element names — those names must come from blueprint captures, not invention. For N user-intent-level actions, the reference target is N+1 \`browser_blueprint\` calls (the first returns full, the rest return deltas). Traces that follow the pattern produce assertions grounded in observable state changes; traces that skip captures fall back to author-inferred assertions and risk citing names that don't exist in the rendered DOM.
|
|
578
614
|
|
|
579
|
-
The rest of the UI workflow stays the same: trace plan, browser auth, navigation, export (\`skyramp_export_zip\`), generation (\`skyramp_ui_test_generation\`), then
|
|
615
|
+
The rest of the UI workflow stays the same: trace plan, browser auth, navigation, export (\`skyramp_export_zip\`), generation (\`skyramp_ui_test_generation\`), then the post-calls the generation result lists (${UI_UTILS_REUSE ? "`skyramp_enhance_assertions`, `skyramp_modularization`, `skyramp_reuse_code`, in that order" : "`skyramp_reuse_code` (when `codeReuse: true`) and `skyramp_enhance_assertions`"}). Capture-act-capture adds blueprint captures alongside the existing steps; it doesn't replace anything.
|
|
580
616
|
- **E2E**: Only if BOTH a backend trace \`.json\` AND a Playwright \`.zip\` already exist in the repo. Without both, move to \`additionalRecommendations\`.
|
|
581
617
|
- Skip smoke tests entirely.
|
|
582
618
|
|
|
@@ -594,8 +630,14 @@ If a test **generation** tool call fails:
|
|
|
594
630
|
If a test **execution** (\`skyramp_execute_test\`) fails for a newly generated test:
|
|
595
631
|
1. Read the error output to diagnose the root cause (4xx on prereq step, assertion mismatch, floating-point precision, 500 from app bug, timeout, etc.).
|
|
596
632
|
2. **Expected failure check (no retry):** If the failure is an assertion error or HTTP error that matches the issue identified in the code analysis (e.g. the test was generated specifically to document a broken endpoint, a UI rendering bug, or a missing validation), then this is the **intended outcome** — the test is correctly catching the real bug. Report it immediately as \`status: "Fail"\` and move on. Do NOT retry.
|
|
633
|
+
|
|
634
|
+
This path also covers an assertion failure that application behavior outside this PR's diff explains — for example child records that survive the deletion of their parent, state inherited when an ID is recycled or reused, or a value that ignores a status the test set. Before you keep such a test red, confirm the cause in the source: read the handler, model, or query that should have done the work, and find the specific operation that is missing or wrong. If you find it, report the test as \`status: "Fail"\` and add an \`issuesFound\` entry for it. Do NOT retry.
|
|
635
|
+
|
|
636
|
+
If you cannot point at the missing or wrong line in application code, the app is not at fault — shared or seeded data, parallel test workers, or setup the test itself never did explain the collision. Treat that as an infrastructure failure: fix it and retry once as in step 3.
|
|
637
|
+
|
|
638
|
+
**If you did confirm the missing or wrong operation in the source, do NOT make the test pass.** Never add a reset, cleanup, or setup call for isolation. Never weaken the assertion — no \`==\` to \`>=\`, no exact value to a range. A failing test whose diagnosis names a pre-existing bug is the most valuable output of this run; a passing version of it reports nothing.
|
|
597
639
|
3. Apply a targeted fix and retry **once** only for **infrastructure failures** — that means exactly **2 total \`skyramp_execute_test\` calls per test file** for these cases. Examples of infrastructure failures worth fixing:
|
|
598
|
-
- Assertion mismatch
|
|
640
|
+
- Assertion mismatch from floating-point precision, or an expected value mis-transcribed from the observed response or computed with an arithmetic slip. If application behavior outside the diff explains the mismatch, it is not an infrastructure failure — use step 2 instead.
|
|
599
641
|
- Import error, syntax error, or missing dependency in the generated test file
|
|
600
642
|
- Connection refused or timeout unrelated to the app under test
|
|
601
643
|
4. If it still fails after the retry, report it as \`status: "Fail"\` with the error details and move on — do NOT edit and re-run a third time. A failing test that documents a real bug is a valid outcome.
|
|
@@ -606,13 +648,13 @@ If a generated UI test fails with a timeout waiting for an element after navigat
|
|
|
606
648
|
2. Add \`await page.locator('[data-testid="some-element"]').waitFor({ state: 'visible', timeout: 10000 });\` for the specific element the test needs.
|
|
607
649
|
Do NOT use \`page.waitForTimeout()\` with fixed delays. Do NOT retry more than once — if the test still fails after this fix, report it as "Fail".
|
|
608
650
|
|
|
609
|
-
**After generation, you MUST do exactly these steps — nothing more, nothing less
|
|
651
|
+
**After generation, you MUST do exactly these steps — nothing more, nothing less** (generation results may add their own CRITICAL NEXT STEPS — e.g. modularize-then-reuse for integration or UI tests generated with \`codeReuse: true\` — follow those too, in the order the result states):
|
|
610
652
|
1. **[MANDATORY] After \`skyramp_integration_test_generation\`**: Call \`skyramp_enhance_assertions\` with \`testFile\` set to the absolute path of the generated integration test file, \`testType: "integration"\`, and \`enhanceType: "generation"\`. Apply every instruction returned to that file.
|
|
611
653
|
2. **[MANDATORY] After \`skyramp_contract_test_generation\` with \`providerMode\`**: Call \`skyramp_enhance_assertions\` with \`testFile\` set to the absolute path of the generated provider contract test file, \`testType: "contract"\`, and \`enhanceType: "generation"\`. Apply every instruction returned to that file.
|
|
612
654
|
${UI_CODE_REUSE_STEP}
|
|
613
|
-
4. **[MANDATORY] After \`skyramp_ui_test_generation\`**:
|
|
614
|
-
5. **Wait**: Do NOT proceed to test execution until steps 1–4 are complete and the verification checklist in the \`skyramp_enhance_assertions\` tool result has been validated for EVERY generated test file.
|
|
615
|
-
Do not make any changes other than the code-reuse refactoring (step 3) and the assertion enhancements described above. For example: do not modify auth headers, cookies, tokens, env vars, or imports that the generation tool already set correctly — those are correct by construction and changing them breaks auth or execution.
|
|
655
|
+
4. **[MANDATORY] After \`skyramp_ui_test_generation\`**: ${UI_ENHANCE_STEP} The HIGH-tier \`possibleAssertions\` from your second \`browser_blueprint\` captures (after each action) during trace recording are in your context — when the enhance instructions ask you to add assertions for state-changing actions, use those grounded candidates first (they contain exact computed values from the DOM delta, e.g. \`toHaveText('Total: $899.98')\`). Only fall back to deriving values from the test file or source code when no HIGH-tier candidate covers the action.
|
|
656
|
+
5. **Wait**: Do NOT proceed to test execution until steps 1–4 (plus any generation-result CRITICAL NEXT STEPS) are complete and the verification checklist in the \`skyramp_enhance_assertions\` tool result has been validated for EVERY generated test file.
|
|
657
|
+
Do not make any changes other than the code-reuse refactoring (step 3 and the generation-result reuse steps) and the assertion enhancements described above. For example: do not modify auth headers, cookies, tokens, env vars, or imports that the generation tool already set correctly — those are correct by construction and changing them breaks auth or execution.
|
|
616
658
|
|
|
617
659
|
**Execution timing:**
|
|
618
660
|
- **beforeStatus** (maintained tests only): execute each maintained test file **once at the start** (before any edits) to capture \`beforeStatus\`. This is the only execution allowed before edits.
|
|
@@ -620,11 +662,13 @@ Do not make any changes other than the code-reuse refactoring (step 3) and the a
|
|
|
620
662
|
- Only report test results for files you actually ran.
|
|
621
663
|
**Auth**: If \`skyramp_analyze_changes\` reports an auth token or \`SKYRAMP_TEST_TOKEN\` is set, pass it in **every** \`skyramp_execute_test\` call from the first attempt — do NOT wait for a 401/403 to discover auth is needed.`;
|
|
622
664
|
}
|
|
623
|
-
const primaryRepoBlock = primaryRepo
|
|
665
|
+
const primaryRepoBlock = primaryRepo
|
|
666
|
+
? `<REPOSITORY>${primaryRepo}</REPOSITORY>\n`
|
|
667
|
+
: "";
|
|
624
668
|
return `<TITLE>${prTitle}</TITLE>
|
|
625
669
|
<DESCRIPTION>${prDescription}</DESCRIPTION>
|
|
626
670
|
${primaryRepoBlock}<REPOSITORY PATH>${repositoryPath}</REPOSITORY PATH>
|
|
627
|
-
${relatedReposBlock}${testsRepoDirBlock}${serviceContext ? serviceContext +
|
|
671
|
+
${relatedReposBlock}${testsRepoDirBlock}${serviceContext ? serviceContext + "\n" : ""}${uiCredentialsBlock ? uiCredentialsBlock + "\n" : ""}## Goal
|
|
628
672
|
|
|
629
673
|
Every test this run delivers must be a usable functional test — one that exercises the running application through its real API or UI surface and that the user can keep running in CI. Optimize for catching real production bugs: business-rule and computed-field errors, data-integrity violations, security-boundary bypasses, broken user journeys. A test that would FAIL if the application's logic were wrong beats several that merely exercise new surface — prefer fewer, higher-signal tests over padded coverage. The tasks below define which tests are in scope for this run; use the Skyramp MCP server tools for all of them.
|
|
630
674
|
|
|
@@ -654,13 +698,15 @@ ${task3CountRule}
|
|
|
654
698
|
|
|
655
699
|
${reportLanguageBlock}Call \`skyramp_submit_report\` with \`stateFile\` (from \`skyramp_analyze_changes\` output) — the stateFile is required for execution outcome tracking, and the report is written beside it. Field names, types, and formats are defined in the tool's parameter schema — follow them exactly.
|
|
656
700
|
|
|
657
|
-
${hasRelatedRepos
|
|
701
|
+
${hasRelatedRepos
|
|
702
|
+
? `
|
|
658
703
|
- **MULTI-REPO attribution**: Set the \`repository\` field (\`owner/repo\`) on EVERY \`newTestsCreated\`, \`testResults\`, \`issuesFound\`, and \`additionalRecommendations\` item — including items about the PRIMARY repo — so each finding is unambiguously attributed. The primary repo's \`repository\` is \`${primaryRepo || "<the primary repo's owner/repo>"}\`; items derived from a related repo's diff (from the \`<related_repositories>\` analysis) carry that repo's \`repository\` value. In \`businessCaseAnalysis\`, include a short per-repo subsection and call out any cross-repo correlations you found.
|
|
659
|
-
`
|
|
704
|
+
`
|
|
705
|
+
: ""}
|
|
660
706
|
- **additionalRecommendations**: AT MOST ${maxRecommendations - maxGenerate} items.
|
|
661
707
|
- For \`testType: "contract"\` entries: **\`primaryEndpoint\` is required** (e.g. \`"GET /api/v1/users/{user_id}"\`). The tool will reject the submission without it — do not omit it or you will be forced to resubmit.
|
|
662
708
|
|
|
663
|
-
${getTraceRecordingPromptText({ outputDir: `${repositoryPath}/.skyramp`, modularize: false })}`;
|
|
709
|
+
${getTraceRecordingPromptText({ outputDir: `${repositoryPath}/.skyramp`, modularize: false, modularizeViaGenerationResult: UI_UTILS_REUSE })}`;
|
|
664
710
|
// Neither path reaches the agent any more: SKYR-4147 made the report derive from the
|
|
665
711
|
// state file's directory, and that directory comes from the environment via
|
|
666
712
|
// runArtifactDir(). Keep it that way — a path the model retypes out of this prose is a
|