@skyramp/mcp 0.3.2-rc.pom-4 → 0.3.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/build/adapters/jestAdapter.d.ts +14 -0
- package/build/adapters/jestAdapter.js +113 -0
- package/build/adapters/mochaAdapter.d.ts +13 -0
- package/build/adapters/mochaAdapter.js +87 -0
- package/build/adapters/playwrightAdapter.d.ts +17 -0
- package/build/adapters/playwrightAdapter.js +182 -0
- package/build/adapters/pytestAdapter.d.ts +15 -0
- package/build/adapters/pytestAdapter.js +108 -0
- package/build/commands/commandLibrary.d.ts +1 -0
- package/build/commands/commandLibrary.js +19 -13
- package/build/commands/localDevTestChangesCommand.d.ts +15 -0
- package/build/commands/localDevTestChangesCommand.js +201 -0
- package/build/index.js +82 -6
- package/build/prompts/code-reuse.js +3 -0
- package/build/prompts/enhance-assertions/sharedAssertionRules.d.ts +1 -0
- package/build/prompts/enhance-assertions/sharedAssertionRules.js +17 -0
- package/build/prompts/initialize-workspace/initializeWorkspacePrompt.js +11 -9
- package/build/prompts/local-dev/local-dev-plan.d.ts +35 -0
- package/build/prompts/local-dev/local-dev-plan.js +429 -0
- package/build/prompts/local-dev/local-dev-prompts.d.ts +4 -0
- package/build/prompts/local-dev/local-dev-prompts.js +190 -0
- package/build/prompts/prompt-utils.d.ts +8 -0
- package/build/prompts/prompt-utils.js +33 -0
- package/build/prompts/sut-setup/modes/adaptWorkflowPrompt.js +70 -7
- package/build/prompts/sut-setup/modes/dockerComposePrompt.js +1 -1
- package/build/prompts/sut-setup/shared.js +19 -17
- package/build/prompts/test-maintenance/drift-analysis-prompt.d.ts +10 -1
- package/build/prompts/test-maintenance/drift-analysis-prompt.js +54 -1
- package/build/prompts/test-recommendation/analysisOutputPrompt.js +21 -29
- package/build/prompts/test-recommendation/scopeAssessment.d.ts +5 -2
- package/build/prompts/test-recommendation/scopeAssessment.js +78 -6
- package/build/prompts/test-recommendation/test-recommendation-prompt.js +41 -4
- package/build/prompts/testbot/testbot-prompts.d.ts +0 -5
- package/build/prompts/testbot/testbot-prompts.js +39 -55
- package/build/recommendation/planRanker.d.ts +15 -2
- package/build/recommendation/planRanker.js +76 -5
- package/build/resources/testbotResource.js +2 -1
- package/build/services/AnalyticsService.d.ts +1 -1
- package/build/services/TestExecutionService.d.ts +2 -1
- package/build/services/TestExecutionService.js +8 -3
- package/build/services/TestGenerationService.d.ts +2 -2
- package/build/services/TestGenerationService.js +39 -21
- package/build/services/containerEnv.js +3 -1
- package/build/tool-phases.js +7 -0
- package/build/tools/code-refactor/codeReuseTool.js +43 -4
- package/build/tools/code-refactor/enhanceAssertionsTool.js +68 -18
- package/build/tools/code-refactor/reuse-outcome.d.ts +109 -0
- package/build/tools/code-refactor/reuse-outcome.js +158 -0
- package/build/tools/code-refactor/reuse-state.d.ts +45 -0
- package/build/tools/code-refactor/reuse-state.js +140 -0
- package/build/tools/enrichTestWithMocksTool.d.ts +28 -0
- package/build/tools/enrichTestWithMocksTool.js +726 -0
- package/build/tools/executeSkyrampTestTool.d.ts +11 -0
- package/build/tools/executeSkyrampTestTool.js +62 -21
- package/build/tools/generate-tests/batchMockGenerationTool.d.ts +106 -0
- package/build/tools/generate-tests/batchMockGenerationTool.js +545 -0
- package/build/tools/generate-tests/generateBatchScenarioRestTool.js +25 -23
- package/build/tools/generate-tests/generateContractRestTool.js +2 -2
- package/build/tools/generate-tests/generateE2ERestTool.d.ts +1 -1
- package/build/tools/generate-tests/generateE2ERestTool.js +1 -1
- package/build/tools/generate-tests/generateLoadRestTool.d.ts +1 -1
- package/build/tools/generate-tests/generateLoadRestTool.js +1 -1
- package/build/tools/generate-tests/generateMockRestTool.d.ts +179 -6
- package/build/tools/generate-tests/generateMockRestTool.js +391 -22
- package/build/tools/generate-tests/loadTestSchema.d.ts +1 -1
- package/build/tools/generate-tests/planGuard.js +2 -22
- package/build/tools/generateEnrichedIntegrationTestTool.d.ts +25 -0
- package/build/tools/generateEnrichedIntegrationTestTool.js +235 -0
- package/build/tools/localDevWorkerComposeTool.d.ts +24 -0
- package/build/tools/localDevWorkerComposeTool.js +264 -0
- package/build/tools/one-click/oneClickTool.d.ts +13 -0
- package/build/tools/one-click/oneClickTool.js +195 -24
- package/build/tools/preflightMockCheckTool.d.ts +2 -0
- package/build/tools/preflightMockCheckTool.js +96 -0
- package/build/tools/queryProxyMocksTool.d.ts +70 -0
- package/build/tools/queryProxyMocksTool.js +522 -0
- package/build/tools/runExistingTestsTool.d.ts +138 -0
- package/build/tools/runExistingTestsTool.js +644 -0
- package/build/tools/submitReportTool.js +30 -1
- package/build/tools/test-management/analyzeChangesTool.d.ts +2 -1
- package/build/tools/test-management/analyzeChangesTool.js +63 -40
- package/build/tools/test-management/analyzeTestHealthTool.d.ts +11 -0
- package/build/tools/test-management/analyzeTestHealthTool.js +63 -1
- package/build/tools/test-management/registerTestPlanTool.js +55 -7
- package/build/tools/trace/startTraceCollectionTool.js +3 -3
- package/build/types/ExternalTestExecution.d.ts +67 -0
- package/build/types/ExternalTestExecution.js +8 -0
- package/build/types/OneClickCommands.d.ts +1 -1
- package/build/types/Recommendation.d.ts +20 -0
- package/build/types/Recommendation.js +32 -6
- package/build/types/RepositoryAnalysis.d.ts +131 -14
- package/build/types/RepositoryAnalysis.js +16 -2
- package/build/types/ReuseOutcome.d.ts +63 -0
- package/build/types/ReuseOutcome.js +33 -0
- package/build/types/TestExecution.d.ts +1 -0
- package/build/types/TestTypes.d.ts +25 -7
- package/build/types/TestTypes.js +25 -7
- package/build/types/TestbotReport.d.ts +7 -0
- package/build/types/index.d.ts +2 -0
- package/build/types/index.js +1 -0
- package/build/utils/AnalysisStateManager.d.ts +33 -0
- package/build/utils/AnalysisStateManager.js +36 -2
- package/build/utils/analyze-openapi.js +18 -1
- package/build/utils/branchDiff.d.ts +17 -1
- package/build/utils/branchDiff.js +99 -14
- package/build/utils/featureFlags.d.ts +31 -0
- package/build/utils/featureFlags.js +37 -0
- package/build/utils/grpcMockValidation.d.ts +1 -0
- package/build/utils/grpcMockValidation.js +49 -0
- package/build/utils/httpMethodValidation.d.ts +4 -0
- package/build/utils/httpMethodValidation.js +15 -0
- package/build/utils/logger.js +1 -1
- package/build/utils/mockCompatibility.d.ts +49 -0
- package/build/utils/mockCompatibility.js +82 -0
- package/build/utils/pom-verify/verify.d.ts +5 -0
- package/build/utils/pom-verify/verify.js +1 -0
- package/build/utils/progress.js +10 -5
- package/build/utils/proxy-terminal.js +3 -3
- package/build/utils/routeParsers.d.ts +3 -9
- package/build/utils/routeParsers.js +79 -4
- package/build/utils/utils.js +2 -2
- package/build/utils/versions.d.ts +4 -3
- package/build/utils/versions.js +3 -1
- package/build/utils/workspaceAuth.d.ts +46 -0
- package/build/utils/workspaceAuth.js +156 -1
- package/build/workspace/frameworks.d.ts +11 -0
- package/build/workspace/frameworks.js +22 -0
- package/build/workspace/testSuites.d.ts +20 -0
- package/build/workspace/testSuites.js +17 -0
- package/build/workspace/workspace.d.ts +206 -24
- package/build/workspace/workspace.js +52 -2
- package/package.json +5 -2
|
@@ -12,6 +12,7 @@ import { toolError, testFileMatches } from "../utils/utils.js";
|
|
|
12
12
|
import { matchesApprovedPlan } from "../utils/planMatchKeys.js";
|
|
13
13
|
import { isTestbotEnabled } from "../utils/featureFlags.js";
|
|
14
14
|
import { findUnbackedClaims, listChangedFiles } from "../utils/reportVerification.js";
|
|
15
|
+
import { rederiveReuseOutcome } from "./code-refactor/reuse-state.js";
|
|
15
16
|
// SKYR-3879 Path B: which testTypes the register-plan checkpoint gates. Mirrors
|
|
16
17
|
// the generation tools actually wired to planGuard (batch-scenario/integration,
|
|
17
18
|
// contract) — UI and E2E are on a separate blueprint-grounded pipeline and are
|
|
@@ -337,6 +338,34 @@ function computeReportMetrics(params) {
|
|
|
337
338
|
issuesLow: String(countBy(params.issuesFound, (i) => i.severity === "low")),
|
|
338
339
|
};
|
|
339
340
|
}
|
|
341
|
+
/**
|
|
342
|
+
* Attach the POM code-reuse outcome for a UI test, matched by the report's
|
|
343
|
+
* `fileName` (the basename skyramp_reuse_code keyed run state on).
|
|
344
|
+
*
|
|
345
|
+
* Server-derived, never supplied by the LLM — the same line this file already
|
|
346
|
+
* draws for testMaintenance's beforeStatus/afterStatus. `skyramp_reuse_code`
|
|
347
|
+
* computes every value in-process (tier-1 detection by selectScopedPoms, the call
|
|
348
|
+
* counts and violations by verifyReuse, the declines from the `// kept inline:`
|
|
349
|
+
* comments in the spec), so asking the agent to read them out of tool text and
|
|
350
|
+
* retype them later would leave the signal prose-mediated — droppable and
|
|
351
|
+
* fabricable, which is the failure this field exists to fix.
|
|
352
|
+
*
|
|
353
|
+
* The counts are RE-DERIVED from the delivered spec rather than read back from
|
|
354
|
+
* state: the execution fix-up can restore `<testFile>.raw.bak` over it afterwards
|
|
355
|
+
* by plain `cp`, which this tool never sees. See rederiveReuseOutcome.
|
|
356
|
+
*
|
|
357
|
+
* Non-UI tests, tests the reuse tool never ran for, and specs whose outcome cannot
|
|
358
|
+
* be re-derived all get nothing — which consumers already treat as "no reuse
|
|
359
|
+
* summary". */
|
|
360
|
+
async function attachReuseOutcome(test, outcomes) {
|
|
361
|
+
if (test.testType !== TestType.UI || !outcomes)
|
|
362
|
+
return test;
|
|
363
|
+
const found = outcomes[path.basename(test.fileName)];
|
|
364
|
+
if (!found)
|
|
365
|
+
return test;
|
|
366
|
+
const reuse = await rederiveReuseOutcome(found);
|
|
367
|
+
return reuse ? { ...test, reuse } : test;
|
|
368
|
+
}
|
|
340
369
|
function deduplicateById(items) {
|
|
341
370
|
const seen = new Set();
|
|
342
371
|
const result = [];
|
|
@@ -582,7 +611,7 @@ export function registerSubmitReportTool(server) {
|
|
|
582
611
|
// generation — downstream scoring scripts don't expect them and fail if
|
|
583
612
|
// they encounter these string fields while traversing the object.
|
|
584
613
|
// Also normalize each item's `repository` (blank → undefined).
|
|
585
|
-
const sanitizedNewTests = dedupedNewTests.map(({ scenarioFile: _sf, traceFile: _tf, frontendTrace: _ft, ...rest }) => normalizeRepository(rest));
|
|
614
|
+
const sanitizedNewTests = await Promise.all(dedupedNewTests.map(({ scenarioFile: _sf, traceFile: _tf, frontendTrace: _ft, ...rest }) => attachReuseOutcome(normalizeRepository(rest), stateData.reuseOutcomes)));
|
|
586
615
|
const report = {
|
|
587
616
|
businessCaseAnalysis: params.businessCaseAnalysis,
|
|
588
617
|
newTestsCreated: sanitizedNewTests,
|
|
@@ -16,11 +16,12 @@ export declare const analyzeChangesInputSchema: {
|
|
|
16
16
|
scope: z.ZodOptional<z.ZodDefault<z.ZodEnum<["full_repo", "branch_diff"]>>>;
|
|
17
17
|
baseBranch: z.ZodOptional<z.ZodString>;
|
|
18
18
|
testDirectory: z.ZodOptional<z.ZodString>;
|
|
19
|
-
topN: z.
|
|
19
|
+
topN: z.ZodDefault<z.ZodOptional<z.ZodNumber>>;
|
|
20
20
|
maxGenerate: z.ZodOptional<z.ZodNumber>;
|
|
21
21
|
prNumber: z.ZodOptional<z.ZodNumber>;
|
|
22
22
|
repository: z.ZodOptional<z.ZodString>;
|
|
23
23
|
testsRepoDir: z.ZodOptional<z.ZodEffects<z.ZodString, string, string>>;
|
|
24
|
+
includeUncommitted: z.ZodDefault<z.ZodOptional<z.ZodBoolean>>;
|
|
24
25
|
};
|
|
25
26
|
export declare const NO_UI_INSTRUCTIONS = "No UI changes detected \u2014 no blueprint capture needed.";
|
|
26
27
|
export declare const NO_RESOLVABLE_URLS_INSTRUCTIONS = "Frontend changes detected but no candidate URLs could be resolved (workspace baseUrl missing or no router files matched). UI recommendations will be source-grounded only.";
|
|
@@ -287,8 +287,8 @@ export const analyzeChangesInputSchema = {
|
|
|
287
287
|
.describe("Directory containing existing tests (auto-detected if omitted)"),
|
|
288
288
|
topN: z
|
|
289
289
|
.number()
|
|
290
|
-
.default(MAX_RECOMMENDATIONS)
|
|
291
290
|
.optional()
|
|
291
|
+
.default(MAX_RECOMMENDATIONS)
|
|
292
292
|
.describe(`Number of ranked test recommendations to generate. Defaults to ${MAX_RECOMMENDATIONS}.`),
|
|
293
293
|
maxGenerate: z
|
|
294
294
|
.number()
|
|
@@ -309,6 +309,11 @@ export const analyzeChangesInputSchema = {
|
|
|
309
309
|
.refine((v) => path.isAbsolute(v), { message: "testsRepoDir must be an absolute path" })
|
|
310
310
|
.optional()
|
|
311
311
|
.describe("Absolute path to a separate test repository clone. When set, existing test discovery scans this directory instead of repositoryPath. Used in cross-repo test delivery mode where tests live in a separate repo."),
|
|
312
|
+
includeUncommitted: z
|
|
313
|
+
.boolean()
|
|
314
|
+
.optional()
|
|
315
|
+
.default(false)
|
|
316
|
+
.describe("When true, diffs the base ref against the working tree (captures uncommitted and unstaged changes). Use for local-dev workflows. Defaults to false (CI mode — committed changes only)."),
|
|
312
317
|
};
|
|
313
318
|
// ── UI blueprint-capture instructions ──
|
|
314
319
|
// Moved here from the former skyramp_ui_analyze_changes pre-flight tool. These
|
|
@@ -391,7 +396,7 @@ export function registerAnalyzeChangesTool(server) {
|
|
|
391
396
|
if (analysisScope === AnalysisScope.CurrentBranchDiff) {
|
|
392
397
|
await sendProgress(10, 100, "Computing branch diff...");
|
|
393
398
|
try {
|
|
394
|
-
diffData = await computeBranchDiff(params.repositoryPath, params.baseBranch);
|
|
399
|
+
diffData = await computeBranchDiff(params.repositoryPath, params.baseBranch, params.includeUncommitted ?? false);
|
|
395
400
|
logger.info("Branch diff computed", {
|
|
396
401
|
currentBranch: diffData.currentBranch,
|
|
397
402
|
baseBranch: diffData.baseBranch,
|
|
@@ -489,7 +494,7 @@ export function registerAnalyzeChangesTool(server) {
|
|
|
489
494
|
const removalCandidateFiles = selectRemovalCandidateFiles(diffData);
|
|
490
495
|
const git = simpleGit(params.repositoryPath);
|
|
491
496
|
const recoveredBaseEndpoints = removalCandidateFiles.length > 0
|
|
492
|
-
? await recoverRemovedEndpointsFromBase(removalCandidateFiles, (file) => git.show([`${diffData.baseBranch}:${file}`]))
|
|
497
|
+
? await recoverRemovedEndpointsFromBase(removalCandidateFiles, (file) => git.show([`${diffData.baseBranch}:${file}`]), params.repositoryPath, diffData.baseBranch)
|
|
493
498
|
: [];
|
|
494
499
|
classifiedEndpoints = classifyEndpointsByChangedFiles(diffData, scannedEndpoints, recoveredBaseEndpoints);
|
|
495
500
|
classifiedEndpoints = {
|
|
@@ -1181,6 +1186,56 @@ export function registerAnalyzeChangesTool(server) {
|
|
|
1181
1186
|
affectedServices: classifiedEndpoints.affectedServices,
|
|
1182
1187
|
summary: "",
|
|
1183
1188
|
} : undefined;
|
|
1189
|
+
// ── Route discovery context for LLM grounding and state persistence ──
|
|
1190
|
+
// fullAnalysis lives only in inMemorySessionStore (for MCP resources
|
|
1191
|
+
// and registerRecommendTestsPrompt). The disk state carries only the
|
|
1192
|
+
// slim fields that downstream tools (health, execute, actions) need.
|
|
1193
|
+
// routerMountContext and candidateRouteFiles are computed here so they
|
|
1194
|
+
// can be persisted to the state file for downstream tools (health, drift).
|
|
1195
|
+
// Without them, analyzeTestHealth would work only off the static catalog
|
|
1196
|
+
// which has wrong paths for nested resources and unsupported frameworks.
|
|
1197
|
+
const routeLikeUnmatchedFiles = [];
|
|
1198
|
+
for (const file of classifiedEndpoints?.unmatchedFiles ?? []) {
|
|
1199
|
+
const routeLike = SOURCE_EXTS.test(file) &&
|
|
1200
|
+
(ROUTE_FILE_PATTERN.test(file) || ROUTE_FILE_BASENAME_PATTERN.test(path.basename(file)));
|
|
1201
|
+
if (routeLike && !(await isGraphQLFile(file, params.repositoryPath))) {
|
|
1202
|
+
routeLikeUnmatchedFiles.push(file);
|
|
1203
|
+
}
|
|
1204
|
+
}
|
|
1205
|
+
const shouldIncludeCandidateRouteFiles = analysisScope !== AnalysisScope.CurrentBranchDiff ||
|
|
1206
|
+
rawRelatedEndpointCount === 0 ||
|
|
1207
|
+
scannedEndpoints.length === 0 ||
|
|
1208
|
+
routeLikeUnmatchedFiles.length > 0;
|
|
1209
|
+
let candidateRouteFiles;
|
|
1210
|
+
if (shouldIncludeCandidateRouteFiles) {
|
|
1211
|
+
candidateRouteFiles = [];
|
|
1212
|
+
for (const file of findCandidateRouteFiles(params.repositoryPath)) {
|
|
1213
|
+
if (!(await isGraphQLFile(file, params.repositoryPath))) {
|
|
1214
|
+
candidateRouteFiles.push(file);
|
|
1215
|
+
}
|
|
1216
|
+
}
|
|
1217
|
+
}
|
|
1218
|
+
// Write the full diff to a temp file before building state so the path
|
|
1219
|
+
// can be persisted and read by analyzeTestHealthTool for per-line detection.
|
|
1220
|
+
let diffFilePath;
|
|
1221
|
+
if (diffData?.diffContent) {
|
|
1222
|
+
diffFilePath = path.join(os.tmpdir(), `skyramp-diff-${sessionId}.diff`);
|
|
1223
|
+
await fs.promises.writeFile(diffFilePath, diffData.diffContent, { encoding: "utf-8", mode: 0o600 });
|
|
1224
|
+
}
|
|
1225
|
+
const routeDiscovery = {
|
|
1226
|
+
candidateFiles: [
|
|
1227
|
+
...(diffData?.changedFiles ?? []),
|
|
1228
|
+
...(candidateRouteFiles ?? []),
|
|
1229
|
+
].filter((file, index, files) => files.indexOf(file) === index),
|
|
1230
|
+
staticHints: scannedEndpoints.map((ep) => ({
|
|
1231
|
+
path: ep.path,
|
|
1232
|
+
methods: ep.methods,
|
|
1233
|
+
sourceFile: ep.sourceFile,
|
|
1234
|
+
})),
|
|
1235
|
+
openApiPaths: specPaths ? [...specPaths] : [],
|
|
1236
|
+
routerMountContext,
|
|
1237
|
+
...(diffFilePath ? { diffFilePath } : {}),
|
|
1238
|
+
};
|
|
1184
1239
|
const fullAnalysis = {
|
|
1185
1240
|
metadata: {
|
|
1186
1241
|
repositoryName: path.basename(params.repositoryPath),
|
|
@@ -1237,6 +1292,7 @@ export function registerAnalyzeChangesTool(server) {
|
|
|
1237
1292
|
hasCoverageReports: false,
|
|
1238
1293
|
relevantExternalTestPaths,
|
|
1239
1294
|
},
|
|
1295
|
+
routeDiscovery,
|
|
1240
1296
|
...(diffContext ? { branchDiffContext: diffContext } : {}),
|
|
1241
1297
|
};
|
|
1242
1298
|
// Store RecommendationState in memory so it's compatible with skyramp_recommend_tests if needed
|
|
@@ -1246,42 +1302,6 @@ export function registerAnalyzeChangesTool(server) {
|
|
|
1246
1302
|
analysis: fullAnalysis,
|
|
1247
1303
|
};
|
|
1248
1304
|
storeSessionData(sessionId, recommendationState);
|
|
1249
|
-
// ── Step 11: Build UnifiedAnalysisState and save ──
|
|
1250
|
-
// fullAnalysis lives only in inMemorySessionStore (for MCP resources
|
|
1251
|
-
// and registerRecommendTestsPrompt). The disk state carries only the
|
|
1252
|
-
// slim fields that downstream tools (health, execute, actions) need.
|
|
1253
|
-
// routerMountContext and candidateRouteFiles are computed here so they
|
|
1254
|
-
// can be persisted to the state file for downstream tools (health, drift).
|
|
1255
|
-
// Without them, analyzeTestHealth would work only off the static catalog
|
|
1256
|
-
// which has wrong paths for nested resources and unsupported frameworks.
|
|
1257
|
-
const routeLikeUnmatchedFiles = [];
|
|
1258
|
-
for (const file of classifiedEndpoints?.unmatchedFiles ?? []) {
|
|
1259
|
-
const routeLike = SOURCE_EXTS.test(file) &&
|
|
1260
|
-
(ROUTE_FILE_PATTERN.test(file) || ROUTE_FILE_BASENAME_PATTERN.test(path.basename(file)));
|
|
1261
|
-
if (routeLike && !(await isGraphQLFile(file, params.repositoryPath))) {
|
|
1262
|
-
routeLikeUnmatchedFiles.push(file);
|
|
1263
|
-
}
|
|
1264
|
-
}
|
|
1265
|
-
const shouldIncludeCandidateRouteFiles = analysisScope !== AnalysisScope.CurrentBranchDiff ||
|
|
1266
|
-
rawRelatedEndpointCount === 0 ||
|
|
1267
|
-
scannedEndpoints.length === 0 ||
|
|
1268
|
-
routeLikeUnmatchedFiles.length > 0;
|
|
1269
|
-
let candidateRouteFiles;
|
|
1270
|
-
if (shouldIncludeCandidateRouteFiles) {
|
|
1271
|
-
candidateRouteFiles = [];
|
|
1272
|
-
for (const file of findCandidateRouteFiles(params.repositoryPath)) {
|
|
1273
|
-
if (!(await isGraphQLFile(file, params.repositoryPath))) {
|
|
1274
|
-
candidateRouteFiles.push(file);
|
|
1275
|
-
}
|
|
1276
|
-
}
|
|
1277
|
-
}
|
|
1278
|
-
// Write the full diff to a temp file before building state so the path
|
|
1279
|
-
// can be persisted and read by analyzeTestHealthTool for per-line detection.
|
|
1280
|
-
let diffFilePath;
|
|
1281
|
-
if (diffData?.diffContent) {
|
|
1282
|
-
diffFilePath = path.join(os.tmpdir(), `skyramp-diff-${sessionId}.diff`);
|
|
1283
|
-
await fs.promises.writeFile(diffFilePath, diffData.diffContent, { encoding: "utf-8", mode: 0o600 });
|
|
1284
|
-
}
|
|
1285
1305
|
// SKYR-3879 Path B: size-capped raw diff text, persisted (not just the
|
|
1286
1306
|
// file path above) so skyramp_register_test_plan can verify a
|
|
1287
1307
|
// discriminator's changedCodeAnchor occurs verbatim in the diff without
|
|
@@ -1367,7 +1387,7 @@ export function registerAnalyzeChangesTool(server) {
|
|
|
1367
1387
|
source: CandidateSource.SERVER,
|
|
1368
1388
|
candidateId: computeCandidateId(scenario),
|
|
1369
1389
|
}));
|
|
1370
|
-
const plan = selectPlan(serverCandidates, budgetCtx);
|
|
1390
|
+
const plan = selectPlan(serverCandidates, { ...budgetCtx, diffText });
|
|
1371
1391
|
approvedPlan = {
|
|
1372
1392
|
planId: crypto.randomUUID(),
|
|
1373
1393
|
createdAt: new Date().toISOString(),
|
|
@@ -1429,6 +1449,7 @@ export function registerAnalyzeChangesTool(server) {
|
|
|
1429
1449
|
sessionId,
|
|
1430
1450
|
routerMountContext,
|
|
1431
1451
|
candidateRouteFiles,
|
|
1452
|
+
routeDiscovery,
|
|
1432
1453
|
relevantExternalTestPaths,
|
|
1433
1454
|
},
|
|
1434
1455
|
};
|
|
@@ -1526,6 +1547,8 @@ export function registerAnalyzeChangesTool(server) {
|
|
|
1526
1547
|
modifiedEndpointCount: classifiedEndpoints?.changedEndpoints.length ?? 0,
|
|
1527
1548
|
removedEndpointCount: classifiedEndpoints?.removedEndpoints.length ?? 0,
|
|
1528
1549
|
endpointCount: skeletonEndpoints.reduce((acc, ep) => acc + ep.methods.length, 0),
|
|
1550
|
+
newEndpoints: (classifiedEndpoints?.newEndpoints ?? []).flatMap((ep) => ep.methods.map((m) => `${m} ${ep.path}`)),
|
|
1551
|
+
modifiedEndpoints: (classifiedEndpoints?.changedEndpoints ?? []).flatMap((ep) => ep.methods.map((m) => `${m} ${ep.path}`)),
|
|
1529
1552
|
},
|
|
1530
1553
|
// Surface uiContext inline so the testbot prompt can iterate
|
|
1531
1554
|
// candidateUiPages and inspect changedFrontendFiles without
|
|
@@ -1,2 +1,13 @@
|
|
|
1
1
|
import { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js";
|
|
2
|
+
/**
|
|
3
|
+
* Names of services whose workspace.yml records a runnable external-test suite
|
|
4
|
+
* (a reader-backed framework — playwright/pytest/jest/vitest/mocha — +
|
|
5
|
+
* a `runtimeDetails.testSuites` entry with a testRunCommand) — i.e. what skyramp_run_existing_tests can
|
|
6
|
+
* actually run. Used to scope the confirm gate so it only fires on repos that
|
|
7
|
+
* went through Test Environment Setup. Checks each service's suites via
|
|
8
|
+
* `resolveSuites` (rather than calling resolveRunConfig) so every
|
|
9
|
+
* `runtimeDetails.testSuites` entry is recognized. Returns [] on any read/parse
|
|
10
|
+
* error so a config problem can never block drift analysis.
|
|
11
|
+
*/
|
|
12
|
+
export declare function runnableTestEnvServices(repositoryPath: string): Promise<string[]>;
|
|
2
13
|
export declare function registerAnalyzeTestHealthTool(server: McpServer): void;
|
|
@@ -5,6 +5,9 @@ import { AnalyticsService } from "../../services/AnalyticsService.js";
|
|
|
5
5
|
import { buildDriftAnalysisPrompt } from "../../prompts/test-maintenance/drift-analysis-prompt.js";
|
|
6
6
|
import { TestSource } from "../../types/TestAnalysis.js";
|
|
7
7
|
import { toolError } from "../../utils/utils.js";
|
|
8
|
+
import { WorkspaceConfigManager } from "../../workspace/workspace.js";
|
|
9
|
+
import { resolveSuites } from "../../workspace/testSuites.js";
|
|
10
|
+
import { frameworkHasReader } from "../runExistingTestsTool.js";
|
|
8
11
|
const TOOL_NAME = "skyramp_analyze_test_health";
|
|
9
12
|
// UI test frameworks. "playwright-ui" and "cypress-ui" are the browser-interaction
|
|
10
13
|
// variants detected when the file uses the page fixture (not just the request fixture).
|
|
@@ -26,6 +29,27 @@ function isUiTest(t) {
|
|
|
26
29
|
const isUiSnap = UI_SNAPSHOT_EXT.test(t.testFile);
|
|
27
30
|
return UI_COMPONENT_FRAMEWORKS.has(fw) || UI_TEST_EXTENSIONS.test(t.testFile) || isUiSnap;
|
|
28
31
|
}
|
|
32
|
+
/**
|
|
33
|
+
* Names of services whose workspace.yml records a runnable external-test suite
|
|
34
|
+
* (a reader-backed framework — playwright/pytest/jest/vitest/mocha — +
|
|
35
|
+
* a `runtimeDetails.testSuites` entry with a testRunCommand) — i.e. what skyramp_run_existing_tests can
|
|
36
|
+
* actually run. Used to scope the confirm gate so it only fires on repos that
|
|
37
|
+
* went through Test Environment Setup. Checks each service's suites via
|
|
38
|
+
* `resolveSuites` (rather than calling resolveRunConfig) so every
|
|
39
|
+
* `runtimeDetails.testSuites` entry is recognized. Returns [] on any read/parse
|
|
40
|
+
* error so a config problem can never block drift analysis.
|
|
41
|
+
*/
|
|
42
|
+
export async function runnableTestEnvServices(repositoryPath) {
|
|
43
|
+
try {
|
|
44
|
+
const config = await new WorkspaceConfigManager(repositoryPath).read();
|
|
45
|
+
return (config.services ?? [])
|
|
46
|
+
.filter((s) => resolveSuites(s).some((suite) => frameworkHasReader(suite.framework) && !!suite.testRunCommand))
|
|
47
|
+
.map((s) => s.serviceName);
|
|
48
|
+
}
|
|
49
|
+
catch {
|
|
50
|
+
return [];
|
|
51
|
+
}
|
|
52
|
+
}
|
|
29
53
|
export function registerAnalyzeTestHealthTool(server) {
|
|
30
54
|
server.registerTool(TOOL_NAME, {
|
|
31
55
|
annotations: {
|
|
@@ -75,6 +99,44 @@ export function registerAnalyzeTestHealthTool(server) {
|
|
|
75
99
|
const skyrampCount = existingTests.filter((t) => t.source !== TestSource.External).length;
|
|
76
100
|
const externalCount = existingTests.length - skyrampCount;
|
|
77
101
|
logger.info(`Loaded ${skyrampCount} Skyramp + ${externalCount} relevant external tests from state file`);
|
|
102
|
+
// ── External-test confirm gate ────────────────────────────────────
|
|
103
|
+
// When the diff touches the repo's OWN [external] tests AND the workspace
|
|
104
|
+
// has a runnable test-env config (runtimeDetails.test*), the drift
|
|
105
|
+
// assessment must be grounded in a real run of those tests
|
|
106
|
+
// (skyramp_run_existing_tests mode:confirm), not the agent's guess from
|
|
107
|
+
// source. Prompt prose alone proved model-fragile — agents read the config
|
|
108
|
+
// and still skipped the confirm, leaving the repo's own tests at
|
|
109
|
+
// afterStatus=Unknown — so enforce ordering here: refuse to produce the
|
|
110
|
+
// drift prompt until a confirm run has been recorded. Loop-safe:
|
|
111
|
+
// skyramp_run_existing_tests persists an externalTestResults record for
|
|
112
|
+
// EVERY outcome (pass/fail/skip/unhealthy), so one confirm call clears the
|
|
113
|
+
// gate. Only a confirm-mode record counts — a verify run (or any future
|
|
114
|
+
// record type) must not bypass the required confirm pass.
|
|
115
|
+
// Scoped by runnableTestEnvServices so it never fires on repos without
|
|
116
|
+
// test-env setup, and fail-open on any config error.
|
|
117
|
+
const confirmAttempted = (stateData.externalTestResults ?? []).some((r) => r.mode === "confirm");
|
|
118
|
+
const runnableServices = externalCount > 0 && !confirmAttempted
|
|
119
|
+
? await runnableTestEnvServices(repositoryPath)
|
|
120
|
+
: [];
|
|
121
|
+
if (runnableServices.length > 0) {
|
|
122
|
+
const externalFiles = existingTests
|
|
123
|
+
.filter((t) => t.source === TestSource.External)
|
|
124
|
+
.map((t) => t.testFile);
|
|
125
|
+
// Cap the enumerated examples — a broad diff can mark dozens of external
|
|
126
|
+
// tests, and the agent already has the full [external] list from
|
|
127
|
+
// skyramp_analyze_changes.
|
|
128
|
+
const shown = externalFiles.slice(0, 8);
|
|
129
|
+
const filesList = shown.map((f) => `"${f}"`).join(", ") +
|
|
130
|
+
(externalFiles.length > shown.length
|
|
131
|
+
? `, …(+${externalFiles.length - shown.length} more)`
|
|
132
|
+
: "");
|
|
133
|
+
// With >1 runnable service the tool needs an explicit `service`.
|
|
134
|
+
const serviceHint = runnableServices.length > 1
|
|
135
|
+
? ` Multiple runnable test services are configured — pass service: "${runnableServices[0]}".`
|
|
136
|
+
: "";
|
|
137
|
+
return toolError(`This repository has a configured test environment (runtimeDetails.test*) and ${externalCount} external test${externalCount === 1 ? "" : "s"} relevant to this diff, but skyramp_run_existing_tests has not run yet. Run the repo's OWN tests BEFORE assessing drift so the assessment is grounded in real pass/fail (otherwise these tests are left at status Unknown). ` +
|
|
138
|
+
`Call skyramp_run_existing_tests with mode: "confirm", stateFile: "${stateManager.getStatePath()}"${args.repository ? `, repository: "${args.repository}"` : ""}, and testSelectors set to the [external] test files skyramp_analyze_changes marked (e.g. ${filesList}).${serviceHint} Then call skyramp_analyze_test_health again.`);
|
|
139
|
+
}
|
|
78
140
|
// Delete stale diff files only — state files must remain for the caller's skyramp_actions call.
|
|
79
141
|
try {
|
|
80
142
|
await StateManager.cleanupOldFiles(24, undefined, []);
|
|
@@ -120,7 +182,7 @@ export function registerAnalyzeTestHealthTool(server) {
|
|
|
120
182
|
const relatedRepoPaths = (await Promise.all(relatedRepoKeys.map(r => stateManager.getRepoRepositoryPath(r)))).filter((p) => typeof p === "string");
|
|
121
183
|
allRepoPaths = [repositoryPath, ...relatedRepoPaths];
|
|
122
184
|
}
|
|
123
|
-
const promptText = buildDriftAnalysisPrompt(stateManager.getStatePath(), apiTests.map((t) => ({ testFile: t.testFile, source: t.source })), uiDriftParams, allRepoPaths);
|
|
185
|
+
const promptText = buildDriftAnalysisPrompt(stateManager.getStatePath(), apiTests.map((t) => ({ testFile: t.testFile, source: t.source })), uiDriftParams, allRepoPaths, stateData?.externalTestResults);
|
|
124
186
|
return {
|
|
125
187
|
structuredContent: { prompt: promptText },
|
|
126
188
|
content: [{ type: "text", text: "Drift analysis prompt generated. Follow the prompt field to assess each test." }],
|
|
@@ -8,7 +8,7 @@ import { toolError } from "../../utils/utils.js";
|
|
|
8
8
|
import { buildApprovedPlanItem } from "../../utils/planMatchKeys.js";
|
|
9
9
|
import { SCENARIO_CATEGORIES, CATEGORY_PRIORITY, Novelty, PriorityTier } from "../../types/TestRecommendation.js";
|
|
10
10
|
import { HttpMethod, TestType } from "../../types/TestTypes.js";
|
|
11
|
-
import { CandidateSource, computeCandidateId, DiscriminatorKind } from "../../types/Recommendation.js";
|
|
11
|
+
import { CandidateSource, computeCandidateId, scenarioMergeKey, DiscriminatorKind } from "../../types/Recommendation.js";
|
|
12
12
|
import { selectPlan } from "../../recommendation/planRanker.js";
|
|
13
13
|
import { validateDiscriminator } from "../../recommendation/discriminators.js";
|
|
14
14
|
import { isAttackSurfaceSecurityBoundary } from "../../prompts/test-recommendation/recommendationShared.js";
|
|
@@ -220,6 +220,39 @@ function renderPlanItem(item, rank) {
|
|
|
220
220
|
const discriminatorNote = item.verifiedDiscriminator ? `, verified discriminator: ${item.verifiedDiscriminator}` : "";
|
|
221
221
|
return `${rank}. [${item.testType}] "${item.scenarioName}" (${item.category}${discriminatorNote}) — ${describeGenerationCall(item)}`;
|
|
222
222
|
}
|
|
223
|
+
/**
|
|
224
|
+
* Whether this run generates UI/E2E tests is a deterministic property of the
|
|
225
|
+
* approved plan, so it is decided here rather than left to the agent to resolve
|
|
226
|
+
* from prompt prose. The prompt owns HOW to record a UI trace and the runtime
|
|
227
|
+
* conditions the server cannot know (app unreachable, unintegrated component);
|
|
228
|
+
* this directive owns WHETHER and HOW MANY.
|
|
229
|
+
*/
|
|
230
|
+
function renderGenerationDirective(plan) {
|
|
231
|
+
const uiCount = plan.generate.filter((item) => item.testType === TestType.UI || item.testType === TestType.E2E).length;
|
|
232
|
+
const nonUICount = plan.generate.length - uiCount;
|
|
233
|
+
const tests = (n) => `${n} test${n === 1 ? "" : "s"}`;
|
|
234
|
+
if (plan.generate.length === 0) {
|
|
235
|
+
return [
|
|
236
|
+
"### Generation directive: GENERATE NOTHING",
|
|
237
|
+
"Create zero new tests of any type this run and proceed to the report. An empty `newTestsCreated` is the CORRECT " +
|
|
238
|
+
"result for this PR — there is no new observable surface to cover. Do not record a browser trace and do not invent " +
|
|
239
|
+
"a spec to have something to report; report your maintenance work and the ADDITIONAL candidates as recommendations instead.",
|
|
240
|
+
];
|
|
241
|
+
}
|
|
242
|
+
if (uiCount === 0) {
|
|
243
|
+
return [
|
|
244
|
+
"### Generation directive: NO UI/E2E GENERATION",
|
|
245
|
+
`This plan allocates no UI/E2E generation. Generate the ${tests(nonUICount)} listed above and create zero UI tests ` +
|
|
246
|
+
"— do not record a browser trace and do not add a UI spec to fill `newTestsCreated`.",
|
|
247
|
+
];
|
|
248
|
+
}
|
|
249
|
+
return [
|
|
250
|
+
`### Generation directive: UI/E2E GENERATION REQUIRED (${uiCount})`,
|
|
251
|
+
`This plan allocates ${tests(uiCount)} of type UI/E2E. You MUST attempt to record and generate each one — do not ` +
|
|
252
|
+
"downgrade them to recommendations while the app is reachable." +
|
|
253
|
+
(nonUICount > 0 ? ` Generate the ${tests(nonUICount)} of other types as well.` : ""),
|
|
254
|
+
];
|
|
255
|
+
}
|
|
223
256
|
function renderPlanText(plan) {
|
|
224
257
|
const lines = [];
|
|
225
258
|
lines.push(`## Approved Test Plan (${plan.planId})`);
|
|
@@ -249,6 +282,8 @@ function renderPlanText(plan) {
|
|
|
249
282
|
lines.push(`- ${demotion.candidateId}: ${demotion.reason}`);
|
|
250
283
|
}
|
|
251
284
|
}
|
|
285
|
+
lines.push("");
|
|
286
|
+
lines.push(...renderGenerationDirective(plan));
|
|
252
287
|
return lines.join("\n");
|
|
253
288
|
}
|
|
254
289
|
// ── Tool registration ───────────────────────────────────────────────────────
|
|
@@ -285,14 +320,27 @@ export function registerRegisterTestPlanTool(server) {
|
|
|
285
320
|
const allScenarios = stateData.repositoryAnalysis?.scenarios ?? [];
|
|
286
321
|
const { candidates: agentCandidates, demotions } = buildAgentCandidates(params.candidates ?? [], diffText);
|
|
287
322
|
const serverCandidates = recoverServerCandidates(stateData.approvedPlan, allScenarios);
|
|
288
|
-
// Merge by
|
|
289
|
-
//
|
|
290
|
-
//
|
|
323
|
+
// Merge by scenario-name identity, NOT the full content-hashed candidateId:
|
|
324
|
+
// the server (analyze_changes) and the agent frequently draft their own
|
|
325
|
+
// independent steps[] for "the same" scenario (identical scenarioName),
|
|
326
|
+
// which hash to different candidateIds — deduping on the full id let
|
|
327
|
+
// both survive as apparent duplicates (SKYR-4026). A pre-seeded server
|
|
328
|
+
// candidate wins over an agent submission for the same scenario name —
|
|
329
|
+
// source: "server" is preserved per the unified-plan design.
|
|
330
|
+
//
|
|
331
|
+
// Uses scenarioMergeKey (untruncated), not scenarioNameSlug — the
|
|
332
|
+
// latter's 48-char cap is fine for a short candidateId prefix combined
|
|
333
|
+
// with a content hash, but bare as a Map key it would silently collapse
|
|
334
|
+
// two distinct long scenario names sharing a common prefix. A missing
|
|
335
|
+
// scenarioName (schema requires one for agent candidates; only a
|
|
336
|
+
// malformed server-recovered scenario could lack one) falls back to
|
|
337
|
+
// the already-unique candidateId rather than a shared literal.
|
|
338
|
+
const mergeKey = (c) => scenarioMergeKey(c.scenario.scenarioName) || c.candidateId;
|
|
291
339
|
const merged = new Map();
|
|
292
340
|
for (const candidate of agentCandidates)
|
|
293
|
-
merged.set(candidate
|
|
341
|
+
merged.set(mergeKey(candidate), candidate);
|
|
294
342
|
for (const candidate of serverCandidates)
|
|
295
|
-
merged.set(candidate
|
|
343
|
+
merged.set(mergeKey(candidate), candidate);
|
|
296
344
|
const allCandidates = [...merged.values()];
|
|
297
345
|
// An empty union would persist an authoritative plan with an empty
|
|
298
346
|
// GENERATE list, which the generation gate then enforces — bricking
|
|
@@ -304,7 +352,7 @@ export function registerRegisterTestPlanTool(server) {
|
|
|
304
352
|
return errorResult;
|
|
305
353
|
}
|
|
306
354
|
const budgetContext = resolveBudgetContext(stateData);
|
|
307
|
-
const result = selectPlan(allCandidates, { ...budgetContext, demotions });
|
|
355
|
+
const result = selectPlan(allCandidates, { ...budgetContext, demotions, diffText });
|
|
308
356
|
const approvedPlan = {
|
|
309
357
|
planId: crypto.randomUUID(),
|
|
310
358
|
createdAt: new Date().toISOString(),
|
|
@@ -5,7 +5,7 @@ import { SkyrampClient } from "@skyramp/skyramp";
|
|
|
5
5
|
import openProxyTerminalTracked from "../../utils/proxy-terminal.js";
|
|
6
6
|
import { getEntryPoint } from "../../utils/telemetry.js";
|
|
7
7
|
import { logger } from "../../utils/logger.js";
|
|
8
|
-
import { basePlaywrightSchema, baseSchema, SESSION_STORAGE_FILENAME, } from "../../types/TestTypes.js";
|
|
8
|
+
import { basePlaywrightSchema, baseSchema, RuntimeEnvironment, SESSION_STORAGE_FILENAME, } from "../../types/TestTypes.js";
|
|
9
9
|
import { AnalyticsService } from "../../services/AnalyticsService.js";
|
|
10
10
|
import { resolveSessionPaths } from "./resolveSessionPaths.js";
|
|
11
11
|
import { setSavedSessionPath } from "./sessionState.js";
|
|
@@ -67,8 +67,8 @@ For detailed documentation visit: https://www.skyramp.dev/docs/load-test/advance
|
|
|
67
67
|
playwrightSaveStoragePath: basePlaywrightSchema.shape.playwrightSaveStoragePath,
|
|
68
68
|
playwrightViewportSize: basePlaywrightSchema.shape.playwrightViewportSize,
|
|
69
69
|
runtime: z
|
|
70
|
-
.
|
|
71
|
-
.default(
|
|
70
|
+
.literal(RuntimeEnvironment.DOCKER)
|
|
71
|
+
.default(RuntimeEnvironment.DOCKER)
|
|
72
72
|
.describe("Runtime environment for trace collection. Currently only 'docker' is supported and is used as the default."),
|
|
73
73
|
include: z
|
|
74
74
|
.array(z.string())
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Neutral result shape for `skyramp_run_existing_tests` — a repo's OWN
|
|
3
|
+
* (external, user-authored) test suite, run at PR time to CONFIRM which failures
|
|
4
|
+
* a change actually causes. Framework-agnostic: the Playwright adapter (and
|
|
5
|
+
* future pytest/jest adapters) all normalize into these types so the
|
|
6
|
+
* maintenance pipeline and report consume one shape.
|
|
7
|
+
*/
|
|
8
|
+
/** `error` (fixture/collection/timeout) is kept distinct from `fail`
|
|
9
|
+
* (assertion) end-to-end — see external-exec plan item 6. */
|
|
10
|
+
export type ExternalTestStatus = "pass" | "fail" | "error" | "skipped";
|
|
11
|
+
export interface ExternalTestResult {
|
|
12
|
+
/** Human-readable identity, e.g. `<file> › <describe…> › <title>`. */
|
|
13
|
+
testId: string;
|
|
14
|
+
/** Spec file the test lives in. */
|
|
15
|
+
file: string;
|
|
16
|
+
status: ExternalTestStatus;
|
|
17
|
+
/** Failure/error text (ANSI-stripped). Present only for fail/error. */
|
|
18
|
+
message?: string;
|
|
19
|
+
durationMs: number;
|
|
20
|
+
}
|
|
21
|
+
export interface ExternalRunSummary {
|
|
22
|
+
/** Tests that produced a real verdict (pass + fail + error); skipped excluded. */
|
|
23
|
+
ran: number;
|
|
24
|
+
failed: number;
|
|
25
|
+
errored: number;
|
|
26
|
+
skipped: number;
|
|
27
|
+
/** True when the candidate set was cut by maxTests/timeout — reported honestly. */
|
|
28
|
+
truncated: boolean;
|
|
29
|
+
}
|
|
30
|
+
/** Structured env-health diagnostics for logging (populated by the framework adapter). */
|
|
31
|
+
export interface ExternalRunDiagnostics {
|
|
32
|
+
/** All top-level Playwright run-level error messages (ANSI-stripped, first line each). */
|
|
33
|
+
runLevelErrors: string[];
|
|
34
|
+
/** A global-setup/teardown (infra) spec reported fail/error — the environment genuinely
|
|
35
|
+
* did not come up (distinct from a product-test failure). */
|
|
36
|
+
infraSpecFailed: boolean;
|
|
37
|
+
/** A run-level error is the `--max-failures` "stopped early" bail — benign for env-health
|
|
38
|
+
* (it means a test failed and the bounded run stopped, NOT that the env is unhealthy). */
|
|
39
|
+
maxFailuresBail: boolean;
|
|
40
|
+
}
|
|
41
|
+
export interface ParsedExternalRun {
|
|
42
|
+
results: ExternalTestResult[];
|
|
43
|
+
summary: ExternalRunSummary;
|
|
44
|
+
/** False when a wall of red is environmental (setup/collection failure), not a
|
|
45
|
+
* PR signal — consumers must NOT treat results as confirmed failures. */
|
|
46
|
+
environmentHealthy: boolean;
|
|
47
|
+
healthDetail?: string;
|
|
48
|
+
/** Structured diagnostics for logging (optional; set by the Playwright adapter). */
|
|
49
|
+
diagnostics?: ExternalRunDiagnostics;
|
|
50
|
+
/** True when the run was skipped because the repo has no runnable external suite
|
|
51
|
+
* (no Playwright service / no `testRunCommand`) — i.e. no external tests to
|
|
52
|
+
* run. Distinct from a green zero-test run: consumers must treat this as "not
|
|
53
|
+
* executed", never "passed". */
|
|
54
|
+
skipped?: boolean;
|
|
55
|
+
/** Why the run was skipped (present only when `skipped`). */
|
|
56
|
+
skipReason?: string;
|
|
57
|
+
}
|
|
58
|
+
/** One persisted run of the external suite, appended to
|
|
59
|
+
* UnifiedAnalysisState.externalTestResults so drift analysis / the report can
|
|
60
|
+
* fold in CONFIRMED failures instead of guesses. */
|
|
61
|
+
export interface ExternalTestRunRecord extends ParsedExternalRun {
|
|
62
|
+
mode: "confirm" | "verify";
|
|
63
|
+
/** ISO timestamp when the run completed. */
|
|
64
|
+
ranAt: string;
|
|
65
|
+
/** owner/repo this run targeted (multi-repo runs); absent → primary. */
|
|
66
|
+
repository?: string;
|
|
67
|
+
}
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Neutral result shape for `skyramp_run_existing_tests` — a repo's OWN
|
|
3
|
+
* (external, user-authored) test suite, run at PR time to CONFIRM which failures
|
|
4
|
+
* a change actually causes. Framework-agnostic: the Playwright adapter (and
|
|
5
|
+
* future pytest/jest adapters) all normalize into these types so the
|
|
6
|
+
* maintenance pipeline and report consume one shape.
|
|
7
|
+
*/
|
|
8
|
+
export {};
|
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
* and executed by the coding assistant following the workflow instructions.
|
|
5
5
|
*/
|
|
6
6
|
/** Unique identifier for a one-click command */
|
|
7
|
-
export type OneClickCommandId = "test_given_endpoint_comprehensively" | "full_repo_scan_recommend_generate_and_execute_top_n_tests";
|
|
7
|
+
export type OneClickCommandId = "test_given_endpoint_comprehensively" | "full_repo_scan_recommend_generate_and_execute_top_n_tests" | "local_dev_test_changes";
|
|
8
8
|
/** Metadata for intent recognition by coding assistants */
|
|
9
9
|
export interface CommandIntentMetadata {
|
|
10
10
|
/** Context indicators describing when to use and when NOT to use this workflow */
|
|
@@ -74,6 +74,26 @@ export interface Budgeter {
|
|
|
74
74
|
readonly name: string;
|
|
75
75
|
select(ranked: Candidate[], ctx: BudgetContext): SelectionResult;
|
|
76
76
|
}
|
|
77
|
+
/**
|
|
78
|
+
* Slug of a scenario name alone (no content hash) — stable across independent
|
|
79
|
+
* drafts of "the same" scenario even when their steps[] differ (e.g. the
|
|
80
|
+
* server's auto-drafted version vs. the agent's own re-drafted version of a
|
|
81
|
+
* scenario with an identical name). Used to dedupe by scenario identity where
|
|
82
|
+
* `computeCandidateId`'s content-sensitive hash would wrongly treat two drafts
|
|
83
|
+
* of the same named scenario as distinct (SKYR-4026).
|
|
84
|
+
*/
|
|
85
|
+
export declare function scenarioNameSlug(scenarioName: string | undefined): string;
|
|
86
|
+
/**
|
|
87
|
+
* Merge-dedup identity key for a scenario name — like {@link scenarioNameSlug}
|
|
88
|
+
* but NOT truncated and with no shared fallback for a missing name. The 48-char
|
|
89
|
+
* truncation in `scenarioNameSlug` is fine for `computeCandidateId` (combined
|
|
90
|
+
* with a content hash for uniqueness), but used bare as a Map key it would
|
|
91
|
+
* silently collapse two distinct scenarios sharing a long common prefix, and
|
|
92
|
+
* collapse every candidate with no scenarioName onto the same key. Returns ""
|
|
93
|
+
* for a missing name so the caller can fall back to something already unique
|
|
94
|
+
* (e.g. candidateId) instead of a shared literal.
|
|
95
|
+
*/
|
|
96
|
+
export declare function scenarioMergeKey(scenarioName: string | undefined): string;
|
|
77
97
|
/**
|
|
78
98
|
* Deterministic candidate id: a slug of the scenario name plus an 8-char content
|
|
79
99
|
* hash. Pure and reproducible — the same scenario always yields the same id, so
|
|
@@ -17,6 +17,37 @@ export var CandidateSource;
|
|
|
17
17
|
CandidateSource["AGENT"] = "agent";
|
|
18
18
|
CandidateSource["SERVER"] = "server";
|
|
19
19
|
})(CandidateSource || (CandidateSource = {}));
|
|
20
|
+
/**
|
|
21
|
+
* Slug of a scenario name alone (no content hash) — stable across independent
|
|
22
|
+
* drafts of "the same" scenario even when their steps[] differ (e.g. the
|
|
23
|
+
* server's auto-drafted version vs. the agent's own re-drafted version of a
|
|
24
|
+
* scenario with an identical name). Used to dedupe by scenario identity where
|
|
25
|
+
* `computeCandidateId`'s content-sensitive hash would wrongly treat two drafts
|
|
26
|
+
* of the same named scenario as distinct (SKYR-4026).
|
|
27
|
+
*/
|
|
28
|
+
export function scenarioNameSlug(scenarioName) {
|
|
29
|
+
return ((scenarioName ?? "")
|
|
30
|
+
.toLowerCase()
|
|
31
|
+
.replace(/[^a-z0-9]+/g, "-")
|
|
32
|
+
.replace(/^-+|-+$/g, "")
|
|
33
|
+
.slice(0, 48) || "scenario");
|
|
34
|
+
}
|
|
35
|
+
/**
|
|
36
|
+
* Merge-dedup identity key for a scenario name — like {@link scenarioNameSlug}
|
|
37
|
+
* but NOT truncated and with no shared fallback for a missing name. The 48-char
|
|
38
|
+
* truncation in `scenarioNameSlug` is fine for `computeCandidateId` (combined
|
|
39
|
+
* with a content hash for uniqueness), but used bare as a Map key it would
|
|
40
|
+
* silently collapse two distinct scenarios sharing a long common prefix, and
|
|
41
|
+
* collapse every candidate with no scenarioName onto the same key. Returns ""
|
|
42
|
+
* for a missing name so the caller can fall back to something already unique
|
|
43
|
+
* (e.g. candidateId) instead of a shared literal.
|
|
44
|
+
*/
|
|
45
|
+
export function scenarioMergeKey(scenarioName) {
|
|
46
|
+
return (scenarioName ?? "")
|
|
47
|
+
.toLowerCase()
|
|
48
|
+
.replace(/[^a-z0-9]+/g, "-")
|
|
49
|
+
.replace(/^-+|-+$/g, "");
|
|
50
|
+
}
|
|
20
51
|
/**
|
|
21
52
|
* Deterministic candidate id: a slug of the scenario name plus an 8-char content
|
|
22
53
|
* hash. Pure and reproducible — the same scenario always yields the same id, so
|
|
@@ -26,12 +57,7 @@ export var CandidateSource;
|
|
|
26
57
|
* cosmetic re-serialization does not change the id while a real edit does.
|
|
27
58
|
*/
|
|
28
59
|
export function computeCandidateId(scenario) {
|
|
29
|
-
const
|
|
30
|
-
const slug = slugSource
|
|
31
|
-
.toLowerCase()
|
|
32
|
-
.replace(/[^a-z0-9]+/g, "-")
|
|
33
|
-
.replace(/^-+|-+$/g, "")
|
|
34
|
-
.slice(0, 48) || "scenario";
|
|
60
|
+
const slug = scenarioNameSlug(scenario.scenarioName);
|
|
35
61
|
const canonical = {
|
|
36
62
|
scenarioName: scenario.scenarioName ?? "",
|
|
37
63
|
category: scenario.category ?? "",
|