@skyramp/mcp 0.3.5 → 0.3.6-rc.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/build/playwright/registerPlaywrightTools.js +92 -30
- package/build/playwright/traceRecordingPrompt.d.ts +6 -0
- package/build/playwright/traceRecordingPrompt.js +6 -2
- package/build/prompts/code-reuse.d.ts +1 -2
- package/build/prompts/code-reuse.js +182 -77
- package/build/prompts/modularization/integration-test-modularization.d.ts +2 -0
- package/build/prompts/modularization/integration-test-modularization.js +83 -41
- package/build/prompts/modularization/render.d.ts +18 -0
- package/build/prompts/modularization/render.js +12 -0
- package/build/prompts/modularization/ui-test-modularization.d.ts +3 -1
- package/build/prompts/modularization/ui-test-modularization.js +89 -47
- package/build/prompts/pom-aware-code-reuse.js +3 -1
- package/build/prompts/shared-helper-policy.d.ts +57 -0
- package/build/prompts/shared-helper-policy.js +135 -0
- package/build/prompts/test-recommendation/diffExecutionPlan.js +62 -56
- package/build/prompts/test-recommendation/fullRepoCatalog.js +19 -8
- package/build/prompts/test-recommendation/recommendationShared.d.ts +28 -6
- package/build/prompts/test-recommendation/recommendationShared.js +90 -16
- package/build/prompts/test-recommendation/registerRecommendTestsPrompt.js +22 -0
- package/build/prompts/test-recommendation/test-recommendation-prompt.d.ts +2 -2
- package/build/prompts/test-recommendation/test-recommendation-prompt.js +3 -3
- package/build/prompts/testbot/testbot-prompts.js +80 -34
- package/build/recommendation/budgeters/shared.js +105 -27
- package/build/recommendation/discriminators.js +13 -2
- package/build/recommendation/planRanker.d.ts +6 -6
- package/build/recommendation/planRanker.js +6 -61
- package/build/services/AnalyticsService.d.ts +7 -0
- package/build/services/AnalyticsService.js +7 -1
- package/build/services/ModularizationService.js +1 -3
- package/build/services/TestDiscoveryService.d.ts +0 -2
- package/build/services/TestDiscoveryService.js +2 -37
- package/build/services/TestGenerationService.d.ts +16 -0
- package/build/services/TestGenerationService.js +86 -10
- package/build/services/containerEnv.js +13 -12
- package/build/tools/code-refactor/codeReuseTool.js +279 -93
- package/build/tools/code-refactor/enhance-state.d.ts +49 -0
- package/build/tools/code-refactor/enhance-state.js +109 -0
- package/build/tools/code-refactor/enhanceAssertionsTool.js +34 -1
- package/build/tools/code-refactor/modularizationTool.js +9 -2
- package/build/tools/code-refactor/reuse-outcome.d.ts +23 -1
- package/build/tools/code-refactor/reuse-outcome.js +14 -4
- package/build/tools/code-refactor/reuse-state.d.ts +127 -5
- package/build/tools/code-refactor/reuse-state.js +628 -16
- package/build/tools/code-refactor/utils-verify-gates.d.ts +26 -0
- package/build/tools/code-refactor/utils-verify-gates.js +100 -0
- package/build/tools/code-refactor/verify-gates.d.ts +2 -1
- package/build/tools/code-refactor/verify-gates.js +90 -25
- package/build/tools/executeSkyrampTestTool.d.ts +19 -0
- package/build/tools/executeSkyrampTestTool.js +158 -8
- package/build/tools/generate-tests/generateBatchScenarioRestTool.js +2 -2
- package/build/tools/generate-tests/generateE2ERestTool.js +16 -0
- package/build/tools/generate-tests/generateUIRestTool.d.ts +1 -0
- package/build/tools/generate-tests/generateUIRestTool.js +22 -0
- package/build/tools/generate-tests/scenarioLint.d.ts +2 -0
- package/build/tools/generate-tests/scenarioLint.js +127 -19
- package/build/tools/generate-tests/trace-reuse-guard.d.ts +20 -0
- package/build/tools/generate-tests/trace-reuse-guard.js +93 -0
- package/build/tools/submitReportTool.d.ts +38 -38
- package/build/tools/submitReportTool.js +411 -114
- package/build/tools/test-management/analyzeChangesTool.d.ts +24 -1
- package/build/tools/test-management/analyzeChangesTool.js +71 -10
- package/build/tools/test-management/analyzeTestHealthTool.js +7 -7
- package/build/tools/test-management/registerTestPlanTool.d.ts +203 -0
- package/build/tools/test-management/registerTestPlanTool.js +70 -12
- package/build/types/Recommendation.d.ts +34 -5
- package/build/types/RepositoryAnalysis.d.ts +133 -114
- package/build/types/RepositoryAnalysis.js +1 -1
- package/build/types/ReuseOutcome.d.ts +102 -6
- package/build/types/ReuseOutcome.js +16 -2
- package/build/types/TestRecommendation.js +21 -3
- package/build/types/TestTypes.js +14 -8
- package/build/types/TestbotReport.d.ts +10 -1
- package/build/types/index.d.ts +2 -2
- package/build/types/index.js +1 -1
- package/build/utils/AnalysisStateManager.d.ts +57 -1
- package/build/utils/AnalysisStateManager.js +54 -5
- package/build/utils/branchDiff.d.ts +10 -0
- package/build/utils/branchDiff.js +28 -0
- package/build/utils/changedRoutes.d.ts +29 -0
- package/build/utils/changedRoutes.js +87 -0
- package/build/utils/featureFlags.d.ts +21 -0
- package/build/utils/featureFlags.js +23 -0
- package/build/utils/frontendIntegration.js +34 -4
- package/build/utils/importerHop.d.ts +2 -8
- package/build/utils/importerHop.js +15 -53
- package/build/utils/pathMatching.d.ts +38 -0
- package/build/utils/pathMatching.js +71 -0
- package/build/utils/pathSignatures.d.ts +22 -0
- package/build/utils/pathSignatures.js +57 -0
- package/build/utils/planMatchKeys.d.ts +16 -3
- package/build/utils/planMatchKeys.js +26 -10
- package/build/utils/pluralization.d.ts +10 -0
- package/build/utils/pluralization.js +18 -0
- package/build/utils/pom-catalog-parse.d.ts +52 -0
- package/build/utils/pom-catalog-parse.js +141 -0
- package/build/utils/pom-scope/selector-extractor.d.ts +12 -0
- package/build/utils/pom-scope/selector-extractor.js +34 -8
- package/build/utils/pom-verify/verify.d.ts +6 -5
- package/build/utils/pom-verify/verify.js +8 -6
- package/build/utils/reportVerification.d.ts +64 -4
- package/build/utils/reportVerification.js +228 -3
- package/build/utils/reuseRouting.d.ts +3 -0
- package/build/utils/reuseRouting.js +50 -0
- package/build/utils/routeParsers.d.ts +2 -0
- package/build/utils/routeParsers.js +65 -8
- package/build/utils/scenarioDrafting.d.ts +1 -1
- package/build/utils/scenarioDrafting.js +57 -45
- package/build/utils/subjectEndpoints.d.ts +19 -0
- package/build/utils/subjectEndpoints.js +98 -0
- package/build/utils/testFileClassification.d.ts +11 -0
- package/build/utils/testFileClassification.js +47 -0
- package/build/utils/uiPageEnumerator.d.ts +45 -19
- package/build/utils/uiPageEnumerator.js +95 -51
- package/build/utils/utils-verify/allow.d.ts +16 -0
- package/build/utils/utils-verify/allow.js +68 -0
- package/build/utils/utils-verify/call-sites.d.ts +34 -0
- package/build/utils/utils-verify/call-sites.js +154 -0
- package/build/utils/utils-verify/index.d.ts +7 -0
- package/build/utils/utils-verify/index.js +7 -0
- package/build/utils/utils-verify/language-spec.d.ts +91 -0
- package/build/utils/utils-verify/language-spec.js +210 -0
- package/build/utils/utils-verify/locate.d.ts +39 -0
- package/build/utils/utils-verify/locate.js +199 -0
- package/build/utils/utils-verify/parse.d.ts +34 -0
- package/build/utils/utils-verify/parse.js +177 -0
- package/build/utils/utils-verify/stage.d.ts +24 -0
- package/build/utils/utils-verify/stage.js +107 -0
- package/build/utils/utils-verify/verify.d.ts +63 -0
- package/build/utils/utils-verify/verify.js +168 -0
- package/build/utils/utils.d.ts +3 -1
- package/build/utils/utils.js +3 -1
- package/build/workspace/workspace.d.ts +32 -32
- package/node_modules/playwright/lib/mcp/skyramp/assertTool.js +9 -5
- package/node_modules/playwright/lib/mcp/skyramp/loadTraceTool.js +16 -0
- package/node_modules/playwright/lib/mcp/skyramp/skyRampImport.js +2 -0
- package/node_modules/playwright/lib/mcp/skyramp/traceRecordingBackend.js +115 -14
- package/node_modules/playwright/lib/mcp/test/skyRampExport.js +13 -1
- package/node_modules/playwright/node_modules/playwright-core/.DS_Store +0 -0
- package/node_modules/playwright/node_modules/playwright-core/lib/server/codegen/skyramp/jsonlReader.js +2 -0
- package/node_modules/playwright/node_modules/playwright-core/lib/vite/htmlReport/index.html +27 -253
- package/node_modules/playwright/node_modules/playwright-core/lib/vite/recorder/assets/{codeMirrorModule-DtudTj_v.js → codeMirrorModule-DJMC4zNo.js} +1 -1
- package/node_modules/playwright/node_modules/playwright-core/lib/vite/recorder/assets/index-BW82eAUI.js +196 -0
- package/node_modules/playwright/node_modules/playwright-core/lib/vite/recorder/index.html +1 -1
- package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/assets/{codeMirrorModule-FNMuBzX1.js → codeMirrorModule-CZfp96qZ.js} +1 -1
- package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/assets/defaultSettingsView-gpLo02E0.js +809 -0
- package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/index.Bq1r1URj.js +2 -0
- package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/index.html +2 -2
- package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/uiMode.VEfqi1qN.js +5 -0
- package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/uiMode.html +2 -2
- package/node_modules/playwright/node_modules/playwright-core/package.json +1 -1
- package/node_modules/playwright/node_modules/playwright-core/src/server/codegen/skyramp/jsonlReader.ts +1 -1
- package/node_modules/playwright/package.json +1 -1
- package/package.json +2 -2
- package/node_modules/playwright/node_modules/playwright-core/lib/vite/recorder/assets/index-BpDwp16L.js +0 -422
- package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/assets/defaultSettingsView-Co9upU5h.js +0 -1035
- package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/index.DXNIQ_dx.js +0 -2
- package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/uiMode.CIKB3XSv.js +0 -5
|
@@ -3,24 +3,27 @@ import { logger } from "../utils/logger.js";
|
|
|
3
3
|
import * as fs from "fs/promises";
|
|
4
4
|
import * as path from "path";
|
|
5
5
|
import { AnalyticsService } from "../services/AnalyticsService.js";
|
|
6
|
-
import { TEST_CATEGORIES, externalCategory } from "../types/TestRecommendation.js";
|
|
6
|
+
import { TEST_CATEGORIES, externalCategory, } from "../types/TestRecommendation.js";
|
|
7
7
|
import { TestType, HttpMethod } from "../types/TestTypes.js";
|
|
8
8
|
import { DriftAction } from "../types/TestAnalysis.js";
|
|
9
9
|
import { TestExecutionStatus } from "../types/TestExecution.js";
|
|
10
10
|
import { IssueFoundCategory } from "../types/TestbotReport.js";
|
|
11
|
-
import { StateManager, runArtifactDir, getTestsRepoDir } from "../utils/AnalysisStateManager.js";
|
|
11
|
+
import { StateManager, runArtifactDir, getTestsRepoDir, } from "../utils/AnalysisStateManager.js";
|
|
12
12
|
import { toolError, testFileMatches } from "../utils/utils.js";
|
|
13
13
|
import { matchesApprovedPlan } from "../utils/planMatchKeys.js";
|
|
14
14
|
import { isTestbotEnabled } from "../utils/featureFlags.js";
|
|
15
|
-
import {
|
|
15
|
+
import { findInvalidSourceCitations, findUnchangedFileClaims, listChangedFiles, listChangedFilesAcross, } from "../utils/reportVerification.js";
|
|
16
16
|
import { getReportLanguage, isEnforcedReportLanguage, findLanguageViolations, findLanguageNearMisses, reportLanguageDisplayName, } from "../utils/reportLanguage.js";
|
|
17
|
-
import { rederiveReuseOutcome } from "./code-refactor/reuse-state.js";
|
|
17
|
+
import { rederiveReuseOutcome, reuseChainSkipped, } from "./code-refactor/reuse-state.js";
|
|
18
18
|
// SKYR-3879 Path B: which testTypes the register-plan checkpoint gates. Mirrors
|
|
19
19
|
// the generation tools actually wired to planGuard (batch-scenario/integration,
|
|
20
20
|
// contract) — UI and E2E are on a separate blueprint-grounded pipeline and are
|
|
21
21
|
// NOT gated at generation time (see planGuard.ts wiring), so they are excluded
|
|
22
22
|
// here too rather than surprising the agent with a report-time-only gate.
|
|
23
|
-
const PLAN_GATED_TEST_TYPES = new Set([
|
|
23
|
+
const PLAN_GATED_TEST_TYPES = new Set([
|
|
24
|
+
TestType.CONTRACT,
|
|
25
|
+
TestType.INTEGRATION,
|
|
26
|
+
]);
|
|
24
27
|
/**
|
|
25
28
|
* Filename of the report, written beside the state file. SKYR-4147: the report path is
|
|
26
29
|
* derived here rather than accepted as a parameter. The caller builds the state file and
|
|
@@ -46,7 +49,9 @@ function planNameCandidates(testId, testType) {
|
|
|
46
49
|
if (!id)
|
|
47
50
|
return [];
|
|
48
51
|
const prefix = `${testType}-`;
|
|
49
|
-
return id.toLowerCase().startsWith(prefix)
|
|
52
|
+
return id.toLowerCase().startsWith(prefix)
|
|
53
|
+
? [id, id.slice(prefix.length)]
|
|
54
|
+
: [id];
|
|
50
55
|
}
|
|
51
56
|
/**
|
|
52
57
|
* Split an `endpoint` field into one {method, path} per endpoint it names.
|
|
@@ -55,14 +60,20 @@ function planNameCandidates(testId, testType) {
|
|
|
55
60
|
* unmatchable path, so such an entry could match nothing (SKYR-4123).
|
|
56
61
|
*/
|
|
57
62
|
function parseEndpointField(endpoint) {
|
|
58
|
-
const parts = (endpoint ?? "")
|
|
63
|
+
const parts = (endpoint ?? "")
|
|
64
|
+
.split(",")
|
|
65
|
+
.map((p) => p.trim())
|
|
66
|
+
.filter(Boolean);
|
|
59
67
|
if (parts.length === 0)
|
|
60
68
|
return [{}];
|
|
61
69
|
return parts.map((part) => {
|
|
62
70
|
const spaceIdx = part.indexOf(" ");
|
|
63
71
|
if (spaceIdx <= 0)
|
|
64
72
|
return { path: part };
|
|
65
|
-
return {
|
|
73
|
+
return {
|
|
74
|
+
method: part.slice(0, spaceIdx),
|
|
75
|
+
path: part.slice(spaceIdx + 1).trim(),
|
|
76
|
+
};
|
|
66
77
|
});
|
|
67
78
|
}
|
|
68
79
|
// Drift actions that actually modify a test file. VERIFY and IGNORE are
|
|
@@ -89,23 +100,33 @@ const repositoryField = z
|
|
|
89
100
|
* (downstream consumers treat absence as "the primary repo"). */
|
|
90
101
|
function normalizeRepository(item) {
|
|
91
102
|
const trimmed = item.repository?.trim();
|
|
92
|
-
return trimmed
|
|
103
|
+
return trimmed
|
|
104
|
+
? { ...item, repository: trimmed }
|
|
105
|
+
: { ...item, repository: undefined };
|
|
93
106
|
}
|
|
94
107
|
// videoPath is deliberately absent from this input contract: it is attached server-side
|
|
95
108
|
// from the run's execution records (see attachVideoPath), and zod strips any the model
|
|
96
109
|
// supplies anyway. SKYR-4156 is what happens when the agent owns that field instead.
|
|
97
110
|
const testResultSchema = z.object({
|
|
98
|
-
testType: z
|
|
99
|
-
|
|
111
|
+
testType: z
|
|
112
|
+
.nativeEnum(TestType)
|
|
113
|
+
.describe("Type of test. Do not include priority or other metadata in this field."),
|
|
114
|
+
endpoint: z
|
|
115
|
+
.string()
|
|
116
|
+
.describe("HTTP verb and path, e.g. 'GET /api/v1/products'"),
|
|
100
117
|
status: z.enum(["Pass", "Fail", "Skipped"]).describe("Test execution result"),
|
|
101
|
-
details: z
|
|
118
|
+
details: z
|
|
119
|
+
.string()
|
|
120
|
+
.describe("One sentence — no embedded newlines, no markdown. e.g. '10.8s, products_contract_test.py' or 'failed: <one-line error summary>, products_contract_test.py'"),
|
|
102
121
|
// Required for every row: each one reports a specific test file the agent ran, so it
|
|
103
122
|
// can always name it. It is what identifies the row server-side — `endpoint` cannot,
|
|
104
123
|
// since several tests routinely exercise one endpoint — and for ui/e2e it is what
|
|
105
124
|
// attaches the recorded video (SKYR-4156). Not included in the report itself.
|
|
106
125
|
testFilePath: z
|
|
107
126
|
.string()
|
|
108
|
-
.refine((p) => path.isAbsolute(p), {
|
|
127
|
+
.refine((p) => path.isAbsolute(p), {
|
|
128
|
+
message: "testFilePath must be an absolute path",
|
|
129
|
+
})
|
|
109
130
|
.describe("Absolute path of the test file this result is for — the same path you passed to skyramp_execute_test's testFile param. Consumers basename it for display."),
|
|
110
131
|
repository: repositoryField,
|
|
111
132
|
});
|
|
@@ -117,21 +138,47 @@ const testResultSchema = z.object({
|
|
|
117
138
|
// parsing. `reasoning` is free-form prose constrained by the Blueprint
|
|
118
139
|
// Citation Invariant — every element cited must appear in `targetElements`.
|
|
119
140
|
// See testbot prompt step 4 for the full populate-and-render rules.
|
|
141
|
+
/**
|
|
142
|
+
* Accept an omitted key as an explicit `null`.
|
|
143
|
+
*
|
|
144
|
+
* SKYR-4208. The blueprint capture no longer carries a key whose value is null,
|
|
145
|
+
* so an element the agent lifts verbatim simply has no `testId`, `stableId` or
|
|
146
|
+
* `contextText`. Consumers of the report still expect all three keys, so the
|
|
147
|
+
* missing one is filled in here rather than asked for in the prompt.
|
|
148
|
+
*/
|
|
149
|
+
function nullWhenAbsent(inner) {
|
|
150
|
+
return inner.nullish().transform((v) => v ?? null);
|
|
151
|
+
}
|
|
120
152
|
export const targetElementSchema = z.object({
|
|
121
|
-
role: z
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
153
|
+
role: z
|
|
154
|
+
.string()
|
|
155
|
+
.describe("ARIA role (e.g. 'button', 'heading', 'textbox', 'link'). Lifted verbatim from the captured blueprint element's `role` field."),
|
|
156
|
+
accessibleName: z
|
|
157
|
+
.string()
|
|
158
|
+
.describe("Computed accessible name (e.g. 'Save changes', 'Order Details'). Lifted verbatim from the captured blueprint element's `accessibleName` field. Must match the bolded name in `reasoning` character-for-character."),
|
|
159
|
+
testId: nullWhenAbsent(z.string()).describe("data-testid attribute value (preferred locator handle), or null when the element has no data-testid. Lifted from the blueprint element's `testId`."),
|
|
160
|
+
stableId: nullWhenAbsent(z.string()).describe("Unique HTML id attribute value (fallback locator when no testId), or null. Lifted from the blueprint element's `stableId`."),
|
|
161
|
+
contextText: nullWhenAbsent(z.array(z.string())).describe("Disambiguating row text for elements inside repeating sections (table rows, list items): the row's surrounding non-interactive text. Lifted from the blueprint repeatingElement's items[].contextText. null for non-repeating elements."),
|
|
162
|
+
mutability: z
|
|
163
|
+
.enum(["mutable", "immutable", "unknown"])
|
|
164
|
+
.optional()
|
|
165
|
+
.describe("Whether the element's content/state is expected to change between captures. 'mutable' elements are behavioral-test targets; 'immutable' are smoke-test targets. Copied from the blueprint element's `mutability`."),
|
|
166
|
+
widgetType: z
|
|
167
|
+
.enum(["native", "custom", "unknown"])
|
|
168
|
+
.optional()
|
|
169
|
+
.describe("'native' = HTML built-ins (button, input). 'custom' or 'unknown' = non-standard composites; fall back to snapshot-driven trial clicks for interaction. Copied from the blueprint element's `widgetType`."),
|
|
128
170
|
});
|
|
129
171
|
// Page metadata for a UI recommendation. Lifted from the BlueprintCapture
|
|
130
172
|
// the agent used to populate `targetElements`. Codegen reads `url` to drive
|
|
131
173
|
// page.goto(); the verifier uses `pageHash` to detect stale captures.
|
|
132
174
|
export const pageContextSchema = z.object({
|
|
133
|
-
url: z
|
|
134
|
-
|
|
175
|
+
url: z
|
|
176
|
+
.string()
|
|
177
|
+
.describe("URL of the page where the test runs. Lifted from BlueprintCapture.url."),
|
|
178
|
+
pageHash: z
|
|
179
|
+
.string()
|
|
180
|
+
.optional()
|
|
181
|
+
.describe("Opaque hash of the captured page state (BlueprintCapture.pageHash). Lets the verifier confirm the recommendation was grounded in a still-current capture."),
|
|
135
182
|
});
|
|
136
183
|
/**
|
|
137
184
|
* SKYR-4193: LLMs habitually emit every key a schema declares, using a
|
|
@@ -165,21 +212,53 @@ function stripNullGroundingFields(val) {
|
|
|
165
212
|
// interface that adds an `implemented: boolean` field. Both describe the same
|
|
166
213
|
// concept (a test recommendation) — the only difference is whether it was
|
|
167
214
|
// generated in this run or left for later. Tracked per Archit's review comment.
|
|
168
|
-
export const newTestSchema = z.preprocess(stripNullGroundingFields, z
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
215
|
+
export const newTestSchema = z.preprocess(stripNullGroundingFields, z
|
|
216
|
+
.object({
|
|
217
|
+
testId: z
|
|
218
|
+
.string()
|
|
219
|
+
.describe("Human-readable kebab-case identifier, e.g. 'contract-get-products' or 'integration-users-orders-workflow'. Format: '<testType>-<method>-<resource>' for single-endpoint tests or '<testType>-<scenario-slug>' for multi-step tests. Must be unique within the report."),
|
|
220
|
+
testType: z
|
|
221
|
+
.nativeEnum(TestType)
|
|
222
|
+
.describe("Type of test created. Do not include priority or other metadata in this field."),
|
|
223
|
+
category: z
|
|
224
|
+
.preprocess((val) => externalCategory(val), z.enum(TEST_CATEGORIES))
|
|
225
|
+
.describe("Test category — critical categories (security_boundary, business_rule, data_integrity, breaking_change) get generation priority over workflow"),
|
|
226
|
+
endpoint: z
|
|
227
|
+
.string()
|
|
228
|
+
.describe("HTTP verb and path, e.g. 'GET /api/v1/products'"),
|
|
173
229
|
fileName: z.string().describe("Name of the generated test file"),
|
|
174
|
-
description: z
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
|
|
230
|
+
description: z
|
|
231
|
+
.string()
|
|
232
|
+
.trim()
|
|
233
|
+
.min(1)
|
|
234
|
+
.describe("What the test does — the steps and assertions, not the bugs it finds. e.g. 'Creates a collection, adds a link, then verifies the link exists'. Do NOT describe expected failures or bugs here — those belong in issuesFound."),
|
|
235
|
+
scenarioFile: z
|
|
236
|
+
.string()
|
|
237
|
+
.optional()
|
|
238
|
+
.describe("Path to the scenario JSON file if one was generated (e.g. 'tests/scenario_collections-links.json')"),
|
|
239
|
+
traceFile: z
|
|
240
|
+
.string()
|
|
241
|
+
.optional()
|
|
242
|
+
.describe("Path to the backend trace file if used or created"),
|
|
243
|
+
frontendTrace: z
|
|
244
|
+
.string()
|
|
245
|
+
.optional()
|
|
246
|
+
.describe("Path to the Playwright/UI trace file if used or created"),
|
|
247
|
+
reasoning: z
|
|
248
|
+
.string()
|
|
249
|
+
.describe("Why this test was created: what production risk it mitigates, what code pattern it targets, or what coverage gap it fills"),
|
|
179
250
|
repository: repositoryField,
|
|
180
|
-
targetElements: z
|
|
181
|
-
|
|
182
|
-
|
|
251
|
+
targetElements: z
|
|
252
|
+
.array(targetElementSchema)
|
|
253
|
+
.min(1)
|
|
254
|
+
.nullable()
|
|
255
|
+
.optional()
|
|
256
|
+
.describe("UI tests only: structured grounding for one or more elements the test targets. Most tests target a single element (array length 1); render-state and multi-step UI tests target several (array length 2+). Each entry must be lifted verbatim from a captured blueprint element. Set to null when blueprint capture failed (also requires '[no-blueprint-data]' marker in BOTH description and reasoning). See Blueprint Citation Invariant in testbot prompt."),
|
|
257
|
+
pageContext: pageContextSchema
|
|
258
|
+
.optional()
|
|
259
|
+
.describe("UI tests only: page metadata for the test. Lifted from the BlueprintCapture used during grounding."),
|
|
260
|
+
})
|
|
261
|
+
.superRefine((rec, ctx) => {
|
|
183
262
|
// targetElements / pageContext are UI-only.
|
|
184
263
|
if (rec.testType !== TestType.UI) {
|
|
185
264
|
for (const field of ["targetElements", "pageContext"]) {
|
|
@@ -210,7 +289,8 @@ export const newTestSchema = z.preprocess(stripNullGroundingFields, z.object({
|
|
|
210
289
|
});
|
|
211
290
|
}
|
|
212
291
|
// pageContext is required when targetElements is an array (grounded), forbidden when null.
|
|
213
|
-
if (Array.isArray(rec.targetElements) &&
|
|
292
|
+
if (Array.isArray(rec.targetElements) &&
|
|
293
|
+
rec.pageContext === undefined) {
|
|
214
294
|
ctx.addIssue({
|
|
215
295
|
code: z.ZodIssueCode.custom,
|
|
216
296
|
path: ["pageContext"],
|
|
@@ -233,7 +313,8 @@ export const newTestSchema = z.preprocess(stripNullGroundingFields, z.object({
|
|
|
233
313
|
message: "reasoning must contain '[no-blueprint-data]' when targetElements is null.",
|
|
234
314
|
});
|
|
235
315
|
}
|
|
236
|
-
if (rec.description &&
|
|
316
|
+
if (rec.description &&
|
|
317
|
+
!rec.description.includes("[no-blueprint-data]")) {
|
|
237
318
|
ctx.addIssue({
|
|
238
319
|
code: z.ZodIssueCode.custom,
|
|
239
320
|
path: ["description"],
|
|
@@ -243,8 +324,18 @@ export const newTestSchema = z.preprocess(stripNullGroundingFields, z.object({
|
|
|
243
324
|
}
|
|
244
325
|
}
|
|
245
326
|
}));
|
|
246
|
-
|
|
247
|
-
|
|
327
|
+
/** A citation string the submit-time check verifies and the report then ships.
|
|
328
|
+
* Normalized HERE, once, so the value the verifier resolved is byte-identical to
|
|
329
|
+
* the value written: the check trimmed its input, so a padded " src/x.py " used to
|
|
330
|
+
* validate against the real path and ship the unusable padded one. A
|
|
331
|
+
* blank/whitespace-only value means "no citation", the same rule `repository` uses. */
|
|
332
|
+
const citationString = z.preprocess((v) => (typeof v === "string" ? v.trim() || undefined : v), z.string().optional());
|
|
333
|
+
const issueFoundSchema = z
|
|
334
|
+
.object({
|
|
335
|
+
description: z
|
|
336
|
+
.string()
|
|
337
|
+
.describe("One-line description. Do NOT prefix with the severity level — severity is a separate field. Include code logic bugs from the diff, test generation/execution failures, and environment misconfiguration. " +
|
|
338
|
+
"When sourceFile or sourceSymbol is set, quote the offending line inside backticks in this description — the exact code, copied from the file, so review tools can find it."),
|
|
248
339
|
severity: z
|
|
249
340
|
.enum(["critical", "high", "medium", "low"])
|
|
250
341
|
.optional()
|
|
@@ -256,36 +347,130 @@ const issueFoundSchema = z.object({
|
|
|
256
347
|
.describe("Issue classification. bug = a product/code defect, e.g. found by a test or in the diff. " +
|
|
257
348
|
"lint = a linter or formatter finding (eslint, flake8, prettier). " +
|
|
258
349
|
"type = a type-check failure (tsc, mypy). " +
|
|
259
|
-
"config = environment or tooling misconfiguration (wrong workspace auth type, missing env var, setup command failure)
|
|
350
|
+
"config = environment or tooling misconfiguration (wrong workspace auth type, missing env var, setup command failure), " +
|
|
351
|
+
"and also every Skyramp tool or environment failure — a generation or execution tool error, an unreachable app, a missing credential, a failed capture. " +
|
|
260
352
|
"The report renders lint/type/config entries in a separate 'Configuration Errors' section so product bugs stay prominent under 'Issues Found'."),
|
|
261
353
|
repository: repositoryField,
|
|
354
|
+
sourceFile: citationString.describe("Path of the application file whose code is missing or wrong, relative to the repository root (e.g. 'src/crud/products.py'). " +
|
|
355
|
+
"REQUIRED when category is 'bug'. " +
|
|
356
|
+
"Cite only a file you actually opened this run: this tool rejects the report if the path does not exist in the repository."),
|
|
357
|
+
sourceSymbol: citationString.describe("The function, method, or identifier inside sourceFile whose code is missing or wrong (e.g. 'delete_product'). " +
|
|
358
|
+
"Optional — a config file or template has no symbol to name. Set it whenever the file has one. " +
|
|
359
|
+
"This tool rejects the report if the text does not appear in the cited file."),
|
|
360
|
+
sourceLine: z
|
|
361
|
+
.number()
|
|
362
|
+
.int()
|
|
363
|
+
.positive()
|
|
364
|
+
.optional()
|
|
365
|
+
.describe("1-based line number in sourceFile. Optional and advisory. " +
|
|
366
|
+
"Give the line of the offending statement itself, not the line of the enclosing function's signature — review views place the finding on exactly this line. " +
|
|
367
|
+
"The tool does not check that the line still holds the cited code, because line numbers drift."),
|
|
368
|
+
})
|
|
369
|
+
.superRefine((issue, ctx) => {
|
|
370
|
+
// A `bug` entry asserts that product code is wrong, so it has to name the code:
|
|
371
|
+
// without a file the entry is prose a reviewer cannot act on, and nothing
|
|
372
|
+
// downstream can place it in the diff. Only the file is required — a config file
|
|
373
|
+
// or a template legitimately has no symbol to name, and line numbers are
|
|
374
|
+
// advisory. The other categories are tooling findings that often have no single
|
|
375
|
+
// source location (an unreachable app, a missing env var), so the requirement is
|
|
376
|
+
// scoped to `bug` alone.
|
|
377
|
+
//
|
|
378
|
+
// The exit when the code cannot be named is NOT to invent a citation: a failure
|
|
379
|
+
// the agent cannot localize is already reported by its `testResults` row, and
|
|
380
|
+
// a tool or environment failure belongs under `config`.
|
|
381
|
+
if (issue.category !== IssueFoundCategory.Bug)
|
|
382
|
+
return;
|
|
383
|
+
if (issue.sourceFile !== undefined)
|
|
384
|
+
return;
|
|
385
|
+
ctx.addIssue({
|
|
386
|
+
code: z.ZodIssueCode.custom,
|
|
387
|
+
path: ["sourceFile"],
|
|
388
|
+
message: `sourceFile is required when category is 'bug' — a bug entry must name the code it is about. ` +
|
|
389
|
+
`Open the file, then set sourceFile (path relative to the repository root). Also set sourceSymbol (the function or identifier inside it) when the file has one, and sourceLine (the line of the offending statement) when you have it — both are optional. ` +
|
|
390
|
+
`If you cannot point at the code: a test failure is already reported by its testResults entry, and a tool or environment failure belongs under category 'config' — do not guess a citation.`,
|
|
391
|
+
});
|
|
262
392
|
});
|
|
263
393
|
const scenarioStepSchema = z.object({
|
|
264
|
-
method: z
|
|
265
|
-
|
|
266
|
-
|
|
267
|
-
|
|
268
|
-
|
|
269
|
-
|
|
394
|
+
method: z
|
|
395
|
+
.nativeEnum(HttpMethod)
|
|
396
|
+
.optional()
|
|
397
|
+
.describe("HTTP method. Required for API steps, omit for UI/E2E actions."),
|
|
398
|
+
path: z
|
|
399
|
+
.string()
|
|
400
|
+
.optional()
|
|
401
|
+
.describe("Endpoint or page path (e.g. '/api/v1/products' or '/products'). Required for API steps, omit for UI actions."),
|
|
402
|
+
description: z
|
|
403
|
+
.string()
|
|
404
|
+
.describe("What this step does, e.g. 'Create a product' or 'Click checkout button and verify confirmation'"),
|
|
405
|
+
expectedStatusCode: z
|
|
406
|
+
.number()
|
|
407
|
+
.optional()
|
|
408
|
+
.describe("Expected HTTP status code, e.g. 200, 201, 404"),
|
|
409
|
+
requestBody: z
|
|
410
|
+
.record(z.any())
|
|
411
|
+
.optional()
|
|
412
|
+
.describe("Example request body with realistic field values"),
|
|
413
|
+
responseBody: z
|
|
414
|
+
.union([z.record(z.any()), z.array(z.any())])
|
|
415
|
+
.optional()
|
|
416
|
+
.describe("Key response fields to verify, e.g. { id: 'number', name: 'string', in_stock: 'boolean?' }. An array for a collection/list endpoint that returns a JSON array."),
|
|
270
417
|
});
|
|
271
|
-
export const additionalRecommendationSchema = z.preprocess(stripNullGroundingFields, z
|
|
272
|
-
|
|
273
|
-
|
|
274
|
-
|
|
275
|
-
|
|
276
|
-
|
|
418
|
+
export const additionalRecommendationSchema = z.preprocess(stripNullGroundingFields, z
|
|
419
|
+
.object({
|
|
420
|
+
testId: z
|
|
421
|
+
.string()
|
|
422
|
+
.describe("Human-readable kebab-case identifier, e.g. 'integration-products-orders-workflow' or 'e2e-checkout-flow'. Format: '<testType>-<scenario-slug>'. Must be unique within the report."),
|
|
423
|
+
testType: z
|
|
424
|
+
.nativeEnum(TestType)
|
|
425
|
+
.describe("Type of test. Do not include priority or other metadata in this field."),
|
|
426
|
+
category: z
|
|
427
|
+
.preprocess((val) => externalCategory(val), z.enum(TEST_CATEGORIES))
|
|
428
|
+
.describe("Test category — critical categories get generation priority over workflow"),
|
|
429
|
+
primaryEndpoint: z
|
|
430
|
+
.string()
|
|
431
|
+
.optional()
|
|
432
|
+
.describe("The focal endpoint this test targets, e.g. 'PATCH /api/v1/orders/{order_id}'. Required for single-step contract tests. For multi-step integration or E2E scenarios, omit — the steps array is the authoritative source of all endpoints involved."),
|
|
433
|
+
scenarioName: z
|
|
434
|
+
.string()
|
|
435
|
+
.optional()
|
|
436
|
+
.describe("Proposed scenario name for future generation, e.g. 'products-orders-workflow'. No file exists yet — this is a suggestion only. Omit if not applicable."),
|
|
277
437
|
// TODO: replace text with max(3) and check for regression
|
|
278
|
-
steps: z
|
|
279
|
-
|
|
280
|
-
|
|
281
|
-
|
|
282
|
-
|
|
283
|
-
|
|
284
|
-
|
|
438
|
+
steps: z
|
|
439
|
+
.array(scenarioStepSchema)
|
|
440
|
+
.describe("Ordered sequence of API/UI steps in this test scenario (at most 3). Each API step must include method and path so the endpoints are explicit; a UI/E2E action step omits them. Include requestBody and responseBody only where they carry something a reader needs — the concrete values a claim rests on, or a list endpoint's array response; omit them otherwise."),
|
|
441
|
+
description: z
|
|
442
|
+
.string()
|
|
443
|
+
.describe("Walkthrough of what the test does — the steps and assertions. For multi-step scenarios, list the endpoints involved. The 'why it is valuable' belongs in reasoning."),
|
|
444
|
+
priority: z
|
|
445
|
+
.preprocess((val) => (typeof val === "string" ? val.toLowerCase() : val), z.enum(["high", "medium", "low"]))
|
|
446
|
+
.describe("Priority level: high, medium, or low. First check diff relevance — does the test target an endpoint changed in this PR? HIGH: diff-relevant security/auth/error tests, cross-resource isolation for diff endpoints, CRUD lifecycle for NEW endpoints in the diff. MEDIUM: diff-relevant business-rule happy paths, multi-resource workflows involving diff endpoints, security/error tests for NON-diff endpoints. LOW: tests targeting only unchanged endpoints, trivially discoverable happy paths duplicating generated tests."),
|
|
447
|
+
openApiSpec: z
|
|
448
|
+
.string()
|
|
449
|
+
.optional()
|
|
450
|
+
.describe("Path to OpenAPI/Swagger spec file if available, e.g. 'openapi.yaml'"),
|
|
451
|
+
backendTrace: z
|
|
452
|
+
.string()
|
|
453
|
+
.optional()
|
|
454
|
+
.describe("Path to backend trace file if available, e.g. 'tests/skyramp-traces.json'. Used by integration and E2E tests."),
|
|
455
|
+
frontendTrace: z
|
|
456
|
+
.string()
|
|
457
|
+
.optional()
|
|
458
|
+
.describe("Path to Playwright/UI trace file if available, e.g. 'tests/skyramp-playwright.zip'. UI tests need this; E2E tests need both frontend and backend traces."),
|
|
459
|
+
reasoning: z
|
|
460
|
+
.string()
|
|
461
|
+
.describe("Why this test is recommended: the specific production risk, business rule, or security boundary it would validate"),
|
|
285
462
|
repository: repositoryField,
|
|
286
|
-
targetElements: z
|
|
287
|
-
|
|
288
|
-
|
|
463
|
+
targetElements: z
|
|
464
|
+
.array(targetElementSchema)
|
|
465
|
+
.min(1)
|
|
466
|
+
.nullable()
|
|
467
|
+
.optional()
|
|
468
|
+
.describe("UI tests only: structured grounding for one or more elements the test targets. Most tests target a single element (array length 1); render-state and multi-step UI tests target several (array length 2+). Each entry must be lifted verbatim from a captured blueprint element. Set to null when blueprint capture failed (also requires '[no-blueprint-data]' marker in BOTH description and reasoning). See Blueprint Citation Invariant in testbot prompt."),
|
|
469
|
+
pageContext: pageContextSchema
|
|
470
|
+
.optional()
|
|
471
|
+
.describe("UI tests only: page metadata for the test. Lifted from the BlueprintCapture used during grounding."),
|
|
472
|
+
})
|
|
473
|
+
.superRefine((rec, ctx) => {
|
|
289
474
|
if (rec.testType === TestType.CONTRACT && !rec.primaryEndpoint) {
|
|
290
475
|
ctx.addIssue({
|
|
291
476
|
code: z.ZodIssueCode.custom,
|
|
@@ -313,7 +498,8 @@ export const additionalRecommendationSchema = z.preprocess(stripNullGroundingFie
|
|
|
313
498
|
message: "targetElements is required for testType: 'ui'. Use null when blueprint capture failed (and add '[no-blueprint-data]' to both description and reasoning).",
|
|
314
499
|
});
|
|
315
500
|
}
|
|
316
|
-
if (Array.isArray(rec.targetElements) &&
|
|
501
|
+
if (Array.isArray(rec.targetElements) &&
|
|
502
|
+
rec.pageContext === undefined) {
|
|
317
503
|
ctx.addIssue({
|
|
318
504
|
code: z.ZodIssueCode.custom,
|
|
319
505
|
path: ["pageContext"],
|
|
@@ -348,19 +534,29 @@ export const additionalRecommendationSchema = z.preprocess(stripNullGroundingFie
|
|
|
348
534
|
// TODO(multi-repo maintenance): no `repository` field yet — see readData() TODO below.
|
|
349
535
|
const testMaintenanceSchema = z.object({
|
|
350
536
|
testType: z.nativeEnum(TestType).describe("Type of test."),
|
|
351
|
-
endpoint: z
|
|
537
|
+
endpoint: z
|
|
538
|
+
.string()
|
|
539
|
+
.describe("HTTP verb and path, e.g. 'GET /api/v1/products'"),
|
|
352
540
|
testFilePath: z
|
|
353
541
|
.string()
|
|
354
|
-
.refine((p) => path.isAbsolute(p), {
|
|
542
|
+
.refine((p) => path.isAbsolute(p), {
|
|
543
|
+
message: "testFilePath must be an absolute path",
|
|
544
|
+
})
|
|
355
545
|
.describe("Absolute path of the test file that was maintained, e.g. '/repo/tests/products_smoke_test.py' — the same path you passed to skyramp_execute_test's testFile param. Consumers should basename this for display."),
|
|
356
|
-
action: z
|
|
546
|
+
action: z
|
|
547
|
+
.nativeEnum(DriftAction)
|
|
548
|
+
.describe("The drift action assigned to this test during maintenance triage."),
|
|
357
549
|
description: z.string().describe("What was changed and why"),
|
|
358
|
-
beforeDetails: z
|
|
550
|
+
beforeDetails: z
|
|
551
|
+
.string()
|
|
552
|
+
.describe("One line only — no embedded newlines, no raw HTTP headers or JSON blobs. " +
|
|
359
553
|
"For passing runs: count and timing, e.g. '4 passed in 15.09s'. " +
|
|
360
554
|
"For failing runs: failure name and one-line root cause, e.g. " +
|
|
361
555
|
"'FAILED test_foo — assert 403 got 200, auth middleware not enforced'. " +
|
|
362
556
|
"Empty string for VERIFY/IGNORE entries where no before-execution was run."),
|
|
363
|
-
afterDetails: z
|
|
557
|
+
afterDetails: z
|
|
558
|
+
.string()
|
|
559
|
+
.describe("One line only — no embedded newlines, no raw HTTP headers or JSON blobs. " +
|
|
364
560
|
"For passing runs: count and timing, e.g. '5 passed in 10.96s'. " +
|
|
365
561
|
"For failing runs: failure name and one-line root cause, e.g. " +
|
|
366
562
|
"'FAILED test_foo — check_schema fails, order_id=1 has discount from prior PATCH test'. " +
|
|
@@ -405,7 +601,9 @@ function computeReportMetrics(params) {
|
|
|
405
601
|
const recommendations = params.additionalRecommendations ?? [];
|
|
406
602
|
const countBy = (items, pred) => items.filter(pred).length;
|
|
407
603
|
const changedMaintenance = (params.testMaintenance ?? []).filter(isMaintenanceChange);
|
|
408
|
-
const maintenanceRecovered = countBy(changedMaintenance, (m) => (m.beforeStatus === TestExecutionStatus.Fail ||
|
|
604
|
+
const maintenanceRecovered = countBy(changedMaintenance, (m) => (m.beforeStatus === TestExecutionStatus.Fail ||
|
|
605
|
+
m.beforeStatus === TestExecutionStatus.Error) &&
|
|
606
|
+
m.afterStatus === TestExecutionStatus.Pass);
|
|
409
607
|
return {
|
|
410
608
|
testsGenerated: String(params.newTestsCreated.length),
|
|
411
609
|
testsMaintained: String(changedMaintenance.length),
|
|
@@ -437,17 +635,35 @@ function computeReportMetrics(params) {
|
|
|
437
635
|
* state: the execution fix-up can restore `<testFile>.raw.bak` over it afterwards
|
|
438
636
|
* by plain `cp`, which this tool never sees. See rederiveReuseOutcome.
|
|
439
637
|
*
|
|
440
|
-
*
|
|
441
|
-
*
|
|
442
|
-
*
|
|
443
|
-
|
|
444
|
-
|
|
638
|
+
* Tests the reuse tool never ran for and specs whose outcome cannot be re-derived
|
|
639
|
+
* get nothing — which consumers already treat as "no reuse summary" — with one
|
|
640
|
+
* exception: a test whose generation handed off a reuse step that was never taken
|
|
641
|
+
* gets `chainSkipped: true` (SKYR-4220), the one claim that is exactly about the
|
|
642
|
+
* reuse tool NOT having run. Applies to every test type: UI tests carry the POM
|
|
643
|
+
* fields, API tests the shared-helper ones. */
|
|
644
|
+
async function attachReuseOutcome(test, outcomes, handOffs) {
|
|
645
|
+
// POM records describe browser specs; a basename collision with an API test's
|
|
646
|
+
// fileName must not attach them there. A utils-path record is attachable anywhere.
|
|
647
|
+
const record = outcomes?.[path.basename(test.fileName)];
|
|
648
|
+
// A utils record carries the test type its verify ran under: a mismatch is a
|
|
649
|
+
// basename collision with another test, not this row's outcome.
|
|
650
|
+
const typeMatches = !record?.utils?.testType || record.utils.testType === test.testType;
|
|
651
|
+
const found = record && typeMatches && (test.testType === TestType.UI || record.utils)
|
|
652
|
+
? record
|
|
653
|
+
: undefined;
|
|
654
|
+
const derived = found ? await rederiveReuseOutcome(found) : undefined;
|
|
655
|
+
// The RAW record: a colliding record of any kind means reuse ran for this basename,
|
|
656
|
+
// and the chain claim must not be made on the filtered view.
|
|
657
|
+
const chainSkipped = await reuseChainSkipped(test.fileName, test.testType, record, handOffs);
|
|
658
|
+
const reuse = derived || chainSkipped
|
|
659
|
+
? { ...(derived ?? {}), ...(chainSkipped ? { chainSkipped } : {}) }
|
|
660
|
+
: undefined;
|
|
661
|
+
if (!reuse)
|
|
445
662
|
return test;
|
|
446
|
-
|
|
447
|
-
|
|
448
|
-
|
|
449
|
-
|
|
450
|
-
return reuse ? { ...test, reuse } : test;
|
|
663
|
+
// A record that re-derives to nothing (e.g. a utils path that wrote no file) must
|
|
664
|
+
// not attach an empty object — consumers treat presence as "reuse ran".
|
|
665
|
+
const hasContent = Object.values(reuse).some((v) => v !== undefined);
|
|
666
|
+
return hasContent ? { ...test, reuse } : test;
|
|
451
667
|
}
|
|
452
668
|
/**
|
|
453
669
|
* Attach the video recorded for this execution, matched by the row's testFilePath
|
|
@@ -461,7 +677,10 @@ async function attachReuseOutcome(test, outcomes) {
|
|
|
461
677
|
* consumers already treat as "no recording for this test".
|
|
462
678
|
*/
|
|
463
679
|
function attachVideoPath(row, videos) {
|
|
464
|
-
return {
|
|
680
|
+
return {
|
|
681
|
+
...row,
|
|
682
|
+
videoPath: videos?.[path.basename(row.testFilePath)]?.videoPath,
|
|
683
|
+
};
|
|
465
684
|
}
|
|
466
685
|
function deduplicateById(items) {
|
|
467
686
|
const seen = new Set();
|
|
@@ -518,8 +737,8 @@ export function registerSubmitReportTool(server) {
|
|
|
518
737
|
.optional()
|
|
519
738
|
.default([])
|
|
520
739
|
.describe("Actionable follow-ups for the PR author. Each entry must be a single-line string (no embedded newlines). " +
|
|
521
|
-
"
|
|
522
|
-
"
|
|
740
|
+
"Do NOT add a next step that repeats an entry in issuesFound. The report renders every issue with its description and source location, and the consumer also posts an inline review comment on the cited line, so a repeated next step shows the same sentence to the author three times. Add a next step for an issue only when it names an action the issue text does not already state. " +
|
|
741
|
+
"Include a next step for every external test (Playwright, Cypress, RTL) assigned REGENERATE or DELETE in testMaintenance — the developer must act on these manually since they cannot be auto-applied (e.g. 'Regenerate frontend/tests/cart_pom.spec.ts — CartLine structure changed, all selectors need re-recording' or 'Delete frontend/tests/homepage.spec.ts — /cart route removed'). " +
|
|
523
742
|
"If multiple tests fail with 404 or connection refused: suggest checking targetSetupCommand/targetReadyCheckCommand. " +
|
|
524
743
|
"If 401/403 on auth endpoints: suggest authTokenCommand. " +
|
|
525
744
|
"When referencing code, use file name and relevant code pattern — no line numbers unless certain."),
|
|
@@ -590,31 +809,62 @@ export function registerSubmitReportTool(server) {
|
|
|
590
809
|
{ path: "businessCaseAnalysis", text: params.businessCaseAnalysis },
|
|
591
810
|
];
|
|
592
811
|
params.newTestsCreated.forEach((t, i) => {
|
|
593
|
-
textFields.push({
|
|
594
|
-
|
|
812
|
+
textFields.push({
|
|
813
|
+
path: `newTestsCreated[${i}].description`,
|
|
814
|
+
text: t.description,
|
|
815
|
+
});
|
|
816
|
+
textFields.push({
|
|
817
|
+
path: `newTestsCreated[${i}].reasoning`,
|
|
818
|
+
text: t.reasoning,
|
|
819
|
+
});
|
|
595
820
|
});
|
|
596
821
|
(params.additionalRecommendations ?? []).forEach((r, i) => {
|
|
597
|
-
textFields.push({
|
|
598
|
-
|
|
822
|
+
textFields.push({
|
|
823
|
+
path: `additionalRecommendations[${i}].description`,
|
|
824
|
+
text: r.description,
|
|
825
|
+
});
|
|
826
|
+
textFields.push({
|
|
827
|
+
path: `additionalRecommendations[${i}].reasoning`,
|
|
828
|
+
text: r.reasoning,
|
|
829
|
+
});
|
|
599
830
|
r.steps.forEach((s, j) => {
|
|
600
|
-
textFields.push({
|
|
831
|
+
textFields.push({
|
|
832
|
+
path: `additionalRecommendations[${i}].steps[${j}].description`,
|
|
833
|
+
text: s.description,
|
|
834
|
+
});
|
|
601
835
|
});
|
|
602
836
|
});
|
|
603
837
|
params.testResults.forEach((t, i) => {
|
|
604
|
-
textFields.push({
|
|
838
|
+
textFields.push({
|
|
839
|
+
path: `testResults[${i}].details`,
|
|
840
|
+
text: t.details,
|
|
841
|
+
});
|
|
605
842
|
});
|
|
606
843
|
params.issuesFound.forEach((f, i) => {
|
|
607
|
-
textFields.push({
|
|
844
|
+
textFields.push({
|
|
845
|
+
path: `issuesFound[${i}].description`,
|
|
846
|
+
text: f.description,
|
|
847
|
+
});
|
|
608
848
|
});
|
|
609
849
|
(params.nextSteps ?? []).forEach((s, i) => {
|
|
610
850
|
textFields.push({ path: `nextSteps[${i}]`, text: s });
|
|
611
851
|
});
|
|
612
852
|
(params.testMaintenanceDetails ?? []).forEach((d, i) => {
|
|
613
|
-
textFields.push({
|
|
614
|
-
|
|
853
|
+
textFields.push({
|
|
854
|
+
path: `testMaintenanceDetails[${i}].beforeDetails`,
|
|
855
|
+
text: d.beforeDetails,
|
|
856
|
+
});
|
|
857
|
+
textFields.push({
|
|
858
|
+
path: `testMaintenanceDetails[${i}].afterDetails`,
|
|
859
|
+
text: d.afterDetails,
|
|
860
|
+
});
|
|
615
861
|
});
|
|
616
|
-
if (params.commitMessage &&
|
|
617
|
-
|
|
862
|
+
if (params.commitMessage &&
|
|
863
|
+
params.commitMessage !== DEFAULT_COMMIT_MESSAGE) {
|
|
864
|
+
textFields.push({
|
|
865
|
+
path: "commitMessage",
|
|
866
|
+
text: params.commitMessage,
|
|
867
|
+
});
|
|
618
868
|
}
|
|
619
869
|
const languageViolations = findLanguageViolations(textFields, reportLanguage);
|
|
620
870
|
if (languageViolations.length > 0) {
|
|
@@ -637,7 +887,9 @@ export function registerSubmitReportTool(server) {
|
|
|
637
887
|
}
|
|
638
888
|
}
|
|
639
889
|
const dedupedNewTests = deduplicateById([...params.newTestsCreated]);
|
|
640
|
-
const dedupedRecommendations = deduplicateById([
|
|
890
|
+
const dedupedRecommendations = deduplicateById([
|
|
891
|
+
...(params.additionalRecommendations ?? []),
|
|
892
|
+
]);
|
|
641
893
|
const stateManager = StateManager.fromStatePath(params.stateFile);
|
|
642
894
|
let stateData;
|
|
643
895
|
try {
|
|
@@ -658,7 +910,8 @@ export function registerSubmitReportTool(server) {
|
|
|
658
910
|
// than silently reporting an empty testMaintenance section indistinguishable from a
|
|
659
911
|
// real "nothing to do". With zero existing tests there was nothing to lose, so
|
|
660
912
|
// undefined is safe there — no round-trip through skyramp_actions required.
|
|
661
|
-
if (stateData.maintenanceVerdicts === undefined &&
|
|
913
|
+
if (stateData.maintenanceVerdicts === undefined &&
|
|
914
|
+
(stateData.existingTests?.length ?? 0) > 0) {
|
|
662
915
|
errorResult = toolError("stateFile has existingTests but no maintenanceVerdicts — skyramp_actions was not called for this run. " +
|
|
663
916
|
"Call skyramp_actions (with recommendations: [] if no existing tests needed action) before skyramp_submit_report.");
|
|
664
917
|
return errorResult;
|
|
@@ -682,19 +935,24 @@ export function registerSubmitReportTool(server) {
|
|
|
682
935
|
return false;
|
|
683
936
|
const endpoints = parseEndpointField(t.endpoint);
|
|
684
937
|
const names = planNameCandidates(t.testId, t.testType);
|
|
685
|
-
return !(names.length > 0 ? names : [undefined]).some((scenarioName) => endpoints.some(({ method, path }) => matchesApprovedPlan(approvedPlan, {
|
|
938
|
+
return !(names.length > 0 ? names : [undefined]).some((scenarioName) => endpoints.some(({ method, path }) => matchesApprovedPlan(approvedPlan, {
|
|
939
|
+
scenarioName,
|
|
940
|
+
testType: t.testType,
|
|
941
|
+
method,
|
|
942
|
+
path,
|
|
943
|
+
})));
|
|
686
944
|
});
|
|
687
945
|
if (unapproved.length > 0) {
|
|
688
946
|
const approvedList = approvedPlan.generate.length > 0
|
|
689
|
-
// Show each item's endpoint keys: without them a rejection that turns on
|
|
690
|
-
|
|
691
|
-
|
|
692
|
-
|
|
693
|
-
|
|
694
|
-
|
|
695
|
-
|
|
696
|
-
|
|
697
|
-
|
|
947
|
+
? // Show each item's endpoint keys: without them a rejection that turns on
|
|
948
|
+
// the endpoint reads as self-contradicting, because the entry's own name
|
|
949
|
+
// is printed in this same list (SKYR-4123).
|
|
950
|
+
approvedPlan.generate
|
|
951
|
+
.map((item) => {
|
|
952
|
+
const eps = item.matchKeys.filter((k) => k.startsWith("ep:"));
|
|
953
|
+
return `[${item.testType}] ${item.scenarioName}${eps.length > 0 ? ` covering ${eps.join(", ")}` : ""}`;
|
|
954
|
+
})
|
|
955
|
+
.join("; ")
|
|
698
956
|
: "(none)";
|
|
699
957
|
errorResult = toolError(`${unapproved.length} newTestsCreated entr${unapproved.length === 1 ? "y" : "ies"} not in the approved plan from ` +
|
|
700
958
|
`skyramp_register_test_plan (plan ${approvedPlan.planId}) — neither its GENERATE list nor its ADDITIONAL backfill pool: ` +
|
|
@@ -745,9 +1003,14 @@ export function registerSubmitReportTool(server) {
|
|
|
745
1003
|
const recorded = stateData.existingTests?.find((t) => testFileMatches(t.testFile, m.testFilePath));
|
|
746
1004
|
const detail = params.testMaintenanceDetails?.find((d) => d.testFilePath === m.testFilePath);
|
|
747
1005
|
const displayName = path.basename(m.testFilePath);
|
|
748
|
-
const defaultBeforeStatus = MAINTENANCE_CHANGE_ACTIONS.has(m.action)
|
|
749
|
-
|
|
750
|
-
:
|
|
1006
|
+
const defaultBeforeStatus = MAINTENANCE_CHANGE_ACTIONS.has(m.action)
|
|
1007
|
+
? TestExecutionStatus.Unknown
|
|
1008
|
+
: TestExecutionStatus.Skipped;
|
|
1009
|
+
const defaultAfterStatus = m.action === DriftAction.Delete
|
|
1010
|
+
? TestExecutionStatus.Skipped
|
|
1011
|
+
: MAINTENANCE_CHANGE_ACTIONS.has(m.action)
|
|
1012
|
+
? TestExecutionStatus.Unknown
|
|
1013
|
+
: TestExecutionStatus.Skipped;
|
|
751
1014
|
const beforeStatus = recorded?.executionBefore?.status ?? defaultBeforeStatus;
|
|
752
1015
|
const afterStatus = recorded?.executionAfter?.status ?? defaultAfterStatus;
|
|
753
1016
|
// Trim before checking — a whitespace-only string is semantically blank and
|
|
@@ -759,7 +1022,13 @@ export function registerSubmitReportTool(server) {
|
|
|
759
1022
|
if (recorded?.executionAfter && !afterDetails)
|
|
760
1023
|
missingDetails.push(`${displayName} (afterDetails)`);
|
|
761
1024
|
logger.info(`${displayName}: before=${beforeStatus} after=${afterStatus}`);
|
|
762
|
-
return {
|
|
1025
|
+
return {
|
|
1026
|
+
...m,
|
|
1027
|
+
beforeDetails,
|
|
1028
|
+
afterDetails,
|
|
1029
|
+
beforeStatus,
|
|
1030
|
+
afterStatus,
|
|
1031
|
+
};
|
|
763
1032
|
});
|
|
764
1033
|
}
|
|
765
1034
|
if (missingDetails.length > 0) {
|
|
@@ -796,7 +1065,7 @@ export function registerSubmitReportTool(server) {
|
|
|
796
1065
|
...Object.values(fullState?.relatedRepos ?? {}).map((section) => section.repositoryPath),
|
|
797
1066
|
])),
|
|
798
1067
|
];
|
|
799
|
-
const unbacked =
|
|
1068
|
+
const unbacked = findUnchangedFileClaims({
|
|
800
1069
|
repoRoot,
|
|
801
1070
|
changedFiles,
|
|
802
1071
|
newTests: dedupedNewTests,
|
|
@@ -815,7 +1084,10 @@ export function registerSubmitReportTool(server) {
|
|
|
815
1084
|
// SKYR-4129. Do NOT offer re-running skyramp_actions to restate the verdict:
|
|
816
1085
|
// this branch only fires when the edit is genuinely absent, so rewriting the
|
|
817
1086
|
// record to match that would be the tamper path, not the fix.
|
|
818
|
-
const changedList = changedFiles
|
|
1087
|
+
const changedList = changedFiles
|
|
1088
|
+
.slice(0, 20)
|
|
1089
|
+
.map((f) => ` - ${f}`)
|
|
1090
|
+
.join("\n");
|
|
819
1091
|
errorResult = toolError(`${unbacked.length} report claim(s) are not backed by any change in the working tree — ` +
|
|
820
1092
|
`Testbot will not report file work that hasn't actually been made. Do NOT make a token edit to the claimed file just to satisfy this check.\n` +
|
|
821
1093
|
`For a newTestsCreated claim: create the file, correct the claim's fileName to the file you actually created (see the changed files below), or remove the claim.\n` +
|
|
@@ -832,12 +1104,36 @@ export function registerSubmitReportTool(server) {
|
|
|
832
1104
|
}
|
|
833
1105
|
}
|
|
834
1106
|
}
|
|
1107
|
+
// Validated against the tool INPUT only, never the state file (a state-file
|
|
1108
|
+
// guard here previously caused a rejection loop). Only entries carrying a
|
|
1109
|
+
// citation are checked, so an ordinary run pays no extra cost.
|
|
1110
|
+
// `repository` is normalized BEFORE the check so the attribution the check
|
|
1111
|
+
// reconciles is the one the report ships, and so the stamp below survives
|
|
1112
|
+
// into the written file — this array is what the report is built from.
|
|
1113
|
+
const issuesFound = params.issuesFound.map(normalizeRepository);
|
|
1114
|
+
if (issuesFound.some((issue) => issue.sourceFile?.trim())) {
|
|
1115
|
+
const citations = await findInvalidSourceCitations({
|
|
1116
|
+
checkouts: await stateManager.listRepoCheckouts(),
|
|
1117
|
+
issues: issuesFound,
|
|
1118
|
+
});
|
|
1119
|
+
if (citations.invalid.length > 0) {
|
|
1120
|
+
errorResult = toolError(citations.invalid.join("\n"));
|
|
1121
|
+
return errorResult;
|
|
1122
|
+
}
|
|
1123
|
+
// The checkout the citation resolved in owns the finding, so the report
|
|
1124
|
+
// states it. Left to the consumer, an unattributed related-repo path is
|
|
1125
|
+
// resolved against the primary checkout and links the wrong repo's file.
|
|
1126
|
+
citations.repository.forEach((repo, i) => {
|
|
1127
|
+
if (repo)
|
|
1128
|
+
issuesFound[i] = { ...issuesFound[i], repository: repo };
|
|
1129
|
+
});
|
|
1130
|
+
}
|
|
835
1131
|
// Strip generation-artifact fields from newTestsCreated before writing.
|
|
836
1132
|
// scenarioFile, traceFile, frontendTrace are internal paths used during
|
|
837
1133
|
// generation — downstream scoring scripts don't expect them and fail if
|
|
838
1134
|
// they encounter these string fields while traversing the object.
|
|
839
1135
|
// Also normalize each item's `repository` (blank → undefined).
|
|
840
|
-
const sanitizedNewTests = await Promise.all(dedupedNewTests.map(({ scenarioFile: _sf, traceFile: _tf, frontendTrace: _ft, ...rest }) => attachReuseOutcome(normalizeRepository(rest), stateData.reuseOutcomes)));
|
|
1136
|
+
const sanitizedNewTests = await Promise.all(dedupedNewTests.map(({ scenarioFile: _sf, traceFile: _tf, frontendTrace: _ft, ...rest }) => attachReuseOutcome(normalizeRepository(rest), stateData.reuseOutcomes, stateData.reuseHandOffs)));
|
|
841
1137
|
const report = {
|
|
842
1138
|
businessCaseAnalysis: params.businessCaseAnalysis,
|
|
843
1139
|
newTestsCreated: sanitizedNewTests,
|
|
@@ -858,9 +1154,10 @@ export function registerSubmitReportTool(server) {
|
|
|
858
1154
|
const { testFilePath: _tfp, ...wire } = attachVideoPath(normalizeRepository(row), stateData.executionVideos);
|
|
859
1155
|
return wire;
|
|
860
1156
|
}),
|
|
861
|
-
issuesFound
|
|
1157
|
+
issuesFound,
|
|
862
1158
|
nextSteps: params.nextSteps ?? [],
|
|
863
|
-
commitMessage: (params.commitMessage ?? "").replace(/[\r\n]+/g, " ").trim() ||
|
|
1159
|
+
commitMessage: (params.commitMessage ?? "").replace(/[\r\n]+/g, " ").trim() ||
|
|
1160
|
+
DEFAULT_COMMIT_MESSAGE,
|
|
864
1161
|
};
|
|
865
1162
|
const reportJson = JSON.stringify(report, null, 2);
|
|
866
1163
|
// Beside the state file, which was read successfully above — so this directory is
|