@skyramp/mcp 0.3.4 → 0.3.6-rc.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (159) hide show
  1. package/build/playwright/registerPlaywrightTools.js +92 -30
  2. package/build/playwright/traceRecordingPrompt.d.ts +6 -0
  3. package/build/playwright/traceRecordingPrompt.js +6 -2
  4. package/build/prompts/code-reuse.d.ts +1 -2
  5. package/build/prompts/code-reuse.js +182 -77
  6. package/build/prompts/modularization/integration-test-modularization.d.ts +2 -0
  7. package/build/prompts/modularization/integration-test-modularization.js +83 -41
  8. package/build/prompts/modularization/render.d.ts +18 -0
  9. package/build/prompts/modularization/render.js +12 -0
  10. package/build/prompts/modularization/ui-test-modularization.d.ts +3 -1
  11. package/build/prompts/modularization/ui-test-modularization.js +89 -47
  12. package/build/prompts/pom-aware-code-reuse.js +3 -1
  13. package/build/prompts/shared-helper-policy.d.ts +57 -0
  14. package/build/prompts/shared-helper-policy.js +135 -0
  15. package/build/prompts/test-recommendation/diffExecutionPlan.js +62 -56
  16. package/build/prompts/test-recommendation/fullRepoCatalog.js +19 -8
  17. package/build/prompts/test-recommendation/recommendationShared.d.ts +28 -6
  18. package/build/prompts/test-recommendation/recommendationShared.js +90 -16
  19. package/build/prompts/test-recommendation/registerRecommendTestsPrompt.js +22 -0
  20. package/build/prompts/test-recommendation/test-recommendation-prompt.d.ts +2 -2
  21. package/build/prompts/test-recommendation/test-recommendation-prompt.js +3 -3
  22. package/build/prompts/testbot/testbot-prompts.js +88 -33
  23. package/build/recommendation/budgeters/shared.js +105 -27
  24. package/build/recommendation/discriminators.js +13 -2
  25. package/build/recommendation/planRanker.d.ts +6 -6
  26. package/build/recommendation/planRanker.js +6 -61
  27. package/build/services/AnalyticsService.d.ts +7 -0
  28. package/build/services/AnalyticsService.js +7 -1
  29. package/build/services/ModularizationService.js +1 -3
  30. package/build/services/TestDiscoveryService.d.ts +0 -2
  31. package/build/services/TestDiscoveryService.js +2 -37
  32. package/build/services/TestGenerationService.d.ts +16 -0
  33. package/build/services/TestGenerationService.js +86 -10
  34. package/build/services/containerEnv.js +13 -12
  35. package/build/tools/code-refactor/codeReuseTool.js +279 -93
  36. package/build/tools/code-refactor/enhance-state.d.ts +49 -0
  37. package/build/tools/code-refactor/enhance-state.js +109 -0
  38. package/build/tools/code-refactor/enhanceAssertionsTool.js +34 -1
  39. package/build/tools/code-refactor/modularizationTool.js +9 -2
  40. package/build/tools/code-refactor/reuse-outcome.d.ts +23 -1
  41. package/build/tools/code-refactor/reuse-outcome.js +14 -4
  42. package/build/tools/code-refactor/reuse-state.d.ts +127 -5
  43. package/build/tools/code-refactor/reuse-state.js +628 -16
  44. package/build/tools/code-refactor/utils-verify-gates.d.ts +26 -0
  45. package/build/tools/code-refactor/utils-verify-gates.js +100 -0
  46. package/build/tools/code-refactor/verify-gates.d.ts +2 -1
  47. package/build/tools/code-refactor/verify-gates.js +90 -25
  48. package/build/tools/executeSkyrampTestTool.d.ts +19 -0
  49. package/build/tools/executeSkyrampTestTool.js +158 -8
  50. package/build/tools/generate-tests/generateBatchScenarioRestTool.js +2 -2
  51. package/build/tools/generate-tests/generateE2ERestTool.js +16 -0
  52. package/build/tools/generate-tests/generateUIRestTool.d.ts +1 -0
  53. package/build/tools/generate-tests/generateUIRestTool.js +22 -0
  54. package/build/tools/generate-tests/scenarioLint.d.ts +2 -0
  55. package/build/tools/generate-tests/scenarioLint.js +127 -19
  56. package/build/tools/generate-tests/trace-reuse-guard.d.ts +20 -0
  57. package/build/tools/generate-tests/trace-reuse-guard.js +93 -0
  58. package/build/tools/submitReportTool.d.ts +38 -38
  59. package/build/tools/submitReportTool.js +487 -104
  60. package/build/tools/test-management/analyzeChangesTool.d.ts +24 -1
  61. package/build/tools/test-management/analyzeChangesTool.js +75 -12
  62. package/build/tools/test-management/analyzeTestHealthTool.js +7 -7
  63. package/build/tools/test-management/registerTestPlanTool.d.ts +203 -0
  64. package/build/tools/test-management/registerTestPlanTool.js +149 -23
  65. package/build/types/Recommendation.d.ts +34 -5
  66. package/build/types/RepositoryAnalysis.d.ts +133 -114
  67. package/build/types/RepositoryAnalysis.js +1 -1
  68. package/build/types/ReuseOutcome.d.ts +102 -6
  69. package/build/types/ReuseOutcome.js +16 -2
  70. package/build/types/TestRecommendation.js +21 -3
  71. package/build/types/TestTypes.js +14 -8
  72. package/build/types/TestbotReport.d.ts +25 -3
  73. package/build/types/index.d.ts +2 -2
  74. package/build/types/index.js +1 -1
  75. package/build/utils/AnalysisStateManager.d.ts +69 -1
  76. package/build/utils/AnalysisStateManager.js +69 -5
  77. package/build/utils/branchDiff.d.ts +10 -0
  78. package/build/utils/branchDiff.js +28 -0
  79. package/build/utils/changedRoutes.d.ts +29 -0
  80. package/build/utils/changedRoutes.js +87 -0
  81. package/build/utils/featureFlags.d.ts +21 -0
  82. package/build/utils/featureFlags.js +23 -0
  83. package/build/utils/frontendIntegration.js +34 -4
  84. package/build/utils/importerHop.d.ts +2 -8
  85. package/build/utils/importerHop.js +15 -53
  86. package/build/utils/pathMatching.d.ts +38 -0
  87. package/build/utils/pathMatching.js +71 -0
  88. package/build/utils/pathSignatures.d.ts +22 -0
  89. package/build/utils/pathSignatures.js +57 -0
  90. package/build/utils/planMatchKeys.d.ts +16 -3
  91. package/build/utils/planMatchKeys.js +26 -10
  92. package/build/utils/pluralization.d.ts +10 -0
  93. package/build/utils/pluralization.js +18 -0
  94. package/build/utils/pom-catalog-parse.d.ts +52 -0
  95. package/build/utils/pom-catalog-parse.js +141 -0
  96. package/build/utils/pom-scope/selector-extractor.d.ts +12 -0
  97. package/build/utils/pom-scope/selector-extractor.js +34 -8
  98. package/build/utils/pom-verify/verify.d.ts +6 -5
  99. package/build/utils/pom-verify/verify.js +8 -6
  100. package/build/utils/reportLanguage.d.ts +43 -0
  101. package/build/utils/reportLanguage.js +125 -0
  102. package/build/utils/reportVerification.d.ts +74 -4
  103. package/build/utils/reportVerification.js +259 -3
  104. package/build/utils/reuseRouting.d.ts +3 -0
  105. package/build/utils/reuseRouting.js +50 -0
  106. package/build/utils/routeParsers.d.ts +2 -0
  107. package/build/utils/routeParsers.js +65 -8
  108. package/build/utils/scenarioDrafting.d.ts +1 -1
  109. package/build/utils/scenarioDrafting.js +57 -45
  110. package/build/utils/subjectEndpoints.d.ts +19 -0
  111. package/build/utils/subjectEndpoints.js +98 -0
  112. package/build/utils/testFileClassification.d.ts +11 -0
  113. package/build/utils/testFileClassification.js +47 -0
  114. package/build/utils/uiPageEnumerator.d.ts +45 -19
  115. package/build/utils/uiPageEnumerator.js +95 -51
  116. package/build/utils/utils-verify/allow.d.ts +16 -0
  117. package/build/utils/utils-verify/allow.js +68 -0
  118. package/build/utils/utils-verify/call-sites.d.ts +34 -0
  119. package/build/utils/utils-verify/call-sites.js +154 -0
  120. package/build/utils/utils-verify/index.d.ts +7 -0
  121. package/build/utils/utils-verify/index.js +7 -0
  122. package/build/utils/utils-verify/language-spec.d.ts +91 -0
  123. package/build/utils/utils-verify/language-spec.js +210 -0
  124. package/build/utils/utils-verify/locate.d.ts +39 -0
  125. package/build/utils/utils-verify/locate.js +199 -0
  126. package/build/utils/utils-verify/parse.d.ts +34 -0
  127. package/build/utils/utils-verify/parse.js +177 -0
  128. package/build/utils/utils-verify/stage.d.ts +24 -0
  129. package/build/utils/utils-verify/stage.js +107 -0
  130. package/build/utils/utils-verify/verify.d.ts +63 -0
  131. package/build/utils/utils-verify/verify.js +168 -0
  132. package/build/utils/utils.d.ts +3 -1
  133. package/build/utils/utils.js +3 -1
  134. package/build/workspace/workspace.d.ts +32 -32
  135. package/node_modules/playwright/lib/mcp/skyramp/assertTool.js +9 -5
  136. package/node_modules/playwright/lib/mcp/skyramp/loadTraceTool.js +16 -0
  137. package/node_modules/playwright/lib/mcp/skyramp/skyRampImport.js +2 -0
  138. package/node_modules/playwright/lib/mcp/skyramp/traceRecordingBackend.js +115 -14
  139. package/node_modules/playwright/lib/mcp/test/skyRampExport.js +13 -1
  140. package/node_modules/playwright/node_modules/playwright-core/.DS_Store +0 -0
  141. package/node_modules/playwright/node_modules/playwright-core/lib/server/codegen/skyramp/jsonlReader.js +2 -0
  142. package/node_modules/playwright/node_modules/playwright-core/lib/vite/htmlReport/index.html +27 -253
  143. package/node_modules/playwright/node_modules/playwright-core/lib/vite/recorder/assets/{codeMirrorModule-DtudTj_v.js → codeMirrorModule-DJMC4zNo.js} +1 -1
  144. package/node_modules/playwright/node_modules/playwright-core/lib/vite/recorder/assets/index-BW82eAUI.js +196 -0
  145. package/node_modules/playwright/node_modules/playwright-core/lib/vite/recorder/index.html +1 -1
  146. package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/assets/{codeMirrorModule-FNMuBzX1.js → codeMirrorModule-CZfp96qZ.js} +1 -1
  147. package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/assets/defaultSettingsView-gpLo02E0.js +809 -0
  148. package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/index.Bq1r1URj.js +2 -0
  149. package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/index.html +2 -2
  150. package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/uiMode.VEfqi1qN.js +5 -0
  151. package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/uiMode.html +2 -2
  152. package/node_modules/playwright/node_modules/playwright-core/package.json +1 -1
  153. package/node_modules/playwright/node_modules/playwright-core/src/server/codegen/skyramp/jsonlReader.ts +1 -1
  154. package/node_modules/playwright/package.json +1 -1
  155. package/package.json +2 -2
  156. package/node_modules/playwright/node_modules/playwright-core/lib/vite/recorder/assets/index-BpDwp16L.js +0 -422
  157. package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/assets/defaultSettingsView-Co9upU5h.js +0 -1035
  158. package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/index.DXNIQ_dx.js +0 -2
  159. package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/uiMode.CIKB3XSv.js +0 -5
@@ -3,23 +3,27 @@ import { logger } from "../utils/logger.js";
3
3
  import * as fs from "fs/promises";
4
4
  import * as path from "path";
5
5
  import { AnalyticsService } from "../services/AnalyticsService.js";
6
- import { TEST_CATEGORIES, externalCategory } from "../types/TestRecommendation.js";
6
+ import { TEST_CATEGORIES, externalCategory, } from "../types/TestRecommendation.js";
7
7
  import { TestType, HttpMethod } from "../types/TestTypes.js";
8
8
  import { DriftAction } from "../types/TestAnalysis.js";
9
9
  import { TestExecutionStatus } from "../types/TestExecution.js";
10
10
  import { IssueFoundCategory } from "../types/TestbotReport.js";
11
- import { StateManager, runArtifactDir } from "../utils/AnalysisStateManager.js";
11
+ import { StateManager, runArtifactDir, getTestsRepoDir, } from "../utils/AnalysisStateManager.js";
12
12
  import { toolError, testFileMatches } from "../utils/utils.js";
13
13
  import { matchesApprovedPlan } from "../utils/planMatchKeys.js";
14
14
  import { isTestbotEnabled } from "../utils/featureFlags.js";
15
- import { findUnbackedClaims, listChangedFiles } from "../utils/reportVerification.js";
16
- import { rederiveReuseOutcome } from "./code-refactor/reuse-state.js";
15
+ import { findInvalidSourceCitations, findUnchangedFileClaims, listChangedFiles, listChangedFilesAcross, } from "../utils/reportVerification.js";
16
+ import { getReportLanguage, isEnforcedReportLanguage, findLanguageViolations, findLanguageNearMisses, reportLanguageDisplayName, } from "../utils/reportLanguage.js";
17
+ import { rederiveReuseOutcome, reuseChainSkipped, } from "./code-refactor/reuse-state.js";
17
18
  // SKYR-3879 Path B: which testTypes the register-plan checkpoint gates. Mirrors
18
19
  // the generation tools actually wired to planGuard (batch-scenario/integration,
19
20
  // contract) — UI and E2E are on a separate blueprint-grounded pipeline and are
20
21
  // NOT gated at generation time (see planGuard.ts wiring), so they are excluded
21
22
  // here too rather than surprising the agent with a report-time-only gate.
22
- const PLAN_GATED_TEST_TYPES = new Set([TestType.CONTRACT, TestType.INTEGRATION]);
23
+ const PLAN_GATED_TEST_TYPES = new Set([
24
+ TestType.CONTRACT,
25
+ TestType.INTEGRATION,
26
+ ]);
23
27
  /**
24
28
  * Filename of the report, written beside the state file. SKYR-4147: the report path is
25
29
  * derived here rather than accepted as a parameter. The caller builds the state file and
@@ -45,7 +49,9 @@ function planNameCandidates(testId, testType) {
45
49
  if (!id)
46
50
  return [];
47
51
  const prefix = `${testType}-`;
48
- return id.toLowerCase().startsWith(prefix) ? [id, id.slice(prefix.length)] : [id];
52
+ return id.toLowerCase().startsWith(prefix)
53
+ ? [id, id.slice(prefix.length)]
54
+ : [id];
49
55
  }
50
56
  /**
51
57
  * Split an `endpoint` field into one {method, path} per endpoint it names.
@@ -54,14 +60,20 @@ function planNameCandidates(testId, testType) {
54
60
  * unmatchable path, so such an entry could match nothing (SKYR-4123).
55
61
  */
56
62
  function parseEndpointField(endpoint) {
57
- const parts = (endpoint ?? "").split(",").map((p) => p.trim()).filter(Boolean);
63
+ const parts = (endpoint ?? "")
64
+ .split(",")
65
+ .map((p) => p.trim())
66
+ .filter(Boolean);
58
67
  if (parts.length === 0)
59
68
  return [{}];
60
69
  return parts.map((part) => {
61
70
  const spaceIdx = part.indexOf(" ");
62
71
  if (spaceIdx <= 0)
63
72
  return { path: part };
64
- return { method: part.slice(0, spaceIdx), path: part.slice(spaceIdx + 1).trim() };
73
+ return {
74
+ method: part.slice(0, spaceIdx),
75
+ path: part.slice(spaceIdx + 1).trim(),
76
+ };
65
77
  });
66
78
  }
67
79
  // Drift actions that actually modify a test file. VERIFY and IGNORE are
@@ -88,23 +100,33 @@ const repositoryField = z
88
100
  * (downstream consumers treat absence as "the primary repo"). */
89
101
  function normalizeRepository(item) {
90
102
  const trimmed = item.repository?.trim();
91
- return trimmed ? { ...item, repository: trimmed } : { ...item, repository: undefined };
103
+ return trimmed
104
+ ? { ...item, repository: trimmed }
105
+ : { ...item, repository: undefined };
92
106
  }
93
107
  // videoPath is deliberately absent from this input contract: it is attached server-side
94
108
  // from the run's execution records (see attachVideoPath), and zod strips any the model
95
109
  // supplies anyway. SKYR-4156 is what happens when the agent owns that field instead.
96
110
  const testResultSchema = z.object({
97
- testType: z.nativeEnum(TestType).describe("Type of test. Do not include priority or other metadata in this field."),
98
- endpoint: z.string().describe("HTTP verb and path, e.g. 'GET /api/v1/products'"),
111
+ testType: z
112
+ .nativeEnum(TestType)
113
+ .describe("Type of test. Do not include priority or other metadata in this field."),
114
+ endpoint: z
115
+ .string()
116
+ .describe("HTTP verb and path, e.g. 'GET /api/v1/products'"),
99
117
  status: z.enum(["Pass", "Fail", "Skipped"]).describe("Test execution result"),
100
- details: z.string().describe("One sentence — no embedded newlines, no markdown. e.g. '10.8s, products_contract_test.py' or 'failed: <one-line error summary>, products_contract_test.py'"),
118
+ details: z
119
+ .string()
120
+ .describe("One sentence — no embedded newlines, no markdown. e.g. '10.8s, products_contract_test.py' or 'failed: <one-line error summary>, products_contract_test.py'"),
101
121
  // Required for every row: each one reports a specific test file the agent ran, so it
102
122
  // can always name it. It is what identifies the row server-side — `endpoint` cannot,
103
123
  // since several tests routinely exercise one endpoint — and for ui/e2e it is what
104
124
  // attaches the recorded video (SKYR-4156). Not included in the report itself.
105
125
  testFilePath: z
106
126
  .string()
107
- .refine((p) => path.isAbsolute(p), { message: "testFilePath must be an absolute path" })
127
+ .refine((p) => path.isAbsolute(p), {
128
+ message: "testFilePath must be an absolute path",
129
+ })
108
130
  .describe("Absolute path of the test file this result is for — the same path you passed to skyramp_execute_test's testFile param. Consumers basename it for display."),
109
131
  repository: repositoryField,
110
132
  });
@@ -116,21 +138,47 @@ const testResultSchema = z.object({
116
138
  // parsing. `reasoning` is free-form prose constrained by the Blueprint
117
139
  // Citation Invariant — every element cited must appear in `targetElements`.
118
140
  // See testbot prompt step 4 for the full populate-and-render rules.
141
+ /**
142
+ * Accept an omitted key as an explicit `null`.
143
+ *
144
+ * SKYR-4208. The blueprint capture no longer carries a key whose value is null,
145
+ * so an element the agent lifts verbatim simply has no `testId`, `stableId` or
146
+ * `contextText`. Consumers of the report still expect all three keys, so the
147
+ * missing one is filled in here rather than asked for in the prompt.
148
+ */
149
+ function nullWhenAbsent(inner) {
150
+ return inner.nullish().transform((v) => v ?? null);
151
+ }
119
152
  export const targetElementSchema = z.object({
120
- role: z.string().describe("ARIA role (e.g. 'button', 'heading', 'textbox', 'link'). Lifted verbatim from the captured blueprint element's `role` field."),
121
- accessibleName: z.string().describe("Computed accessible name (e.g. 'Save changes', 'Order Details'). Lifted verbatim from the captured blueprint element's `accessibleName` field. Must match the bolded name in `reasoning` character-for-character."),
122
- testId: z.string().nullable().describe("data-testid attribute value (preferred locator handle), or null when the element has no data-testid. Lifted from the blueprint element's `testId`."),
123
- stableId: z.string().nullable().describe("Unique HTML id attribute value (fallback locator when no testId), or null. Lifted from the blueprint element's `stableId`."),
124
- contextText: z.array(z.string()).nullable().describe("Disambiguating row text for elements inside repeating sections (table rows, list items): the row's surrounding non-interactive text. Lifted from the blueprint repeatingElement's items[].contextText. null for non-repeating elements."),
125
- mutability: z.enum(["mutable", "immutable", "unknown"]).optional().describe("Whether the element's content/state is expected to change between captures. 'mutable' elements are behavioral-test targets; 'immutable' are smoke-test targets. Copied from the blueprint element's `mutability`."),
126
- widgetType: z.enum(["native", "custom", "unknown"]).optional().describe("'native' = HTML built-ins (button, input). 'custom' or 'unknown' = non-standard composites; fall back to snapshot-driven trial clicks for interaction. Copied from the blueprint element's `widgetType`."),
153
+ role: z
154
+ .string()
155
+ .describe("ARIA role (e.g. 'button', 'heading', 'textbox', 'link'). Lifted verbatim from the captured blueprint element's `role` field."),
156
+ accessibleName: z
157
+ .string()
158
+ .describe("Computed accessible name (e.g. 'Save changes', 'Order Details'). Lifted verbatim from the captured blueprint element's `accessibleName` field. Must match the bolded name in `reasoning` character-for-character."),
159
+ testId: nullWhenAbsent(z.string()).describe("data-testid attribute value (preferred locator handle), or null when the element has no data-testid. Lifted from the blueprint element's `testId`."),
160
+ stableId: nullWhenAbsent(z.string()).describe("Unique HTML id attribute value (fallback locator when no testId), or null. Lifted from the blueprint element's `stableId`."),
161
+ contextText: nullWhenAbsent(z.array(z.string())).describe("Disambiguating row text for elements inside repeating sections (table rows, list items): the row's surrounding non-interactive text. Lifted from the blueprint repeatingElement's items[].contextText. null for non-repeating elements."),
162
+ mutability: z
163
+ .enum(["mutable", "immutable", "unknown"])
164
+ .optional()
165
+ .describe("Whether the element's content/state is expected to change between captures. 'mutable' elements are behavioral-test targets; 'immutable' are smoke-test targets. Copied from the blueprint element's `mutability`."),
166
+ widgetType: z
167
+ .enum(["native", "custom", "unknown"])
168
+ .optional()
169
+ .describe("'native' = HTML built-ins (button, input). 'custom' or 'unknown' = non-standard composites; fall back to snapshot-driven trial clicks for interaction. Copied from the blueprint element's `widgetType`."),
127
170
  });
128
171
  // Page metadata for a UI recommendation. Lifted from the BlueprintCapture
129
172
  // the agent used to populate `targetElements`. Codegen reads `url` to drive
130
173
  // page.goto(); the verifier uses `pageHash` to detect stale captures.
131
174
  export const pageContextSchema = z.object({
132
- url: z.string().describe("URL of the page where the test runs. Lifted from BlueprintCapture.url."),
133
- pageHash: z.string().optional().describe("Opaque hash of the captured page state (BlueprintCapture.pageHash). Lets the verifier confirm the recommendation was grounded in a still-current capture."),
175
+ url: z
176
+ .string()
177
+ .describe("URL of the page where the test runs. Lifted from BlueprintCapture.url."),
178
+ pageHash: z
179
+ .string()
180
+ .optional()
181
+ .describe("Opaque hash of the captured page state (BlueprintCapture.pageHash). Lets the verifier confirm the recommendation was grounded in a still-current capture."),
134
182
  });
135
183
  /**
136
184
  * SKYR-4193: LLMs habitually emit every key a schema declares, using a
@@ -164,21 +212,53 @@ function stripNullGroundingFields(val) {
164
212
  // interface that adds an `implemented: boolean` field. Both describe the same
165
213
  // concept (a test recommendation) — the only difference is whether it was
166
214
  // generated in this run or left for later. Tracked per Archit's review comment.
167
- export const newTestSchema = z.preprocess(stripNullGroundingFields, z.object({
168
- testId: z.string().describe("Human-readable kebab-case identifier, e.g. 'contract-get-products' or 'integration-users-orders-workflow'. Format: '<testType>-<method>-<resource>' for single-endpoint tests or '<testType>-<scenario-slug>' for multi-step tests. Must be unique within the report."),
169
- testType: z.nativeEnum(TestType).describe("Type of test created. Do not include priority or other metadata in this field."),
170
- category: z.preprocess((val) => externalCategory(val), z.enum(TEST_CATEGORIES)).describe("Test category — critical categories (security_boundary, business_rule, data_integrity, breaking_change) get generation priority over workflow"),
171
- endpoint: z.string().describe("HTTP verb and path, e.g. 'GET /api/v1/products'"),
215
+ export const newTestSchema = z.preprocess(stripNullGroundingFields, z
216
+ .object({
217
+ testId: z
218
+ .string()
219
+ .describe("Human-readable kebab-case identifier, e.g. 'contract-get-products' or 'integration-users-orders-workflow'. Format: '<testType>-<method>-<resource>' for single-endpoint tests or '<testType>-<scenario-slug>' for multi-step tests. Must be unique within the report."),
220
+ testType: z
221
+ .nativeEnum(TestType)
222
+ .describe("Type of test created. Do not include priority or other metadata in this field."),
223
+ category: z
224
+ .preprocess((val) => externalCategory(val), z.enum(TEST_CATEGORIES))
225
+ .describe("Test category — critical categories (security_boundary, business_rule, data_integrity, breaking_change) get generation priority over workflow"),
226
+ endpoint: z
227
+ .string()
228
+ .describe("HTTP verb and path, e.g. 'GET /api/v1/products'"),
172
229
  fileName: z.string().describe("Name of the generated test file"),
173
- description: z.string().trim().min(1).describe("What the test does — the steps and assertions, not the bugs it finds. e.g. 'Creates a collection, adds a link, then verifies the link exists'. Do NOT describe expected failures or bugs here — those belong in issuesFound."),
174
- scenarioFile: z.string().optional().describe("Path to the scenario JSON file if one was generated (e.g. 'tests/scenario_collections-links.json')"),
175
- traceFile: z.string().optional().describe("Path to the backend trace file if used or created"),
176
- frontendTrace: z.string().optional().describe("Path to the Playwright/UI trace file if used or created"),
177
- reasoning: z.string().describe("Why this test was created: what production risk it mitigates, what code pattern it targets, or what coverage gap it fills"),
230
+ description: z
231
+ .string()
232
+ .trim()
233
+ .min(1)
234
+ .describe("What the test does — the steps and assertions, not the bugs it finds. e.g. 'Creates a collection, adds a link, then verifies the link exists'. Do NOT describe expected failures or bugs here — those belong in issuesFound."),
235
+ scenarioFile: z
236
+ .string()
237
+ .optional()
238
+ .describe("Path to the scenario JSON file if one was generated (e.g. 'tests/scenario_collections-links.json')"),
239
+ traceFile: z
240
+ .string()
241
+ .optional()
242
+ .describe("Path to the backend trace file if used or created"),
243
+ frontendTrace: z
244
+ .string()
245
+ .optional()
246
+ .describe("Path to the Playwright/UI trace file if used or created"),
247
+ reasoning: z
248
+ .string()
249
+ .describe("Why this test was created: what production risk it mitigates, what code pattern it targets, or what coverage gap it fills"),
178
250
  repository: repositoryField,
179
- targetElements: z.array(targetElementSchema).min(1).nullable().optional().describe("UI tests only: structured grounding for one or more elements the test targets. Most tests target a single element (array length 1); render-state and multi-step UI tests target several (array length 2+). Each entry must be lifted verbatim from a captured blueprint element. Set to null when blueprint capture failed (also requires '[no-blueprint-data]' marker in BOTH description and reasoning). See Blueprint Citation Invariant in testbot prompt."),
180
- pageContext: pageContextSchema.optional().describe("UI tests only: page metadata for the test. Lifted from the BlueprintCapture used during grounding."),
181
- }).superRefine((rec, ctx) => {
251
+ targetElements: z
252
+ .array(targetElementSchema)
253
+ .min(1)
254
+ .nullable()
255
+ .optional()
256
+ .describe("UI tests only: structured grounding for one or more elements the test targets. Most tests target a single element (array length 1); render-state and multi-step UI tests target several (array length 2+). Each entry must be lifted verbatim from a captured blueprint element. Set to null when blueprint capture failed (also requires '[no-blueprint-data]' marker in BOTH description and reasoning). See Blueprint Citation Invariant in testbot prompt."),
257
+ pageContext: pageContextSchema
258
+ .optional()
259
+ .describe("UI tests only: page metadata for the test. Lifted from the BlueprintCapture used during grounding."),
260
+ })
261
+ .superRefine((rec, ctx) => {
182
262
  // targetElements / pageContext are UI-only.
183
263
  if (rec.testType !== TestType.UI) {
184
264
  for (const field of ["targetElements", "pageContext"]) {
@@ -209,7 +289,8 @@ export const newTestSchema = z.preprocess(stripNullGroundingFields, z.object({
209
289
  });
210
290
  }
211
291
  // pageContext is required when targetElements is an array (grounded), forbidden when null.
212
- if (Array.isArray(rec.targetElements) && rec.pageContext === undefined) {
292
+ if (Array.isArray(rec.targetElements) &&
293
+ rec.pageContext === undefined) {
213
294
  ctx.addIssue({
214
295
  code: z.ZodIssueCode.custom,
215
296
  path: ["pageContext"],
@@ -232,7 +313,8 @@ export const newTestSchema = z.preprocess(stripNullGroundingFields, z.object({
232
313
  message: "reasoning must contain '[no-blueprint-data]' when targetElements is null.",
233
314
  });
234
315
  }
235
- if (rec.description && !rec.description.includes("[no-blueprint-data]")) {
316
+ if (rec.description &&
317
+ !rec.description.includes("[no-blueprint-data]")) {
236
318
  ctx.addIssue({
237
319
  code: z.ZodIssueCode.custom,
238
320
  path: ["description"],
@@ -242,8 +324,18 @@ export const newTestSchema = z.preprocess(stripNullGroundingFields, z.object({
242
324
  }
243
325
  }
244
326
  }));
245
- const issueFoundSchema = z.object({
246
- description: z.string().describe("One-line description. Do NOT prefix with the severity level — severity is a separate field. Include code logic bugs from the diff, test generation/execution failures, and environment misconfiguration."),
327
+ /** A citation string the submit-time check verifies and the report then ships.
328
+ * Normalized HERE, once, so the value the verifier resolved is byte-identical to
329
+ * the value written: the check trimmed its input, so a padded " src/x.py " used to
330
+ * validate against the real path and ship the unusable padded one. A
331
+ * blank/whitespace-only value means "no citation", the same rule `repository` uses. */
332
+ const citationString = z.preprocess((v) => (typeof v === "string" ? v.trim() || undefined : v), z.string().optional());
333
+ const issueFoundSchema = z
334
+ .object({
335
+ description: z
336
+ .string()
337
+ .describe("One-line description. Do NOT prefix with the severity level — severity is a separate field. Include code logic bugs from the diff, test generation/execution failures, and environment misconfiguration. " +
338
+ "When sourceFile or sourceSymbol is set, quote the offending line inside backticks in this description — the exact code, copied from the file, so review tools can find it."),
247
339
  severity: z
248
340
  .enum(["critical", "high", "medium", "low"])
249
341
  .optional()
@@ -255,36 +347,130 @@ const issueFoundSchema = z.object({
255
347
  .describe("Issue classification. bug = a product/code defect, e.g. found by a test or in the diff. " +
256
348
  "lint = a linter or formatter finding (eslint, flake8, prettier). " +
257
349
  "type = a type-check failure (tsc, mypy). " +
258
- "config = environment or tooling misconfiguration (wrong workspace auth type, missing env var, setup command failure). " +
350
+ "config = environment or tooling misconfiguration (wrong workspace auth type, missing env var, setup command failure), " +
351
+ "and also every Skyramp tool or environment failure — a generation or execution tool error, an unreachable app, a missing credential, a failed capture. " +
259
352
  "The report renders lint/type/config entries in a separate 'Configuration Errors' section so product bugs stay prominent under 'Issues Found'."),
260
353
  repository: repositoryField,
354
+ sourceFile: citationString.describe("Path of the application file whose code is missing or wrong, relative to the repository root (e.g. 'src/crud/products.py'). " +
355
+ "REQUIRED when category is 'bug'. " +
356
+ "Cite only a file you actually opened this run: this tool rejects the report if the path does not exist in the repository."),
357
+ sourceSymbol: citationString.describe("The function, method, or identifier inside sourceFile whose code is missing or wrong (e.g. 'delete_product'). " +
358
+ "Optional — a config file or template has no symbol to name. Set it whenever the file has one. " +
359
+ "This tool rejects the report if the text does not appear in the cited file."),
360
+ sourceLine: z
361
+ .number()
362
+ .int()
363
+ .positive()
364
+ .optional()
365
+ .describe("1-based line number in sourceFile. Optional and advisory. " +
366
+ "Give the line of the offending statement itself, not the line of the enclosing function's signature — review views place the finding on exactly this line. " +
367
+ "The tool does not check that the line still holds the cited code, because line numbers drift."),
368
+ })
369
+ .superRefine((issue, ctx) => {
370
+ // A `bug` entry asserts that product code is wrong, so it has to name the code:
371
+ // without a file the entry is prose a reviewer cannot act on, and nothing
372
+ // downstream can place it in the diff. Only the file is required — a config file
373
+ // or a template legitimately has no symbol to name, and line numbers are
374
+ // advisory. The other categories are tooling findings that often have no single
375
+ // source location (an unreachable app, a missing env var), so the requirement is
376
+ // scoped to `bug` alone.
377
+ //
378
+ // The exit when the code cannot be named is NOT to invent a citation: a failure
379
+ // the agent cannot localize is already reported by its `testResults` row, and
380
+ // a tool or environment failure belongs under `config`.
381
+ if (issue.category !== IssueFoundCategory.Bug)
382
+ return;
383
+ if (issue.sourceFile !== undefined)
384
+ return;
385
+ ctx.addIssue({
386
+ code: z.ZodIssueCode.custom,
387
+ path: ["sourceFile"],
388
+ message: `sourceFile is required when category is 'bug' — a bug entry must name the code it is about. ` +
389
+ `Open the file, then set sourceFile (path relative to the repository root). Also set sourceSymbol (the function or identifier inside it) when the file has one, and sourceLine (the line of the offending statement) when you have it — both are optional. ` +
390
+ `If you cannot point at the code: a test failure is already reported by its testResults entry, and a tool or environment failure belongs under category 'config' — do not guess a citation.`,
391
+ });
261
392
  });
262
393
  const scenarioStepSchema = z.object({
263
- method: z.nativeEnum(HttpMethod).optional().describe("HTTP method. Required for API steps, omit for UI/E2E actions."),
264
- path: z.string().optional().describe("Endpoint or page path (e.g. '/api/v1/products' or '/products'). Required for API steps, omit for UI actions."),
265
- description: z.string().describe("What this step does, e.g. 'Create a product' or 'Click checkout button and verify confirmation'"),
266
- expectedStatusCode: z.number().optional().describe("Expected HTTP status code, e.g. 200, 201, 404"),
267
- requestBody: z.record(z.any()).optional().describe("Example request body with realistic field values"),
268
- responseBody: z.record(z.any()).optional().describe("Key response fields to verify, e.g. { id: 'number', name: 'string', in_stock: 'boolean?' }"),
394
+ method: z
395
+ .nativeEnum(HttpMethod)
396
+ .optional()
397
+ .describe("HTTP method. Required for API steps, omit for UI/E2E actions."),
398
+ path: z
399
+ .string()
400
+ .optional()
401
+ .describe("Endpoint or page path (e.g. '/api/v1/products' or '/products'). Required for API steps, omit for UI actions."),
402
+ description: z
403
+ .string()
404
+ .describe("What this step does, e.g. 'Create a product' or 'Click checkout button and verify confirmation'"),
405
+ expectedStatusCode: z
406
+ .number()
407
+ .optional()
408
+ .describe("Expected HTTP status code, e.g. 200, 201, 404"),
409
+ requestBody: z
410
+ .record(z.any())
411
+ .optional()
412
+ .describe("Example request body with realistic field values"),
413
+ responseBody: z
414
+ .union([z.record(z.any()), z.array(z.any())])
415
+ .optional()
416
+ .describe("Key response fields to verify, e.g. { id: 'number', name: 'string', in_stock: 'boolean?' }. An array for a collection/list endpoint that returns a JSON array."),
269
417
  });
270
- export const additionalRecommendationSchema = z.preprocess(stripNullGroundingFields, z.object({
271
- testId: z.string().describe("Human-readable kebab-case identifier, e.g. 'integration-products-orders-workflow' or 'e2e-checkout-flow'. Format: '<testType>-<scenario-slug>'. Must be unique within the report."),
272
- testType: z.nativeEnum(TestType).describe("Type of test. Do not include priority or other metadata in this field."),
273
- category: z.preprocess((val) => externalCategory(val), z.enum(TEST_CATEGORIES)).describe("Test category — critical categories get generation priority over workflow"),
274
- primaryEndpoint: z.string().optional().describe("The focal endpoint this test targets, e.g. 'PATCH /api/v1/orders/{order_id}'. Required for single-step contract tests. For multi-step integration or E2E scenarios, omit — the steps array is the authoritative source of all endpoints involved."),
275
- scenarioName: z.string().optional().describe("Proposed scenario name for future generation, e.g. 'products-orders-workflow'. No file exists yet — this is a suggestion only. Omit if not applicable."),
418
+ export const additionalRecommendationSchema = z.preprocess(stripNullGroundingFields, z
419
+ .object({
420
+ testId: z
421
+ .string()
422
+ .describe("Human-readable kebab-case identifier, e.g. 'integration-products-orders-workflow' or 'e2e-checkout-flow'. Format: '<testType>-<scenario-slug>'. Must be unique within the report."),
423
+ testType: z
424
+ .nativeEnum(TestType)
425
+ .describe("Type of test. Do not include priority or other metadata in this field."),
426
+ category: z
427
+ .preprocess((val) => externalCategory(val), z.enum(TEST_CATEGORIES))
428
+ .describe("Test category — critical categories get generation priority over workflow"),
429
+ primaryEndpoint: z
430
+ .string()
431
+ .optional()
432
+ .describe("The focal endpoint this test targets, e.g. 'PATCH /api/v1/orders/{order_id}'. Required for single-step contract tests. For multi-step integration or E2E scenarios, omit — the steps array is the authoritative source of all endpoints involved."),
433
+ scenarioName: z
434
+ .string()
435
+ .optional()
436
+ .describe("Proposed scenario name for future generation, e.g. 'products-orders-workflow'. No file exists yet — this is a suggestion only. Omit if not applicable."),
276
437
  // TODO: replace text with max(3) and check for regression
277
- steps: z.array(scenarioStepSchema).describe("Ordered sequence of API/UI steps in this test scenario (at most 3). Each step must include method and path so the endpoints are explicit. Omit requestBody and responseBody from steps."),
278
- description: z.string().describe("Walkthrough of what the test does — the steps and assertions. For multi-step scenarios, list the endpoints involved. The 'why it is valuable' belongs in reasoning."),
279
- priority: z.preprocess((val) => (typeof val === "string" ? val.toLowerCase() : val), z.enum(["high", "medium", "low"])).describe("Priority level: high, medium, or low. First check diff relevance — does the test target an endpoint changed in this PR? HIGH: diff-relevant security/auth/error tests, cross-resource isolation for diff endpoints, CRUD lifecycle for NEW endpoints in the diff. MEDIUM: diff-relevant business-rule happy paths, multi-resource workflows involving diff endpoints, security/error tests for NON-diff endpoints. LOW: tests targeting only unchanged endpoints, trivially discoverable happy paths duplicating generated tests."),
280
- openApiSpec: z.string().optional().describe("Path to OpenAPI/Swagger spec file if available, e.g. 'openapi.yaml'"),
281
- backendTrace: z.string().optional().describe("Path to backend trace file if available, e.g. 'tests/skyramp-traces.json'. Used by integration and E2E tests."),
282
- frontendTrace: z.string().optional().describe("Path to Playwright/UI trace file if available, e.g. 'tests/skyramp-playwright.zip'. UI tests need this; E2E tests need both frontend and backend traces."),
283
- reasoning: z.string().describe("Why this test is recommended: the specific production risk, business rule, or security boundary it would validate"),
438
+ steps: z
439
+ .array(scenarioStepSchema)
440
+ .describe("Ordered sequence of API/UI steps in this test scenario (at most 3). Each API step must include method and path so the endpoints are explicit; a UI/E2E action step omits them. Include requestBody and responseBody only where they carry something a reader needs — the concrete values a claim rests on, or a list endpoint's array response; omit them otherwise."),
441
+ description: z
442
+ .string()
443
+ .describe("Walkthrough of what the test does — the steps and assertions. For multi-step scenarios, list the endpoints involved. The 'why it is valuable' belongs in reasoning."),
444
+ priority: z
445
+ .preprocess((val) => (typeof val === "string" ? val.toLowerCase() : val), z.enum(["high", "medium", "low"]))
446
+ .describe("Priority level: high, medium, or low. First check diff relevance — does the test target an endpoint changed in this PR? HIGH: diff-relevant security/auth/error tests, cross-resource isolation for diff endpoints, CRUD lifecycle for NEW endpoints in the diff. MEDIUM: diff-relevant business-rule happy paths, multi-resource workflows involving diff endpoints, security/error tests for NON-diff endpoints. LOW: tests targeting only unchanged endpoints, trivially discoverable happy paths duplicating generated tests."),
447
+ openApiSpec: z
448
+ .string()
449
+ .optional()
450
+ .describe("Path to OpenAPI/Swagger spec file if available, e.g. 'openapi.yaml'"),
451
+ backendTrace: z
452
+ .string()
453
+ .optional()
454
+ .describe("Path to backend trace file if available, e.g. 'tests/skyramp-traces.json'. Used by integration and E2E tests."),
455
+ frontendTrace: z
456
+ .string()
457
+ .optional()
458
+ .describe("Path to Playwright/UI trace file if available, e.g. 'tests/skyramp-playwright.zip'. UI tests need this; E2E tests need both frontend and backend traces."),
459
+ reasoning: z
460
+ .string()
461
+ .describe("Why this test is recommended: the specific production risk, business rule, or security boundary it would validate"),
284
462
  repository: repositoryField,
285
- targetElements: z.array(targetElementSchema).min(1).nullable().optional().describe("UI tests only: structured grounding for one or more elements the test targets. Most tests target a single element (array length 1); render-state and multi-step UI tests target several (array length 2+). Each entry must be lifted verbatim from a captured blueprint element. Set to null when blueprint capture failed (also requires '[no-blueprint-data]' marker in BOTH description and reasoning). See Blueprint Citation Invariant in testbot prompt."),
286
- pageContext: pageContextSchema.optional().describe("UI tests only: page metadata for the test. Lifted from the BlueprintCapture used during grounding."),
287
- }).superRefine((rec, ctx) => {
463
+ targetElements: z
464
+ .array(targetElementSchema)
465
+ .min(1)
466
+ .nullable()
467
+ .optional()
468
+ .describe("UI tests only: structured grounding for one or more elements the test targets. Most tests target a single element (array length 1); render-state and multi-step UI tests target several (array length 2+). Each entry must be lifted verbatim from a captured blueprint element. Set to null when blueprint capture failed (also requires '[no-blueprint-data]' marker in BOTH description and reasoning). See Blueprint Citation Invariant in testbot prompt."),
469
+ pageContext: pageContextSchema
470
+ .optional()
471
+ .describe("UI tests only: page metadata for the test. Lifted from the BlueprintCapture used during grounding."),
472
+ })
473
+ .superRefine((rec, ctx) => {
288
474
  if (rec.testType === TestType.CONTRACT && !rec.primaryEndpoint) {
289
475
  ctx.addIssue({
290
476
  code: z.ZodIssueCode.custom,
@@ -312,7 +498,8 @@ export const additionalRecommendationSchema = z.preprocess(stripNullGroundingFie
312
498
  message: "targetElements is required for testType: 'ui'. Use null when blueprint capture failed (and add '[no-blueprint-data]' to both description and reasoning).",
313
499
  });
314
500
  }
315
- if (Array.isArray(rec.targetElements) && rec.pageContext === undefined) {
501
+ if (Array.isArray(rec.targetElements) &&
502
+ rec.pageContext === undefined) {
316
503
  ctx.addIssue({
317
504
  code: z.ZodIssueCode.custom,
318
505
  path: ["pageContext"],
@@ -347,19 +534,29 @@ export const additionalRecommendationSchema = z.preprocess(stripNullGroundingFie
347
534
  // TODO(multi-repo maintenance): no `repository` field yet — see readData() TODO below.
348
535
  const testMaintenanceSchema = z.object({
349
536
  testType: z.nativeEnum(TestType).describe("Type of test."),
350
- endpoint: z.string().describe("HTTP verb and path, e.g. 'GET /api/v1/products'"),
537
+ endpoint: z
538
+ .string()
539
+ .describe("HTTP verb and path, e.g. 'GET /api/v1/products'"),
351
540
  testFilePath: z
352
541
  .string()
353
- .refine((p) => path.isAbsolute(p), { message: "testFilePath must be an absolute path" })
542
+ .refine((p) => path.isAbsolute(p), {
543
+ message: "testFilePath must be an absolute path",
544
+ })
354
545
  .describe("Absolute path of the test file that was maintained, e.g. '/repo/tests/products_smoke_test.py' — the same path you passed to skyramp_execute_test's testFile param. Consumers should basename this for display."),
355
- action: z.nativeEnum(DriftAction).describe("The drift action assigned to this test during maintenance triage."),
546
+ action: z
547
+ .nativeEnum(DriftAction)
548
+ .describe("The drift action assigned to this test during maintenance triage."),
356
549
  description: z.string().describe("What was changed and why"),
357
- beforeDetails: z.string().describe("One line only — no embedded newlines, no raw HTTP headers or JSON blobs. " +
550
+ beforeDetails: z
551
+ .string()
552
+ .describe("One line only — no embedded newlines, no raw HTTP headers or JSON blobs. " +
358
553
  "For passing runs: count and timing, e.g. '4 passed in 15.09s'. " +
359
554
  "For failing runs: failure name and one-line root cause, e.g. " +
360
555
  "'FAILED test_foo — assert 403 got 200, auth middleware not enforced'. " +
361
556
  "Empty string for VERIFY/IGNORE entries where no before-execution was run."),
362
- afterDetails: z.string().describe("One line only — no embedded newlines, no raw HTTP headers or JSON blobs. " +
557
+ afterDetails: z
558
+ .string()
559
+ .describe("One line only — no embedded newlines, no raw HTTP headers or JSON blobs. " +
363
560
  "For passing runs: count and timing, e.g. '5 passed in 10.96s'. " +
364
561
  "For failing runs: failure name and one-line root cause, e.g. " +
365
562
  "'FAILED test_foo — check_schema fails, order_id=1 has discount from prior PATCH test'. " +
@@ -404,7 +601,9 @@ function computeReportMetrics(params) {
404
601
  const recommendations = params.additionalRecommendations ?? [];
405
602
  const countBy = (items, pred) => items.filter(pred).length;
406
603
  const changedMaintenance = (params.testMaintenance ?? []).filter(isMaintenanceChange);
407
- const maintenanceRecovered = countBy(changedMaintenance, (m) => (m.beforeStatus === TestExecutionStatus.Fail || m.beforeStatus === TestExecutionStatus.Error) && m.afterStatus === TestExecutionStatus.Pass);
604
+ const maintenanceRecovered = countBy(changedMaintenance, (m) => (m.beforeStatus === TestExecutionStatus.Fail ||
605
+ m.beforeStatus === TestExecutionStatus.Error) &&
606
+ m.afterStatus === TestExecutionStatus.Pass);
408
607
  return {
409
608
  testsGenerated: String(params.newTestsCreated.length),
410
609
  testsMaintained: String(changedMaintenance.length),
@@ -436,17 +635,35 @@ function computeReportMetrics(params) {
436
635
  * state: the execution fix-up can restore `<testFile>.raw.bak` over it afterwards
437
636
  * by plain `cp`, which this tool never sees. See rederiveReuseOutcome.
438
637
  *
439
- * Non-UI tests, tests the reuse tool never ran for, and specs whose outcome cannot
440
- * be re-derived all get nothing — which consumers already treat as "no reuse
441
- * summary". */
442
- async function attachReuseOutcome(test, outcomes) {
443
- if (test.testType !== TestType.UI || !outcomes)
444
- return test;
445
- const found = outcomes[path.basename(test.fileName)];
446
- if (!found)
638
+ * Tests the reuse tool never ran for and specs whose outcome cannot be re-derived
639
+ * get nothing — which consumers already treat as "no reuse summary" — with one
640
+ * exception: a test whose generation handed off a reuse step that was never taken
641
+ * gets `chainSkipped: true` (SKYR-4220), the one claim that is exactly about the
642
+ * reuse tool NOT having run. Applies to every test type: UI tests carry the POM
643
+ * fields, API tests the shared-helper ones. */
644
+ async function attachReuseOutcome(test, outcomes, handOffs) {
645
+ // POM records describe browser specs; a basename collision with an API test's
646
+ // fileName must not attach them there. A utils-path record is attachable anywhere.
647
+ const record = outcomes?.[path.basename(test.fileName)];
648
+ // A utils record carries the test type its verify ran under: a mismatch is a
649
+ // basename collision with another test, not this row's outcome.
650
+ const typeMatches = !record?.utils?.testType || record.utils.testType === test.testType;
651
+ const found = record && typeMatches && (test.testType === TestType.UI || record.utils)
652
+ ? record
653
+ : undefined;
654
+ const derived = found ? await rederiveReuseOutcome(found) : undefined;
655
+ // The RAW record: a colliding record of any kind means reuse ran for this basename,
656
+ // and the chain claim must not be made on the filtered view.
657
+ const chainSkipped = await reuseChainSkipped(test.fileName, test.testType, record, handOffs);
658
+ const reuse = derived || chainSkipped
659
+ ? { ...(derived ?? {}), ...(chainSkipped ? { chainSkipped } : {}) }
660
+ : undefined;
661
+ if (!reuse)
447
662
  return test;
448
- const reuse = await rederiveReuseOutcome(found);
449
- return reuse ? { ...test, reuse } : test;
663
+ // A record that re-derives to nothing (e.g. a utils path that wrote no file) must
664
+ // not attach an empty object — consumers treat presence as "reuse ran".
665
+ const hasContent = Object.values(reuse).some((v) => v !== undefined);
666
+ return hasContent ? { ...test, reuse } : test;
450
667
  }
451
668
  /**
452
669
  * Attach the video recorded for this execution, matched by the row's testFilePath
@@ -460,7 +677,10 @@ async function attachReuseOutcome(test, outcomes) {
460
677
  * consumers already treat as "no recording for this test".
461
678
  */
462
679
  function attachVideoPath(row, videos) {
463
- return { ...row, videoPath: videos?.[path.basename(row.testFilePath)]?.videoPath };
680
+ return {
681
+ ...row,
682
+ videoPath: videos?.[path.basename(row.testFilePath)]?.videoPath,
683
+ };
464
684
  }
465
685
  function deduplicateById(items) {
466
686
  const seen = new Set();
@@ -517,8 +737,8 @@ export function registerSubmitReportTool(server) {
517
737
  .optional()
518
738
  .default([])
519
739
  .describe("Actionable follow-ups for the PR author. Each entry must be a single-line string (no embedded newlines). " +
520
- "Include a next step for every critical/high severity issue in issuesFound. No next steps for low-severity issues. " +
521
- "Also include a next step for every external test (Playwright, Cypress, RTL) assigned REGENERATE or DELETE in testMaintenance — the developer must act on these manually since they cannot be auto-applied (e.g. 'Regenerate frontend/tests/cart_pom.spec.ts — CartLine structure changed, all selectors need re-recording' or 'Delete frontend/tests/homepage.spec.ts — /cart route removed'). " +
740
+ "Do NOT add a next step that repeats an entry in issuesFound. The report renders every issue with its description and source location, and the consumer also posts an inline review comment on the cited line, so a repeated next step shows the same sentence to the author three times. Add a next step for an issue only when it names an action the issue text does not already state. " +
741
+ "Include a next step for every external test (Playwright, Cypress, RTL) assigned REGENERATE or DELETE in testMaintenance — the developer must act on these manually since they cannot be auto-applied (e.g. 'Regenerate frontend/tests/cart_pom.spec.ts — CartLine structure changed, all selectors need re-recording' or 'Delete frontend/tests/homepage.spec.ts — /cart route removed'). " +
522
742
  "If multiple tests fail with 404 or connection refused: suggest checking targetSetupCommand/targetReadyCheckCommand. " +
523
743
  "If 401/403 on auth endpoints: suggest authTokenCommand. " +
524
744
  "When referencing code, use file name and relevant code pattern — no line numbers unless certain."),
@@ -543,6 +763,10 @@ export function registerSubmitReportTool(server) {
543
763
  }, async (params) => {
544
764
  const startTime = Date.now();
545
765
  let errorResult;
766
+ // Residual-leak telemetry (SKYR-4185): filled by the language guardrail
767
+ // below when a report language is enforced, spread into the analytics
768
+ // event in the finally. Empty (no properties) for non-enforced runs.
769
+ const languageTelemetry = {};
546
770
  // The report goes next to the state file, so in a Testbot run the state file has to
547
771
  // be in the run directory — that is where the action reads the report back from.
548
772
  // Two ways that breaks, both ending in a report nobody reads while this tool says
@@ -567,8 +791,105 @@ export function registerSubmitReportTool(server) {
567
791
  `earlier run or one typed by hand.`);
568
792
  }
569
793
  }
794
+ // SKYR-4185: report-language guardrail. The language was captured when the
795
+ // testbot prompt was rendered (see reportLanguage.ts) — the instruction to
796
+ // write free-text fields in it is prose the agent sometimes ignores
797
+ // (letsramp/api-insight#215 shipped English testResults[].details in an
798
+ // otherwise-Japanese report). Reject rather than ship a half-translated
799
+ // report; the listed paths tell the agent exactly what to rewrite. Fields
800
+ // the agent does not author here (testMaintenance description from
801
+ // skyramp_actions verdicts, the server default commitMessage) are exempt —
802
+ // rejecting those would loop, since resubmitting cannot change them.
803
+ const reportLanguage = getReportLanguage();
804
+ if (reportLanguage && !isEnforcedReportLanguage(reportLanguage)) {
805
+ logger.warning(`Report language '${reportLanguage}' has no enforcement rules — skipping the language guardrail`);
806
+ }
807
+ if (isEnforcedReportLanguage(reportLanguage)) {
808
+ const textFields = [
809
+ { path: "businessCaseAnalysis", text: params.businessCaseAnalysis },
810
+ ];
811
+ params.newTestsCreated.forEach((t, i) => {
812
+ textFields.push({
813
+ path: `newTestsCreated[${i}].description`,
814
+ text: t.description,
815
+ });
816
+ textFields.push({
817
+ path: `newTestsCreated[${i}].reasoning`,
818
+ text: t.reasoning,
819
+ });
820
+ });
821
+ (params.additionalRecommendations ?? []).forEach((r, i) => {
822
+ textFields.push({
823
+ path: `additionalRecommendations[${i}].description`,
824
+ text: r.description,
825
+ });
826
+ textFields.push({
827
+ path: `additionalRecommendations[${i}].reasoning`,
828
+ text: r.reasoning,
829
+ });
830
+ r.steps.forEach((s, j) => {
831
+ textFields.push({
832
+ path: `additionalRecommendations[${i}].steps[${j}].description`,
833
+ text: s.description,
834
+ });
835
+ });
836
+ });
837
+ params.testResults.forEach((t, i) => {
838
+ textFields.push({
839
+ path: `testResults[${i}].details`,
840
+ text: t.details,
841
+ });
842
+ });
843
+ params.issuesFound.forEach((f, i) => {
844
+ textFields.push({
845
+ path: `issuesFound[${i}].description`,
846
+ text: f.description,
847
+ });
848
+ });
849
+ (params.nextSteps ?? []).forEach((s, i) => {
850
+ textFields.push({ path: `nextSteps[${i}]`, text: s });
851
+ });
852
+ (params.testMaintenanceDetails ?? []).forEach((d, i) => {
853
+ textFields.push({
854
+ path: `testMaintenanceDetails[${i}].beforeDetails`,
855
+ text: d.beforeDetails,
856
+ });
857
+ textFields.push({
858
+ path: `testMaintenanceDetails[${i}].afterDetails`,
859
+ text: d.afterDetails,
860
+ });
861
+ });
862
+ if (params.commitMessage &&
863
+ params.commitMessage !== DEFAULT_COMMIT_MESSAGE) {
864
+ textFields.push({
865
+ path: "commitMessage",
866
+ text: params.commitMessage,
867
+ });
868
+ }
869
+ const languageViolations = findLanguageViolations(textFields, reportLanguage);
870
+ if (languageViolations.length > 0) {
871
+ const languageName = reportLanguageDisplayName(reportLanguage);
872
+ errorResult = toolError(`This run's report language is ${languageName}, but ${languageViolations.length} free-text field(s) are written in English:\n` +
873
+ languageViolations.map((v) => ` - ${v}`).join("\n") +
874
+ `\nRewrite ONLY the listed fields in ${languageName} and resubmit the report. ` +
875
+ `Keep code identifiers, endpoint paths, file names, test IDs, enum values ` +
876
+ `(Pass/Fail/Skipped, severity values, testType), and anything inside backticks untranslated.`);
877
+ return errorResult;
878
+ }
879
+ // The sub-threshold band (a single stray English word) is accepted by
880
+ // design; count what ships so the leniency's real-world leak rate is
881
+ // visible in Amplitude before anyone tightens the threshold.
882
+ const nearMisses = findLanguageNearMisses(textFields, reportLanguage);
883
+ languageTelemetry.reportLanguage = reportLanguage;
884
+ languageTelemetry.languageNearMissCount = String(nearMisses.length);
885
+ if (nearMisses.length > 0) {
886
+ logger.info(`Report language ${reportLanguage}: accepting ${nearMisses.length} field(s) with a single stray English word`, { fields: nearMisses });
887
+ }
888
+ }
570
889
  const dedupedNewTests = deduplicateById([...params.newTestsCreated]);
571
- const dedupedRecommendations = deduplicateById([...(params.additionalRecommendations ?? [])]);
890
+ const dedupedRecommendations = deduplicateById([
891
+ ...(params.additionalRecommendations ?? []),
892
+ ]);
572
893
  const stateManager = StateManager.fromStatePath(params.stateFile);
573
894
  let stateData;
574
895
  try {
@@ -589,7 +910,8 @@ export function registerSubmitReportTool(server) {
589
910
  // than silently reporting an empty testMaintenance section indistinguishable from a
590
911
  // real "nothing to do". With zero existing tests there was nothing to lose, so
591
912
  // undefined is safe there — no round-trip through skyramp_actions required.
592
- if (stateData.maintenanceVerdicts === undefined && (stateData.existingTests?.length ?? 0) > 0) {
913
+ if (stateData.maintenanceVerdicts === undefined &&
914
+ (stateData.existingTests?.length ?? 0) > 0) {
593
915
  errorResult = toolError("stateFile has existingTests but no maintenanceVerdicts — skyramp_actions was not called for this run. " +
594
916
  "Call skyramp_actions (with recommendations: [] if no existing tests needed action) before skyramp_submit_report.");
595
917
  return errorResult;
@@ -613,19 +935,24 @@ export function registerSubmitReportTool(server) {
613
935
  return false;
614
936
  const endpoints = parseEndpointField(t.endpoint);
615
937
  const names = planNameCandidates(t.testId, t.testType);
616
- return !(names.length > 0 ? names : [undefined]).some((scenarioName) => endpoints.some(({ method, path }) => matchesApprovedPlan(approvedPlan, { scenarioName, testType: t.testType, method, path })));
938
+ return !(names.length > 0 ? names : [undefined]).some((scenarioName) => endpoints.some(({ method, path }) => matchesApprovedPlan(approvedPlan, {
939
+ scenarioName,
940
+ testType: t.testType,
941
+ method,
942
+ path,
943
+ })));
617
944
  });
618
945
  if (unapproved.length > 0) {
619
946
  const approvedList = approvedPlan.generate.length > 0
620
- // Show each item's endpoint keys: without them a rejection that turns on
621
- // the endpoint reads as self-contradicting, because the entry's own name
622
- // is printed in this same list (SKYR-4123).
623
- ? approvedPlan.generate
624
- .map((item) => {
625
- const eps = item.matchKeys.filter((k) => k.startsWith("ep:"));
626
- return `[${item.testType}] ${item.scenarioName}${eps.length > 0 ? ` covering ${eps.join(", ")}` : ""}`;
627
- })
628
- .join("; ")
947
+ ? // Show each item's endpoint keys: without them a rejection that turns on
948
+ // the endpoint reads as self-contradicting, because the entry's own name
949
+ // is printed in this same list (SKYR-4123).
950
+ approvedPlan.generate
951
+ .map((item) => {
952
+ const eps = item.matchKeys.filter((k) => k.startsWith("ep:"));
953
+ return `[${item.testType}] ${item.scenarioName}${eps.length > 0 ? ` covering ${eps.join(", ")}` : ""}`;
954
+ })
955
+ .join("; ")
629
956
  : "(none)";
630
957
  errorResult = toolError(`${unapproved.length} newTestsCreated entr${unapproved.length === 1 ? "y" : "ies"} not in the approved plan from ` +
631
958
  `skyramp_register_test_plan (plan ${approvedPlan.planId}) — neither its GENERATE list nor its ADDITIONAL backfill pool: ` +
@@ -676,9 +1003,14 @@ export function registerSubmitReportTool(server) {
676
1003
  const recorded = stateData.existingTests?.find((t) => testFileMatches(t.testFile, m.testFilePath));
677
1004
  const detail = params.testMaintenanceDetails?.find((d) => d.testFilePath === m.testFilePath);
678
1005
  const displayName = path.basename(m.testFilePath);
679
- const defaultBeforeStatus = MAINTENANCE_CHANGE_ACTIONS.has(m.action) ? TestExecutionStatus.Unknown : TestExecutionStatus.Skipped;
680
- const defaultAfterStatus = m.action === DriftAction.Delete ? TestExecutionStatus.Skipped
681
- : MAINTENANCE_CHANGE_ACTIONS.has(m.action) ? TestExecutionStatus.Unknown : TestExecutionStatus.Skipped;
1006
+ const defaultBeforeStatus = MAINTENANCE_CHANGE_ACTIONS.has(m.action)
1007
+ ? TestExecutionStatus.Unknown
1008
+ : TestExecutionStatus.Skipped;
1009
+ const defaultAfterStatus = m.action === DriftAction.Delete
1010
+ ? TestExecutionStatus.Skipped
1011
+ : MAINTENANCE_CHANGE_ACTIONS.has(m.action)
1012
+ ? TestExecutionStatus.Unknown
1013
+ : TestExecutionStatus.Skipped;
682
1014
  const beforeStatus = recorded?.executionBefore?.status ?? defaultBeforeStatus;
683
1015
  const afterStatus = recorded?.executionAfter?.status ?? defaultAfterStatus;
684
1016
  // Trim before checking — a whitespace-only string is semantically blank and
@@ -690,7 +1022,13 @@ export function registerSubmitReportTool(server) {
690
1022
  if (recorded?.executionAfter && !afterDetails)
691
1023
  missingDetails.push(`${displayName} (afterDetails)`);
692
1024
  logger.info(`${displayName}: before=${beforeStatus} after=${afterStatus}`);
693
- return { ...m, beforeDetails, afterDetails, beforeStatus, afterStatus };
1025
+ return {
1026
+ ...m,
1027
+ beforeDetails,
1028
+ afterDetails,
1029
+ beforeStatus,
1030
+ afterStatus,
1031
+ };
694
1032
  });
695
1033
  }
696
1034
  if (missingDetails.length > 0) {
@@ -710,8 +1048,24 @@ export function registerSubmitReportTool(server) {
710
1048
  const repoRoot = fullState?.metadata?.repositoryPath;
711
1049
  if (repoRoot && repoRoot !== "unknown") {
712
1050
  try {
713
- const changedFiles = await listChangedFiles(repoRoot);
714
- const unbacked = findUnbackedClaims({
1051
+ // A run writes tests into up to three kinds of trees: the primary
1052
+ // checkout, the tests-repo checkout (testRepoPath), and related
1053
+ // repos. Scanning only the primary rejected a real, executed UI
1054
+ // test in the tests-repo checkout as "unbacked", forcing the agent
1055
+ // to demote it to additionalRecommendations (SKYR-4204 reopen,
1056
+ // run 32510028428).
1057
+ // The primary scan keeps its throwing form: an unscannable primary
1058
+ // degrades the whole check to a pass (SKYR-3883 semantics) via the
1059
+ // catch below. The extra trees are best-effort — a missing or
1060
+ // non-git one contributes nothing.
1061
+ const changedFiles = [
1062
+ ...(await listChangedFiles(repoRoot)),
1063
+ ...(await listChangedFilesAcross([
1064
+ getTestsRepoDir(),
1065
+ ...Object.values(fullState?.relatedRepos ?? {}).map((section) => section.repositoryPath),
1066
+ ])),
1067
+ ];
1068
+ const unbacked = findUnchangedFileClaims({
715
1069
  repoRoot,
716
1070
  changedFiles,
717
1071
  newTests: dedupedNewTests,
@@ -730,7 +1084,10 @@ export function registerSubmitReportTool(server) {
730
1084
  // SKYR-4129. Do NOT offer re-running skyramp_actions to restate the verdict:
731
1085
  // this branch only fires when the edit is genuinely absent, so rewriting the
732
1086
  // record to match that would be the tamper path, not the fix.
733
- const changedList = changedFiles.slice(0, 20).map((f) => ` - ${f}`).join("\n");
1087
+ const changedList = changedFiles
1088
+ .slice(0, 20)
1089
+ .map((f) => ` - ${f}`)
1090
+ .join("\n");
734
1091
  errorResult = toolError(`${unbacked.length} report claim(s) are not backed by any change in the working tree — ` +
735
1092
  `Testbot will not report file work that hasn't actually been made. Do NOT make a token edit to the claimed file just to satisfy this check.\n` +
736
1093
  `For a newTestsCreated claim: create the file, correct the claim's fileName to the file you actually created (see the changed files below), or remove the claim.\n` +
@@ -747,12 +1104,36 @@ export function registerSubmitReportTool(server) {
747
1104
  }
748
1105
  }
749
1106
  }
1107
+ // Validated against the tool INPUT only, never the state file (a state-file
1108
+ // guard here previously caused a rejection loop). Only entries carrying a
1109
+ // citation are checked, so an ordinary run pays no extra cost.
1110
+ // `repository` is normalized BEFORE the check so the attribution the check
1111
+ // reconciles is the one the report ships, and so the stamp below survives
1112
+ // into the written file — this array is what the report is built from.
1113
+ const issuesFound = params.issuesFound.map(normalizeRepository);
1114
+ if (issuesFound.some((issue) => issue.sourceFile?.trim())) {
1115
+ const citations = await findInvalidSourceCitations({
1116
+ checkouts: await stateManager.listRepoCheckouts(),
1117
+ issues: issuesFound,
1118
+ });
1119
+ if (citations.invalid.length > 0) {
1120
+ errorResult = toolError(citations.invalid.join("\n"));
1121
+ return errorResult;
1122
+ }
1123
+ // The checkout the citation resolved in owns the finding, so the report
1124
+ // states it. Left to the consumer, an unattributed related-repo path is
1125
+ // resolved against the primary checkout and links the wrong repo's file.
1126
+ citations.repository.forEach((repo, i) => {
1127
+ if (repo)
1128
+ issuesFound[i] = { ...issuesFound[i], repository: repo };
1129
+ });
1130
+ }
750
1131
  // Strip generation-artifact fields from newTestsCreated before writing.
751
1132
  // scenarioFile, traceFile, frontendTrace are internal paths used during
752
1133
  // generation — downstream scoring scripts don't expect them and fail if
753
1134
  // they encounter these string fields while traversing the object.
754
1135
  // Also normalize each item's `repository` (blank → undefined).
755
- const sanitizedNewTests = await Promise.all(dedupedNewTests.map(({ scenarioFile: _sf, traceFile: _tf, frontendTrace: _ft, ...rest }) => attachReuseOutcome(normalizeRepository(rest), stateData.reuseOutcomes)));
1136
+ const sanitizedNewTests = await Promise.all(dedupedNewTests.map(({ scenarioFile: _sf, traceFile: _tf, frontendTrace: _ft, ...rest }) => attachReuseOutcome(normalizeRepository(rest), stateData.reuseOutcomes, stateData.reuseHandOffs)));
756
1137
  const report = {
757
1138
  businessCaseAnalysis: params.businessCaseAnalysis,
758
1139
  newTestsCreated: sanitizedNewTests,
@@ -773,9 +1154,10 @@ export function registerSubmitReportTool(server) {
773
1154
  const { testFilePath: _tfp, ...wire } = attachVideoPath(normalizeRepository(row), stateData.executionVideos);
774
1155
  return wire;
775
1156
  }),
776
- issuesFound: params.issuesFound.map(normalizeRepository),
1157
+ issuesFound,
777
1158
  nextSteps: params.nextSteps ?? [],
778
- commitMessage: (params.commitMessage ?? "").replace(/[\r\n]+/g, " ").trim() || DEFAULT_COMMIT_MESSAGE,
1159
+ commitMessage: (params.commitMessage ?? "").replace(/[\r\n]+/g, " ").trim() ||
1160
+ DEFAULT_COMMIT_MESSAGE,
779
1161
  };
780
1162
  const reportJson = JSON.stringify(report, null, 2);
781
1163
  // Beside the state file, which was read successfully above — so this directory is
@@ -815,6 +1197,7 @@ export function registerSubmitReportTool(server) {
815
1197
  testResultCount: String(params.testResults.length),
816
1198
  payloadBytes: String(reportJson.length),
817
1199
  ...computeReportMetrics({ ...params, testMaintenance }),
1200
+ ...languageTelemetry,
818
1201
  }).catch(() => { });
819
1202
  }
820
1203
  });