@skyramp/mcp 0.3.4 → 0.3.6-rc.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (159) hide show
  1. package/build/playwright/registerPlaywrightTools.js +92 -30
  2. package/build/playwright/traceRecordingPrompt.d.ts +6 -0
  3. package/build/playwright/traceRecordingPrompt.js +6 -2
  4. package/build/prompts/code-reuse.d.ts +1 -2
  5. package/build/prompts/code-reuse.js +182 -77
  6. package/build/prompts/modularization/integration-test-modularization.d.ts +2 -0
  7. package/build/prompts/modularization/integration-test-modularization.js +83 -41
  8. package/build/prompts/modularization/render.d.ts +18 -0
  9. package/build/prompts/modularization/render.js +12 -0
  10. package/build/prompts/modularization/ui-test-modularization.d.ts +3 -1
  11. package/build/prompts/modularization/ui-test-modularization.js +89 -47
  12. package/build/prompts/pom-aware-code-reuse.js +3 -1
  13. package/build/prompts/shared-helper-policy.d.ts +57 -0
  14. package/build/prompts/shared-helper-policy.js +135 -0
  15. package/build/prompts/test-recommendation/diffExecutionPlan.js +62 -56
  16. package/build/prompts/test-recommendation/fullRepoCatalog.js +19 -8
  17. package/build/prompts/test-recommendation/recommendationShared.d.ts +28 -6
  18. package/build/prompts/test-recommendation/recommendationShared.js +90 -16
  19. package/build/prompts/test-recommendation/registerRecommendTestsPrompt.js +22 -0
  20. package/build/prompts/test-recommendation/test-recommendation-prompt.d.ts +2 -2
  21. package/build/prompts/test-recommendation/test-recommendation-prompt.js +3 -3
  22. package/build/prompts/testbot/testbot-prompts.js +88 -33
  23. package/build/recommendation/budgeters/shared.js +105 -27
  24. package/build/recommendation/discriminators.js +13 -2
  25. package/build/recommendation/planRanker.d.ts +6 -6
  26. package/build/recommendation/planRanker.js +6 -61
  27. package/build/services/AnalyticsService.d.ts +7 -0
  28. package/build/services/AnalyticsService.js +7 -1
  29. package/build/services/ModularizationService.js +1 -3
  30. package/build/services/TestDiscoveryService.d.ts +0 -2
  31. package/build/services/TestDiscoveryService.js +2 -37
  32. package/build/services/TestGenerationService.d.ts +16 -0
  33. package/build/services/TestGenerationService.js +86 -10
  34. package/build/services/containerEnv.js +13 -12
  35. package/build/tools/code-refactor/codeReuseTool.js +279 -93
  36. package/build/tools/code-refactor/enhance-state.d.ts +49 -0
  37. package/build/tools/code-refactor/enhance-state.js +109 -0
  38. package/build/tools/code-refactor/enhanceAssertionsTool.js +34 -1
  39. package/build/tools/code-refactor/modularizationTool.js +9 -2
  40. package/build/tools/code-refactor/reuse-outcome.d.ts +23 -1
  41. package/build/tools/code-refactor/reuse-outcome.js +14 -4
  42. package/build/tools/code-refactor/reuse-state.d.ts +127 -5
  43. package/build/tools/code-refactor/reuse-state.js +628 -16
  44. package/build/tools/code-refactor/utils-verify-gates.d.ts +26 -0
  45. package/build/tools/code-refactor/utils-verify-gates.js +100 -0
  46. package/build/tools/code-refactor/verify-gates.d.ts +2 -1
  47. package/build/tools/code-refactor/verify-gates.js +90 -25
  48. package/build/tools/executeSkyrampTestTool.d.ts +19 -0
  49. package/build/tools/executeSkyrampTestTool.js +158 -8
  50. package/build/tools/generate-tests/generateBatchScenarioRestTool.js +2 -2
  51. package/build/tools/generate-tests/generateE2ERestTool.js +16 -0
  52. package/build/tools/generate-tests/generateUIRestTool.d.ts +1 -0
  53. package/build/tools/generate-tests/generateUIRestTool.js +22 -0
  54. package/build/tools/generate-tests/scenarioLint.d.ts +2 -0
  55. package/build/tools/generate-tests/scenarioLint.js +127 -19
  56. package/build/tools/generate-tests/trace-reuse-guard.d.ts +20 -0
  57. package/build/tools/generate-tests/trace-reuse-guard.js +93 -0
  58. package/build/tools/submitReportTool.d.ts +38 -38
  59. package/build/tools/submitReportTool.js +487 -104
  60. package/build/tools/test-management/analyzeChangesTool.d.ts +24 -1
  61. package/build/tools/test-management/analyzeChangesTool.js +75 -12
  62. package/build/tools/test-management/analyzeTestHealthTool.js +7 -7
  63. package/build/tools/test-management/registerTestPlanTool.d.ts +203 -0
  64. package/build/tools/test-management/registerTestPlanTool.js +149 -23
  65. package/build/types/Recommendation.d.ts +34 -5
  66. package/build/types/RepositoryAnalysis.d.ts +133 -114
  67. package/build/types/RepositoryAnalysis.js +1 -1
  68. package/build/types/ReuseOutcome.d.ts +102 -6
  69. package/build/types/ReuseOutcome.js +16 -2
  70. package/build/types/TestRecommendation.js +21 -3
  71. package/build/types/TestTypes.js +14 -8
  72. package/build/types/TestbotReport.d.ts +25 -3
  73. package/build/types/index.d.ts +2 -2
  74. package/build/types/index.js +1 -1
  75. package/build/utils/AnalysisStateManager.d.ts +69 -1
  76. package/build/utils/AnalysisStateManager.js +69 -5
  77. package/build/utils/branchDiff.d.ts +10 -0
  78. package/build/utils/branchDiff.js +28 -0
  79. package/build/utils/changedRoutes.d.ts +29 -0
  80. package/build/utils/changedRoutes.js +87 -0
  81. package/build/utils/featureFlags.d.ts +21 -0
  82. package/build/utils/featureFlags.js +23 -0
  83. package/build/utils/frontendIntegration.js +34 -4
  84. package/build/utils/importerHop.d.ts +2 -8
  85. package/build/utils/importerHop.js +15 -53
  86. package/build/utils/pathMatching.d.ts +38 -0
  87. package/build/utils/pathMatching.js +71 -0
  88. package/build/utils/pathSignatures.d.ts +22 -0
  89. package/build/utils/pathSignatures.js +57 -0
  90. package/build/utils/planMatchKeys.d.ts +16 -3
  91. package/build/utils/planMatchKeys.js +26 -10
  92. package/build/utils/pluralization.d.ts +10 -0
  93. package/build/utils/pluralization.js +18 -0
  94. package/build/utils/pom-catalog-parse.d.ts +52 -0
  95. package/build/utils/pom-catalog-parse.js +141 -0
  96. package/build/utils/pom-scope/selector-extractor.d.ts +12 -0
  97. package/build/utils/pom-scope/selector-extractor.js +34 -8
  98. package/build/utils/pom-verify/verify.d.ts +6 -5
  99. package/build/utils/pom-verify/verify.js +8 -6
  100. package/build/utils/reportLanguage.d.ts +43 -0
  101. package/build/utils/reportLanguage.js +125 -0
  102. package/build/utils/reportVerification.d.ts +74 -4
  103. package/build/utils/reportVerification.js +259 -3
  104. package/build/utils/reuseRouting.d.ts +3 -0
  105. package/build/utils/reuseRouting.js +50 -0
  106. package/build/utils/routeParsers.d.ts +2 -0
  107. package/build/utils/routeParsers.js +65 -8
  108. package/build/utils/scenarioDrafting.d.ts +1 -1
  109. package/build/utils/scenarioDrafting.js +57 -45
  110. package/build/utils/subjectEndpoints.d.ts +19 -0
  111. package/build/utils/subjectEndpoints.js +98 -0
  112. package/build/utils/testFileClassification.d.ts +11 -0
  113. package/build/utils/testFileClassification.js +47 -0
  114. package/build/utils/uiPageEnumerator.d.ts +45 -19
  115. package/build/utils/uiPageEnumerator.js +95 -51
  116. package/build/utils/utils-verify/allow.d.ts +16 -0
  117. package/build/utils/utils-verify/allow.js +68 -0
  118. package/build/utils/utils-verify/call-sites.d.ts +34 -0
  119. package/build/utils/utils-verify/call-sites.js +154 -0
  120. package/build/utils/utils-verify/index.d.ts +7 -0
  121. package/build/utils/utils-verify/index.js +7 -0
  122. package/build/utils/utils-verify/language-spec.d.ts +91 -0
  123. package/build/utils/utils-verify/language-spec.js +210 -0
  124. package/build/utils/utils-verify/locate.d.ts +39 -0
  125. package/build/utils/utils-verify/locate.js +199 -0
  126. package/build/utils/utils-verify/parse.d.ts +34 -0
  127. package/build/utils/utils-verify/parse.js +177 -0
  128. package/build/utils/utils-verify/stage.d.ts +24 -0
  129. package/build/utils/utils-verify/stage.js +107 -0
  130. package/build/utils/utils-verify/verify.d.ts +63 -0
  131. package/build/utils/utils-verify/verify.js +168 -0
  132. package/build/utils/utils.d.ts +3 -1
  133. package/build/utils/utils.js +3 -1
  134. package/build/workspace/workspace.d.ts +32 -32
  135. package/node_modules/playwright/lib/mcp/skyramp/assertTool.js +9 -5
  136. package/node_modules/playwright/lib/mcp/skyramp/loadTraceTool.js +16 -0
  137. package/node_modules/playwright/lib/mcp/skyramp/skyRampImport.js +2 -0
  138. package/node_modules/playwright/lib/mcp/skyramp/traceRecordingBackend.js +115 -14
  139. package/node_modules/playwright/lib/mcp/test/skyRampExport.js +13 -1
  140. package/node_modules/playwright/node_modules/playwright-core/.DS_Store +0 -0
  141. package/node_modules/playwright/node_modules/playwright-core/lib/server/codegen/skyramp/jsonlReader.js +2 -0
  142. package/node_modules/playwright/node_modules/playwright-core/lib/vite/htmlReport/index.html +27 -253
  143. package/node_modules/playwright/node_modules/playwright-core/lib/vite/recorder/assets/{codeMirrorModule-DtudTj_v.js → codeMirrorModule-DJMC4zNo.js} +1 -1
  144. package/node_modules/playwright/node_modules/playwright-core/lib/vite/recorder/assets/index-BW82eAUI.js +196 -0
  145. package/node_modules/playwright/node_modules/playwright-core/lib/vite/recorder/index.html +1 -1
  146. package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/assets/{codeMirrorModule-FNMuBzX1.js → codeMirrorModule-CZfp96qZ.js} +1 -1
  147. package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/assets/defaultSettingsView-gpLo02E0.js +809 -0
  148. package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/index.Bq1r1URj.js +2 -0
  149. package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/index.html +2 -2
  150. package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/uiMode.VEfqi1qN.js +5 -0
  151. package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/uiMode.html +2 -2
  152. package/node_modules/playwright/node_modules/playwright-core/package.json +1 -1
  153. package/node_modules/playwright/node_modules/playwright-core/src/server/codegen/skyramp/jsonlReader.ts +1 -1
  154. package/node_modules/playwright/package.json +1 -1
  155. package/package.json +2 -2
  156. package/node_modules/playwright/node_modules/playwright-core/lib/vite/recorder/assets/index-BpDwp16L.js +0 -422
  157. package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/assets/defaultSettingsView-Co9upU5h.js +0 -1035
  158. package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/index.DXNIQ_dx.js +0 -2
  159. package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/uiMode.CIKB3XSv.js +0 -5
@@ -87,7 +87,7 @@ export const scenarioStepSchema = z.object({
87
87
  interactionType: z.enum(["success", "error", "edge-case"]),
88
88
  requestBody: z.record(z.any()).optional(),
89
89
  queryParams: z.record(z.any()).optional(),
90
- responseBody: z.record(z.any()).optional(),
90
+ responseBody: z.union([z.record(z.any()), z.array(z.any())]).optional(),
91
91
  expectedStatusCode: z.number(),
92
92
  expectedResponseFields: z.array(z.string()).optional(),
93
93
  chainsFrom: z
@@ -1,6 +1,9 @@
1
1
  /**
2
- * The POM code-reuse shape carried in a TestbotReport — the wire contract between
3
- * this server and the consumers that render it (test-bot.git).
2
+ * The code-reuse observability shape carried in a TestbotReport — the wire contract
3
+ * between this server and the consumers that render it (test-bot.git). It covers
4
+ * both reuse paths: page-object (POM) reuse for browser tests, and shared-helper
5
+ * (SkyrampUtils) reuse for API tests, plus whether the generation chain that leads
6
+ * to reuse was followed at all.
4
7
  *
5
8
  * Lives in types/ rather than beside the derivation logic in
6
9
  * tools/code-refactor/reuse-outcome.ts, and is re-exported from types/index.ts,
@@ -45,11 +48,72 @@ export interface ReuseSkippedEntry {
45
48
  declinedBy?: ReuseDeclinedBy;
46
49
  reason: string;
47
50
  }
51
+ /** A raw locator left inline in the delivered spec that a catalogued POM member
52
+ * covers — the calls the spec did NOT make, which the call counts cannot express.
53
+ *
54
+ * Purely informational: nothing gates on it. A page object may legitimately not
55
+ * model a given assertion, so an entry here is a prompt to look, never a verdict
56
+ * that the spec is wrong.
57
+ *
58
+ * Overlaps with {@link ReuseOutcome.skipped} on purpose, so a reader who sees the
59
+ * same page object in both lists is not looking at two problems. A decline names a
60
+ * FILE (`// kept inline: <file basename>`), never a locator, so filtering this list
61
+ * by it would drop every miss in that file — including the ones the agent never
62
+ * considered. Read `skipped` as the reasons the agent gave, and this list as what
63
+ * the delivered spec still carries. */
64
+ export interface ReuseMissedEntry {
65
+ /** The selector token exactly as it appears in the delivered spec. */
66
+ locator: string;
67
+ /** Basename of the catalog's `Import:` path. Deliberately EXTENSION-LESS
68
+ * (`functionsPage`), unlike {@link ReuseSkippedEntry.pageObject}, which carries
69
+ * the extension (`functionsPage.js`) because it is parsed out of the agent's
70
+ * `// kept inline: <file basename>` comment. Resolving the catalog's
71
+ * extension-less import against disk would mean probing `.ts`/`.js` and inventing
72
+ * new failure modes for a cosmetic gain, so the two are documented rather than
73
+ * unified. */
74
+ pageObject: string;
75
+ /** `<ClassName>.<member>` as the catalog names it, e.g. `FunctionsPage.getRowByName`. */
76
+ member: string;
77
+ }
78
+ /** Outcome of the shared-helper (SkyrampUtils) verification for a test. A separate
79
+ * enum from {@link ReuseVerificationOutcome} on purpose: the utils path has no
80
+ * page-object layer to skip on, and reusing `skipped-no-pom` to mean "no helper
81
+ * file was written" would be a mislabelled member. */
82
+ export declare enum HelperVerificationOutcome {
83
+ Passed = "passed",
84
+ Failed = "failed",
85
+ /** The test neither wrote nor imports a shared utils file. */
86
+ NoUtilsFile = "no-utils-file"
87
+ }
88
+ /** Shared-helper reuse for an API test, re-derived from the delivered files at report
89
+ * time — nothing here is read back from what the agent or the verify pass recorded.
90
+ * Absent altogether when no utils file exists for the test (the same omission rule
91
+ * as the POM fields at zero candidates: a zero says nothing "no file" does not). */
92
+ export interface HelperReuseOutcome {
93
+ /** BASENAME of the shared utils file the test imports from or this run wrote —
94
+ * found by its header, never by a conventional name (the reuse prompt allows "a
95
+ * new file with a different name" once the conventional one is large). Several
96
+ * files are joined with `, `. */
97
+ utilsFile: string;
98
+ /** Helpers the delivered test imports from that file that the file defines. */
99
+ helpersImported?: number;
100
+ /** Inline request calls in OTHER Skyramp-generated tests beside this one that a
101
+ * helper in `utilsFile` already wraps — reuse that was available and not taken.
102
+ * Omitted at zero. Informational, like {@link ReuseOutcome.missedReuse}. */
103
+ siblingInlineCallSites?: number;
104
+ /** Whether the delivered utils file holds the shared-helper invariants: one helper
105
+ * per method+path, status-code-only assertions, method+resource names — with
106
+ * documented declines (`reuse-verify: allow …`) counted as holding. */
107
+ verification?: HelperVerificationOutcome;
108
+ }
48
109
  /**
49
- * What lands in a report's `reuse` field, for a UI test whose generation ran POM
50
- * code reuse. Every member is optional: consumers must render from whatever
51
- * subset is present, and omit the summary entirely when the object is absent (as
52
- * it is for non-UI tests, and in reports from versions predating the field).
110
+ * What lands in a report's `reuse` field for a generated test. Every member is
111
+ * optional: consumers must render from whatever subset is present, and omit the
112
+ * summary entirely when the object is absent (as it is for tests whose generation
113
+ * ran no reuse step, and in reports from versions predating the field). POM fields
114
+ * appear for browser tests that took the POM path; `helpers` for tests on the
115
+ * SkyrampUtils path; `chainSkipped` for either when the reuse step was owed and
116
+ * never called.
53
117
  *
54
118
  * `candidatesDetected` counts FILES while the call counts count CALLS — an
55
119
  * indicator, never a fraction. Do not render the pair as a ratio or percentage.
@@ -58,6 +122,38 @@ export interface ReuseOutcome {
58
122
  candidatesDetected?: number;
59
123
  callsReused?: number;
60
124
  callsVerified?: number;
125
+ /** Calls the verifier could not statically check — dynamic locators, chained
126
+ * `.click()` on a property, anything not resolvable without executing.
127
+ *
128
+ * `callsVerified + callsUnverifiable === callsReused` holds, so a renderer can
129
+ * say "9 reused, 6 verified, 3 not statically verifiable" instead of implying
130
+ * three failures. Carried as its own number rather than left to the renderer to
131
+ * subtract, so a consumer cannot arrive at a different one.
132
+ *
133
+ * It equals `unverifiableCalls.length` too — the verifier puts every call it
134
+ * examines in exactly one of the two buckets — but derive it from the counts,
135
+ * since the list is omitted when empty. */
136
+ callsUnverifiable?: number;
137
+ /** The unverifiable calls by name. Unrecognized bindings are NOT included — they
138
+ * are not calls, and counting them here is what made the old single list
139
+ * unusable. */
140
+ unverifiableCalls?: string[];
141
+ /** Raw selector tokens still inline in the delivered spec. The denominator the
142
+ * call counts lack: `callsReused: 2` alone cannot distinguish thorough reuse
143
+ * from a spec that reused login and ignored everything else. */
144
+ rawLocatorsInline?: number;
145
+ /** The subset of those raw locators that a catalogued POM member covers.
146
+ * Informational; see {@link ReuseMissedEntry}. */
147
+ missedReuse?: ReuseMissedEntry[];
61
148
  verification?: ReuseVerificationOutcome;
62
149
  skipped?: ReuseSkippedEntry[];
150
+ /** Shared-helper (SkyrampUtils) reuse. See {@link HelperReuseOutcome}. */
151
+ helpers?: HelperReuseOutcome;
152
+ /** Present — and only ever `true` — when generation handed off a modularize→reuse
153
+ * step for this test and no reuse pass was recorded for it. Since the execution
154
+ * checkpoint refuses to run such a test until reuse is called, this can only
155
+ * appear when the agent walked past that checkpoint to the report: it is a
156
+ * direct signal of gate evasion, not a general "did not happen". Omitted (never
157
+ * `false`) when the chain was followed. */
158
+ chainSkipped?: true;
63
159
  }
@@ -1,6 +1,9 @@
1
1
  /**
2
- * The POM code-reuse shape carried in a TestbotReport — the wire contract between
3
- * this server and the consumers that render it (test-bot.git).
2
+ * The code-reuse observability shape carried in a TestbotReport — the wire contract
3
+ * between this server and the consumers that render it (test-bot.git). It covers
4
+ * both reuse paths: page-object (POM) reuse for browser tests, and shared-helper
5
+ * (SkyrampUtils) reuse for API tests, plus whether the generation chain that leads
6
+ * to reuse was followed at all.
4
7
  *
5
8
  * Lives in types/ rather than beside the derivation logic in
6
9
  * tools/code-refactor/reuse-outcome.ts, and is re-exported from types/index.ts,
@@ -31,3 +34,14 @@ export var ReuseVerificationOutcome;
31
34
  /** No page-object layer existed to reuse. */
32
35
  ReuseVerificationOutcome["SkippedNoPom"] = "skipped-no-pom";
33
36
  })(ReuseVerificationOutcome || (ReuseVerificationOutcome = {}));
37
+ /** Outcome of the shared-helper (SkyrampUtils) verification for a test. A separate
38
+ * enum from {@link ReuseVerificationOutcome} on purpose: the utils path has no
39
+ * page-object layer to skip on, and reusing `skipped-no-pom` to mean "no helper
40
+ * file was written" would be a mislabelled member. */
41
+ export var HelperVerificationOutcome;
42
+ (function (HelperVerificationOutcome) {
43
+ HelperVerificationOutcome["Passed"] = "passed";
44
+ HelperVerificationOutcome["Failed"] = "failed";
45
+ /** The test neither wrote nor imports a shared utils file. */
46
+ HelperVerificationOutcome["NoUtilsFile"] = "no-utils-file";
47
+ })(HelperVerificationOutcome || (HelperVerificationOutcome = {}));
@@ -15,7 +15,7 @@ export var Novelty;
15
15
  })(Novelty || (Novelty = {}));
16
16
  /** Internal-only categories (not submitted to tools). */
17
17
  const INTERNAL_CATEGORIES = [
18
- "new_endpoint", // CRITICAL - diff-direct scenarios always fill GENERATE slots first
18
+ "new_endpoint", // MEDIUM - diff-direct scenario; where a test came from, not a guarantee of a slot
19
19
  "bug_caught", // CRITICAL - tests targeting a specific <bug_found> flaw identified during enrichment
20
20
  ];
21
21
  /** External categories valid for tool submissions, ordered by priority. */
@@ -39,9 +39,27 @@ export const SCENARIO_CATEGORIES = [...INTERNAL_CATEGORIES, ...CATEGORIES];
39
39
  export const TEST_CATEGORIES = CATEGORIES;
40
40
  /** Priority assignment for each category. */
41
41
  export const CATEGORY_PRIORITY = {
42
- new_endpoint: PriorityTier.CRITICAL,
42
+ // CRITICAL means "this test targets a specific identified flaw" — only bug_caught
43
+ // asserts that. new_endpoint says where a test came from (a template fired on an
44
+ // endpoint the diff added), not that it earns a guaranteed slot. Reserving the
45
+ // top tier for it let server-drafted templates outrank anything an agent said,
46
+ // by label rather than merit. (The registration enum is SCENARIO_CATEGORIES, so
47
+ // an agent CAN submit new_endpoint; externalCategory maps it to crud/LOW for
48
+ // report and merge purposes only.)
49
+ //
50
+ // It sits BELOW the categories that state what a test proves, not level with
51
+ // them. Level was still too high: the last rank key is the candidateId, so among
52
+ // same-tier candidates the ALPHABET decided. Template names start with the
53
+ // resource (billing-, bookings-, export-, members-) and an agent names a scenario
54
+ // for what it checks, so on cc15-org-reviewer-role all 8 templates outranked all
55
+ // 12 candidates the agent wrote, and 2 of the 3 GENERATE slots went to the letter
56
+ // "b" beating "o". MEDIUM, not LOW: a template on an endpoint the diff just added
57
+ // is still worth more than a crud test on an endpoint that was always there. The
58
+ // floor survives — when the agent offers nothing above MEDIUM, templates still
59
+ // fill GENERATE.
60
+ new_endpoint: PriorityTier.MEDIUM,
43
61
  bug_caught: PriorityTier.CRITICAL, // tests targeting a <bug_found> flaw — always in GENERATE
44
- business_rule: PriorityTier.HIGH, // formula/business-logic bugs are high priority but CRITICAL is reserved for new-endpoint diff-direct scenarios
62
+ business_rule: PriorityTier.HIGH, // formula/business-logic bugs are high priority
45
63
  security_boundary: PriorityTier.HIGH,
46
64
  data_integrity: PriorityTier.HIGH,
47
65
  breaking_change: PriorityTier.HIGH,
@@ -163,7 +163,9 @@ export const basePlaywrightSchema = z.object({
163
163
  playwrightViewportSize: z
164
164
  .union([
165
165
  z.enum(["", "hd", "full-hd", "2k"]),
166
- z.string().regex(/^\d+,\d+$/, "Custom viewport must be 'WIDTH,HEIGHT' format (e.g., '1920,1080')"),
166
+ z
167
+ .string()
168
+ .regex(/^\d+,\d+$/, "Custom viewport must be 'WIDTH,HEIGHT' format (e.g., '1920,1080')"),
167
169
  ])
168
170
  .default("")
169
171
  .describe("Viewport size for playwright browser. THE VALUE MUST BE IN THE FORMAT ['', 'hd', 'full-hd', '2k', 'x,y' e.g. '1920,1080' for 1920x1080 resolution]. DEFAULT VALUE IS ''. If set to '', the browser will use its default viewport size."),
@@ -215,10 +217,10 @@ export const baseTestSchema = {
215
217
  queryParams: z
216
218
  .string()
217
219
  .default("")
218
- .describe("MUST be string of comma separated values like 'id=1,name=John' for URL query parameters. "
219
- + "Workspace-configured api.defaultQueryParams (if set) are merged in automatically, with any "
220
- + "api.queryParamOverrides entry whose pathPattern matches this endpoint layered on top — "
221
- + "no need to repeat them here. An explicit value for the same key here overrides both."),
220
+ .describe("MUST be string of comma separated values like 'id=1,name=John' for URL query parameters. " +
221
+ "Workspace-configured api.defaultQueryParams (if set) are merged in automatically, with any " +
222
+ "api.queryParamOverrides entry whose pathPattern matches this endpoint layered on top — " +
223
+ "no need to repeat them here. An explicit value for the same key here overrides both."),
222
224
  formParams: z
223
225
  .string()
224
226
  .default("")
@@ -235,7 +237,9 @@ export const baseTestSchema = {
235
237
  JSON.parse(val);
236
238
  return true;
237
239
  }
238
- catch { /* not JSON */ }
240
+ catch {
241
+ /* not JSON */
242
+ }
239
243
  const trimmed = val.trim();
240
244
  // Accept common YAML patterns: document separator, mappings (key: val), sequences (- item)
241
245
  if (trimmed.startsWith("---"))
@@ -254,7 +258,9 @@ export const baseTestSchema = {
254
258
  responseStatusCode: z
255
259
  .string()
256
260
  .default("")
257
- .refine((val) => !val || /^\d{3}$/.test(val), { message: "Must be a 3-digit HTTP status code (e.g., '200', '404') or empty string" })
261
+ .refine((val) => !val || /^[1-5]\d{2}$/.test(val), {
262
+ message: "Must be a valid HTTP status code in 100-599 (e.g., '200', '404') or empty string",
263
+ })
258
264
  .describe("Expected HTTP response status code (e.g., '200', '201', '404'). DO NOT ASSUME STATUS CODE IF NOT PROVIDED"),
259
265
  ...baseSchema.shape,
260
266
  };
@@ -262,7 +268,7 @@ export const codeRefactoringSchema = z.object({
262
268
  codeReuse: z
263
269
  .boolean()
264
270
  .default(false)
265
- .describe("Reuse existing code in the generated test — for TS/JS Playwright tests this refactors the test to use the repo's existing Page Object Model"),
271
+ .describe("Reuse existing code in the generated test — consolidates helper functions shared with sibling Skyramp-generated tests into a SkyrampUtils file; for TS/JS Playwright tests with POM reuse enabled it instead refactors the test to use the repo's existing Page Object Model"),
266
272
  modularizeCode: z
267
273
  .boolean()
268
274
  .default(false)
@@ -14,7 +14,9 @@ export declare enum IssueFoundCategory {
14
14
  /**
15
15
  * Shape of the JSON report written by skyramp_submit_report and read by testbot
16
16
  * for rendering as Markdown. All fields mirror the corresponding Zod schemas in
17
- * submitReportTool.ts — keep the two in sync.
17
+ * submitReportTool.ts — keep the two in sync. This type describes what a READER
18
+ * may encounter across MCP versions; the Zod schemas define what the current
19
+ * producer must submit, so a field can be required on submit but optional here.
18
20
  */
19
21
  export interface TestbotReport {
20
22
  businessCaseAnalysis: string;
@@ -25,6 +27,9 @@ export interface TestbotReport {
25
27
  fileName: string;
26
28
  reasoning: string;
27
29
  description: string;
30
+ /** `owner/repo` attribution in multi-repo runs (SKYR-3786). Absent = the
31
+ * primary repo, or a single-repo run. */
32
+ repository?: string;
28
33
  scenarioFile?: string;
29
34
  traceFile?: string;
30
35
  frontendTrace?: string;
@@ -53,6 +58,8 @@ export interface TestbotReport {
53
58
  status: "Pass" | "Fail" | "Skipped";
54
59
  details: string;
55
60
  videoPath?: string;
61
+ /** See newTestsCreated[].repository. */
62
+ repository?: string;
56
63
  }[];
57
64
  additionalRecommendations?: {
58
65
  testId: string;
@@ -66,7 +73,7 @@ export interface TestbotReport {
66
73
  description: string;
67
74
  expectedStatusCode?: number;
68
75
  requestBody?: Record<string, unknown>;
69
- responseBody?: Record<string, unknown>;
76
+ responseBody?: Record<string, unknown> | unknown[];
70
77
  }[];
71
78
  description: string;
72
79
  priority: "high" | "medium" | "low";
@@ -74,11 +81,26 @@ export interface TestbotReport {
74
81
  openApiSpec?: string;
75
82
  backendTrace?: string;
76
83
  frontendTrace?: string;
84
+ /** See newTestsCreated[].repository. */
85
+ repository?: string;
77
86
  }[];
78
87
  issuesFound: {
79
88
  description: string;
80
89
  severity?: "critical" | "high" | "medium" | "low";
81
- category: IssueFoundCategory;
90
+ /** Required by the submit_report schema since 0.3.4; absent in reports
91
+ * written by older MCP versions. Readers treat absence as Bug. */
92
+ category?: IssueFoundCategory;
93
+ /** Where the defect lives, checked against the checkout when the report is
94
+ * submitted. The submit schema requires sourceFile on a `bug` entry; symbol
95
+ * and line stay optional there (a config file or template has no symbol, and
96
+ * line numbers are advisory). All three are optional here because a report
97
+ * from an older MCP version has none of them, and a lint/type/config entry
98
+ * never needs them. */
99
+ sourceFile?: string;
100
+ sourceSymbol?: string;
101
+ sourceLine?: number;
102
+ /** See newTestsCreated[].repository. */
103
+ repository?: string;
82
104
  }[];
83
105
  nextSteps: string[];
84
106
  commitMessage: string;
@@ -3,6 +3,6 @@ export { DriftAction } from "./TestAnalysis.js";
3
3
  export { TestType, HttpMethod } from "./TestTypes.js";
4
4
  export type { TestbotReport } from "./TestbotReport.js";
5
5
  export { IssueFoundCategory } from "./TestbotReport.js";
6
- export { ReuseDeclinedBy, ReuseVerificationOutcome } from "./ReuseOutcome.js";
7
- export type { ReuseOutcome, ReuseSkippedEntry } from "./ReuseOutcome.js";
6
+ export { ReuseDeclinedBy, ReuseVerificationOutcome, HelperVerificationOutcome, } from "./ReuseOutcome.js";
7
+ export type { ReuseOutcome, ReuseSkippedEntry, ReuseMissedEntry, HelperReuseOutcome, } from "./ReuseOutcome.js";
8
8
  export type { RelatedRepository, TestbotPromptOptions, } from "./TestbotPromptOptions.js";
@@ -2,4 +2,4 @@ export { TestExecutionStatus } from "./TestExecution.js";
2
2
  export { DriftAction } from "./TestAnalysis.js";
3
3
  export { TestType, HttpMethod } from "./TestTypes.js";
4
4
  export { IssueFoundCategory } from "./TestbotReport.js";
5
- export { ReuseDeclinedBy, ReuseVerificationOutcome } from "./ReuseOutcome.js";
5
+ export { ReuseDeclinedBy, ReuseVerificationOutcome, HelperVerificationOutcome, } from "./ReuseOutcome.js";
@@ -8,8 +8,21 @@ import type { FrontendFileIntegration } from "../types/FrontendIntegration.js";
8
8
  import type { ApprovedPlanItem } from "../types/Recommendation.js";
9
9
  import type { ExternalTestRunRecord } from "../types/ExternalTestExecution.js";
10
10
  import type { VideoRecord } from "../types/TestExecution.js";
11
+ import type { RepoCheckout } from "./reportVerification.js";
11
12
  export type { CandidateUiPage } from "./uiPageEnumerator.js";
12
13
  export declare function setTestsRepoDir(dir: string | undefined): void;
14
+ /**
15
+ * Record the run's tests repo without ever clearing a known value.
16
+ *
17
+ * The tests repo is a RUN-scoped fact, but analyze_changes is called once per
18
+ * repository and only the PRIMARY call carries `testsRepoDir` — a related
19
+ * repo's call omits it. An unconditional set therefore wiped the primary's
20
+ * value on the second call of every multi-repo run, and by report time the
21
+ * tests repo was unknown: a UI test generated there looked unbacked by any
22
+ * working tree and the agent demoted it out of newTestsCreated (SKYR-4204,
23
+ * run 32536569467). Callers that mean to clear it use setTestsRepoDir.
24
+ */
25
+ export declare function rememberTestsRepoDir(dir: string | undefined): void;
13
26
  export declare function getTestsRepoDir(): string | undefined;
14
27
  /** Filename of the run-scoped analysis state file under `runArtifactDir()`.
15
28
  * Single-sourced so the constructor that WRITES there and `resolveRunStatePath`
@@ -54,6 +67,20 @@ export declare function clearActiveRunStatePath(): void;
54
67
  * decide where to look for it.
55
68
  */
56
69
  export declare function resolveRunStatePath(explicitPath?: string): string | undefined;
70
+ /**
71
+ * The run's state file if this process is actually serving a run, else undefined.
72
+ *
73
+ * Distinct from {@link resolveRunStatePath}, which answers "where would the file
74
+ * be": the fallback above builds a path from `$RUNNER_TEMP`, and GitHub Actions
75
+ * sets RUNNER_TEMP on every job — including this repo's own CI, where a resolvable
76
+ * path is not a run. Callers that gate BEHAVIOUR on being in a run must use this
77
+ * one; callers that only want to write state can keep using the path, because a
78
+ * write to a file that holds no run data is already a no-op.
79
+ *
80
+ * Returns the path rather than a boolean so a caller can also key per-run state on
81
+ * it, and lives here because this is the module that decides where the file goes.
82
+ */
83
+ export declare function currentRunStateFile(): string | undefined;
57
84
  export declare function registerSession(sessionId: string, stateFilePath: string): void;
58
85
  export declare function getSessionFilePath(sessionId: string): string | undefined;
59
86
  export declare function getRegisteredSessions(): ReadonlyMap<string, string>;
@@ -114,7 +141,12 @@ export interface UiAnalysisContext {
114
141
  * when it has its own pre-ranked GENERATE items). `generate` is the mandated,
115
142
  * final list generation tools gate against (see planGuard.ts); `additional` is
116
143
  * report-only; `demotions` carries discriminator claims that failed structural
117
- * verification (the candidate itself is never dropped, only the claim).
144
+ * verification — the demotion drops the CLAIM, never the candidate, but it does
145
+ * not keep the candidate either: budgeting can still cut it, so the same id may
146
+ * appear in both `demotions` and `dropped`;
147
+ * `dropped` carries candidates the selection stage removed outright from
148
+ * `generate`/`additional` (SKYR-4214) — the visible record of what got cut
149
+ * and why, since neither list says anything about what is missing from it.
118
150
  */
119
151
  export interface ApprovedPlan {
120
152
  planId: string;
@@ -125,6 +157,10 @@ export interface ApprovedPlan {
125
157
  candidateId: string;
126
158
  reason: string;
127
159
  }>;
160
+ dropped: Array<{
161
+ candidateId: string;
162
+ reason: string;
163
+ }>;
128
164
  }
129
165
  /**
130
166
  * Budget inputs computed once by skyramp_analyze_changes (from the same
@@ -142,6 +178,15 @@ export interface PlanBudgetContext {
142
178
  hasFrontendChanges: boolean;
143
179
  /** Serialized form of the BudgetContext.externalCoverage Set. */
144
180
  externalCoverageKeys: string[];
181
+ /** Whether the PR diff changes any test file; see BudgetContext. */
182
+ diffChangesTestFiles?: boolean;
183
+ }
184
+ export interface ReuseHandOff {
185
+ testType: string;
186
+ language: string;
187
+ /** The framework the hand-off named, so a refusal can hand back a call the
188
+ * reuse tool's schema accepts (`framework` is required there). */
189
+ framework: string;
145
190
  }
146
191
  /**
147
192
  * Unified state data combining test discovery + endpoint scanning
@@ -186,6 +231,19 @@ export interface UnifiedAnalysisState {
186
231
  * which computes every value in-process; skyramp_submit_report merges it into
187
232
  * the report's `reuse` field. Never supplied by the LLM. */
188
233
  reuseOutcomes?: Record<string, ReuseRecord>;
234
+ /** SKYR-4262. Assertion-enhancement obligations per spec, keyed by test-file
235
+ * BASENAME (same key as `reuseOutcomes`). Written in-process by
236
+ * skyramp_enhance_assertions when it hands out instructions; checked by
237
+ * skyramp_execute_test, which blocks when the spec is byte-identical to its
238
+ * state at handout time (the agent acknowledged the instructions but never
239
+ * acted on them). Never supplied by the LLM. */
240
+ enhanceOutcomes?: Record<string, import("../tools/code-refactor/enhance-state.js").EnhanceRecord>;
241
+ /** Generation hand-offs that owe a modularize→reuse pass, keyed by the ABSOLUTE
242
+ * PATH of each file the generation call wrote (found by snapshotting `outputDir`
243
+ * before and after codegen — the agent, not the tool, chooses the file name).
244
+ * Written by the generation service at the exact site that emits the hand-off, so
245
+ * the debt has the same server-derived trust as `reuseOutcomes`. */
246
+ reuseHandOffs?: Record<string, ReuseHandOff>;
189
247
  /**
190
248
  * SKYR-4156. Recorded video per executed browser test, keyed by test-file
191
249
  * BASENAME (the same key `reuseOutcomes` uses, so matching needs no path
@@ -319,6 +377,16 @@ export declare class StateManager<T = any> {
319
377
  getRepoRepositoryPath(repo?: string): Promise<string | undefined>;
320
378
  /** List the related repo keys present in the run-scoped file (empty if single-repo). */
321
379
  listRelatedRepos(): Promise<string[]>;
380
+ /**
381
+ * Every checkout this run has — the primary plus each related repo — each with
382
+ * the owner/repo it holds, for callers that resolve a path against "any repo in
383
+ * the run" rather than one named repo. `relatedRepos` is keyed by owner/repo, so
384
+ * the key IS that checkout's name; the primary's name comes from its metadata and
385
+ * can be absent. Reads the state file once, and drops both an absent path and the
386
+ * `"unknown"` placeholder metadata carries when the path was never resolved, so
387
+ * every returned entry is a path worth trying.
388
+ */
389
+ listRepoCheckouts(): Promise<RepoCheckout[]>;
322
390
  getStatePath(): string;
323
391
  getSessionId(): string;
324
392
  getStateType(): StateType;
@@ -19,6 +19,21 @@ let _testsRepoDir;
19
19
  export function setTestsRepoDir(dir) {
20
20
  _testsRepoDir = dir;
21
21
  }
22
+ /**
23
+ * Record the run's tests repo without ever clearing a known value.
24
+ *
25
+ * The tests repo is a RUN-scoped fact, but analyze_changes is called once per
26
+ * repository and only the PRIMARY call carries `testsRepoDir` — a related
27
+ * repo's call omits it. An unconditional set therefore wiped the primary's
28
+ * value on the second call of every multi-repo run, and by report time the
29
+ * tests repo was unknown: a UI test generated there looked unbacked by any
30
+ * working tree and the agent demoted it out of newTestsCreated (SKYR-4204,
31
+ * run 32536569467). Callers that mean to clear it use setTestsRepoDir.
32
+ */
33
+ export function rememberTestsRepoDir(dir) {
34
+ if (dir)
35
+ _testsRepoDir = dir;
36
+ }
22
37
  export function getTestsRepoDir() {
23
38
  return _testsRepoDir;
24
39
  }
@@ -98,6 +113,25 @@ export function resolveRunStatePath(explicitPath) {
98
113
  }
99
114
  return undefined;
100
115
  }
116
+ /**
117
+ * The run's state file if this process is actually serving a run, else undefined.
118
+ *
119
+ * Distinct from {@link resolveRunStatePath}, which answers "where would the file
120
+ * be": the fallback above builds a path from `$RUNNER_TEMP`, and GitHub Actions
121
+ * sets RUNNER_TEMP on every job — including this repo's own CI, where a resolvable
122
+ * path is not a run. Callers that gate BEHAVIOUR on being in a run must use this
123
+ * one; callers that only want to write state can keep using the path, because a
124
+ * write to a file that holds no run data is already a no-op.
125
+ *
126
+ * Returns the path rather than a boolean so a caller can also key per-run state on
127
+ * it, and lives here because this is the module that decides where the file goes.
128
+ */
129
+ export function currentRunStateFile() {
130
+ const stateFile = resolveRunStatePath();
131
+ return stateFile !== undefined && fs.existsSync(stateFile)
132
+ ? stateFile
133
+ : undefined;
134
+ }
101
135
  /**
102
136
  * In-memory session store: sessionId → { data, storedAt }.
103
137
  * Eliminates the need for the LLM to read/write state files on disk.
@@ -119,8 +153,7 @@ function evictStaleSessions() {
119
153
  }
120
154
  }
121
155
  if (inMemorySessionStore.size > MAX_SESSIONS) {
122
- const sorted = [...inMemorySessionStore.entries()]
123
- .sort((a, b) => a[1].storedAt - b[1].storedAt);
156
+ const sorted = [...inMemorySessionStore.entries()].sort((a, b) => a[1].storedAt - b[1].storedAt);
124
157
  const toDrop = sorted.slice(0, sorted.length - MAX_SESSIONS);
125
158
  for (const [id] of toDrop) {
126
159
  inMemorySessionStore.delete(id);
@@ -157,7 +190,9 @@ export function normalizeRecommendationState(data) {
157
190
  if (!data.analysis && data.apiEndpoints) {
158
191
  logger.info("Detected unwrapped RepositoryAnalysis — adapting");
159
192
  return {
160
- repositoryPath: data.repositoryPath || data.metadata?.repositoryName || "unknown",
193
+ repositoryPath: data.repositoryPath ||
194
+ data.metadata?.repositoryName ||
195
+ "unknown",
161
196
  analysisScope: data.analysisScope || data.metadata?.analysisScope,
162
197
  analysis: data,
163
198
  };
@@ -305,7 +340,9 @@ export class StateManager {
305
340
  },
306
341
  ...(existingRelated ? { relatedRepos: existingRelated } : {}),
307
342
  };
308
- await fs.promises.mkdir(path.dirname(this.stateFile), { recursive: true });
343
+ await fs.promises.mkdir(path.dirname(this.stateFile), {
344
+ recursive: true,
345
+ });
309
346
  await fs.promises.writeFile(this.stateFile, JSON.stringify(state, null, 2), "utf-8");
310
347
  logger.debug(`Wrote data to state file: ${this.stateFile}`);
311
348
  }
@@ -414,6 +451,31 @@ export class StateManager {
414
451
  const full = await this.readFullState();
415
452
  return Object.keys(full?.relatedRepos ?? {});
416
453
  }
454
+ /**
455
+ * Every checkout this run has — the primary plus each related repo — each with
456
+ * the owner/repo it holds, for callers that resolve a path against "any repo in
457
+ * the run" rather than one named repo. `relatedRepos` is keyed by owner/repo, so
458
+ * the key IS that checkout's name; the primary's name comes from its metadata and
459
+ * can be absent. Reads the state file once, and drops both an absent path and the
460
+ * `"unknown"` placeholder metadata carries when the path was never resolved, so
461
+ * every returned entry is a path worth trying.
462
+ */
463
+ async listRepoCheckouts() {
464
+ const full = await this.readFullState();
465
+ const checkouts = [
466
+ {
467
+ root: full?.metadata.repositoryPath,
468
+ repository: full?.metadata.repository,
469
+ primary: true,
470
+ },
471
+ ...Object.entries(full?.relatedRepos ?? {}).map(([repository, section]) => ({
472
+ root: section.repositoryPath,
473
+ repository,
474
+ primary: false,
475
+ })),
476
+ ];
477
+ return checkouts.flatMap((c) => c.root && c.root !== "unknown" ? [{ ...c, root: c.root }] : []);
478
+ }
417
479
  getStatePath() {
418
480
  return this.stateFile;
419
481
  }
@@ -457,7 +519,9 @@ export class StateManager {
457
519
  */
458
520
  static async cleanupOldFiles(maxAgeHours = 24, stateDir, stateTypes) {
459
521
  const baseDir = stateDir || runArtifactDir() || os.tmpdir();
460
- const files = await fs.promises.readdir(baseDir).catch(() => []);
522
+ const files = await fs.promises
523
+ .readdir(baseDir)
524
+ .catch(() => []);
461
525
  const statePrefixes = stateTypes
462
526
  ? stateTypes.map((t) => STATE_FILE_PREFIXES[t])
463
527
  : Object.values(STATE_FILE_PREFIXES);
@@ -15,6 +15,16 @@ export interface BranchDiffData {
15
15
  * Always uses the `b/` form so renames return the new path.
16
16
  */
17
17
  export declare function parseChangedFilesFromDiff(rawDiff: string): string[];
18
+ /**
19
+ * Splits a unified diff into per-file added-line arrays, keyed by the `b/`
20
+ * (post-change) path. Exported so other file->resource-affinity callers
21
+ * (e.g. `pathAffinityClassification.ts`'s fallback, SKYR-3857) can derive the
22
+ * same per-file diff hunks this module uses, without re-implementing the
23
+ * diff-header walk. Lives here rather than in `importerHop.ts`, its original
24
+ * home, so callers that `importerHop.ts` itself imports can reach it without
25
+ * an import cycle.
26
+ */
27
+ export declare function sliceAddedLinesByFile(diffContent: string): Map<string, string[]>;
18
28
  /**
19
29
  * Normalize and validate a caller-provided base branch/ref before passing it
20
30
  * to git. Reject option-like or ambiguous values so the ref is always treated
@@ -13,6 +13,34 @@ export function parseChangedFilesFromDiff(rawDiff) {
13
13
  }
14
14
  return out;
15
15
  }
16
+ /**
17
+ * Splits a unified diff into per-file added-line arrays, keyed by the `b/`
18
+ * (post-change) path. Exported so other file->resource-affinity callers
19
+ * (e.g. `pathAffinityClassification.ts`'s fallback, SKYR-3857) can derive the
20
+ * same per-file diff hunks this module uses, without re-implementing the
21
+ * diff-header walk. Lives here rather than in `importerHop.ts`, its original
22
+ * home, so callers that `importerHop.ts` itself imports can reach it without
23
+ * an import cycle.
24
+ */
25
+ export function sliceAddedLinesByFile(diffContent) {
26
+ const sections = new Map();
27
+ let currentFile = null;
28
+ for (const line of diffContent.split("\n")) {
29
+ const headerMatch = /^diff --git a\/.+ b\/(.+)$/.exec(line);
30
+ if (headerMatch) {
31
+ currentFile = headerMatch[1];
32
+ continue;
33
+ }
34
+ if (!currentFile)
35
+ continue;
36
+ if (line.startsWith("+") && !line.startsWith("+++")) {
37
+ const arr = sections.get(currentFile) ?? [];
38
+ arr.push(line);
39
+ sections.set(currentFile, arr);
40
+ }
41
+ }
42
+ return sections;
43
+ }
16
44
  /** Parse diff headers to find newly created and deleted files. */
17
45
  function isShaLikeRef(ref) {
18
46
  return /^[0-9a-f]{7,40}$/i.test(ref);