@skyramp/mcp 0.3.6-rc.2.ac20 → 0.3.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (103) hide show
  1. package/build/adapters/jestAdapter.js +0 -3
  2. package/build/adapters/mochaAdapter.js +0 -2
  3. package/build/adapters/playwrightAdapter.js +0 -3
  4. package/build/adapters/pytestAdapter.js +0 -12
  5. package/build/prompts/code-reuse.js +17 -2
  6. package/build/prompts/enhance-assertions/sharedAssertionRules.js +1 -1
  7. package/build/prompts/initialize-workspace/initializeWorkspacePrompt.js +2 -1
  8. package/build/prompts/modularization/ui-test-modularization.js +9 -6
  9. package/build/prompts/pom-aware-code-reuse.js +1 -1
  10. package/build/prompts/shared-helper-policy.js +5 -5
  11. package/build/prompts/testbot/testbot-prompts.js +1 -1
  12. package/build/services/TestGenerationService.js +28 -1
  13. package/build/tools/code-refactor/assertion-state.d.ts +91 -0
  14. package/build/tools/code-refactor/assertion-state.js +375 -0
  15. package/build/tools/code-refactor/codeReuseTool.js +6 -4
  16. package/build/tools/code-refactor/enhanceAssertionsTool.js +73 -18
  17. package/build/tools/code-refactor/retrofit-state.d.ts +53 -0
  18. package/build/tools/code-refactor/retrofit-state.js +162 -0
  19. package/build/tools/code-refactor/reuse-outcome.d.ts +7 -0
  20. package/build/tools/code-refactor/reuse-state.d.ts +9 -0
  21. package/build/tools/code-refactor/reuse-state.js +42 -4
  22. package/build/tools/code-refactor/utils-verify-gates.js +69 -15
  23. package/build/tools/executeSkyrampTestTool.js +19 -14
  24. package/build/tools/generate-tests/batchMockGenerationTool.js +25 -0
  25. package/build/tools/generateEnrichedIntegrationTestTool.js +10 -0
  26. package/build/tools/runExistingTestsTool.d.ts +2 -34
  27. package/build/tools/runExistingTestsTool.js +4 -104
  28. package/build/tools/submitReportTool.js +87 -134
  29. package/build/tools/workspace/initializeWorkspaceTool.js +99 -27
  30. package/build/types/AssertionOutcome.d.ts +68 -0
  31. package/build/types/AssertionOutcome.js +1 -0
  32. package/build/types/ExternalTestExecution.d.ts +1 -67
  33. package/build/types/ReuseOutcome.d.ts +16 -0
  34. package/build/types/TestTypes.d.ts +4 -0
  35. package/build/types/TestTypes.js +8 -0
  36. package/build/types/TestbotReport.d.ts +13 -0
  37. package/build/types/index.d.ts +1 -1
  38. package/build/utils/AnalysisStateManager.d.ts +20 -7
  39. package/build/utils/assertion-verify/api-shared-lints.d.ts +5 -0
  40. package/build/utils/assertion-verify/api-shared-lints.js +315 -0
  41. package/build/utils/assertion-verify/contract-lints.d.ts +3 -0
  42. package/build/utils/assertion-verify/contract-lints.js +87 -0
  43. package/build/utils/assertion-verify/format.d.ts +5 -0
  44. package/build/utils/assertion-verify/format.js +65 -0
  45. package/build/utils/assertion-verify/helper-imports.d.ts +6 -0
  46. package/build/utils/assertion-verify/helper-imports.js +178 -0
  47. package/build/utils/assertion-verify/index.d.ts +3 -0
  48. package/build/utils/assertion-verify/index.js +7 -0
  49. package/build/utils/assertion-verify/integration-lints.d.ts +3 -0
  50. package/build/utils/assertion-verify/integration-lints.js +36 -0
  51. package/build/utils/assertion-verify/js-regex-blank.d.ts +1 -0
  52. package/build/utils/assertion-verify/js-regex-blank.js +153 -0
  53. package/build/utils/assertion-verify/lint-types.d.ts +33 -0
  54. package/build/utils/assertion-verify/lint-types.js +57 -0
  55. package/build/utils/assertion-verify/marker.d.ts +27 -0
  56. package/build/utils/assertion-verify/marker.js +61 -0
  57. package/build/utils/assertion-verify/metrics.d.ts +30 -0
  58. package/build/utils/assertion-verify/metrics.js +341 -0
  59. package/build/utils/assertion-verify/python-strip.d.ts +6 -0
  60. package/build/utils/assertion-verify/python-strip.js +75 -0
  61. package/build/utils/assertion-verify/strip-dispatch.d.ts +19 -0
  62. package/build/utils/assertion-verify/strip-dispatch.js +42 -0
  63. package/build/utils/assertion-verify/ui-lints.d.ts +8 -0
  64. package/build/utils/assertion-verify/ui-lints.js +244 -0
  65. package/build/utils/assertion-verify/verify.d.ts +61 -0
  66. package/build/utils/assertion-verify/verify.js +215 -0
  67. package/build/utils/executorWorkDir.d.ts +36 -0
  68. package/build/utils/executorWorkDir.js +77 -0
  69. package/build/utils/featureFlags.d.ts +12 -2
  70. package/build/utils/featureFlags.js +33 -3
  71. package/build/utils/reportVerification.d.ts +4 -0
  72. package/build/utils/reportVerification.js +32 -4
  73. package/build/utils/utils-verify/allow.d.ts +22 -4
  74. package/build/utils/utils-verify/allow.js +8 -2
  75. package/build/utils/utils-verify/call-sites.d.ts +40 -1
  76. package/build/utils/utils-verify/call-sites.js +196 -30
  77. package/build/utils/utils-verify/importers.d.ts +31 -0
  78. package/build/utils/utils-verify/importers.js +78 -0
  79. package/build/utils/utils-verify/index.d.ts +1 -0
  80. package/build/utils/utils-verify/index.js +1 -0
  81. package/build/utils/utils-verify/language-spec.d.ts +13 -2
  82. package/build/utils/utils-verify/language-spec.js +12 -2
  83. package/build/utils/utils-verify/parse.d.ts +31 -3
  84. package/build/utils/utils-verify/parse.js +190 -9
  85. package/build/utils/utils-verify/retrofit-equivalence.d.ts +43 -0
  86. package/build/utils/utils-verify/retrofit-equivalence.js +218 -0
  87. package/build/utils/utils-verify/stage.d.ts +6 -0
  88. package/build/utils/utils-verify/stage.js +12 -2
  89. package/build/utils/utils-verify/verify.d.ts +54 -4
  90. package/build/utils/utils-verify/verify.js +224 -12
  91. package/node_modules/playwright/node_modules/playwright-core/lib/generated/injectedScriptSource.js +1 -1
  92. package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/assets/{codeMirrorModule-CZfp96qZ.js → codeMirrorModule-LNgEKtdV.js} +1 -1
  93. package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/assets/{defaultSettingsView-gpLo02E0.js → defaultSettingsView-Bwr1eMKC.js} +135 -135
  94. package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/{index.Bq1r1URj.js → index.-Id052Lr.js} +1 -1
  95. package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/index.html +2 -2
  96. package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/{uiMode.VEfqi1qN.js → uiMode.BPopbasy.js} +1 -1
  97. package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/uiMode.html +2 -2
  98. package/node_modules/playwright/node_modules/playwright-core/package.json +1 -1
  99. package/node_modules/playwright/node_modules/playwright-core/src/generated/injectedScriptSource.ts +1 -1
  100. package/node_modules/playwright/package.json +1 -1
  101. package/package.json +2 -2
  102. package/build/tools/code-refactor/enhance-state.d.ts +0 -49
  103. package/build/tools/code-refactor/enhance-state.js +0 -109
@@ -0,0 +1,244 @@
1
+ import { escapeRegExp } from "../regex.js";
2
+ import { strippedSources } from "./strip-dispatch.js";
3
+ import { balancedCloseIndex, identifierRe, inScope, interpolationRe, lineOfOffset, } from "./lint-types.js";
4
+ /**
5
+ * Deterministic UI (Playwright) lints — the machine-checkable slice of the D5
6
+ * rubric's UI dims:
7
+ * critical_assertions → expect import source, pageerror listener asserted,
8
+ * dead response captures, repeated elements w/o count
9
+ * computed_values → tautologies here; exact-vs-visibility is enforced by
10
+ * the strength score + weak-additions gate in verify.ts
11
+ * post_edit → unasserted trailing state-changing action
12
+ */
13
+ const PLAYWRIGHT_EXPECT_IMPORT_RES = [
14
+ /import\s*(?:type\s*)?\{([^}]*)\}\s*from\s*['"]@playwright\/test['"]/g,
15
+ /(?:const|let|var)\s*\{([^}]*)\}\s*=\s*require\s*\(\s*['"]@playwright\/test['"]\s*\)/g,
16
+ ];
17
+ /** `expect` imported from @playwright/test — banned flatly by the UI rules. */
18
+ function lintPlaywrightExpectImport(commentless) {
19
+ for (const re of PLAYWRIGHT_EXPECT_IMPORT_RES) {
20
+ re.lastIndex = 0;
21
+ let m;
22
+ while ((m = re.exec(commentless)) !== null) {
23
+ if (/\bexpect\b/.test(m[1])) {
24
+ return [
25
+ {
26
+ rule: "expect-import-source",
27
+ severity: "hard",
28
+ line: lineOfOffset(commentless, m.index),
29
+ message: "`expect` is imported from '@playwright/test'.",
30
+ remediation: "Import `expect` from '@skyramp/skyramp' instead (keep `test` on the playwright line).",
31
+ },
32
+ ];
33
+ }
34
+ }
35
+ }
36
+ return [];
37
+ }
38
+ // Any page-ish handle counts: specs keep a raw handle (`rawPage`) for
39
+ // assertions because expect() rejects the SkyrampPage proxy.
40
+ const PAGEERROR_LISTENER_RE = /[\w$]*[Pp]age\s*\.\s*on\s*\(\s*['"]pageerror['"]/;
41
+ const ERRORS_LENGTH_ASSERT_RES = [
42
+ /toHaveLength\s*\(\s*0\s*\)/,
43
+ // `(?:,[^)]*)?` admits expect's optional message argument.
44
+ /\.length\s*(?:,[^)]*)?\)\s*\.\s*(?:toBe|toEqual|toStrictEqual)\s*\(\s*0\s*\)/,
45
+ /(?:toEqual|toStrictEqual)\s*\(\s*\[\s*\]\s*\)/,
46
+ ];
47
+ /** pageerror listener registered but its errors array never asserted (hard);
48
+ * listener absent entirely stays warn-only — helpers/fixtures may register it,
49
+ * and forcing a structural insertion conflicts with autoApply's
50
+ * only-add-assertion-lines constraint. Promotion pending run evidence. */
51
+ function lintPageErrors(commentless) {
52
+ const listener = PAGEERROR_LISTENER_RE.exec(commentless);
53
+ if (!listener) {
54
+ return [
55
+ {
56
+ rule: "pageerror-listener-absent",
57
+ severity: "warn",
58
+ message: "No `page.on('pageerror', ...)` listener found in this file.",
59
+ remediation: "Register the listener before the first navigation and assert the collected errors array is empty at the end of the test.",
60
+ },
61
+ ];
62
+ }
63
+ if (ERRORS_LENGTH_ASSERT_RES.some((re) => re.test(commentless)))
64
+ return [];
65
+ return [
66
+ {
67
+ rule: "pageerror-unasserted",
68
+ severity: "hard",
69
+ line: lineOfOffset(commentless, listener.index),
70
+ message: "A `page.on('pageerror', ...)` listener collects errors, but the errors array is never asserted.",
71
+ remediation: "Add `expect(errors).toHaveLength(0)` (or equivalent) at the end of the test.",
72
+ },
73
+ ];
74
+ }
75
+ // Codegen emits BOTH shapes: direct (`const r = await page.waitForResponse(…)`)
76
+ // and promise-then-await (`const p = page.waitForResponse(…); …; const r = await p;`).
77
+ // Matching only the direct form made this lint a no-op on real codegen output.
78
+ const DIRECT_RESPONSE_BINDING_RE = /(?:const|let|var)\s+([A-Za-z_$][\w$]*)\s*=\s*await\s+[\w$]*[Pp]age\s*\.\s*waitForResponse\s*\(/g;
79
+ const PROMISE_BINDING_RE = /(?:const|let|var)\s+([A-Za-z_$][\w$]*)\s*=\s*[\w$]*[Pp]age\s*\.\s*waitForResponse\s*\(/g;
80
+ /** A captured `waitForResponse` response that is never referenced again.
81
+ * Takes the raw comment-stripped source too: template-literal interpolations
82
+ * are blanked in `stripped`, so a use like `` `order-${resp.headers()...}` ``
83
+ * is only visible there.
84
+ *
85
+ * WARN, not hard — measured against skyramp's gen_reference UI goldens,
86
+ * pristine codegen emits 3–9 unused `const rN = await rPromiseN` captures per
87
+ * file (63 across the corpus): many are synchronization-only waits whose
88
+ * bodies aren't meaningfully assertable, and a hard finding per capture would
89
+ * dominate every UI verify loop with no marker escape. The rubric pressure to
90
+ * assert captured responses stays via this warn plus the strength gate. */
91
+ function lintDeadResponseCaptures(stripped, commentless) {
92
+ const findings = [];
93
+ // Response bindings from the direct shape…
94
+ const bindings = [];
95
+ let m;
96
+ DIRECT_RESPONSE_BINDING_RE.lastIndex = 0;
97
+ while ((m = DIRECT_RESPONSE_BINDING_RE.exec(stripped)) !== null) {
98
+ const close = balancedCloseIndex(stripped, m.index + m[0].length - 1);
99
+ bindings.push({
100
+ varName: m[1],
101
+ index: m.index,
102
+ end: close === -1 ? m.index + m[0].length : close + 1,
103
+ });
104
+ }
105
+ // …plus the promise-then-await shape: resolve `const r = await <promiseVar>`
106
+ // for every promise binding, and treat a promise that is never referenced at
107
+ // all as a dead capture itself.
108
+ // (The promise pattern cannot match the direct form — `await ` between `=`
109
+ // and the page handle breaks its contiguous match — so no overlap handling.)
110
+ PROMISE_BINDING_RE.lastIndex = 0;
111
+ while ((m = PROMISE_BINDING_RE.exec(stripped)) !== null) {
112
+ const promiseVar = m[1];
113
+ const close = balancedCloseIndex(stripped, m.index + m[0].length - 1);
114
+ const restStart = close === -1 ? m.index + m[0].length : close + 1;
115
+ const rest = stripped.slice(restStart);
116
+ const awaited = new RegExp(`(?:const|let|var)\\s+([A-Za-z_$][\\w$]*)\\s*=\\s*await\\s+${escapeRegExp(promiseVar)}(?![\\w$])`).exec(rest);
117
+ if (awaited) {
118
+ bindings.push({
119
+ varName: awaited[1],
120
+ index: restStart + awaited.index,
121
+ end: restStart + awaited.index + awaited[0].length,
122
+ });
123
+ }
124
+ else if (!identifierRe(promiseVar).test(rest)) {
125
+ bindings.push({ varName: promiseVar, index: m.index, end: restStart });
126
+ }
127
+ }
128
+ for (const { varName, index, end } of bindings) {
129
+ const rest = stripped.slice(end);
130
+ const usedInInterpolation = interpolationRe(varName, "typescript").test(commentless.slice(end));
131
+ // Identifier-boundary lookarounds, not \b: `\b` breaks on `$`-prefixed
132
+ // names ($resp), silently turning this into a guaranteed false positive.
133
+ if (!identifierRe(varName).test(rest) && !usedInInterpolation) {
134
+ findings.push({
135
+ rule: "dead-response-capture",
136
+ severity: "warn",
137
+ line: lineOfOffset(stripped, index),
138
+ message: `\`${varName}\` captures a network response via waitForResponse but is never used.`,
139
+ remediation: `Assert at least one status, body, or header field on \`${varName}\` (e.g. \`expect(${varName}.status()).toBe(200)\`).`,
140
+ });
141
+ }
142
+ }
143
+ return findings;
144
+ }
145
+ // getByText('X') asserted toHaveText('X') with the identical literal. Restricted
146
+ // to getByText — getByRole+name asserted with the same text is a sanctioned
147
+ // pattern in the rules' own examples. Misses chains with links in between
148
+ // (.first() etc.) by design: a miss is fine, a false positive is not.
149
+ // The literal groups exclude `\` from the any-char alternative so the two
150
+ // alternatives are disjoint — the ambiguous form backtracks exponentially on
151
+ // escape-heavy literals (ReDoS that would hang the whole MCP server).
152
+ const TAUTOLOGY_RE = /(?:getByText|get_by_text)\s*\(\s*(['"`])((?:\\.|(?!\1)[^\\\n])*)\1[^)]*\)\s*\)?\s*\.\s*(?:toHaveText|toContainText|to_have_text|to_contain_text)\s*\(\s*(['"`])((?:\\.|(?!\3)[^\\\n])*)\3/g;
153
+ function lintTautologies(commentless, opts) {
154
+ const findings = [];
155
+ let m;
156
+ TAUTOLOGY_RE.lastIndex = 0;
157
+ while ((m = TAUTOLOGY_RE.exec(commentless)) !== null) {
158
+ const line = lineOfOffset(commentless, m.index);
159
+ if (m[2] === m[4] && inScope(line, opts)) {
160
+ findings.push({
161
+ rule: "tautological-assertion",
162
+ severity: "hard",
163
+ line,
164
+ message: `Tautology: element located by text ${JSON.stringify(m[2])} is asserted to have that same text.`,
165
+ remediation: "Assert a different knowable property (count, value, attribute), or locate the element by a structural selector and keep the text assertion.",
166
+ });
167
+ }
168
+ }
169
+ return findings;
170
+ }
171
+ // Signals that the page renders repeated elements (rows, list items, nth/first
172
+ // access), which the rubric expects paired with an exact toHaveCount(N).
173
+ const REPEATED_ELEMENT_RE = /getByRole\s*\(\s*['"](?:row|listitem|cell|option)['"]|\.\s*nth\s*\(|\.\s*first\s*\(\s*\)|\.\s*last\s*\(\s*\)/;
174
+ const HAS_COUNT_RE = /toHaveCount\s*\(|to_have_count\s*\(/;
175
+ /** Repeated/collection elements are exercised but no exact count is asserted.
176
+ * Warn-only: whether a list truly renders is not knowable from the spec alone
177
+ * (the D5 judge has a matching N/A condition for the same reason). */
178
+ function lintRepeatedElementsWithoutCount(commentless) {
179
+ const repeated = REPEATED_ELEMENT_RE.exec(commentless);
180
+ if (!repeated || HAS_COUNT_RE.test(commentless))
181
+ return [];
182
+ return [
183
+ {
184
+ rule: "repeated-elements-without-count",
185
+ severity: "warn",
186
+ line: lineOfOffset(commentless, repeated.index),
187
+ message: "The test works with repeated/collection elements but never asserts an exact count.",
188
+ remediation: "Add `toHaveCount(N)` (N from the trace or rendered DOM) plus at least one per-item content assertion.",
189
+ },
190
+ ];
191
+ }
192
+ // State-changing interactions; a test whose LAST such action is never followed
193
+ // by an assertion leaves the post-action state unverified (post_edit dim).
194
+ const ACTION_RE = /\.\s*(?:click|fill|press|check|uncheck|selectOption|select_option|set_input_files|setInputFiles)\s*\(/g;
195
+ const EXPECT_SITE_RE = /\bexpect(?:\s*\.\s*soft)?\s*\(/g;
196
+ function lintUnassertedTrailingAction(stripped) {
197
+ let lastAction = -1;
198
+ let m;
199
+ ACTION_RE.lastIndex = 0;
200
+ while ((m = ACTION_RE.exec(stripped)) !== null)
201
+ lastAction = m.index;
202
+ if (lastAction === -1)
203
+ return [];
204
+ let lastExpect = -1;
205
+ EXPECT_SITE_RE.lastIndex = 0;
206
+ while ((m = EXPECT_SITE_RE.exec(stripped)) !== null)
207
+ lastExpect = m.index;
208
+ if (lastExpect > lastAction)
209
+ return [];
210
+ return [
211
+ {
212
+ rule: "unasserted-trailing-action",
213
+ severity: "warn",
214
+ line: lineOfOffset(stripped, lastAction),
215
+ message: "The test's final state-changing action is never followed by an assertion — the updated state goes unverified.",
216
+ remediation: "After the last action, assert the visible updated outcome (`toHaveText`/`toHaveValue`/`toBeChecked`/`toHaveCount`) reflecting the edit.",
217
+ },
218
+ ];
219
+ }
220
+ /** UI (Playwright) lints. Python (playwright-python) UI specs deliberately get
221
+ * the site-scoped lints only: the structural rules' patterns and accept-lists
222
+ * are JS/TS-specific (imports, waitForResponse shapes, expect matcher accept
223
+ * forms), and porting them without a python golden corpus to validate against
224
+ * would recreate the false-block class review round 2 removed. Pinned by test. */
225
+ export function lintUiSpec(raw, language, opts) {
226
+ if (language === "java")
227
+ return [];
228
+ const maintenance = opts?.scopeLines !== undefined;
229
+ const { commentless, stripped } = strippedSources(raw, language);
230
+ const findings = [];
231
+ if (language !== "python" && !maintenance) {
232
+ // Import lint is generation-only too: maintenance updates existing customer
233
+ // specs, which legitimately import expect from @playwright/test (the repo
234
+ // may not even depend on @skyramp/skyramp), and the maintenance scope
235
+ // forbids touching imports.
236
+ findings.push(...lintPlaywrightExpectImport(commentless));
237
+ findings.push(...lintPageErrors(commentless));
238
+ findings.push(...lintDeadResponseCaptures(stripped, commentless));
239
+ findings.push(...lintRepeatedElementsWithoutCount(commentless));
240
+ findings.push(...lintUnassertedTrailingAction(stripped));
241
+ }
242
+ findings.push(...lintTautologies(commentless, opts));
243
+ return findings;
244
+ }
@@ -0,0 +1,61 @@
1
+ import { type AssertionMetrics } from "./metrics.js";
2
+ import type { LintFinding } from "./lint-types.js";
3
+ export type AssertionEnhanceType = "generation" | "maintenance";
4
+ /** Snapshot taken when the enhancement instructions were handed out. */
5
+ export interface AssertionBaseline {
6
+ fileSha256: string;
7
+ count?: number;
8
+ strengthScore?: number;
9
+ fingerprints?: string[];
10
+ /** Copy of the generated file at hand-out time (run artifact dir). */
11
+ baselineFilePath?: string;
12
+ }
13
+ export interface AssertionVerifyResult {
14
+ ok: boolean;
15
+ enhanceType: AssertionEnhanceType;
16
+ /** False when verifying stateless (no baseline) — differential gates skipped. */
17
+ baselinePresent: boolean;
18
+ /** Maintenance verified with an empty assertion diff — nothing was added,
19
+ * changed, or removed, so there was nothing to verify. */
20
+ maintenanceNoAssertionChanges?: boolean;
21
+ hashUnchanged: boolean;
22
+ /** Baseline assertions whose SUBJECT lost all coverage — genuine removals
23
+ * (sample of 5). Replaced assertions (subject still asserted) never appear. */
24
+ removedFingerprints?: string[];
25
+ /** Total genuine removals — lets the report say "showing 5 of N" so the
26
+ * agent restores everything in ONE pass instead of discovering the rest
27
+ * on the next verify round. */
28
+ removedCount?: number;
29
+ /** Baseline assertions superseded by a new assertion on the same subject. */
30
+ replacedCount?: number;
31
+ /** Baseline assertions whose subject now lives in a locally imported helper
32
+ * file (modularization) — sanctioned, never counted as removed. */
33
+ movedToHelperCount?: number;
34
+ /** Copy of the generated file at hand-out time, for diff/restore remediation. */
35
+ baselineFilePath?: string;
36
+ /** Current − baseline strength score; undefined without a baseline. */
37
+ strengthDelta?: number;
38
+ strengthGateFailed: boolean;
39
+ /** Assertions were added, but every one is existence/visibility-tier — the
40
+ * computed-values rubric wants exact matchers. Runs in generation AND
41
+ * maintenance (it requires additions to exist, so value-only maintenance
42
+ * fixes cannot trip it). Marker-clearable, like the strength gate. */
43
+ weakAdditionsOnly: boolean;
44
+ hardFindings: LintFinding[];
45
+ warnings: LintFinding[];
46
+ markerReason?: string;
47
+ currentMetrics?: AssertionMetrics;
48
+ }
49
+ export declare function sha256Of(content: string): string;
50
+ /**
51
+ * Deterministic verification that the enhancement instructions were applied:
52
+ * hash delta, no assertions removed, rubric lints clean, and (generation only)
53
+ * strength strictly increased unless a documented `assertions complete` marker
54
+ * declines it. Reads the file as delivered — throws if it cannot be read.
55
+ */
56
+ export declare function verifyAssertionEnhancement(params: {
57
+ testFile: string;
58
+ testType: string;
59
+ enhanceType: AssertionEnhanceType;
60
+ baseline?: AssertionBaseline;
61
+ }): Promise<AssertionVerifyResult>;
@@ -0,0 +1,215 @@
1
+ import { createHash } from "crypto";
2
+ import { readFile } from "fs/promises";
3
+ import { computeAssertionMetrics, detectAssertionLanguage, subjectOf, } from "./metrics.js";
4
+ import { lintUiSpec } from "./ui-lints.js";
5
+ import { lintIntegrationTest } from "./integration-lints.js";
6
+ import { lintContractTest } from "./contract-lints.js";
7
+ import { findAssertionsCompleteMarker, findUnparsableAssertionsComplete, markerCommentToken, } from "./marker.js";
8
+ import { importedHelperSubjects } from "./helper-imports.js";
9
+ export function sha256Of(content) {
10
+ return createHash("sha256").update(content).digest("hex");
11
+ }
12
+ /** Current sites whose fingerprint occurs more often now than in the baseline
13
+ * multiset — the assertions ADDED since hand-out (includes the new side of a
14
+ * replacement). Powers maintenance lint scoping and the weak-additions gate. */
15
+ function addedSites(current, baselineFingerprints) {
16
+ const budget = new Map();
17
+ for (const fp of baselineFingerprints)
18
+ budget.set(fp, (budget.get(fp) ?? 0) + 1);
19
+ const added = [];
20
+ for (const site of current.sites) {
21
+ const left = budget.get(site.fingerprint) ?? 0;
22
+ if (left > 0)
23
+ budget.set(site.fingerprint, left - 1);
24
+ else
25
+ added.push(site);
26
+ }
27
+ return added;
28
+ }
29
+ /**
30
+ * Deterministic verification that the enhancement instructions were applied:
31
+ * hash delta, no assertions removed, rubric lints clean, and (generation only)
32
+ * strength strictly increased unless a documented `assertions complete` marker
33
+ * declines it. Reads the file as delivered — throws if it cannot be read.
34
+ */
35
+ export async function verifyAssertionEnhancement(params) {
36
+ const { testFile, testType, enhanceType, baseline } = params;
37
+ const content = await readFile(testFile, "utf8");
38
+ const language = detectAssertionLanguage(testFile);
39
+ const hashUnchanged = baseline !== undefined && sha256Of(content) === baseline.fileSha256;
40
+ // Unknown language: only the hash gate is computable — everything else fails open.
41
+ if (language === undefined) {
42
+ return {
43
+ ok: !(hashUnchanged && enhanceType === "generation"),
44
+ enhanceType,
45
+ baselinePresent: baseline !== undefined,
46
+ hashUnchanged,
47
+ strengthGateFailed: false,
48
+ weakAdditionsOnly: false,
49
+ hardFindings: [],
50
+ warnings: [],
51
+ };
52
+ }
53
+ const metrics = computeAssertionMetrics(content, language);
54
+ const countDecreasedBy = baseline?.count !== undefined
55
+ ? Math.max(0, baseline.count - metrics.count)
56
+ : 0;
57
+ // Correlate what disappeared since the baseline. A missing fingerprint whose
58
+ // SUBJECT is still asserted (by any current site) was REPLACED — a sanctioned
59
+ // enhancement (usually weak matcher → strong). A subject asserted inside a
60
+ // locally imported helper file was MOVED (modularization) — also sanctioned.
61
+ // A KNOWN subject that lost all coverage counts as removed regardless of the
62
+ // raw count (a same-count remove-A-add-B edit must not slip through);
63
+ // unrecoverable subjects stay conservative and count only when the raw count
64
+ // actually decreased.
65
+ let removedFingerprints;
66
+ let removedCount;
67
+ let replacedCount;
68
+ let movedToHelperCount;
69
+ if (baseline?.fingerprints !== undefined) {
70
+ const current = new Map();
71
+ for (const fp of metrics.fingerprints) {
72
+ current.set(fp, (current.get(fp) ?? 0) + 1);
73
+ }
74
+ const missing = baseline.fingerprints.filter((fp) => {
75
+ const left = current.get(fp) ?? 0;
76
+ if (left > 0) {
77
+ current.set(fp, left - 1);
78
+ return false;
79
+ }
80
+ return true;
81
+ });
82
+ const currentSubjects = new Set(metrics.fingerprints.map(subjectOf).filter((s) => s !== undefined));
83
+ let removed = missing.filter((fp) => {
84
+ const subject = subjectOf(fp);
85
+ if (subject === undefined)
86
+ return countDecreasedBy > 0;
87
+ return !currentSubjects.has(subject);
88
+ });
89
+ // Only pay the helper-file scan when something actually looks removed.
90
+ if (removed.length > 0) {
91
+ const helperSubjects = await importedHelperSubjects(testFile, content);
92
+ if (helperSubjects.size > 0) {
93
+ const stillRemoved = removed.filter((fp) => {
94
+ const subject = subjectOf(fp);
95
+ return subject === undefined || !helperSubjects.has(subject);
96
+ });
97
+ movedToHelperCount = removed.length - stillRemoved.length;
98
+ removed = stillRemoved;
99
+ }
100
+ }
101
+ replacedCount = missing.length - removed.length - (movedToHelperCount ?? 0);
102
+ if (removed.length > 0) {
103
+ removedFingerprints = removed.slice(0, 5);
104
+ removedCount = removed.length;
105
+ }
106
+ }
107
+ const strengthDelta = baseline?.strengthScore !== undefined
108
+ ? metrics.strengthScore - baseline.strengthScore
109
+ : undefined;
110
+ const added = baseline?.fingerprints !== undefined
111
+ ? addedSites(metrics, baseline.fingerprints)
112
+ : undefined;
113
+ // Maintenance scopes site lints to assertions added since the baseline; the
114
+ // baseline fingerprints are what make that diff computable.
115
+ const scopeLines = enhanceType === "maintenance" && added !== undefined
116
+ ? new Set(added.map((s) => s.line))
117
+ : undefined;
118
+ const lintOpts = scopeLines !== undefined ? { scopeLines } : undefined;
119
+ // Per-test-type lints; Java is never linted or strength-gated — count+hash
120
+ // only. Test types with no lint module (smoke/fuzz/load/e2e — the hand-out
121
+ // path refuses them) get NO lints rather than silently inheriting
122
+ // integration's: those rules were not written for them.
123
+ const findings = testType === "ui"
124
+ ? lintUiSpec(content, language, lintOpts)
125
+ : testType === "contract"
126
+ ? lintContractTest(content, language, lintOpts)
127
+ : testType === "integration"
128
+ ? lintIntegrationTest(content, language, lintOpts)
129
+ : [];
130
+ const marker = findAssertionsCompleteMarker(content, testFile, language);
131
+ const hardFindings = findings.filter((f) => f.severity === "hard");
132
+ const commentToken = markerCommentToken(language);
133
+ for (const line of findUnparsableAssertionsComplete(content, language)) {
134
+ hardFindings.push({
135
+ rule: "malformed-assertions-complete-marker",
136
+ severity: "hard",
137
+ message: `Unreadable marker: \`${line}\``,
138
+ remediation: `Use \`${commentToken} assertions complete: <test-file basename> — <reason>\` (this language's comment token; em dash, or a hyphen with a space on BOTH sides), with a real reason.`,
139
+ });
140
+ }
141
+ // Removals are NEVER marker-clearable: a marker that excused them would let
142
+ // "delete the failing assertion + one comment line" verify green — the exact
143
+ // case this gate exists to catch. The sanctioned coverage-moves (verify
144
+ // before modularization per the mandated steps; passed-verdict skip at
145
+ // execute time) don't need the exemption, and a post-modularize failure is
146
+ // always fixable by re-adding direct assertions beside the helper calls.
147
+ // (Derived, not stored: removedFingerprints !== undefined IS the gate.)
148
+ const removalGateFailed = removedFingerprints !== undefined;
149
+ // A decline marker is accepted on a GENERATION file only when work
150
+ // actually happened (something added or replaced). Without this, a
151
+ // marker-only comment edit was a full bypass: the comment changes the hash,
152
+ // the marker cleared both strength-family gates, and an untouched file
153
+ // trips nothing else — the exact "tests already have the key assertions"
154
+ // no-op the gate exists to catch (review finding on ae8b397e). The marker
155
+ // can excuse INSUFFICIENT strengthening ("only existence is knowable"); it
156
+ // cannot excuse ABSENT strengthening. Maintenance keeps marker semantics
157
+ // unchanged (its strength gates don't run; value-only fixes are legitimate).
158
+ const markerAccepted = marker !== undefined &&
159
+ (enhanceType !== "generation" ||
160
+ (added?.length ?? 0) > 0 ||
161
+ (replacedCount ?? 0) > 0);
162
+ const strengthGateFailed = enhanceType === "generation" &&
163
+ language !== "java" &&
164
+ strengthDelta !== undefined &&
165
+ strengthDelta <= 0 &&
166
+ !markerAccepted;
167
+ // Computed-values rubric: additions must not be exclusively existence/
168
+ // visibility-tier. A replacement's new side counts as an added site, so a
169
+ // weak→strong upgrade satisfies this. Marker-clearable (strength family).
170
+ // Unlike the strength-increase gate this runs in MAINTENANCE too: it is
171
+ // conditional on additions existing, so a value-only drift fix (no added
172
+ // sites) can never trip it — only genuinely weak new assertions do.
173
+ const weakAdditionsOnly = language !== "java" &&
174
+ added !== undefined &&
175
+ added.length > 0 &&
176
+ added.every((s) => s.weight === 1) &&
177
+ !markerAccepted;
178
+ // The hash gate ("byte-identical → nothing applied") is generation-only: in
179
+ // maintenance the file edits precede the enhance call, and "no assertion
180
+ // changes needed" is a legitimate outcome — the assertion-diff below is what
181
+ // decides whether there is anything to verify at all.
182
+ const hashGateFailed = hashUnchanged && enhanceType === "generation";
183
+ // Maintenance with an EMPTY assertion diff (nothing added, changed, or
184
+ // removed since hand-out): there is no assertion work to verify — pass, and
185
+ // let the report say so honestly.
186
+ const maintenanceNoAssertionChanges = enhanceType === "maintenance" &&
187
+ baseline?.fingerprints !== undefined &&
188
+ (added?.length ?? 0) === 0 &&
189
+ replacedCount === 0 &&
190
+ (removedCount ?? 0) === 0 &&
191
+ (movedToHelperCount ?? 0) === 0;
192
+ return {
193
+ ok: !hashGateFailed &&
194
+ !removalGateFailed &&
195
+ hardFindings.length === 0 &&
196
+ !strengthGateFailed &&
197
+ !weakAdditionsOnly,
198
+ maintenanceNoAssertionChanges,
199
+ enhanceType,
200
+ baselinePresent: baseline !== undefined,
201
+ hashUnchanged,
202
+ removedFingerprints,
203
+ removedCount,
204
+ replacedCount,
205
+ movedToHelperCount,
206
+ baselineFilePath: baseline?.baselineFilePath,
207
+ strengthDelta,
208
+ strengthGateFailed,
209
+ weakAdditionsOnly,
210
+ hardFindings,
211
+ warnings: findings.filter((f) => f.severity === "warn"),
212
+ markerReason: markerAccepted ? marker?.reason : undefined,
213
+ currentMetrics: metrics,
214
+ };
215
+ }
@@ -0,0 +1,36 @@
1
+ /**
2
+ * The executor's own working area inside a repo — run videos, trace zips and
3
+ * other per-run artefacts. Delivered test files do not belong here: the run
4
+ * cleans and rewrites it, and `git add -- <testDirectory>` on a directory that
5
+ * encloses it commits every .webm.
6
+ */
7
+ export declare const EXECUTOR_WORK_DIR = ".skyramp";
8
+ export declare const EXECUTOR_VIDEOS_DIR = ".skyramp/videos";
9
+ /**
10
+ * Whether `p` is the executor's working area or sits inside it. Matches on whole
11
+ * path segments, so a sibling such as `.skyramp-tests` is not a hit, and on the
12
+ * canonical path, so a symlink pointing into `.skyramp` cannot spell its way past
13
+ * the check.
14
+ */
15
+ export declare function isInsideExecutorWorkDir(p: string): boolean;
16
+ /**
17
+ * Refusal text for a caller that asked for output inside the working area. Names
18
+ * the field and the replacement so the agent can correct itself without help.
19
+ */
20
+ export declare function executorWorkDirRefusal(destination: string): string;
21
+ /**
22
+ * Every path a generation call could write to, given an outputDir and an optional
23
+ * caller-supplied file name. `output` can carry `../` segments, so outputDir alone
24
+ * is not enough to judge.
25
+ *
26
+ * Two compositions, because the writers disagree on an ABSOLUTE `output`:
27
+ * `generateEnrichedIntegrationTestTool` joins it (`path.join(outputDir, output)`,
28
+ * which keeps outputDir as a prefix), while `path.resolve` discards outputDir
29
+ * entirely. A guard cannot pick one and be right for both, so it judges both and
30
+ * refuses if either lands in the working area.
31
+ *
32
+ * outputDir is always judged on its own as well. An `output` of `../tests/x.py`
33
+ * resolves clear of the working area while outputDir stays `.skyramp`, and
34
+ * outputDir is what codegen receives as the directory to work in.
35
+ */
36
+ export declare function generationTargets(outputDir: string, output?: string): string[];
@@ -0,0 +1,77 @@
1
+ import fs from "fs";
2
+ import path from "path";
3
+ /**
4
+ * The executor's own working area inside a repo — run videos, trace zips and
5
+ * other per-run artefacts. Delivered test files do not belong here: the run
6
+ * cleans and rewrites it, and `git add -- <testDirectory>` on a directory that
7
+ * encloses it commits every .webm.
8
+ */
9
+ export const EXECUTOR_WORK_DIR = ".skyramp";
10
+ export const EXECUTOR_VIDEOS_DIR = `${EXECUTOR_WORK_DIR}/videos`;
11
+ /**
12
+ * Canonical form of `p`, where `p` may not exist yet: the deepest ancestor that
13
+ * DOES exist is resolved through symlinks and the missing remainder is appended
14
+ * back.
15
+ *
16
+ * `utils-verify/locate.ts` (realpath) and `code-refactor/reuse-state.ts` (canon)
17
+ * both fall back to the lexical path when the target is missing. A generation
18
+ * outputDir usually IS missing, and the lexical path is what a symlinked parent
19
+ * hides behind — so neither is usable here.
20
+ */
21
+ function realpathAllowingMissing(p) {
22
+ let current = path.resolve(p);
23
+ const missing = [];
24
+ for (;;) {
25
+ try {
26
+ return path.join(fs.realpathSync(current), ...[...missing].reverse());
27
+ }
28
+ catch {
29
+ const parent = path.dirname(current);
30
+ if (parent === current)
31
+ return path.resolve(p);
32
+ missing.push(path.basename(current));
33
+ current = parent;
34
+ }
35
+ }
36
+ }
37
+ /**
38
+ * Whether `p` is the executor's working area or sits inside it. Matches on whole
39
+ * path segments, so a sibling such as `.skyramp-tests` is not a hit, and on the
40
+ * canonical path, so a symlink pointing into `.skyramp` cannot spell its way past
41
+ * the check.
42
+ */
43
+ export function isInsideExecutorWorkDir(p) {
44
+ return realpathAllowingMissing(p).split(path.sep).includes(EXECUTOR_WORK_DIR);
45
+ }
46
+ /**
47
+ * Refusal text for a caller that asked for output inside the working area. Names
48
+ * the field and the replacement so the agent can correct itself without help.
49
+ */
50
+ export function executorWorkDirRefusal(destination) {
51
+ return `Refusing to write into "${destination}": it is inside ${EXECUTOR_WORK_DIR}, the Skyramp executor's working area (run videos and trace zips). Files written there are not delivered as tests. Set outputDir — and any output file name — so the result lands in the testDirectory of the service under test, read from ${EXECUTOR_WORK_DIR}/workspace.yml, or tests/skyramp when that service declares none. If that testDirectory is itself inside ${EXECUTOR_WORK_DIR}, it is wrong: re-run skyramp_init_workspace with a corrected one.`;
52
+ }
53
+ /**
54
+ * Every path a generation call could write to, given an outputDir and an optional
55
+ * caller-supplied file name. `output` can carry `../` segments, so outputDir alone
56
+ * is not enough to judge.
57
+ *
58
+ * Two compositions, because the writers disagree on an ABSOLUTE `output`:
59
+ * `generateEnrichedIntegrationTestTool` joins it (`path.join(outputDir, output)`,
60
+ * which keeps outputDir as a prefix), while `path.resolve` discards outputDir
61
+ * entirely. A guard cannot pick one and be right for both, so it judges both and
62
+ * refuses if either lands in the working area.
63
+ *
64
+ * outputDir is always judged on its own as well. An `output` of `../tests/x.py`
65
+ * resolves clear of the working area while outputDir stays `.skyramp`, and
66
+ * outputDir is what codegen receives as the directory to work in.
67
+ */
68
+ export function generationTargets(outputDir, output) {
69
+ const dir = path.resolve(outputDir);
70
+ if (!output)
71
+ return [dir];
72
+ return [
73
+ dir,
74
+ path.resolve(outputDir, output),
75
+ path.resolve(path.join(outputDir, output)),
76
+ ];
77
+ }
@@ -4,7 +4,8 @@
4
4
  * Each helper in this module reads a `SKYRAMP_FEATURE_*` env var and returns
5
5
  * a boolean. These helpers treat a flag as ON only when the env var is
6
6
  * exactly `"1"`; anything else (unset, empty, "0", "true", "false", etc.) is
7
- * treated as OFF.
7
+ * treated as OFF. The utils-reuse flag is additionally ON for the orgs in
8
+ * `UTILS_REUSE_DEFAULT_ORGS` regardless of the env vars.
8
9
  */
9
10
  /**
10
11
  * Gates BOTH consumer-side contract tests AND the SDK's "default mode"
@@ -55,6 +56,14 @@ export declare function isOneClickEnabled(): boolean;
55
56
  * Gates the POM-aware path of `skyramp_reuse_code`
56
57
  * (SKYRAMP_FEATURE_POM_REUSE=1).
57
58
  *
59
+ * Deliberately NOT defaulted on for `UTILS_REUSE_DEFAULT_ORGS`: for UI tests
60
+ * the POM path supersedes the modularize-first flow (see
61
+ * `isModularizeFirstTarget` in utils/reuseRouting.ts), so an org-level POM
62
+ * default would stop those orgs' UI tests from ever seeding a SkyrampUtils
63
+ * file — the very thing the utils org default exists to provide (SKYR-4290).
64
+ * And an explicit-"0"-override escape hatch cannot work either, because
65
+ * testbot normalizes an unset flag to a literal "0" before forwarding.
66
+ *
58
67
  * When OFF (default), TypeScript/JavaScript + Playwright targets are treated
59
68
  * like every other target: `skyramp_reuse_code` returns the SkyrampUtils
60
69
  * consolidation prompt, and none of the POM machinery runs — no deterministic
@@ -68,7 +77,8 @@ export declare function isOneClickEnabled(): boolean;
68
77
  export declare function isPomReuseEnabled(): boolean;
69
78
  /**
70
79
  * Gates SkyrampUtils code reuse for integration tests on the testbot path, and
71
- * for UI tests whenever the POM path is off (SKYRAMP_FEATURE_UTILS_REUSE=1).
80
+ * for UI tests whenever the POM path is off (SKYRAMP_FEATURE_UTILS_REUSE=1,
81
+ * or a `UTILS_REUSE_DEFAULT_ORGS` org).
72
82
  *
73
83
  * When OFF (default), the testbot prompt does not set `codeReuse` on
74
84
  * `skyramp_integration_test_generation` and its post-generation