@skyramp/mcp 0.3.6-rc.2.ac20 → 0.3.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (103) hide show
  1. package/build/adapters/jestAdapter.js +0 -3
  2. package/build/adapters/mochaAdapter.js +0 -2
  3. package/build/adapters/playwrightAdapter.js +0 -3
  4. package/build/adapters/pytestAdapter.js +0 -12
  5. package/build/prompts/code-reuse.js +17 -2
  6. package/build/prompts/enhance-assertions/sharedAssertionRules.js +1 -1
  7. package/build/prompts/initialize-workspace/initializeWorkspacePrompt.js +2 -1
  8. package/build/prompts/modularization/ui-test-modularization.js +9 -6
  9. package/build/prompts/pom-aware-code-reuse.js +1 -1
  10. package/build/prompts/shared-helper-policy.js +5 -5
  11. package/build/prompts/testbot/testbot-prompts.js +1 -1
  12. package/build/services/TestGenerationService.js +28 -1
  13. package/build/tools/code-refactor/assertion-state.d.ts +91 -0
  14. package/build/tools/code-refactor/assertion-state.js +375 -0
  15. package/build/tools/code-refactor/codeReuseTool.js +6 -4
  16. package/build/tools/code-refactor/enhanceAssertionsTool.js +73 -18
  17. package/build/tools/code-refactor/retrofit-state.d.ts +53 -0
  18. package/build/tools/code-refactor/retrofit-state.js +162 -0
  19. package/build/tools/code-refactor/reuse-outcome.d.ts +7 -0
  20. package/build/tools/code-refactor/reuse-state.d.ts +9 -0
  21. package/build/tools/code-refactor/reuse-state.js +42 -4
  22. package/build/tools/code-refactor/utils-verify-gates.js +69 -15
  23. package/build/tools/executeSkyrampTestTool.js +19 -14
  24. package/build/tools/generate-tests/batchMockGenerationTool.js +25 -0
  25. package/build/tools/generateEnrichedIntegrationTestTool.js +10 -0
  26. package/build/tools/runExistingTestsTool.d.ts +2 -34
  27. package/build/tools/runExistingTestsTool.js +4 -104
  28. package/build/tools/submitReportTool.js +87 -134
  29. package/build/tools/workspace/initializeWorkspaceTool.js +99 -27
  30. package/build/types/AssertionOutcome.d.ts +68 -0
  31. package/build/types/AssertionOutcome.js +1 -0
  32. package/build/types/ExternalTestExecution.d.ts +1 -67
  33. package/build/types/ReuseOutcome.d.ts +16 -0
  34. package/build/types/TestTypes.d.ts +4 -0
  35. package/build/types/TestTypes.js +8 -0
  36. package/build/types/TestbotReport.d.ts +13 -0
  37. package/build/types/index.d.ts +1 -1
  38. package/build/utils/AnalysisStateManager.d.ts +20 -7
  39. package/build/utils/assertion-verify/api-shared-lints.d.ts +5 -0
  40. package/build/utils/assertion-verify/api-shared-lints.js +315 -0
  41. package/build/utils/assertion-verify/contract-lints.d.ts +3 -0
  42. package/build/utils/assertion-verify/contract-lints.js +87 -0
  43. package/build/utils/assertion-verify/format.d.ts +5 -0
  44. package/build/utils/assertion-verify/format.js +65 -0
  45. package/build/utils/assertion-verify/helper-imports.d.ts +6 -0
  46. package/build/utils/assertion-verify/helper-imports.js +178 -0
  47. package/build/utils/assertion-verify/index.d.ts +3 -0
  48. package/build/utils/assertion-verify/index.js +7 -0
  49. package/build/utils/assertion-verify/integration-lints.d.ts +3 -0
  50. package/build/utils/assertion-verify/integration-lints.js +36 -0
  51. package/build/utils/assertion-verify/js-regex-blank.d.ts +1 -0
  52. package/build/utils/assertion-verify/js-regex-blank.js +153 -0
  53. package/build/utils/assertion-verify/lint-types.d.ts +33 -0
  54. package/build/utils/assertion-verify/lint-types.js +57 -0
  55. package/build/utils/assertion-verify/marker.d.ts +27 -0
  56. package/build/utils/assertion-verify/marker.js +61 -0
  57. package/build/utils/assertion-verify/metrics.d.ts +30 -0
  58. package/build/utils/assertion-verify/metrics.js +341 -0
  59. package/build/utils/assertion-verify/python-strip.d.ts +6 -0
  60. package/build/utils/assertion-verify/python-strip.js +75 -0
  61. package/build/utils/assertion-verify/strip-dispatch.d.ts +19 -0
  62. package/build/utils/assertion-verify/strip-dispatch.js +42 -0
  63. package/build/utils/assertion-verify/ui-lints.d.ts +8 -0
  64. package/build/utils/assertion-verify/ui-lints.js +244 -0
  65. package/build/utils/assertion-verify/verify.d.ts +61 -0
  66. package/build/utils/assertion-verify/verify.js +215 -0
  67. package/build/utils/executorWorkDir.d.ts +36 -0
  68. package/build/utils/executorWorkDir.js +77 -0
  69. package/build/utils/featureFlags.d.ts +12 -2
  70. package/build/utils/featureFlags.js +33 -3
  71. package/build/utils/reportVerification.d.ts +4 -0
  72. package/build/utils/reportVerification.js +32 -4
  73. package/build/utils/utils-verify/allow.d.ts +22 -4
  74. package/build/utils/utils-verify/allow.js +8 -2
  75. package/build/utils/utils-verify/call-sites.d.ts +40 -1
  76. package/build/utils/utils-verify/call-sites.js +196 -30
  77. package/build/utils/utils-verify/importers.d.ts +31 -0
  78. package/build/utils/utils-verify/importers.js +78 -0
  79. package/build/utils/utils-verify/index.d.ts +1 -0
  80. package/build/utils/utils-verify/index.js +1 -0
  81. package/build/utils/utils-verify/language-spec.d.ts +13 -2
  82. package/build/utils/utils-verify/language-spec.js +12 -2
  83. package/build/utils/utils-verify/parse.d.ts +31 -3
  84. package/build/utils/utils-verify/parse.js +190 -9
  85. package/build/utils/utils-verify/retrofit-equivalence.d.ts +43 -0
  86. package/build/utils/utils-verify/retrofit-equivalence.js +218 -0
  87. package/build/utils/utils-verify/stage.d.ts +6 -0
  88. package/build/utils/utils-verify/stage.js +12 -2
  89. package/build/utils/utils-verify/verify.d.ts +54 -4
  90. package/build/utils/utils-verify/verify.js +224 -12
  91. package/node_modules/playwright/node_modules/playwright-core/lib/generated/injectedScriptSource.js +1 -1
  92. package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/assets/{codeMirrorModule-CZfp96qZ.js → codeMirrorModule-LNgEKtdV.js} +1 -1
  93. package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/assets/{defaultSettingsView-gpLo02E0.js → defaultSettingsView-Bwr1eMKC.js} +135 -135
  94. package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/{index.Bq1r1URj.js → index.-Id052Lr.js} +1 -1
  95. package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/index.html +2 -2
  96. package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/{uiMode.VEfqi1qN.js → uiMode.BPopbasy.js} +1 -1
  97. package/node_modules/playwright/node_modules/playwright-core/lib/vite/traceViewer/uiMode.html +2 -2
  98. package/node_modules/playwright/node_modules/playwright-core/package.json +1 -1
  99. package/node_modules/playwright/node_modules/playwright-core/src/generated/injectedScriptSource.ts +1 -1
  100. package/node_modules/playwright/package.json +1 -1
  101. package/package.json +2 -2
  102. package/build/tools/code-refactor/enhance-state.d.ts +0 -49
  103. package/build/tools/code-refactor/enhance-state.js +0 -109
@@ -440,95 +440,6 @@ const defaultDeps = {
440
440
  },
441
441
  now: () => new Date().toISOString(),
442
442
  };
443
- /** Worst outcome wins for a file. `error` (never reached its assertions) outranks
444
- * `fail` so a collection failure is never reported as an assertion failure, and
445
- * `skipped` sits below everything: it only wins when nothing else ran. */
446
- const STATUS_RANK = {
447
- skipped: 0,
448
- pass: 1,
449
- fail: 2,
450
- error: 3,
451
- };
452
- /**
453
- * Reduce a run's flat result list to one outcome per spec file.
454
- *
455
- * This is the only place that knows BOTH what the framework reported and what the
456
- * run asked it for, which is what `complete` needs. Deriving it later from the
457
- * results alone is not possible: a file re-run for one test looks identical to a
458
- * file whose other tests silently vanished.
459
- *
460
- * `partialFiles` are the files a `file::test title` selector narrowed to a subset of
461
- * their tests — the one narrowing this tool applies itself, and so the only one it
462
- * can state with certainty.
463
- */
464
- export function summarizeFilesInRun(results, opts) {
465
- const byFile = new Map();
466
- for (const r of results) {
467
- if (typeof r.file !== "string" || r.file === "")
468
- continue;
469
- const key = r.absoluteFile ?? r.file;
470
- byFile.set(key, [...(byFile.get(key) ?? []), r]);
471
- }
472
- return [...byFile.values()].map((group) => {
473
- const counts = { pass: 0, fail: 0, error: 0, skipped: 0 };
474
- for (const r of group)
475
- counts[r.status] = (counts[r.status] ?? 0) + 1;
476
- const status = group.reduce((a, b) => STATUS_RANK[b.status] > STATUS_RANK[a.status] ? b : a).status;
477
- const first = group[0];
478
- return {
479
- file: first.file,
480
- ...(first.absoluteFile ? { absoluteFile: first.absoluteFile } : {}),
481
- status,
482
- counts,
483
- complete: !isPartial(first, opts.partialFiles),
484
- errors: group
485
- .filter((r) => r.status === "fail" || r.status === "error")
486
- .map((r) => `${r.testId}: ${r.message ?? r.status}`),
487
- durationMs: group.reduce((sum, r) => sum + (r.durationMs ?? 0), 0),
488
- };
489
- });
490
- }
491
- /** A selector spelling and a result spelling need not agree — the selector is
492
- * repo-relative, the result may be absolute or framework-root-relative — so match
493
- * on a path-segment suffix in either direction. */
494
- function isPartial(r, partialFiles) {
495
- if (partialFiles.size === 0)
496
- return false;
497
- const candidates = [r.file, r.absoluteFile].filter((c) => typeof c === "string" && c !== "");
498
- for (const sel of partialFiles) {
499
- for (const c of candidates) {
500
- if (c === sel || c.endsWith("/" + sel) || sel.endsWith("/" + c))
501
- return true;
502
- }
503
- }
504
- return false;
505
- }
506
- /** Placeholder identity for a record that names no suite because none resolved.
507
- * Deliberately not a valid suiteIdOf output, so it can never collide with one. */
508
- export const NO_RUNNABLE_SUITE = "(no runnable suite)";
509
- /** Stable identity for ONE configured suite. A service can declare several
510
- * (workspace.yml testSuites), so its name alone collides — two suites under one
511
- * service would share a slot and the later would mask the other instead of both
512
- * counting. */
513
- export function suiteIdOf(cfg) {
514
- return `${cfg.serviceName}::${cfg.framework}::${cfg.testRunCommand}`;
515
- }
516
- /** The files a selector narrowed to individual tests. A bare `file` selector runs
517
- * the whole file; `file::test title` does not. */
518
- export function partialFilesFrom(selectors) {
519
- const out = new Set();
520
- for (const sel of selectors) {
521
- const [file, ...rest] = sel.split("::");
522
- // A trailing `::` names no title, so every builder drops it and the whole file
523
- // runs. Treating it as partial would throw away a real status.
524
- if (rest.join("::").trim() === "")
525
- continue;
526
- const trimmed = file.trim();
527
- if (trimmed)
528
- out.add(trimmed);
529
- }
530
- return out;
531
- }
532
443
  /** Appends a run record into UnifiedAnalysisState.externalTestResults for the
533
444
  * given repo section. Best-effort: a state failure never fails the run. */
534
445
  async function persistExternalRun(stateFile, repository, record) {
@@ -552,7 +463,7 @@ async function persistExternalRun(stateFile, repository, record) {
552
463
  * fully injectable for tests.
553
464
  */
554
465
  export async function runExternalTests(params, deps = defaultDeps) {
555
- const mode = params.mode;
466
+ const mode = params.mode ?? "confirm";
556
467
  const selectors = params.testSelectors ?? [];
557
468
  let effective = selectors;
558
469
  let truncated = false;
@@ -575,11 +486,6 @@ export async function runExternalTests(params, deps = defaultDeps) {
575
486
  if (params.stateFile) {
576
487
  await persistExternalRun(params.stateFile, params.repository, {
577
488
  ...skippedRun,
578
- files: [],
579
- // No run config resolved, so there is no suite to identify — and a service
580
- // name here would read as one. The reader never keys on it (the record is
581
- // skipped, and skipped records are dropped first), so name it for what it is.
582
- suite: NO_RUNNABLE_SUITE,
583
489
  mode,
584
490
  ranAt: deps.now(),
585
491
  repository: params.repository,
@@ -612,8 +518,6 @@ export async function runExternalTests(params, deps = defaultDeps) {
612
518
  if (params.stateFile) {
613
519
  await persistExternalRun(params.stateFile, params.repository, {
614
520
  ...routedOut,
615
- files: [],
616
- suite: suiteIdOf(cfg),
617
521
  mode,
618
522
  ranAt: deps.now(),
619
523
  repository: params.repository,
@@ -638,13 +542,8 @@ export async function runExternalTests(params, deps = defaultDeps) {
638
542
  };
639
543
  perSuite.push(suiteResult);
640
544
  if (params.stateFile) {
641
- const files = summarizeFilesInRun(suiteResult.results, {
642
- partialFiles: partialFilesFrom(routing.kept),
643
- });
644
545
  await persistExternalRun(params.stateFile, params.repository, {
645
546
  ...suiteResult,
646
- files,
647
- suite: suiteIdOf(cfg),
648
547
  mode,
649
548
  ranAt: deps.now(),
650
549
  repository: params.repository,
@@ -676,7 +575,8 @@ export function registerRunExistingTestsTool(server) {
676
575
  .describe("Spec files to run (paths relative to the run command's cwd), optionally `file::test title` to also filter by name. Omit to run the whole configured suite."),
677
576
  mode: z
678
577
  .enum(["confirm", "verify"])
679
- .describe("REQUIRED, and the only thing that marks a run as before or after the edit: confirm = the pre-edit baseline (a scoped run establishing which tests the change breaks; disables the suite's fail-fast cap, e.g. Playwright's maxFailures, so ALL breakages are counted), verify = the re-run after you edited. Passing confirm for a post-edit run overwrites the baseline with the fixed result, and the report then shows a repair as having never been broken."),
578
+ .optional()
579
+ .describe("confirm (default): scoped run to establish which tests a change breaks; disables the suite's fail-fast cap (e.g. Playwright's maxFailures) so ALL breakages are counted. verify: re-run after an edit."),
680
580
  project: z
681
581
  .string()
682
582
  .optional()
@@ -735,7 +635,7 @@ export function registerRunExistingTestsTool(server) {
735
635
  AnalyticsService.pushMCPToolEvent(TOOL_NAME, errorResult, {
736
636
  workspacePath: params.workspacePath,
737
637
  service: params.service ?? "",
738
- mode: params.mode,
638
+ mode: params.mode ?? "confirm",
739
639
  }).catch((err) => {
740
640
  logger.warning("Analytics event failed", { error: String(err) });
741
641
  });
@@ -12,9 +12,11 @@ import { StateManager, runArtifactDir, getTestsRepoDir, } from "../utils/Analysi
12
12
  import { toolError, testFileMatches } from "../utils/utils.js";
13
13
  import { matchesApprovedPlan } from "../utils/planMatchKeys.js";
14
14
  import { isTestbotEnabled } from "../utils/featureFlags.js";
15
- import { findInvalidSourceCitations, findUnchangedFileClaims, listChangedFiles, listChangedFilesAcross, } from "../utils/reportVerification.js";
15
+ import { findInvalidSourceCitations, findUnchangedFileClaims, listChangedFiles, listChangedFilesAcross, listChangedFilesAbs, } from "../utils/reportVerification.js";
16
16
  import { getReportLanguage, isEnforcedReportLanguage, findLanguageViolations, findLanguageNearMisses, reportLanguageDisplayName, } from "../utils/reportLanguage.js";
17
- import { rederiveReuseOutcome, reuseChainSkipped, } from "./code-refactor/reuse-state.js";
17
+ import { canonicalTestPath, findAssertionRecordByFileName, rederiveAssertionOutcome, } from "./code-refactor/assertion-state.js";
18
+ import { rederiveReuseOutcome, reuseChainSkipped, samePath, } from "./code-refactor/reuse-state.js";
19
+ import { retrofitGate, } from "./code-refactor/retrofit-state.js";
18
20
  // SKYR-3879 Path B: which testTypes the register-plan checkpoint gates. Mirrors
19
21
  // the generation tools actually wired to planGuard (batch-scenario/integration,
20
22
  // contract) — UI and E2E are on a separate blueprint-grounded pipeline and are
@@ -84,96 +86,6 @@ const MAINTENANCE_CHANGE_ACTIONS = new Set([
84
86
  DriftAction.Regenerate,
85
87
  DriftAction.Delete,
86
88
  ]);
87
- const EXTERNAL_TO_EXECUTION_STATUS = {
88
- pass: TestExecutionStatus.Pass,
89
- fail: TestExecutionStatus.Fail,
90
- error: TestExecutionStatus.Error,
91
- };
92
- /** Does this run's file outcome describe `knownPath`?
93
- *
94
- * `absoluteFile` is the framework's own path resolved against the root it reported,
95
- * so it names one file and is compared exactly. An outcome without one is refused:
96
- * a relative spelling is a suffix, and a suffix cannot separate
97
- * `packages/a/tests/x.spec.ts` from `packages/b/tests/x.spec.ts`. Guessing there and
98
- * guarding the guess is what this deliberately does not do — a missing status is a
99
- * row that reads Unknown, which is what it reads today; a wrong one is a false claim
100
- * about someone's tests.
101
- *
102
- * The cost is real and accepted: a suite running inside a container reports the
103
- * container's root, which cannot equal the host path discovery recorded, so those
104
- * repos get no status until the run records a root the report can line up. */
105
- function outcomeMatches(outcome, knownPath) {
106
- return (typeof outcome.absoluteFile === "string" &&
107
- path.resolve(outcome.absoluteFile) === path.resolve(knownPath));
108
- }
109
- /** One line describing what the suite saw, for a row whose status came from the run
110
- * rather than from an execution the agent drafted a summary for. Without it the
111
- * rendered cell is a bare `Error` or `Pass` with nothing behind it. */
112
- function describeOutcome(o) {
113
- // Defended like the record reads around it: a malformed outcome must not throw
114
- // out of the handler and take the whole report submission with it.
115
- const c = o.counts ?? { pass: 0, fail: 0, error: 0, skipped: 0 };
116
- const parts = [
117
- c.pass ? `${c.pass} passed` : "",
118
- c.fail ? `${c.fail} failed` : "",
119
- c.error ? `${c.error} errored` : "",
120
- c.skipped ? `${c.skipped} skipped` : "",
121
- ].filter(Boolean);
122
- const head = `${parts.join(", ") || "no tests reported"} in the repo's own suite`;
123
- // Adapter messages carry stack traces, so collapse whitespace before truncating:
124
- // an embedded newline would break the maintenance row this lands in.
125
- const first = (o.errors ?? [])[0]?.replace(/\s+/g, " ").trim();
126
- return first ? `${head} — ${first.slice(0, 200)}` : head;
127
- }
128
- function externalStatusesFor(records, knownPath) {
129
- // LATEST per suite, then WORST across suites. Both halves matter: a suite re-run
130
- // after another edit supersedes its own earlier verdict, so an obsolete failure
131
- // must not outlive the fix that repaired it; but two different suites can each
132
- // own the same file, and a pass in one must not mask a failure in the other.
133
- const RANK = { pass: 0, fail: 1, error: 2 };
134
- const statusIn = (mode) => {
135
- const latestPerSuite = new Map();
136
- for (const record of records ?? []) {
137
- if (record.mode !== mode || record.skipped || !record.environmentHealthy)
138
- continue;
139
- // A suite that stopped on --max-failures left tests unrun and does not say
140
- // which, so no file it touched can stand as covered. The adapter reports this;
141
- // it is not a guess. (The tool forces --max-failures=0 for confirm, so this is
142
- // reachable on a verify re-run under the repo's own cap.)
143
- if (record.diagnostics?.maxFailuresBail)
144
- continue;
145
- const hit = (record.files ?? []).find((o) => outcomeMatches(o, knownPath));
146
- if (!hit)
147
- continue;
148
- const key = record.suite ?? "";
149
- const prior = latestPerSuite.get(key);
150
- // `ranAt` is an ISO timestamp, so a lexical compare is chronological. The
151
- // newest run wins BEFORE it is judged: a later partial re-run supersedes an
152
- // earlier complete one, and must leave the row unknown rather than let the
153
- // stale complete outcome stand in for it.
154
- if (!prior || record.ranAt >= prior.at)
155
- latestPerSuite.set(key, { at: record.ranAt, outcome: hit });
156
- }
157
- let worst;
158
- for (const { outcome } of latestPerSuite.values()) {
159
- // `skipped` means every test in the file was skipped — it was not exercised,
160
- // and TestExecutionStatus.Skipped means a deliberate decision not to run it.
161
- if (!outcome.complete || outcome.status === "skipped")
162
- continue;
163
- if (!worst || RANK[outcome.status] > RANK[worst.status])
164
- worst = outcome;
165
- }
166
- return worst;
167
- };
168
- const before = statusIn("confirm");
169
- const after = statusIn("verify");
170
- return {
171
- before: before && EXTERNAL_TO_EXECUTION_STATUS[before.status],
172
- after: after && EXTERNAL_TO_EXECUTION_STATUS[after.status],
173
- beforeDetail: before && describeOutcome(before),
174
- afterDetail: after && describeOutcome(after),
175
- };
176
- }
177
89
  const TOOL_NAME = "skyramp_submit_report";
178
90
  const DEFAULT_COMMIT_MESSAGE = "Added recommendations by Skyramp Testbot.";
179
91
  // Per-repo attribution. In a multi-repo run, every report item carries the
@@ -643,14 +555,14 @@ const testMaintenanceSchema = z.object({
643
555
  "For passing runs: count and timing, e.g. '4 passed in 15.09s'. " +
644
556
  "For failing runs: failure name and one-line root cause, e.g. " +
645
557
  "'FAILED test_foo — assert 403 got 200, auth middleware not enforced'. " +
646
- "Pass an empty string when you ran nothing yourself — the server then fills this from the repo suite's own run — and for VERIFY/IGNORE entries where nothing ran at all. Omit the whole entry, never this field."),
558
+ "Empty string for VERIFY/IGNORE entries where no before-execution was run."),
647
559
  afterDetails: z
648
560
  .string()
649
561
  .describe("One line only — no embedded newlines, no raw HTTP headers or JSON blobs. " +
650
562
  "For passing runs: count and timing, e.g. '5 passed in 10.96s'. " +
651
563
  "For failing runs: failure name and one-line root cause, e.g. " +
652
564
  "'FAILED test_foo — check_schema fails, order_id=1 has discount from prior PATCH test'. " +
653
- "Pass an empty string when you ran nothing yourself — the server then fills this from the repo suite's own run — and for VERIFY/IGNORE/DELETE entries where nothing ran at all. Omit the whole entry, never this field."),
565
+ "Empty string for VERIFY/IGNORE/DELETE entries where no after-execution was run."),
654
566
  // Server-populated from stateFile execution records — never supplied by the LLM.
655
567
  beforeStatus: z.nativeEnum(TestExecutionStatus),
656
568
  afterStatus: z.nativeEnum(TestExecutionStatus),
@@ -731,7 +643,26 @@ function computeReportMetrics(params) {
731
643
  * gets `chainSkipped: true` (SKYR-4220), the one claim that is exactly about the
732
644
  * reuse tool NOT having run. Applies to every test type: UI tests carry the POM
733
645
  * fields, API tests the shared-helper ones. */
734
- async function attachReuseOutcome(test, outcomes, handOffs) {
646
+ /** Attach the assertion-enhancement summary to a report row.
647
+ *
648
+ * How the row finds its record: records are keyed by the test file's full
649
+ * path, but a report row only carries a file NAME. So the match compares
650
+ * basenames, then checks testType and repository. When more than one record
651
+ * still matches, the row gets NO summary — attaching the wrong spec's
652
+ * numbers is worse than attaching none.
653
+ *
654
+ * The numbers are server-derived and re-computed from the delivered file at
655
+ * report time — the report narrative is LLM-authored, these numbers are not.
656
+ * What a reader can conclude: `executionCount: 0` = generated but never
657
+ * executed; no `assertions` field at all = the enhance tool never ran for
658
+ * the file and it never executed. */
659
+ async function attachAssertionOutcome(test, outcomes, checkouts) {
660
+ const record = findAssertionRecordByFileName(outcomes, test, checkouts);
661
+ if (!record)
662
+ return test;
663
+ return { ...test, assertions: await rederiveAssertionOutcome(record) };
664
+ }
665
+ async function attachReuseOutcome(test, outcomes, handOffs, retrofits = []) {
735
666
  // POM records describe browser specs; a basename collision with an API test's
736
667
  // fileName must not attach them there. A utils-path record is attachable anywhere.
737
668
  const record = outcomes?.[path.basename(test.fileName)];
@@ -742,6 +673,18 @@ async function attachReuseOutcome(test, outcomes, handOffs) {
742
673
  ? record
743
674
  : undefined;
744
675
  const derived = found ? await rederiveReuseOutcome(found) : undefined;
676
+ // Pre-existing generated tests this spec's reuse pass rewired (SKYR-4276 A4): the
677
+ // report names each with its recorded execution, so a reviewer sees that the
678
+ // module became a dependency of code they already owned — and that it still runs.
679
+ if (derived?.helpers && found?.testFilePath) {
680
+ const specPath = found.testFilePath;
681
+ const mine = retrofits.filter((r) => samePath(r.testFile, specPath));
682
+ if (mine.length > 0)
683
+ derived.helpers.retrofits = mine.map((r) => ({
684
+ file: path.basename(r.file),
685
+ ...(r.execution ? { execution: r.execution } : {}),
686
+ }));
687
+ }
745
688
  // The RAW record: a colliding record of any kind means reuse ran for this basename,
746
689
  // and the chain claim must not be made on the filtered view.
747
690
  const chainSkipped = await reuseChainSkipped(test.fileName, test.testType, record, handOffs);
@@ -1101,42 +1044,12 @@ export function registerSubmitReportTool(server) {
1101
1044
  : MAINTENANCE_CHANGE_ACTIONS.has(m.action)
1102
1045
  ? TestExecutionStatus.Unknown
1103
1046
  : TestExecutionStatus.Skipped;
1104
- // A suite run answers only for a row whose default is Unknown. VERIFY and
1105
- // IGNORE are Skipped by decision, not for want of a result, and testbot
1106
- // drops both from the rendered table.
1107
- const external = externalStatusesFor(stateData.externalTestResults, m.testFilePath);
1108
- const beforeStatus = recorded?.executionBefore?.status ??
1109
- (defaultBeforeStatus === TestExecutionStatus.Unknown
1110
- ? external.before
1111
- : undefined) ??
1112
- defaultBeforeStatus;
1113
- const afterStatus = recorded?.executionAfter?.status ??
1114
- (defaultAfterStatus === TestExecutionStatus.Unknown
1115
- ? external.after
1116
- : undefined) ??
1117
- defaultAfterStatus;
1047
+ const beforeStatus = recorded?.executionBefore?.status ?? defaultBeforeStatus;
1048
+ const afterStatus = recorded?.executionAfter?.status ?? defaultAfterStatus;
1118
1049
  // Trim before checking — a whitespace-only string is semantically blank and
1119
1050
  // must not satisfy the "drafted a summary" requirement.
1120
- // The agent's drafted summary wins; otherwise a status that came from
1121
- // the suite run carries the run's own line, so the rendered cell is
1122
- // never a bare verdict with nothing behind it.
1123
- //
1124
- // Only when there is NO recorded execution. A file the agent ran itself
1125
- // owes a drafted summary, and filling it here would satisfy the check
1126
- // below on the agent's behalf whenever the two statuses happened to
1127
- // agree — silently dropping the requirement.
1128
- //
1129
- // And only where the STATUS came from the suite too. A VERIFY or DELETE
1130
- // row keeps its Skipped default, so a line saying "1 passed in the repo's
1131
- // own suite" beside it contradicts the very cell it annotates.
1132
- const beforeFromSuite = defaultBeforeStatus === TestExecutionStatus.Unknown &&
1133
- !recorded?.executionBefore;
1134
- const afterFromSuite = defaultAfterStatus === TestExecutionStatus.Unknown &&
1135
- !recorded?.executionAfter;
1136
- const beforeDetails = detail?.beforeDetails?.trim() ||
1137
- (beforeFromSuite ? (external.beforeDetail ?? "") : "");
1138
- const afterDetails = detail?.afterDetails?.trim() ||
1139
- (afterFromSuite ? (external.afterDetail ?? "") : "");
1051
+ const beforeDetails = detail?.beforeDetails?.trim() ?? "";
1052
+ const afterDetails = detail?.afterDetails?.trim() ?? "";
1140
1053
  if (recorded?.executionBefore && !beforeDetails)
1141
1054
  missingDetails.push(`${displayName} (beforeDetails)`);
1142
1055
  if (recorded?.executionAfter && !afterDetails)
@@ -1156,6 +1069,11 @@ export function registerSubmitReportTool(server) {
1156
1069
  "Add a testMaintenanceDetails entry with the missing field(s), drafted from the execution output you already saw.");
1157
1070
  return errorResult;
1158
1071
  }
1072
+ // SKYR-4276 A4: retrofits (pre-existing generated tests the reuse pass edited)
1073
+ // that still stand in the working tree, resolved once here and attached to the
1074
+ // rows below; an unexecuted one refuses the report (inside the SKYR-3883 block,
1075
+ // which already enumerates the working tree).
1076
+ let retrofitViews = [];
1159
1077
  // SKYR-3883: in a testbot run, refuse to ship a report that claims file work
1160
1078
  // the working tree doesn't reflect. The delivery step can only commit what the
1161
1079
  // agent actually created/edited, so a report claiming otherwise erodes trust
@@ -1216,6 +1134,17 @@ export function registerSubmitReportTool(server) {
1216
1134
  `Files with actual working-tree changes:\n${changedList || " (none)"}`);
1217
1135
  return errorResult;
1218
1136
  }
1137
+ const changedFilesAbs = await listChangedFilesAbs([
1138
+ repoRoot,
1139
+ getTestsRepoDir(),
1140
+ ...Object.values(fullState?.relatedRepos ?? {}).map((section) => section.repositoryPath),
1141
+ ]);
1142
+ const gate = await retrofitGate(stateData, changedFilesAbs, repoRoot);
1143
+ retrofitViews = gate.views;
1144
+ if (gate.refusal) {
1145
+ errorResult = toolError(gate.refusal);
1146
+ return errorResult;
1147
+ }
1219
1148
  }
1220
1149
  catch (err) {
1221
1150
  // Never block a valid report because verification itself failed (path not a
@@ -1253,7 +1182,16 @@ export function registerSubmitReportTool(server) {
1253
1182
  // generation — downstream scoring scripts don't expect them and fail if
1254
1183
  // they encounter these string fields while traversing the object.
1255
1184
  // Also normalize each item's `repository` (blank → undefined).
1256
- const sanitizedNewTests = await Promise.all(dedupedNewTests.map(({ scenarioFile: _sf, traceFile: _tf, frontendTrace: _ft, ...rest }) => attachReuseOutcome(normalizeRepository(rest), stateData.reuseOutcomes, stateData.reuseHandOffs)));
1185
+ // Checkout roots let the assertion-record matcher verify a candidate
1186
+ // record actually lives in the row's repo (basename collisions across
1187
+ // repos must not publish one spec's proof-of-work under another's name).
1188
+ const assertionCheckouts = await stateManager
1189
+ .listRepoCheckouts()
1190
+ .catch(() => []);
1191
+ const sanitizedNewTests = await Promise.all(dedupedNewTests.map(async ({ scenarioFile: _sf, traceFile: _tf, frontendTrace: _ft, ...rest }) => {
1192
+ const row = await attachReuseOutcome(normalizeRepository(rest), stateData.reuseOutcomes, stateData.reuseHandOffs, retrofitViews);
1193
+ return attachAssertionOutcome(row, stateData.assertionOutcomes ?? {}, assertionCheckouts);
1194
+ }));
1257
1195
  const report = {
1258
1196
  businessCaseAnalysis: params.businessCaseAnalysis,
1259
1197
  newTestsCreated: sanitizedNewTests,
@@ -1262,10 +1200,25 @@ export function registerSubmitReportTool(server) {
1262
1200
  // an internal-only field, needed for matching but never meant to reach the report.
1263
1201
  // TODO(multi-repo maintenance): map(normalizeRepository) once testMaintenanceSchema
1264
1202
  // has a repository field (see TODO above).
1265
- testMaintenance: testMaintenance?.map(({ testFilePath, ...row }) => ({
1266
- ...row,
1267
- fileName: path.basename(testFilePath),
1268
- })),
1203
+ testMaintenance: testMaintenance
1204
+ ? await Promise.all(testMaintenance.map(async ({ testFilePath, ...row }) => {
1205
+ // Maintenance rows still carry the absolute path here, so the
1206
+ // assertion summary uses an EXACT canonical-path lookup — no
1207
+ // basename ambiguity. This is what carries the maintenance
1208
+ // honesty labels (nothing-to-verify vs verified) into the
1209
+ // report instead of leaving them as tool text the agent can
1210
+ // paraphrase.
1211
+ const record = stateData.assertionOutcomes?.[canonicalTestPath(testFilePath)];
1212
+ const assertions = record
1213
+ ? await rederiveAssertionOutcome(record)
1214
+ : undefined;
1215
+ return {
1216
+ ...row,
1217
+ fileName: path.basename(testFilePath),
1218
+ ...(assertions ? { assertions } : {}),
1219
+ };
1220
+ }))
1221
+ : undefined,
1269
1222
  // videoPath is filled from the run's execution records; testFilePath is the
1270
1223
  // match-only key and is stripped from the wire format, the same line drawn for
1271
1224
  // testMaintenance's own testFilePath above (downstream scoring scripts traverse
@@ -1,12 +1,68 @@
1
1
  import { z } from "zod";
2
2
  import { WorkspaceConfigManager, serviceSchema, } from "../../workspace/workspace.js";
3
3
  import fs from "fs/promises";
4
+ import fsSync from "fs";
5
+ import path from "path";
4
6
  import yaml from "js-yaml";
5
7
  import { logger } from "../../utils/logger.js";
6
8
  import { AnalyticsService } from "../../services/AnalyticsService.js";
7
9
  import { SKYRAMP_IMAGE_VERSION } from "../../utils/versions.js";
8
10
  import { validateAndConsumeScanToken } from "./initScanWorkspaceTool.js";
9
11
  import { upsertServicesByRepo } from "./serviceUpsert.js";
12
+ import { isInsideDir } from "../../utils/gitStaging.js";
13
+ import { EXECUTOR_VIDEOS_DIR, EXECUTOR_WORK_DIR, isInsideExecutorWorkDir, } from "../../utils/executorWorkDir.js";
14
+ // Test delivery stages each service's whole testDirectory with
15
+ // `git add -- <testDirectory>`, so a testDirectory that encloses the executor's
16
+ // videos commits every .webm. This tool runs before any test is generated, so it
17
+ // is the only point where the choice can still be redirected.
18
+ /**
19
+ * First service whose `testDirectory` collides with the executor's working area,
20
+ * or undefined when none does. Two ways to collide, and both have to be rejected
21
+ * here: the directory sits inside `.skyramp` (which the generation tools refuse
22
+ * to write into, so accepting it would deadlock the agent between the two
23
+ * checks), or it encloses `.skyramp/videos` (`.`, "" or the repo root), which
24
+ * makes delivery commit every video. Relative paths resolve against
25
+ * `workspacePath`; an absolute one is matched on its own segments, so another
26
+ * repo's `.skyramp` is caught too.
27
+ *
28
+ * A directory that is itself a git repository root is rejected for the same
29
+ * reason, whichever repo it belongs to. The enclosure test above is anchored on
30
+ * the PRIMARY workspace's videos path, so a multi-repo entry naming another
31
+ * repo's root — `/work/repo-b` while workspacePath is `/work/repo-a` — slips
32
+ * past it: repo A's videos are not inside repo B. The executor writes
33
+ * `.skyramp/videos` at a repo root, so being one is the collision.
34
+ */
35
+ function badTestDirectoryResult(bad) {
36
+ return {
37
+ content: [
38
+ {
39
+ type: "text",
40
+ text: `Service "${bad.serviceName}" sets testDirectory to "${bad.testDirectory}", which collides with ${EXECUTOR_WORK_DIR}, the executor's working area: a directory inside it is never delivered, and one that encloses ${EXECUTOR_VIDEOS_DIR}/ makes delivery commit every execution video. Use a dedicated test path — the service's framework-configured test directory, or tests/skyramp. If this service came from the existing ${EXECUTOR_WORK_DIR}/workspace.yml, re-run with force: true and the full corrected services array.`,
41
+ },
42
+ ],
43
+ isError: true,
44
+ };
45
+ }
46
+ /** Whether `dir` is the top of a git checkout. `.git` is a directory in a normal
47
+ * clone and a file in a worktree, so existence is the test. */
48
+ function isGitRepositoryRoot(dir) {
49
+ return fsSync.existsSync(path.join(dir, ".git"));
50
+ }
51
+ function findBadTestDirectory(workspacePath, services) {
52
+ const workspaceVideos = path.resolve(workspacePath, EXECUTOR_VIDEOS_DIR);
53
+ for (const svc of services) {
54
+ const dir = svc.testDirectory;
55
+ if (dir === undefined)
56
+ continue;
57
+ const resolved = path.resolve(workspacePath, dir);
58
+ if (isInsideExecutorWorkDir(resolved) ||
59
+ isInsideDir(workspaceVideos, resolved) ||
60
+ isGitRepositoryRoot(resolved)) {
61
+ return { serviceName: svc.serviceName, testDirectory: dir };
62
+ }
63
+ }
64
+ return undefined;
65
+ }
10
66
  function getExecutorVersion() {
11
67
  return SKYRAMP_IMAGE_VERSION;
12
68
  }
@@ -92,6 +148,40 @@ export function registerInitializeWorkspaceTool(server) {
92
148
  isError: false,
93
149
  };
94
150
  }
151
+ if (!params.services || params.services.length === 0) {
152
+ return {
153
+ content: [
154
+ {
155
+ type: "text",
156
+ text: "No services provided. Follow the scanning instructions from skyramp_init_scan to discover services, then call this tool again.",
157
+ },
158
+ ],
159
+ isError: true,
160
+ };
161
+ }
162
+ // Both checks above and below are input-only, so they run before
163
+ // validateAndConsumeScanToken: that
164
+ // token is single-use, and a rejection after it would make the corrected
165
+ // retry this error asks for fail with "Invalid or expired scanToken". It
166
+ // is also before initialize()/updateMetadata(), which both write
167
+ // workspace.yml (workspace.ts:431, :460) — a later rejection would leave
168
+ // an initialized file behind and the retry would then short-circuit as
169
+ // "already initialized".
170
+ const badIncoming = findBadTestDirectory(workspacePath, params.services);
171
+ if (badIncoming)
172
+ return badTestDirectoryResult(badIncoming);
173
+ // Merge mode writes the MERGED set, so a bad testDirectory already in the
174
+ // file would be re-persisted. Checked on the merge, not on the existing
175
+ // services alone, so replacing a bad entry with a corrected one of the
176
+ // same identity is still the way out. Reading the file is side-effect
177
+ // free, so this rejection belongs up here with the other two rather than
178
+ // after the token is spent.
179
+ const mergeInto = params.merge && alreadyExists ? await manager.read() : undefined;
180
+ if (mergeInto) {
181
+ const badMerged = findBadTestDirectory(workspacePath, upsertServicesByRepo([...(mergeInto.services ?? [])], params.services));
182
+ if (badMerged)
183
+ return badTestDirectoryResult(badMerged);
184
+ }
95
185
  // scanToken is required for fresh init; for edits (force:true on an
96
186
  // existing workspace.yml) the caller supplies the full services array
97
187
  // directly and we skip the token check.
@@ -120,33 +210,15 @@ export function registerInitializeWorkspaceTool(server) {
120
210
  };
121
211
  }
122
212
  }
123
- if (!params.services || params.services.length === 0) {
124
- return {
125
- content: [
126
- {
127
- type: "text",
128
- text: "No services provided. Follow the scanning instructions from skyramp_init_scan to discover services, then call this tool again.",
129
- },
130
- ],
131
- isError: true,
132
- };
133
- }
134
- // Initialize (fresh) or read (merge into existing). Merge mode preserves
135
- // the existing primary repo identity + previously-registered services;
136
- // initialize() would reset both, dropping other repos' services.
137
- let config;
138
- if (params.merge && alreadyExists) {
139
- config = await manager.read();
140
- config = await manager.updateMetadata({
141
- executorVersion: getExecutorVersion(),
142
- });
143
- }
144
- else {
145
- config = await manager.initialize();
146
- config = await manager.updateMetadata({
147
- executorVersion: getExecutorVersion(),
148
- });
149
- }
213
+ // Merge mode skips initialize(): it preserves the existing primary repo
214
+ // identity + previously-registered services, both of which initialize()
215
+ // would reset, dropping other repos' services. The file itself was
216
+ // already read above, for the merged-set check.
217
+ if (!mergeInto)
218
+ await manager.initialize();
219
+ const config = await manager.updateMetadata({
220
+ executorVersion: getExecutorVersion(),
221
+ });
150
222
  // Batch all service upserts: update in memory, write once (direct YAML
151
223
  // write avoids N+1 reads and a mid-loop validation error). Upsert is keyed
152
224
  // on the composite (repo, serviceName) — see upsertServicesByRepo — so a