@skyramp/mcp 0.3.6-rc.1 → 0.3.6-rc.2.ac20

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,3 +1,4 @@
1
+ import * as path from "path";
1
2
  /**
2
3
  * Shared jest + vitest adapter for skyramp_run_existing_tests. vitest's json
3
4
  * reporter is jest-compatible, so both use ONE parser; only the reporter flag
@@ -40,6 +41,7 @@ export function parseJestJson(report, opts) {
40
41
  results.push({
41
42
  testId: `${name} › (file failed to run)`,
42
43
  file: name,
44
+ ...(path.isAbsolute(name) ? { absoluteFile: name } : {}),
43
45
  status: "error",
44
46
  message: stripVTControlCharacters(raw).trim() || undefined,
45
47
  durationMs: 0,
@@ -55,6 +57,7 @@ export function parseJestJson(report, opts) {
55
57
  results.push({
56
58
  testId: `${name} › ${titlePath}`,
57
59
  file: name,
60
+ ...(path.isAbsolute(name) ? { absoluteFile: name } : {}),
58
61
  status,
59
62
  message,
60
63
  durationMs: Math.round(a.duration ?? 0),
@@ -1,3 +1,4 @@
1
+ import * as path from "path";
1
2
  /**
2
3
  * Mocha adapter for skyramp_run_existing_tests. Normalizes `mocha --reporter
3
4
  * json` (emitted to stdout) into the neutral shape. Mocha does not tag
@@ -29,6 +30,7 @@ function toResult(t, status) {
29
30
  return {
30
31
  testId: `${file} › ${full}`,
31
32
  file,
33
+ ...(path.isAbsolute(file) ? { absoluteFile: file } : {}),
32
34
  status,
33
35
  message: status === "fail" || status === "error" ? messageOf(t) : undefined,
34
36
  durationMs: Math.round(t.duration ?? 0),
@@ -12,6 +12,7 @@
12
12
  * builds the flags appended to the suite's `testRunCommand`. The IO shell that
13
13
  * actually spawns the run lives in `runExistingTestsTool`.
14
14
  */
15
+ import * as path from "path";
15
16
  import { stripVTControlCharacters } from "util";
16
17
  /**
17
18
  * A Playwright project (or spec file) is "infra" when it exists to bring up /
@@ -67,6 +68,7 @@ function messageOf(test) {
67
68
  }
68
69
  export function parsePlaywrightJson(report, opts) {
69
70
  const rep = (report ?? {});
71
+ const rootDir = rep.config?.rootDir;
70
72
  const collected = [];
71
73
  for (const fileSuite of rep.suites ?? []) {
72
74
  collectSpecs(fileSuite, fileSuite.file ?? "", [], collected);
@@ -118,6 +120,7 @@ export function parsePlaywrightJson(report, opts) {
118
120
  results.push({
119
121
  testId,
120
122
  file,
123
+ ...(rootDir ? { absoluteFile: path.resolve(rootDir, file) } : {}),
121
124
  status,
122
125
  message: status === "fail" || status === "error" ? messageOf(test) : undefined,
123
126
  durationMs,
@@ -1,3 +1,4 @@
1
+ import * as path from "path";
1
2
  /**
2
3
  * pytest adapter for skyramp_run_existing_tests. Normalizes the
3
4
  * `pytest-json-report` (`--json-report`) output into the neutral result shape.
@@ -39,6 +40,8 @@ function durationMsOf(test) {
39
40
  export function parsePytestJson(report, opts) {
40
41
  const rep = (report ?? {});
41
42
  const tests = rep.tests ?? [];
43
+ const root = rep.root;
44
+ const abs = (f) => (root ? { absoluteFile: path.resolve(root, f) } : {});
42
45
  // Real test outcomes.
43
46
  const testResults = tests.map((t) => {
44
47
  const status = mapOutcome(t.outcome);
@@ -46,6 +49,7 @@ export function parsePytestJson(report, opts) {
46
49
  return {
47
50
  testId: nodeid,
48
51
  file: fileOf(nodeid),
52
+ ...abs(fileOf(nodeid)),
49
53
  status,
50
54
  message: status === "fail" || status === "error" ? messageOf(t) : undefined,
51
55
  durationMs: durationMsOf(t),
@@ -61,6 +65,7 @@ export function parsePytestJson(report, opts) {
61
65
  return {
62
66
  testId: `${nodeid} › (collection error)`,
63
67
  file: fileOf(nodeid),
68
+ ...abs(fileOf(nodeid)),
64
69
  status: "error",
65
70
  message: lastLine(c.longrepr),
66
71
  durationMs: 0,
@@ -72,6 +77,13 @@ export function parsePytestJson(report, opts) {
72
77
  // When collection failed AND no real test ran, the suite as a whole could not be
73
78
  // collected — environmental (a wall of red), not PR signal. When some real tests
74
79
  // DID run, the collection errors above stand as ordinary PR-signal error results.
80
+ // NOTE: a collection error names the file that failed to import, but says nothing
81
+ // about WHY. A missing settings module, a database that is down or any shared
82
+ // dependency failure surfaces per-module, landing on exactly the files the run
83
+ // selected — indistinguishable here from a file the PR itself broke. So a run that
84
+ // collected nothing stays unhealthy even when scoped, and the file keeps its
85
+ // Unknown baseline. Treating it as PR signal would let an environment outage be
86
+ // reported as a test the change broke, and then repaired.
75
87
  if (collectionErrors.length > 0 && testResults.length === 0) {
76
88
  environmentHealthy = false;
77
89
  const first = collectionErrors[0];
@@ -212,7 +212,7 @@ ${maintenanceBeforeExecStep}
212
212
 
213
213
  e. Call \`skyramp_actions\` with \`stateFile\` (from \`skyramp_analyze_changes\` output) and apply the edits it returns.
214
214
 
215
- f. Verify external-test fixes. **This step is not optional and it is the easiest one to forget — you have just edited files in step 2(e), so come back here before you move on to anything else.** It applies whenever step 2(a) reported a real pass/fail result for a file you then edited. It does NOT apply when step 2(a) returned \`skipped: true\` or \`ran: 0\` for every suite — there is no baseline to compare against, so say so in your report instead of re-running. When it applies: re-run those \`[external]\` files with \`skyramp_run_existing_tests\` (\`mode: "verify"\`, \`stateFile\`) and record each file's result as its \`afterStatus\`. Editing an \`[external]\` file that step 2(a) confirmed failing and NOT re-running it leaves your own fix unverified — you would be reporting a repair you never saw work. A still-failing verify is surfaced in the report — do not loop.
215
+ f. Verify external-test fixes. **This step is not optional and it is the easiest one to forget — you have just edited files in step 2(e), so come back here before you move on to anything else.** It applies whenever step 2(a) reported a real pass/fail result for a file you then edited. It does NOT apply when step 2(a) returned \`skipped: true\` or \`ran: 0\` for every suite — there is no baseline to compare against, so say so in your report instead of re-running. When it applies: re-run those \`[external]\` files with \`skyramp_run_existing_tests\` (\`mode: "verify"\`, \`stateFile\`) — the server reads each file's result back as its \`afterStatus\`. Editing an \`[external]\` file that step 2(a) confirmed failing and NOT re-running it leaves your own fix unverified — you would be reporting a repair you never saw work. A still-failing verify is surfaced in the report — do not loop.
216
216
 
217
217
  3. **Code review:** Find the logic bugs in the code that this change touches. Read the implementation of each changed endpoint: the route handler, and the functions that it calls to read or write data. For a changed screen, read the component and the functions that it calls. Read these files even when the diff does not contain them — a defect often sits in the code that the change depends on. Report each finding in \`issuesFound\` with a severity, and say which file and line holds it. Common patterns to flag:
218
218
  - Computed fields not recalculated after mutation (e.g. \`total_amount\` unchanged after items are added/removed)
@@ -1,6 +1,6 @@
1
1
  import { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js";
2
2
  import { WorkspaceConfig } from "../workspace/workspace.js";
3
- import { ParsedExternalRun } from "../types/ExternalTestExecution.js";
3
+ import { ExternalFileOutcome, ExternalTestResult, ParsedExternalRun } from "../types/ExternalTestExecution.js";
4
4
  import { SUPPORTED_FRAMEWORKS } from "../workspace/frameworks.js";
5
5
  /** Test-run lifecycle resolved from a service's workspace.yml runtimeDetails. */
6
6
  export interface ResolvedRunConfig {
@@ -42,7 +42,9 @@ export interface RunExistingTestsParams {
42
42
  workspacePath: string;
43
43
  service?: string;
44
44
  testSelectors?: string[];
45
- mode?: "confirm" | "verify";
45
+ /** Required: the only thing that marks a record as the pre-edit baseline or the
46
+ * post-edit re-run. A wrong label is read as fact by the report. */
47
+ mode: "confirm" | "verify";
46
48
  project?: string;
47
49
  stateFile?: string;
48
50
  repository?: string;
@@ -128,6 +130,36 @@ export declare function aggregateRuns(parsed: ParsedExternalRun[]): ParsedExtern
128
130
  */
129
131
  export declare function runExternalSuite(spec: ReaderSpec, cfg: ResolvedRunConfig, opts: PlaywrightRunOpts): Promise<ParsedExternalRun>;
130
132
  export declare function runPlaywright(cfg: ResolvedRunConfig, opts: PlaywrightRunOpts): Promise<ParsedExternalRun>;
133
+ /**
134
+ * Reduce a run's flat result list to one outcome per spec file.
135
+ *
136
+ * This is the only place that knows BOTH what the framework reported and what the
137
+ * run asked it for, which is what `complete` needs. Deriving it later from the
138
+ * results alone is not possible: a file re-run for one test looks identical to a
139
+ * file whose other tests silently vanished.
140
+ *
141
+ * `partialFiles` are the files a `file::test title` selector narrowed to a subset of
142
+ * their tests — the one narrowing this tool applies itself, and so the only one it
143
+ * can state with certainty.
144
+ */
145
+ export declare function summarizeFilesInRun(results: ExternalTestResult[], opts: {
146
+ partialFiles: Set<string>;
147
+ }): ExternalFileOutcome[];
148
+ /** Placeholder identity for a record that names no suite because none resolved.
149
+ * Deliberately not a valid suiteIdOf output, so it can never collide with one. */
150
+ export declare const NO_RUNNABLE_SUITE = "(no runnable suite)";
151
+ /** Stable identity for ONE configured suite. A service can declare several
152
+ * (workspace.yml testSuites), so its name alone collides — two suites under one
153
+ * service would share a slot and the later would mask the other instead of both
154
+ * counting. */
155
+ export declare function suiteIdOf(cfg: {
156
+ serviceName: string;
157
+ framework: string;
158
+ testRunCommand: string;
159
+ }): string;
160
+ /** The files a selector narrowed to individual tests. A bare `file` selector runs
161
+ * the whole file; `file::test title` does not. */
162
+ export declare function partialFilesFrom(selectors: string[]): Set<string>;
131
163
  /**
132
164
  * Orchestrates an external test run: resolve config → apply maxTests cap → run →
133
165
  * persist a record to the state file. Pure of MCP/registration concerns and
@@ -440,6 +440,95 @@ const defaultDeps = {
440
440
  },
441
441
  now: () => new Date().toISOString(),
442
442
  };
443
+ /** Worst outcome wins for a file. `error` (never reached its assertions) outranks
444
+ * `fail` so a collection failure is never reported as an assertion failure, and
445
+ * `skipped` sits below everything: it only wins when nothing else ran. */
446
+ const STATUS_RANK = {
447
+ skipped: 0,
448
+ pass: 1,
449
+ fail: 2,
450
+ error: 3,
451
+ };
452
+ /**
453
+ * Reduce a run's flat result list to one outcome per spec file.
454
+ *
455
+ * This is the only place that knows BOTH what the framework reported and what the
456
+ * run asked it for, which is what `complete` needs. Deriving it later from the
457
+ * results alone is not possible: a file re-run for one test looks identical to a
458
+ * file whose other tests silently vanished.
459
+ *
460
+ * `partialFiles` are the files a `file::test title` selector narrowed to a subset of
461
+ * their tests — the one narrowing this tool applies itself, and so the only one it
462
+ * can state with certainty.
463
+ */
464
+ export function summarizeFilesInRun(results, opts) {
465
+ const byFile = new Map();
466
+ for (const r of results) {
467
+ if (typeof r.file !== "string" || r.file === "")
468
+ continue;
469
+ const key = r.absoluteFile ?? r.file;
470
+ byFile.set(key, [...(byFile.get(key) ?? []), r]);
471
+ }
472
+ return [...byFile.values()].map((group) => {
473
+ const counts = { pass: 0, fail: 0, error: 0, skipped: 0 };
474
+ for (const r of group)
475
+ counts[r.status] = (counts[r.status] ?? 0) + 1;
476
+ const status = group.reduce((a, b) => STATUS_RANK[b.status] > STATUS_RANK[a.status] ? b : a).status;
477
+ const first = group[0];
478
+ return {
479
+ file: first.file,
480
+ ...(first.absoluteFile ? { absoluteFile: first.absoluteFile } : {}),
481
+ status,
482
+ counts,
483
+ complete: !isPartial(first, opts.partialFiles),
484
+ errors: group
485
+ .filter((r) => r.status === "fail" || r.status === "error")
486
+ .map((r) => `${r.testId}: ${r.message ?? r.status}`),
487
+ durationMs: group.reduce((sum, r) => sum + (r.durationMs ?? 0), 0),
488
+ };
489
+ });
490
+ }
491
+ /** A selector spelling and a result spelling need not agree — the selector is
492
+ * repo-relative, the result may be absolute or framework-root-relative — so match
493
+ * on a path-segment suffix in either direction. */
494
+ function isPartial(r, partialFiles) {
495
+ if (partialFiles.size === 0)
496
+ return false;
497
+ const candidates = [r.file, r.absoluteFile].filter((c) => typeof c === "string" && c !== "");
498
+ for (const sel of partialFiles) {
499
+ for (const c of candidates) {
500
+ if (c === sel || c.endsWith("/" + sel) || sel.endsWith("/" + c))
501
+ return true;
502
+ }
503
+ }
504
+ return false;
505
+ }
506
+ /** Placeholder identity for a record that names no suite because none resolved.
507
+ * Deliberately not a valid suiteIdOf output, so it can never collide with one. */
508
+ export const NO_RUNNABLE_SUITE = "(no runnable suite)";
509
+ /** Stable identity for ONE configured suite. A service can declare several
510
+ * (workspace.yml testSuites), so its name alone collides — two suites under one
511
+ * service would share a slot and the later would mask the other instead of both
512
+ * counting. */
513
+ export function suiteIdOf(cfg) {
514
+ return `${cfg.serviceName}::${cfg.framework}::${cfg.testRunCommand}`;
515
+ }
516
+ /** The files a selector narrowed to individual tests. A bare `file` selector runs
517
+ * the whole file; `file::test title` does not. */
518
+ export function partialFilesFrom(selectors) {
519
+ const out = new Set();
520
+ for (const sel of selectors) {
521
+ const [file, ...rest] = sel.split("::");
522
+ // A trailing `::` names no title, so every builder drops it and the whole file
523
+ // runs. Treating it as partial would throw away a real status.
524
+ if (rest.join("::").trim() === "")
525
+ continue;
526
+ const trimmed = file.trim();
527
+ if (trimmed)
528
+ out.add(trimmed);
529
+ }
530
+ return out;
531
+ }
443
532
  /** Appends a run record into UnifiedAnalysisState.externalTestResults for the
444
533
  * given repo section. Best-effort: a state failure never fails the run. */
445
534
  async function persistExternalRun(stateFile, repository, record) {
@@ -463,7 +552,7 @@ async function persistExternalRun(stateFile, repository, record) {
463
552
  * fully injectable for tests.
464
553
  */
465
554
  export async function runExternalTests(params, deps = defaultDeps) {
466
- const mode = params.mode ?? "confirm";
555
+ const mode = params.mode;
467
556
  const selectors = params.testSelectors ?? [];
468
557
  let effective = selectors;
469
558
  let truncated = false;
@@ -486,6 +575,11 @@ export async function runExternalTests(params, deps = defaultDeps) {
486
575
  if (params.stateFile) {
487
576
  await persistExternalRun(params.stateFile, params.repository, {
488
577
  ...skippedRun,
578
+ files: [],
579
+ // No run config resolved, so there is no suite to identify — and a service
580
+ // name here would read as one. The reader never keys on it (the record is
581
+ // skipped, and skipped records are dropped first), so name it for what it is.
582
+ suite: NO_RUNNABLE_SUITE,
489
583
  mode,
490
584
  ranAt: deps.now(),
491
585
  repository: params.repository,
@@ -518,6 +612,8 @@ export async function runExternalTests(params, deps = defaultDeps) {
518
612
  if (params.stateFile) {
519
613
  await persistExternalRun(params.stateFile, params.repository, {
520
614
  ...routedOut,
615
+ files: [],
616
+ suite: suiteIdOf(cfg),
521
617
  mode,
522
618
  ranAt: deps.now(),
523
619
  repository: params.repository,
@@ -542,8 +638,13 @@ export async function runExternalTests(params, deps = defaultDeps) {
542
638
  };
543
639
  perSuite.push(suiteResult);
544
640
  if (params.stateFile) {
641
+ const files = summarizeFilesInRun(suiteResult.results, {
642
+ partialFiles: partialFilesFrom(routing.kept),
643
+ });
545
644
  await persistExternalRun(params.stateFile, params.repository, {
546
645
  ...suiteResult,
646
+ files,
647
+ suite: suiteIdOf(cfg),
547
648
  mode,
548
649
  ranAt: deps.now(),
549
650
  repository: params.repository,
@@ -575,8 +676,7 @@ export function registerRunExistingTestsTool(server) {
575
676
  .describe("Spec files to run (paths relative to the run command's cwd), optionally `file::test title` to also filter by name. Omit to run the whole configured suite."),
576
677
  mode: z
577
678
  .enum(["confirm", "verify"])
578
- .optional()
579
- .describe("confirm (default): scoped run to establish which tests a change breaks; disables the suite's fail-fast cap (e.g. Playwright's maxFailures) so ALL breakages are counted. verify: re-run after an edit."),
679
+ .describe("REQUIRED, and the only thing that marks a run as before or after the edit: confirm = the pre-edit baseline (a scoped run establishing which tests the change breaks; disables the suite's fail-fast cap, e.g. Playwright's maxFailures, so ALL breakages are counted), verify = the re-run after you edited. Passing confirm for a post-edit run overwrites the baseline with the fixed result, and the report then shows a repair as having never been broken."),
580
680
  project: z
581
681
  .string()
582
682
  .optional()
@@ -635,7 +735,7 @@ export function registerRunExistingTestsTool(server) {
635
735
  AnalyticsService.pushMCPToolEvent(TOOL_NAME, errorResult, {
636
736
  workspacePath: params.workspacePath,
637
737
  service: params.service ?? "",
638
- mode: params.mode ?? "confirm",
738
+ mode: params.mode,
639
739
  }).catch((err) => {
640
740
  logger.warning("Analytics event failed", { error: String(err) });
641
741
  });
@@ -84,6 +84,96 @@ const MAINTENANCE_CHANGE_ACTIONS = new Set([
84
84
  DriftAction.Regenerate,
85
85
  DriftAction.Delete,
86
86
  ]);
87
+ const EXTERNAL_TO_EXECUTION_STATUS = {
88
+ pass: TestExecutionStatus.Pass,
89
+ fail: TestExecutionStatus.Fail,
90
+ error: TestExecutionStatus.Error,
91
+ };
92
+ /** Does this run's file outcome describe `knownPath`?
93
+ *
94
+ * `absoluteFile` is the framework's own path resolved against the root it reported,
95
+ * so it names one file and is compared exactly. An outcome without one is refused:
96
+ * a relative spelling is a suffix, and a suffix cannot separate
97
+ * `packages/a/tests/x.spec.ts` from `packages/b/tests/x.spec.ts`. Guessing there and
98
+ * guarding the guess is what this deliberately does not do — a missing status is a
99
+ * row that reads Unknown, which is what it reads today; a wrong one is a false claim
100
+ * about someone's tests.
101
+ *
102
+ * The cost is real and accepted: a suite running inside a container reports the
103
+ * container's root, which cannot equal the host path discovery recorded, so those
104
+ * repos get no status until the run records a root the report can line up. */
105
+ function outcomeMatches(outcome, knownPath) {
106
+ return (typeof outcome.absoluteFile === "string" &&
107
+ path.resolve(outcome.absoluteFile) === path.resolve(knownPath));
108
+ }
109
+ /** One line describing what the suite saw, for a row whose status came from the run
110
+ * rather than from an execution the agent drafted a summary for. Without it the
111
+ * rendered cell is a bare `Error` or `Pass` with nothing behind it. */
112
+ function describeOutcome(o) {
113
+ // Defended like the record reads around it: a malformed outcome must not throw
114
+ // out of the handler and take the whole report submission with it.
115
+ const c = o.counts ?? { pass: 0, fail: 0, error: 0, skipped: 0 };
116
+ const parts = [
117
+ c.pass ? `${c.pass} passed` : "",
118
+ c.fail ? `${c.fail} failed` : "",
119
+ c.error ? `${c.error} errored` : "",
120
+ c.skipped ? `${c.skipped} skipped` : "",
121
+ ].filter(Boolean);
122
+ const head = `${parts.join(", ") || "no tests reported"} in the repo's own suite`;
123
+ // Adapter messages carry stack traces, so collapse whitespace before truncating:
124
+ // an embedded newline would break the maintenance row this lands in.
125
+ const first = (o.errors ?? [])[0]?.replace(/\s+/g, " ").trim();
126
+ return first ? `${head} — ${first.slice(0, 200)}` : head;
127
+ }
128
+ function externalStatusesFor(records, knownPath) {
129
+ // LATEST per suite, then WORST across suites. Both halves matter: a suite re-run
130
+ // after another edit supersedes its own earlier verdict, so an obsolete failure
131
+ // must not outlive the fix that repaired it; but two different suites can each
132
+ // own the same file, and a pass in one must not mask a failure in the other.
133
+ const RANK = { pass: 0, fail: 1, error: 2 };
134
+ const statusIn = (mode) => {
135
+ const latestPerSuite = new Map();
136
+ for (const record of records ?? []) {
137
+ if (record.mode !== mode || record.skipped || !record.environmentHealthy)
138
+ continue;
139
+ // A suite that stopped on --max-failures left tests unrun and does not say
140
+ // which, so no file it touched can stand as covered. The adapter reports this;
141
+ // it is not a guess. (The tool forces --max-failures=0 for confirm, so this is
142
+ // reachable on a verify re-run under the repo's own cap.)
143
+ if (record.diagnostics?.maxFailuresBail)
144
+ continue;
145
+ const hit = (record.files ?? []).find((o) => outcomeMatches(o, knownPath));
146
+ if (!hit)
147
+ continue;
148
+ const key = record.suite ?? "";
149
+ const prior = latestPerSuite.get(key);
150
+ // `ranAt` is an ISO timestamp, so a lexical compare is chronological. The
151
+ // newest run wins BEFORE it is judged: a later partial re-run supersedes an
152
+ // earlier complete one, and must leave the row unknown rather than let the
153
+ // stale complete outcome stand in for it.
154
+ if (!prior || record.ranAt >= prior.at)
155
+ latestPerSuite.set(key, { at: record.ranAt, outcome: hit });
156
+ }
157
+ let worst;
158
+ for (const { outcome } of latestPerSuite.values()) {
159
+ // `skipped` means every test in the file was skipped — it was not exercised,
160
+ // and TestExecutionStatus.Skipped means a deliberate decision not to run it.
161
+ if (!outcome.complete || outcome.status === "skipped")
162
+ continue;
163
+ if (!worst || RANK[outcome.status] > RANK[worst.status])
164
+ worst = outcome;
165
+ }
166
+ return worst;
167
+ };
168
+ const before = statusIn("confirm");
169
+ const after = statusIn("verify");
170
+ return {
171
+ before: before && EXTERNAL_TO_EXECUTION_STATUS[before.status],
172
+ after: after && EXTERNAL_TO_EXECUTION_STATUS[after.status],
173
+ beforeDetail: before && describeOutcome(before),
174
+ afterDetail: after && describeOutcome(after),
175
+ };
176
+ }
87
177
  const TOOL_NAME = "skyramp_submit_report";
88
178
  const DEFAULT_COMMIT_MESSAGE = "Added recommendations by Skyramp Testbot.";
89
179
  // Per-repo attribution. In a multi-repo run, every report item carries the
@@ -553,14 +643,14 @@ const testMaintenanceSchema = z.object({
553
643
  "For passing runs: count and timing, e.g. '4 passed in 15.09s'. " +
554
644
  "For failing runs: failure name and one-line root cause, e.g. " +
555
645
  "'FAILED test_foo — assert 403 got 200, auth middleware not enforced'. " +
556
- "Empty string for VERIFY/IGNORE entries where no before-execution was run."),
646
+ "Pass an empty string when you ran nothing yourself — the server then fills this from the repo suite's own run — and for VERIFY/IGNORE entries where nothing ran at all. Omit the whole entry, never this field."),
557
647
  afterDetails: z
558
648
  .string()
559
649
  .describe("One line only — no embedded newlines, no raw HTTP headers or JSON blobs. " +
560
650
  "For passing runs: count and timing, e.g. '5 passed in 10.96s'. " +
561
651
  "For failing runs: failure name and one-line root cause, e.g. " +
562
652
  "'FAILED test_foo — check_schema fails, order_id=1 has discount from prior PATCH test'. " +
563
- "Empty string for VERIFY/IGNORE/DELETE entries where no after-execution was run."),
653
+ "Pass an empty string when you ran nothing yourself — the server then fills this from the repo suite's own run — and for VERIFY/IGNORE/DELETE entries where nothing ran at all. Omit the whole entry, never this field."),
564
654
  // Server-populated from stateFile execution records — never supplied by the LLM.
565
655
  beforeStatus: z.nativeEnum(TestExecutionStatus),
566
656
  afterStatus: z.nativeEnum(TestExecutionStatus),
@@ -1011,12 +1101,42 @@ export function registerSubmitReportTool(server) {
1011
1101
  : MAINTENANCE_CHANGE_ACTIONS.has(m.action)
1012
1102
  ? TestExecutionStatus.Unknown
1013
1103
  : TestExecutionStatus.Skipped;
1014
- const beforeStatus = recorded?.executionBefore?.status ?? defaultBeforeStatus;
1015
- const afterStatus = recorded?.executionAfter?.status ?? defaultAfterStatus;
1104
+ // A suite run answers only for a row whose default is Unknown. VERIFY and
1105
+ // IGNORE are Skipped by decision, not for want of a result, and testbot
1106
+ // drops both from the rendered table.
1107
+ const external = externalStatusesFor(stateData.externalTestResults, m.testFilePath);
1108
+ const beforeStatus = recorded?.executionBefore?.status ??
1109
+ (defaultBeforeStatus === TestExecutionStatus.Unknown
1110
+ ? external.before
1111
+ : undefined) ??
1112
+ defaultBeforeStatus;
1113
+ const afterStatus = recorded?.executionAfter?.status ??
1114
+ (defaultAfterStatus === TestExecutionStatus.Unknown
1115
+ ? external.after
1116
+ : undefined) ??
1117
+ defaultAfterStatus;
1016
1118
  // Trim before checking — a whitespace-only string is semantically blank and
1017
1119
  // must not satisfy the "drafted a summary" requirement.
1018
- const beforeDetails = detail?.beforeDetails?.trim() ?? "";
1019
- const afterDetails = detail?.afterDetails?.trim() ?? "";
1120
+ // The agent's drafted summary wins; otherwise a status that came from
1121
+ // the suite run carries the run's own line, so the rendered cell is
1122
+ // never a bare verdict with nothing behind it.
1123
+ //
1124
+ // Only when there is NO recorded execution. A file the agent ran itself
1125
+ // owes a drafted summary, and filling it here would satisfy the check
1126
+ // below on the agent's behalf whenever the two statuses happened to
1127
+ // agree — silently dropping the requirement.
1128
+ //
1129
+ // And only where the STATUS came from the suite too. A VERIFY or DELETE
1130
+ // row keeps its Skipped default, so a line saying "1 passed in the repo's
1131
+ // own suite" beside it contradicts the very cell it annotates.
1132
+ const beforeFromSuite = defaultBeforeStatus === TestExecutionStatus.Unknown &&
1133
+ !recorded?.executionBefore;
1134
+ const afterFromSuite = defaultAfterStatus === TestExecutionStatus.Unknown &&
1135
+ !recorded?.executionAfter;
1136
+ const beforeDetails = detail?.beforeDetails?.trim() ||
1137
+ (beforeFromSuite ? (external.beforeDetail ?? "") : "");
1138
+ const afterDetails = detail?.afterDetails?.trim() ||
1139
+ (afterFromSuite ? (external.afterDetail ?? "") : "");
1020
1140
  if (recorded?.executionBefore && !beforeDetails)
1021
1141
  missingDetails.push(`${displayName} (beforeDetails)`);
1022
1142
  if (recorded?.executionAfter && !afterDetails)
@@ -11,8 +11,15 @@ export type ExternalTestStatus = "pass" | "fail" | "error" | "skipped";
11
11
  export interface ExternalTestResult {
12
12
  /** Human-readable identity, e.g. `<file> › <describe…> › <title>`. */
13
13
  testId: string;
14
- /** Spec file the test lives in. */
14
+ /** Spec file the test lives in, in the framework's own spelling — usually absolute
15
+ * for jest and mocha, relative to the framework's root for Playwright and pytest.
16
+ * No framework guarantees which, so never assume: use `absoluteFile` when present. */
15
17
  file: string;
18
+ /** `file` resolved against the root the framework reported, when it reported one.
19
+ * A relative spelling alone cannot identify a file: `tests/Button.test.tsx` fits
20
+ * every package that has one. Consumers matching a result to a known path must
21
+ * prefer this and compare exactly. */
22
+ absoluteFile?: string;
16
23
  status: ExternalTestStatus;
17
24
  /** Failure/error text (ANSI-stripped). Present only for fail/error. */
18
25
  message?: string;
@@ -55,10 +62,69 @@ export interface ParsedExternalRun {
55
62
  /** Why the run was skipped (present only when `skipped`). */
56
63
  skipReason?: string;
57
64
  }
65
+ /**
66
+ * What one run established about ONE spec file. Computed by the run itself, which
67
+ * knows what it asked the framework for; the report then reads this instead of
68
+ * re-deriving coverage from the flat result list, where the facts below are no
69
+ * longer recoverable.
70
+ */
71
+ export interface ExternalFileOutcome {
72
+ /** The framework's own spelling, kept for diagnostics. */
73
+ file: string;
74
+ /** `file` resolved against the root the framework reported, when it reported one.
75
+ * Present here exactly when it was present on the results. */
76
+ absoluteFile?: string;
77
+ /** Worst outcome across the file's tests: error > fail > pass. `skipped` means
78
+ * every test in it was skipped, so the file was not exercised at all. */
79
+ status: ExternalTestStatus;
80
+ /** How many tests produced each outcome. Rendered into the maintenance row's
81
+ * before/after detail when the agent drafted none. */
82
+ counts: {
83
+ pass: number;
84
+ fail: number;
85
+ error: number;
86
+ skipped: number;
87
+ };
88
+ /**
89
+ * Did this run exercise every TEST in the file? False when a `file::test title`
90
+ * selector narrowed it — the one narrowing this tool applies per file, and so the
91
+ * only one it can state with certainty. A partial run says nothing about the tests
92
+ * it did not reach, so it cannot stand as the file's status.
93
+ *
94
+ * Two narrowings are deliberately NOT reflected here, and the reader compensates
95
+ * for one of them:
96
+ * - A `file::title` selector also becomes a run-WIDE --grep / -t on every
97
+ * framework except pytest, so a co-selected file ran only its matching tests
98
+ * while this says complete. Known and accepted.
99
+ * - A suite that bailed on --max-failures abandoned tests it cannot name. That one
100
+ * IS reported, in `diagnostics.maxFailuresBail`, and the report skips such a
101
+ * record outright rather than reading any file from it.
102
+ *
103
+ * A Playwright `project` scope does NOT make a file incomplete: every test in it
104
+ * still ran, in one project rather than all of them. That is a different axis, and
105
+ * discarding the status over it throws away the run's whole verdict — measured on
106
+ * eval run 33280302496, where it turned a real Error -> Pass into Unknown.
107
+ *
108
+ * Known gap: a repo whose runner bails on its own (a jest or mocha `bail` in
109
+ * config rather than on the command line) is not detectable from the report,
110
+ * and only Playwright reports a --max-failures bail back to us.
111
+ */
112
+ complete: boolean;
113
+ /** Failure/error text for the tests that did not pass. */
114
+ errors: string[];
115
+ durationMs: number;
116
+ }
58
117
  /** One persisted run of the external suite, appended to
59
118
  * UnifiedAnalysisState.externalTestResults so drift analysis / the report can
60
119
  * fold in CONFIRMED failures instead of guesses. */
61
120
  export interface ExternalTestRunRecord extends ParsedExternalRun {
121
+ /** Per-file outcome, one entry per spec file this run produced results for.
122
+ * Empty for a skipped run. */
123
+ files: ExternalFileOutcome[];
124
+ /** Which configured suite produced this record. A repo can declare several, and
125
+ * without this a re-run of ONE suite cannot be told from a second suite's first
126
+ * run — so a superseded failure would outlive the fix that repaired it. */
127
+ suite: string;
62
128
  mode: "confirm" | "verify";
63
129
  /** ISO timestamp when the run completed. */
64
130
  ranAt: string;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@skyramp/mcp",
3
- "version": "0.3.6-rc.1",
3
+ "version": "0.3.6-rc.2.ac20",
4
4
  "main": "build/index.js",
5
5
  "exports": {
6
6
  ".": "./build/index.js",