@skyramp/mcp 0.3.6-rc.1 → 0.3.6-rc.2.ac20
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/build/adapters/jestAdapter.js +3 -0
- package/build/adapters/mochaAdapter.js +2 -0
- package/build/adapters/playwrightAdapter.js +3 -0
- package/build/adapters/pytestAdapter.js +12 -0
- package/build/prompts/testbot/testbot-prompts.js +1 -1
- package/build/tools/runExistingTestsTool.d.ts +34 -2
- package/build/tools/runExistingTestsTool.js +104 -4
- package/build/tools/submitReportTool.js +126 -6
- package/build/types/ExternalTestExecution.d.ts +67 -1
- package/package.json +1 -1
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import * as path from "path";
|
|
1
2
|
/**
|
|
2
3
|
* Shared jest + vitest adapter for skyramp_run_existing_tests. vitest's json
|
|
3
4
|
* reporter is jest-compatible, so both use ONE parser; only the reporter flag
|
|
@@ -40,6 +41,7 @@ export function parseJestJson(report, opts) {
|
|
|
40
41
|
results.push({
|
|
41
42
|
testId: `${name} › (file failed to run)`,
|
|
42
43
|
file: name,
|
|
44
|
+
...(path.isAbsolute(name) ? { absoluteFile: name } : {}),
|
|
43
45
|
status: "error",
|
|
44
46
|
message: stripVTControlCharacters(raw).trim() || undefined,
|
|
45
47
|
durationMs: 0,
|
|
@@ -55,6 +57,7 @@ export function parseJestJson(report, opts) {
|
|
|
55
57
|
results.push({
|
|
56
58
|
testId: `${name} › ${titlePath}`,
|
|
57
59
|
file: name,
|
|
60
|
+
...(path.isAbsolute(name) ? { absoluteFile: name } : {}),
|
|
58
61
|
status,
|
|
59
62
|
message,
|
|
60
63
|
durationMs: Math.round(a.duration ?? 0),
|
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import * as path from "path";
|
|
1
2
|
/**
|
|
2
3
|
* Mocha adapter for skyramp_run_existing_tests. Normalizes `mocha --reporter
|
|
3
4
|
* json` (emitted to stdout) into the neutral shape. Mocha does not tag
|
|
@@ -29,6 +30,7 @@ function toResult(t, status) {
|
|
|
29
30
|
return {
|
|
30
31
|
testId: `${file} › ${full}`,
|
|
31
32
|
file,
|
|
33
|
+
...(path.isAbsolute(file) ? { absoluteFile: file } : {}),
|
|
32
34
|
status,
|
|
33
35
|
message: status === "fail" || status === "error" ? messageOf(t) : undefined,
|
|
34
36
|
durationMs: Math.round(t.duration ?? 0),
|
|
@@ -12,6 +12,7 @@
|
|
|
12
12
|
* builds the flags appended to the suite's `testRunCommand`. The IO shell that
|
|
13
13
|
* actually spawns the run lives in `runExistingTestsTool`.
|
|
14
14
|
*/
|
|
15
|
+
import * as path from "path";
|
|
15
16
|
import { stripVTControlCharacters } from "util";
|
|
16
17
|
/**
|
|
17
18
|
* A Playwright project (or spec file) is "infra" when it exists to bring up /
|
|
@@ -67,6 +68,7 @@ function messageOf(test) {
|
|
|
67
68
|
}
|
|
68
69
|
export function parsePlaywrightJson(report, opts) {
|
|
69
70
|
const rep = (report ?? {});
|
|
71
|
+
const rootDir = rep.config?.rootDir;
|
|
70
72
|
const collected = [];
|
|
71
73
|
for (const fileSuite of rep.suites ?? []) {
|
|
72
74
|
collectSpecs(fileSuite, fileSuite.file ?? "", [], collected);
|
|
@@ -118,6 +120,7 @@ export function parsePlaywrightJson(report, opts) {
|
|
|
118
120
|
results.push({
|
|
119
121
|
testId,
|
|
120
122
|
file,
|
|
123
|
+
...(rootDir ? { absoluteFile: path.resolve(rootDir, file) } : {}),
|
|
121
124
|
status,
|
|
122
125
|
message: status === "fail" || status === "error" ? messageOf(test) : undefined,
|
|
123
126
|
durationMs,
|
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import * as path from "path";
|
|
1
2
|
/**
|
|
2
3
|
* pytest adapter for skyramp_run_existing_tests. Normalizes the
|
|
3
4
|
* `pytest-json-report` (`--json-report`) output into the neutral result shape.
|
|
@@ -39,6 +40,8 @@ function durationMsOf(test) {
|
|
|
39
40
|
export function parsePytestJson(report, opts) {
|
|
40
41
|
const rep = (report ?? {});
|
|
41
42
|
const tests = rep.tests ?? [];
|
|
43
|
+
const root = rep.root;
|
|
44
|
+
const abs = (f) => (root ? { absoluteFile: path.resolve(root, f) } : {});
|
|
42
45
|
// Real test outcomes.
|
|
43
46
|
const testResults = tests.map((t) => {
|
|
44
47
|
const status = mapOutcome(t.outcome);
|
|
@@ -46,6 +49,7 @@ export function parsePytestJson(report, opts) {
|
|
|
46
49
|
return {
|
|
47
50
|
testId: nodeid,
|
|
48
51
|
file: fileOf(nodeid),
|
|
52
|
+
...abs(fileOf(nodeid)),
|
|
49
53
|
status,
|
|
50
54
|
message: status === "fail" || status === "error" ? messageOf(t) : undefined,
|
|
51
55
|
durationMs: durationMsOf(t),
|
|
@@ -61,6 +65,7 @@ export function parsePytestJson(report, opts) {
|
|
|
61
65
|
return {
|
|
62
66
|
testId: `${nodeid} › (collection error)`,
|
|
63
67
|
file: fileOf(nodeid),
|
|
68
|
+
...abs(fileOf(nodeid)),
|
|
64
69
|
status: "error",
|
|
65
70
|
message: lastLine(c.longrepr),
|
|
66
71
|
durationMs: 0,
|
|
@@ -72,6 +77,13 @@ export function parsePytestJson(report, opts) {
|
|
|
72
77
|
// When collection failed AND no real test ran, the suite as a whole could not be
|
|
73
78
|
// collected — environmental (a wall of red), not PR signal. When some real tests
|
|
74
79
|
// DID run, the collection errors above stand as ordinary PR-signal error results.
|
|
80
|
+
// NOTE: a collection error names the file that failed to import, but says nothing
|
|
81
|
+
// about WHY. A missing settings module, a database that is down or any shared
|
|
82
|
+
// dependency failure surfaces per-module, landing on exactly the files the run
|
|
83
|
+
// selected — indistinguishable here from a file the PR itself broke. So a run that
|
|
84
|
+
// collected nothing stays unhealthy even when scoped, and the file keeps its
|
|
85
|
+
// Unknown baseline. Treating it as PR signal would let an environment outage be
|
|
86
|
+
// reported as a test the change broke, and then repaired.
|
|
75
87
|
if (collectionErrors.length > 0 && testResults.length === 0) {
|
|
76
88
|
environmentHealthy = false;
|
|
77
89
|
const first = collectionErrors[0];
|
|
@@ -212,7 +212,7 @@ ${maintenanceBeforeExecStep}
|
|
|
212
212
|
|
|
213
213
|
e. Call \`skyramp_actions\` with \`stateFile\` (from \`skyramp_analyze_changes\` output) and apply the edits it returns.
|
|
214
214
|
|
|
215
|
-
f. Verify external-test fixes. **This step is not optional and it is the easiest one to forget — you have just edited files in step 2(e), so come back here before you move on to anything else.** It applies whenever step 2(a) reported a real pass/fail result for a file you then edited. It does NOT apply when step 2(a) returned \`skipped: true\` or \`ran: 0\` for every suite — there is no baseline to compare against, so say so in your report instead of re-running. When it applies: re-run those \`[external]\` files with \`skyramp_run_existing_tests\` (\`mode: "verify"\`, \`stateFile\`)
|
|
215
|
+
f. Verify external-test fixes. **This step is not optional and it is the easiest one to forget — you have just edited files in step 2(e), so come back here before you move on to anything else.** It applies whenever step 2(a) reported a real pass/fail result for a file you then edited. It does NOT apply when step 2(a) returned \`skipped: true\` or \`ran: 0\` for every suite — there is no baseline to compare against, so say so in your report instead of re-running. When it applies: re-run those \`[external]\` files with \`skyramp_run_existing_tests\` (\`mode: "verify"\`, \`stateFile\`) — the server reads each file's result back as its \`afterStatus\`. Editing an \`[external]\` file that step 2(a) confirmed failing and NOT re-running it leaves your own fix unverified — you would be reporting a repair you never saw work. A still-failing verify is surfaced in the report — do not loop.
|
|
216
216
|
|
|
217
217
|
3. **Code review:** Find the logic bugs in the code that this change touches. Read the implementation of each changed endpoint: the route handler, and the functions that it calls to read or write data. For a changed screen, read the component and the functions that it calls. Read these files even when the diff does not contain them — a defect often sits in the code that the change depends on. Report each finding in \`issuesFound\` with a severity, and say which file and line holds it. Common patterns to flag:
|
|
218
218
|
- Computed fields not recalculated after mutation (e.g. \`total_amount\` unchanged after items are added/removed)
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js";
|
|
2
2
|
import { WorkspaceConfig } from "../workspace/workspace.js";
|
|
3
|
-
import { ParsedExternalRun } from "../types/ExternalTestExecution.js";
|
|
3
|
+
import { ExternalFileOutcome, ExternalTestResult, ParsedExternalRun } from "../types/ExternalTestExecution.js";
|
|
4
4
|
import { SUPPORTED_FRAMEWORKS } from "../workspace/frameworks.js";
|
|
5
5
|
/** Test-run lifecycle resolved from a service's workspace.yml runtimeDetails. */
|
|
6
6
|
export interface ResolvedRunConfig {
|
|
@@ -42,7 +42,9 @@ export interface RunExistingTestsParams {
|
|
|
42
42
|
workspacePath: string;
|
|
43
43
|
service?: string;
|
|
44
44
|
testSelectors?: string[];
|
|
45
|
-
|
|
45
|
+
/** Required: the only thing that marks a record as the pre-edit baseline or the
|
|
46
|
+
* post-edit re-run. A wrong label is read as fact by the report. */
|
|
47
|
+
mode: "confirm" | "verify";
|
|
46
48
|
project?: string;
|
|
47
49
|
stateFile?: string;
|
|
48
50
|
repository?: string;
|
|
@@ -128,6 +130,36 @@ export declare function aggregateRuns(parsed: ParsedExternalRun[]): ParsedExtern
|
|
|
128
130
|
*/
|
|
129
131
|
export declare function runExternalSuite(spec: ReaderSpec, cfg: ResolvedRunConfig, opts: PlaywrightRunOpts): Promise<ParsedExternalRun>;
|
|
130
132
|
export declare function runPlaywright(cfg: ResolvedRunConfig, opts: PlaywrightRunOpts): Promise<ParsedExternalRun>;
|
|
133
|
+
/**
|
|
134
|
+
* Reduce a run's flat result list to one outcome per spec file.
|
|
135
|
+
*
|
|
136
|
+
* This is the only place that knows BOTH what the framework reported and what the
|
|
137
|
+
* run asked it for, which is what `complete` needs. Deriving it later from the
|
|
138
|
+
* results alone is not possible: a file re-run for one test looks identical to a
|
|
139
|
+
* file whose other tests silently vanished.
|
|
140
|
+
*
|
|
141
|
+
* `partialFiles` are the files a `file::test title` selector narrowed to a subset of
|
|
142
|
+
* their tests — the one narrowing this tool applies itself, and so the only one it
|
|
143
|
+
* can state with certainty.
|
|
144
|
+
*/
|
|
145
|
+
export declare function summarizeFilesInRun(results: ExternalTestResult[], opts: {
|
|
146
|
+
partialFiles: Set<string>;
|
|
147
|
+
}): ExternalFileOutcome[];
|
|
148
|
+
/** Placeholder identity for a record that names no suite because none resolved.
|
|
149
|
+
* Deliberately not a valid suiteIdOf output, so it can never collide with one. */
|
|
150
|
+
export declare const NO_RUNNABLE_SUITE = "(no runnable suite)";
|
|
151
|
+
/** Stable identity for ONE configured suite. A service can declare several
|
|
152
|
+
* (workspace.yml testSuites), so its name alone collides — two suites under one
|
|
153
|
+
* service would share a slot and the later would mask the other instead of both
|
|
154
|
+
* counting. */
|
|
155
|
+
export declare function suiteIdOf(cfg: {
|
|
156
|
+
serviceName: string;
|
|
157
|
+
framework: string;
|
|
158
|
+
testRunCommand: string;
|
|
159
|
+
}): string;
|
|
160
|
+
/** The files a selector narrowed to individual tests. A bare `file` selector runs
|
|
161
|
+
* the whole file; `file::test title` does not. */
|
|
162
|
+
export declare function partialFilesFrom(selectors: string[]): Set<string>;
|
|
131
163
|
/**
|
|
132
164
|
* Orchestrates an external test run: resolve config → apply maxTests cap → run →
|
|
133
165
|
* persist a record to the state file. Pure of MCP/registration concerns and
|
|
@@ -440,6 +440,95 @@ const defaultDeps = {
|
|
|
440
440
|
},
|
|
441
441
|
now: () => new Date().toISOString(),
|
|
442
442
|
};
|
|
443
|
+
/** Worst outcome wins for a file. `error` (never reached its assertions) outranks
|
|
444
|
+
* `fail` so a collection failure is never reported as an assertion failure, and
|
|
445
|
+
* `skipped` sits below everything: it only wins when nothing else ran. */
|
|
446
|
+
const STATUS_RANK = {
|
|
447
|
+
skipped: 0,
|
|
448
|
+
pass: 1,
|
|
449
|
+
fail: 2,
|
|
450
|
+
error: 3,
|
|
451
|
+
};
|
|
452
|
+
/**
|
|
453
|
+
* Reduce a run's flat result list to one outcome per spec file.
|
|
454
|
+
*
|
|
455
|
+
* This is the only place that knows BOTH what the framework reported and what the
|
|
456
|
+
* run asked it for, which is what `complete` needs. Deriving it later from the
|
|
457
|
+
* results alone is not possible: a file re-run for one test looks identical to a
|
|
458
|
+
* file whose other tests silently vanished.
|
|
459
|
+
*
|
|
460
|
+
* `partialFiles` are the files a `file::test title` selector narrowed to a subset of
|
|
461
|
+
* their tests — the one narrowing this tool applies itself, and so the only one it
|
|
462
|
+
* can state with certainty.
|
|
463
|
+
*/
|
|
464
|
+
export function summarizeFilesInRun(results, opts) {
|
|
465
|
+
const byFile = new Map();
|
|
466
|
+
for (const r of results) {
|
|
467
|
+
if (typeof r.file !== "string" || r.file === "")
|
|
468
|
+
continue;
|
|
469
|
+
const key = r.absoluteFile ?? r.file;
|
|
470
|
+
byFile.set(key, [...(byFile.get(key) ?? []), r]);
|
|
471
|
+
}
|
|
472
|
+
return [...byFile.values()].map((group) => {
|
|
473
|
+
const counts = { pass: 0, fail: 0, error: 0, skipped: 0 };
|
|
474
|
+
for (const r of group)
|
|
475
|
+
counts[r.status] = (counts[r.status] ?? 0) + 1;
|
|
476
|
+
const status = group.reduce((a, b) => STATUS_RANK[b.status] > STATUS_RANK[a.status] ? b : a).status;
|
|
477
|
+
const first = group[0];
|
|
478
|
+
return {
|
|
479
|
+
file: first.file,
|
|
480
|
+
...(first.absoluteFile ? { absoluteFile: first.absoluteFile } : {}),
|
|
481
|
+
status,
|
|
482
|
+
counts,
|
|
483
|
+
complete: !isPartial(first, opts.partialFiles),
|
|
484
|
+
errors: group
|
|
485
|
+
.filter((r) => r.status === "fail" || r.status === "error")
|
|
486
|
+
.map((r) => `${r.testId}: ${r.message ?? r.status}`),
|
|
487
|
+
durationMs: group.reduce((sum, r) => sum + (r.durationMs ?? 0), 0),
|
|
488
|
+
};
|
|
489
|
+
});
|
|
490
|
+
}
|
|
491
|
+
/** A selector spelling and a result spelling need not agree — the selector is
|
|
492
|
+
* repo-relative, the result may be absolute or framework-root-relative — so match
|
|
493
|
+
* on a path-segment suffix in either direction. */
|
|
494
|
+
function isPartial(r, partialFiles) {
|
|
495
|
+
if (partialFiles.size === 0)
|
|
496
|
+
return false;
|
|
497
|
+
const candidates = [r.file, r.absoluteFile].filter((c) => typeof c === "string" && c !== "");
|
|
498
|
+
for (const sel of partialFiles) {
|
|
499
|
+
for (const c of candidates) {
|
|
500
|
+
if (c === sel || c.endsWith("/" + sel) || sel.endsWith("/" + c))
|
|
501
|
+
return true;
|
|
502
|
+
}
|
|
503
|
+
}
|
|
504
|
+
return false;
|
|
505
|
+
}
|
|
506
|
+
/** Placeholder identity for a record that names no suite because none resolved.
|
|
507
|
+
* Deliberately not a valid suiteIdOf output, so it can never collide with one. */
|
|
508
|
+
export const NO_RUNNABLE_SUITE = "(no runnable suite)";
|
|
509
|
+
/** Stable identity for ONE configured suite. A service can declare several
|
|
510
|
+
* (workspace.yml testSuites), so its name alone collides — two suites under one
|
|
511
|
+
* service would share a slot and the later would mask the other instead of both
|
|
512
|
+
* counting. */
|
|
513
|
+
export function suiteIdOf(cfg) {
|
|
514
|
+
return `${cfg.serviceName}::${cfg.framework}::${cfg.testRunCommand}`;
|
|
515
|
+
}
|
|
516
|
+
/** The files a selector narrowed to individual tests. A bare `file` selector runs
|
|
517
|
+
* the whole file; `file::test title` does not. */
|
|
518
|
+
export function partialFilesFrom(selectors) {
|
|
519
|
+
const out = new Set();
|
|
520
|
+
for (const sel of selectors) {
|
|
521
|
+
const [file, ...rest] = sel.split("::");
|
|
522
|
+
// A trailing `::` names no title, so every builder drops it and the whole file
|
|
523
|
+
// runs. Treating it as partial would throw away a real status.
|
|
524
|
+
if (rest.join("::").trim() === "")
|
|
525
|
+
continue;
|
|
526
|
+
const trimmed = file.trim();
|
|
527
|
+
if (trimmed)
|
|
528
|
+
out.add(trimmed);
|
|
529
|
+
}
|
|
530
|
+
return out;
|
|
531
|
+
}
|
|
443
532
|
/** Appends a run record into UnifiedAnalysisState.externalTestResults for the
|
|
444
533
|
* given repo section. Best-effort: a state failure never fails the run. */
|
|
445
534
|
async function persistExternalRun(stateFile, repository, record) {
|
|
@@ -463,7 +552,7 @@ async function persistExternalRun(stateFile, repository, record) {
|
|
|
463
552
|
* fully injectable for tests.
|
|
464
553
|
*/
|
|
465
554
|
export async function runExternalTests(params, deps = defaultDeps) {
|
|
466
|
-
const mode = params.mode
|
|
555
|
+
const mode = params.mode;
|
|
467
556
|
const selectors = params.testSelectors ?? [];
|
|
468
557
|
let effective = selectors;
|
|
469
558
|
let truncated = false;
|
|
@@ -486,6 +575,11 @@ export async function runExternalTests(params, deps = defaultDeps) {
|
|
|
486
575
|
if (params.stateFile) {
|
|
487
576
|
await persistExternalRun(params.stateFile, params.repository, {
|
|
488
577
|
...skippedRun,
|
|
578
|
+
files: [],
|
|
579
|
+
// No run config resolved, so there is no suite to identify — and a service
|
|
580
|
+
// name here would read as one. The reader never keys on it (the record is
|
|
581
|
+
// skipped, and skipped records are dropped first), so name it for what it is.
|
|
582
|
+
suite: NO_RUNNABLE_SUITE,
|
|
489
583
|
mode,
|
|
490
584
|
ranAt: deps.now(),
|
|
491
585
|
repository: params.repository,
|
|
@@ -518,6 +612,8 @@ export async function runExternalTests(params, deps = defaultDeps) {
|
|
|
518
612
|
if (params.stateFile) {
|
|
519
613
|
await persistExternalRun(params.stateFile, params.repository, {
|
|
520
614
|
...routedOut,
|
|
615
|
+
files: [],
|
|
616
|
+
suite: suiteIdOf(cfg),
|
|
521
617
|
mode,
|
|
522
618
|
ranAt: deps.now(),
|
|
523
619
|
repository: params.repository,
|
|
@@ -542,8 +638,13 @@ export async function runExternalTests(params, deps = defaultDeps) {
|
|
|
542
638
|
};
|
|
543
639
|
perSuite.push(suiteResult);
|
|
544
640
|
if (params.stateFile) {
|
|
641
|
+
const files = summarizeFilesInRun(suiteResult.results, {
|
|
642
|
+
partialFiles: partialFilesFrom(routing.kept),
|
|
643
|
+
});
|
|
545
644
|
await persistExternalRun(params.stateFile, params.repository, {
|
|
546
645
|
...suiteResult,
|
|
646
|
+
files,
|
|
647
|
+
suite: suiteIdOf(cfg),
|
|
547
648
|
mode,
|
|
548
649
|
ranAt: deps.now(),
|
|
549
650
|
repository: params.repository,
|
|
@@ -575,8 +676,7 @@ export function registerRunExistingTestsTool(server) {
|
|
|
575
676
|
.describe("Spec files to run (paths relative to the run command's cwd), optionally `file::test title` to also filter by name. Omit to run the whole configured suite."),
|
|
576
677
|
mode: z
|
|
577
678
|
.enum(["confirm", "verify"])
|
|
578
|
-
.
|
|
579
|
-
.describe("confirm (default): scoped run to establish which tests a change breaks; disables the suite's fail-fast cap (e.g. Playwright's maxFailures) so ALL breakages are counted. verify: re-run after an edit."),
|
|
679
|
+
.describe("REQUIRED, and the only thing that marks a run as before or after the edit: confirm = the pre-edit baseline (a scoped run establishing which tests the change breaks; disables the suite's fail-fast cap, e.g. Playwright's maxFailures, so ALL breakages are counted), verify = the re-run after you edited. Passing confirm for a post-edit run overwrites the baseline with the fixed result, and the report then shows a repair as having never been broken."),
|
|
580
680
|
project: z
|
|
581
681
|
.string()
|
|
582
682
|
.optional()
|
|
@@ -635,7 +735,7 @@ export function registerRunExistingTestsTool(server) {
|
|
|
635
735
|
AnalyticsService.pushMCPToolEvent(TOOL_NAME, errorResult, {
|
|
636
736
|
workspacePath: params.workspacePath,
|
|
637
737
|
service: params.service ?? "",
|
|
638
|
-
mode: params.mode
|
|
738
|
+
mode: params.mode,
|
|
639
739
|
}).catch((err) => {
|
|
640
740
|
logger.warning("Analytics event failed", { error: String(err) });
|
|
641
741
|
});
|
|
@@ -84,6 +84,96 @@ const MAINTENANCE_CHANGE_ACTIONS = new Set([
|
|
|
84
84
|
DriftAction.Regenerate,
|
|
85
85
|
DriftAction.Delete,
|
|
86
86
|
]);
|
|
87
|
+
const EXTERNAL_TO_EXECUTION_STATUS = {
|
|
88
|
+
pass: TestExecutionStatus.Pass,
|
|
89
|
+
fail: TestExecutionStatus.Fail,
|
|
90
|
+
error: TestExecutionStatus.Error,
|
|
91
|
+
};
|
|
92
|
+
/** Does this run's file outcome describe `knownPath`?
|
|
93
|
+
*
|
|
94
|
+
* `absoluteFile` is the framework's own path resolved against the root it reported,
|
|
95
|
+
* so it names one file and is compared exactly. An outcome without one is refused:
|
|
96
|
+
* a relative spelling is a suffix, and a suffix cannot separate
|
|
97
|
+
* `packages/a/tests/x.spec.ts` from `packages/b/tests/x.spec.ts`. Guessing there and
|
|
98
|
+
* guarding the guess is what this deliberately does not do — a missing status is a
|
|
99
|
+
* row that reads Unknown, which is what it reads today; a wrong one is a false claim
|
|
100
|
+
* about someone's tests.
|
|
101
|
+
*
|
|
102
|
+
* The cost is real and accepted: a suite running inside a container reports the
|
|
103
|
+
* container's root, which cannot equal the host path discovery recorded, so those
|
|
104
|
+
* repos get no status until the run records a root the report can line up. */
|
|
105
|
+
function outcomeMatches(outcome, knownPath) {
|
|
106
|
+
return (typeof outcome.absoluteFile === "string" &&
|
|
107
|
+
path.resolve(outcome.absoluteFile) === path.resolve(knownPath));
|
|
108
|
+
}
|
|
109
|
+
/** One line describing what the suite saw, for a row whose status came from the run
|
|
110
|
+
* rather than from an execution the agent drafted a summary for. Without it the
|
|
111
|
+
* rendered cell is a bare `Error` or `Pass` with nothing behind it. */
|
|
112
|
+
function describeOutcome(o) {
|
|
113
|
+
// Defended like the record reads around it: a malformed outcome must not throw
|
|
114
|
+
// out of the handler and take the whole report submission with it.
|
|
115
|
+
const c = o.counts ?? { pass: 0, fail: 0, error: 0, skipped: 0 };
|
|
116
|
+
const parts = [
|
|
117
|
+
c.pass ? `${c.pass} passed` : "",
|
|
118
|
+
c.fail ? `${c.fail} failed` : "",
|
|
119
|
+
c.error ? `${c.error} errored` : "",
|
|
120
|
+
c.skipped ? `${c.skipped} skipped` : "",
|
|
121
|
+
].filter(Boolean);
|
|
122
|
+
const head = `${parts.join(", ") || "no tests reported"} in the repo's own suite`;
|
|
123
|
+
// Adapter messages carry stack traces, so collapse whitespace before truncating:
|
|
124
|
+
// an embedded newline would break the maintenance row this lands in.
|
|
125
|
+
const first = (o.errors ?? [])[0]?.replace(/\s+/g, " ").trim();
|
|
126
|
+
return first ? `${head} — ${first.slice(0, 200)}` : head;
|
|
127
|
+
}
|
|
128
|
+
function externalStatusesFor(records, knownPath) {
|
|
129
|
+
// LATEST per suite, then WORST across suites. Both halves matter: a suite re-run
|
|
130
|
+
// after another edit supersedes its own earlier verdict, so an obsolete failure
|
|
131
|
+
// must not outlive the fix that repaired it; but two different suites can each
|
|
132
|
+
// own the same file, and a pass in one must not mask a failure in the other.
|
|
133
|
+
const RANK = { pass: 0, fail: 1, error: 2 };
|
|
134
|
+
const statusIn = (mode) => {
|
|
135
|
+
const latestPerSuite = new Map();
|
|
136
|
+
for (const record of records ?? []) {
|
|
137
|
+
if (record.mode !== mode || record.skipped || !record.environmentHealthy)
|
|
138
|
+
continue;
|
|
139
|
+
// A suite that stopped on --max-failures left tests unrun and does not say
|
|
140
|
+
// which, so no file it touched can stand as covered. The adapter reports this;
|
|
141
|
+
// it is not a guess. (The tool forces --max-failures=0 for confirm, so this is
|
|
142
|
+
// reachable on a verify re-run under the repo's own cap.)
|
|
143
|
+
if (record.diagnostics?.maxFailuresBail)
|
|
144
|
+
continue;
|
|
145
|
+
const hit = (record.files ?? []).find((o) => outcomeMatches(o, knownPath));
|
|
146
|
+
if (!hit)
|
|
147
|
+
continue;
|
|
148
|
+
const key = record.suite ?? "";
|
|
149
|
+
const prior = latestPerSuite.get(key);
|
|
150
|
+
// `ranAt` is an ISO timestamp, so a lexical compare is chronological. The
|
|
151
|
+
// newest run wins BEFORE it is judged: a later partial re-run supersedes an
|
|
152
|
+
// earlier complete one, and must leave the row unknown rather than let the
|
|
153
|
+
// stale complete outcome stand in for it.
|
|
154
|
+
if (!prior || record.ranAt >= prior.at)
|
|
155
|
+
latestPerSuite.set(key, { at: record.ranAt, outcome: hit });
|
|
156
|
+
}
|
|
157
|
+
let worst;
|
|
158
|
+
for (const { outcome } of latestPerSuite.values()) {
|
|
159
|
+
// `skipped` means every test in the file was skipped — it was not exercised,
|
|
160
|
+
// and TestExecutionStatus.Skipped means a deliberate decision not to run it.
|
|
161
|
+
if (!outcome.complete || outcome.status === "skipped")
|
|
162
|
+
continue;
|
|
163
|
+
if (!worst || RANK[outcome.status] > RANK[worst.status])
|
|
164
|
+
worst = outcome;
|
|
165
|
+
}
|
|
166
|
+
return worst;
|
|
167
|
+
};
|
|
168
|
+
const before = statusIn("confirm");
|
|
169
|
+
const after = statusIn("verify");
|
|
170
|
+
return {
|
|
171
|
+
before: before && EXTERNAL_TO_EXECUTION_STATUS[before.status],
|
|
172
|
+
after: after && EXTERNAL_TO_EXECUTION_STATUS[after.status],
|
|
173
|
+
beforeDetail: before && describeOutcome(before),
|
|
174
|
+
afterDetail: after && describeOutcome(after),
|
|
175
|
+
};
|
|
176
|
+
}
|
|
87
177
|
const TOOL_NAME = "skyramp_submit_report";
|
|
88
178
|
const DEFAULT_COMMIT_MESSAGE = "Added recommendations by Skyramp Testbot.";
|
|
89
179
|
// Per-repo attribution. In a multi-repo run, every report item carries the
|
|
@@ -553,14 +643,14 @@ const testMaintenanceSchema = z.object({
|
|
|
553
643
|
"For passing runs: count and timing, e.g. '4 passed in 15.09s'. " +
|
|
554
644
|
"For failing runs: failure name and one-line root cause, e.g. " +
|
|
555
645
|
"'FAILED test_foo — assert 403 got 200, auth middleware not enforced'. " +
|
|
556
|
-
"
|
|
646
|
+
"Pass an empty string when you ran nothing yourself — the server then fills this from the repo suite's own run — and for VERIFY/IGNORE entries where nothing ran at all. Omit the whole entry, never this field."),
|
|
557
647
|
afterDetails: z
|
|
558
648
|
.string()
|
|
559
649
|
.describe("One line only — no embedded newlines, no raw HTTP headers or JSON blobs. " +
|
|
560
650
|
"For passing runs: count and timing, e.g. '5 passed in 10.96s'. " +
|
|
561
651
|
"For failing runs: failure name and one-line root cause, e.g. " +
|
|
562
652
|
"'FAILED test_foo — check_schema fails, order_id=1 has discount from prior PATCH test'. " +
|
|
563
|
-
"
|
|
653
|
+
"Pass an empty string when you ran nothing yourself — the server then fills this from the repo suite's own run — and for VERIFY/IGNORE/DELETE entries where nothing ran at all. Omit the whole entry, never this field."),
|
|
564
654
|
// Server-populated from stateFile execution records — never supplied by the LLM.
|
|
565
655
|
beforeStatus: z.nativeEnum(TestExecutionStatus),
|
|
566
656
|
afterStatus: z.nativeEnum(TestExecutionStatus),
|
|
@@ -1011,12 +1101,42 @@ export function registerSubmitReportTool(server) {
|
|
|
1011
1101
|
: MAINTENANCE_CHANGE_ACTIONS.has(m.action)
|
|
1012
1102
|
? TestExecutionStatus.Unknown
|
|
1013
1103
|
: TestExecutionStatus.Skipped;
|
|
1014
|
-
|
|
1015
|
-
|
|
1104
|
+
// A suite run answers only for a row whose default is Unknown. VERIFY and
|
|
1105
|
+
// IGNORE are Skipped by decision, not for want of a result, and testbot
|
|
1106
|
+
// drops both from the rendered table.
|
|
1107
|
+
const external = externalStatusesFor(stateData.externalTestResults, m.testFilePath);
|
|
1108
|
+
const beforeStatus = recorded?.executionBefore?.status ??
|
|
1109
|
+
(defaultBeforeStatus === TestExecutionStatus.Unknown
|
|
1110
|
+
? external.before
|
|
1111
|
+
: undefined) ??
|
|
1112
|
+
defaultBeforeStatus;
|
|
1113
|
+
const afterStatus = recorded?.executionAfter?.status ??
|
|
1114
|
+
(defaultAfterStatus === TestExecutionStatus.Unknown
|
|
1115
|
+
? external.after
|
|
1116
|
+
: undefined) ??
|
|
1117
|
+
defaultAfterStatus;
|
|
1016
1118
|
// Trim before checking — a whitespace-only string is semantically blank and
|
|
1017
1119
|
// must not satisfy the "drafted a summary" requirement.
|
|
1018
|
-
|
|
1019
|
-
|
|
1120
|
+
// The agent's drafted summary wins; otherwise a status that came from
|
|
1121
|
+
// the suite run carries the run's own line, so the rendered cell is
|
|
1122
|
+
// never a bare verdict with nothing behind it.
|
|
1123
|
+
//
|
|
1124
|
+
// Only when there is NO recorded execution. A file the agent ran itself
|
|
1125
|
+
// owes a drafted summary, and filling it here would satisfy the check
|
|
1126
|
+
// below on the agent's behalf whenever the two statuses happened to
|
|
1127
|
+
// agree — silently dropping the requirement.
|
|
1128
|
+
//
|
|
1129
|
+
// And only where the STATUS came from the suite too. A VERIFY or DELETE
|
|
1130
|
+
// row keeps its Skipped default, so a line saying "1 passed in the repo's
|
|
1131
|
+
// own suite" beside it contradicts the very cell it annotates.
|
|
1132
|
+
const beforeFromSuite = defaultBeforeStatus === TestExecutionStatus.Unknown &&
|
|
1133
|
+
!recorded?.executionBefore;
|
|
1134
|
+
const afterFromSuite = defaultAfterStatus === TestExecutionStatus.Unknown &&
|
|
1135
|
+
!recorded?.executionAfter;
|
|
1136
|
+
const beforeDetails = detail?.beforeDetails?.trim() ||
|
|
1137
|
+
(beforeFromSuite ? (external.beforeDetail ?? "") : "");
|
|
1138
|
+
const afterDetails = detail?.afterDetails?.trim() ||
|
|
1139
|
+
(afterFromSuite ? (external.afterDetail ?? "") : "");
|
|
1020
1140
|
if (recorded?.executionBefore && !beforeDetails)
|
|
1021
1141
|
missingDetails.push(`${displayName} (beforeDetails)`);
|
|
1022
1142
|
if (recorded?.executionAfter && !afterDetails)
|
|
@@ -11,8 +11,15 @@ export type ExternalTestStatus = "pass" | "fail" | "error" | "skipped";
|
|
|
11
11
|
export interface ExternalTestResult {
|
|
12
12
|
/** Human-readable identity, e.g. `<file> › <describe…> › <title>`. */
|
|
13
13
|
testId: string;
|
|
14
|
-
/** Spec file the test lives in
|
|
14
|
+
/** Spec file the test lives in, in the framework's own spelling — usually absolute
|
|
15
|
+
* for jest and mocha, relative to the framework's root for Playwright and pytest.
|
|
16
|
+
* No framework guarantees which, so never assume: use `absoluteFile` when present. */
|
|
15
17
|
file: string;
|
|
18
|
+
/** `file` resolved against the root the framework reported, when it reported one.
|
|
19
|
+
* A relative spelling alone cannot identify a file: `tests/Button.test.tsx` fits
|
|
20
|
+
* every package that has one. Consumers matching a result to a known path must
|
|
21
|
+
* prefer this and compare exactly. */
|
|
22
|
+
absoluteFile?: string;
|
|
16
23
|
status: ExternalTestStatus;
|
|
17
24
|
/** Failure/error text (ANSI-stripped). Present only for fail/error. */
|
|
18
25
|
message?: string;
|
|
@@ -55,10 +62,69 @@ export interface ParsedExternalRun {
|
|
|
55
62
|
/** Why the run was skipped (present only when `skipped`). */
|
|
56
63
|
skipReason?: string;
|
|
57
64
|
}
|
|
65
|
+
/**
|
|
66
|
+
* What one run established about ONE spec file. Computed by the run itself, which
|
|
67
|
+
* knows what it asked the framework for; the report then reads this instead of
|
|
68
|
+
* re-deriving coverage from the flat result list, where the facts below are no
|
|
69
|
+
* longer recoverable.
|
|
70
|
+
*/
|
|
71
|
+
export interface ExternalFileOutcome {
|
|
72
|
+
/** The framework's own spelling, kept for diagnostics. */
|
|
73
|
+
file: string;
|
|
74
|
+
/** `file` resolved against the root the framework reported, when it reported one.
|
|
75
|
+
* Present here exactly when it was present on the results. */
|
|
76
|
+
absoluteFile?: string;
|
|
77
|
+
/** Worst outcome across the file's tests: error > fail > pass. `skipped` means
|
|
78
|
+
* every test in it was skipped, so the file was not exercised at all. */
|
|
79
|
+
status: ExternalTestStatus;
|
|
80
|
+
/** How many tests produced each outcome. Rendered into the maintenance row's
|
|
81
|
+
* before/after detail when the agent drafted none. */
|
|
82
|
+
counts: {
|
|
83
|
+
pass: number;
|
|
84
|
+
fail: number;
|
|
85
|
+
error: number;
|
|
86
|
+
skipped: number;
|
|
87
|
+
};
|
|
88
|
+
/**
|
|
89
|
+
* Did this run exercise every TEST in the file? False when a `file::test title`
|
|
90
|
+
* selector narrowed it — the one narrowing this tool applies per file, and so the
|
|
91
|
+
* only one it can state with certainty. A partial run says nothing about the tests
|
|
92
|
+
* it did not reach, so it cannot stand as the file's status.
|
|
93
|
+
*
|
|
94
|
+
* Two narrowings are deliberately NOT reflected here, and the reader compensates
|
|
95
|
+
* for one of them:
|
|
96
|
+
* - A `file::title` selector also becomes a run-WIDE --grep / -t on every
|
|
97
|
+
* framework except pytest, so a co-selected file ran only its matching tests
|
|
98
|
+
* while this says complete. Known and accepted.
|
|
99
|
+
* - A suite that bailed on --max-failures abandoned tests it cannot name. That one
|
|
100
|
+
* IS reported, in `diagnostics.maxFailuresBail`, and the report skips such a
|
|
101
|
+
* record outright rather than reading any file from it.
|
|
102
|
+
*
|
|
103
|
+
* A Playwright `project` scope does NOT make a file incomplete: every test in it
|
|
104
|
+
* still ran, in one project rather than all of them. That is a different axis, and
|
|
105
|
+
* discarding the status over it throws away the run's whole verdict — measured on
|
|
106
|
+
* eval run 33280302496, where it turned a real Error -> Pass into Unknown.
|
|
107
|
+
*
|
|
108
|
+
* Known gap: a repo whose runner bails on its own (a jest or mocha `bail` in
|
|
109
|
+
* config rather than on the command line) is not detectable from the report,
|
|
110
|
+
* and only Playwright reports a --max-failures bail back to us.
|
|
111
|
+
*/
|
|
112
|
+
complete: boolean;
|
|
113
|
+
/** Failure/error text for the tests that did not pass. */
|
|
114
|
+
errors: string[];
|
|
115
|
+
durationMs: number;
|
|
116
|
+
}
|
|
58
117
|
/** One persisted run of the external suite, appended to
|
|
59
118
|
* UnifiedAnalysisState.externalTestResults so drift analysis / the report can
|
|
60
119
|
* fold in CONFIRMED failures instead of guesses. */
|
|
61
120
|
export interface ExternalTestRunRecord extends ParsedExternalRun {
|
|
121
|
+
/** Per-file outcome, one entry per spec file this run produced results for.
|
|
122
|
+
* Empty for a skipped run. */
|
|
123
|
+
files: ExternalFileOutcome[];
|
|
124
|
+
/** Which configured suite produced this record. A repo can declare several, and
|
|
125
|
+
* without this a re-run of ONE suite cannot be told from a second suite's first
|
|
126
|
+
* run — so a superseded failure would outlive the fix that repaired it. */
|
|
127
|
+
suite: string;
|
|
62
128
|
mode: "confirm" | "verify";
|
|
63
129
|
/** ISO timestamp when the run completed. */
|
|
64
130
|
ranAt: string;
|