@skyramp/mcp 0.4.1-rc.1 → 0.4.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/build/prompts/enhance-assertions/contractProviderAssertionsPrompt.js +2 -1
- package/build/prompts/enhance-assertions/integrationAssertionsPrompt.js +2 -1
- package/build/prompts/enhance-assertions/sharedAssertionRules.d.ts +1 -1
- package/build/prompts/enhance-assertions/sharedAssertionRules.js +41 -22
- package/build/prompts/enhance-assertions/uiAssertionsPrompt.js +17 -9
- package/build/prompts/test-recommendation/diffExecutionPlan.js +0 -2
- package/build/prompts/test-recommendation/test-recommendation-prompt.js +8 -2
- package/build/prompts/testbot/testbot-prompts.js +12 -7
- package/build/recommendation/registerPlan.d.ts +5 -1
- package/build/recommendation/registerPlan.js +5 -0
- package/build/recommendation/types.d.ts +25 -2
- package/build/recommendation/verifierContracts.d.ts +16 -4
- package/build/recommendation/verifierContracts.js +20 -4
- package/build/recommendation/verifiers/coverage.d.ts +10 -0
- package/build/recommendation/verifiers/coverage.js +144 -22
- package/build/recommendation/verifiers/deliveredMatchesPlan.d.ts +22 -0
- package/build/recommendation/verifiers/deliveredMatchesPlan.js +43 -0
- package/build/recommendation/verifiers/existingCoverage.js +53 -0
- package/build/recommendation/verifiers/expectedValueSourced.js +111 -14
- package/build/services/TestGenerationService.js +3 -1
- package/build/tools/code-refactor/codeReuseTool.js +1 -1
- package/build/tools/code-refactor/reuse-outcome.d.ts +1 -1
- package/build/tools/code-refactor/reuse-state.d.ts +85 -7
- package/build/tools/code-refactor/reuse-state.js +239 -34
- package/build/tools/code-refactor/utils-verify-gates.d.ts +5 -0
- package/build/tools/code-refactor/utils-verify-gates.js +103 -11
- package/build/tools/generate-tests/generateBatchScenarioRestTool.js +2 -1
- package/build/tools/submitReportTool.js +259 -38
- package/build/tools/test-management/actionsTool.js +5 -0
- package/build/tools/test-management/analyzeChangesTool.d.ts +53 -0
- package/build/tools/test-management/analyzeChangesTool.js +55 -2
- package/build/tools/test-management/registerTestPlanTool.d.ts +2 -1
- package/build/tools/test-management/registerTestPlanTool.js +29 -14
- package/build/types/ReuseOutcome.d.ts +73 -7
- package/build/types/TestAnalysis.d.ts +6 -0
- package/build/types/TestbotReport.d.ts +15 -1
- package/build/utils/AnalysisStateManager.d.ts +7 -1
- package/build/utils/AnalysisStateManager.js +5 -1
- package/build/utils/assertion-verify/api-shared-lints.js +71 -34
- package/build/utils/assertion-verify/format.js +2 -2
- package/build/utils/assertion-verify/helper-imports.d.ts +7 -0
- package/build/utils/assertion-verify/helper-imports.js +119 -27
- package/build/utils/assertion-verify/lint-types.d.ts +31 -2
- package/build/utils/assertion-verify/lint-types.js +66 -0
- package/build/utils/assertion-verify/metrics.d.ts +13 -0
- package/build/utils/assertion-verify/metrics.js +16 -0
- package/build/utils/assertion-verify/verify.d.ts +11 -6
- package/build/utils/assertion-verify/verify.js +56 -15
- package/build/utils/canonicalJson.d.ts +11 -0
- package/build/utils/canonicalJson.js +17 -0
- package/build/utils/utils-verify/action-key.d.ts +27 -0
- package/build/utils/utils-verify/action-key.js +292 -0
- package/build/utils/utils-verify/allow.d.ts +8 -1
- package/build/utils/utils-verify/allow.js +14 -1
- package/build/utils/utils-verify/call-sites.d.ts +76 -8
- package/build/utils/utils-verify/call-sites.js +256 -70
- package/build/utils/utils-verify/language-spec.d.ts +3 -2
- package/build/utils/utils-verify/parse.d.ts +22 -3
- package/build/utils/utils-verify/parse.js +123 -52
- package/build/utils/utils-verify/verify.d.ts +33 -3
- package/build/utils/utils-verify/verify.js +126 -12
- package/build/utils/workspaceAuth.d.ts +59 -19
- package/build/utils/workspaceAuth.js +228 -31
- package/package.json +1 -1
- package/plugin/prompts/generate-tests/execution-plan.md +1 -1
- package/plugin/prompts/generate-tests/generation.md +1 -0
- package/plugin/prompts/plan-tests.md +33 -16
|
@@ -13,7 +13,7 @@ import { StateManager, runArtifactDir, getTestsRepoDir, resolveRunStatePath, } f
|
|
|
13
13
|
import { toolError, testFileMatches } from "../utils/utils.js";
|
|
14
14
|
import { isTestbotEnabled, } from "../utils/featureFlags.js";
|
|
15
15
|
import { answerFor, unknownAnswerObjections } from "../recommendation/answers.js";
|
|
16
|
-
import { checkDeliveredMatchesPlan } from "../recommendation/verifiers/deliveredMatchesPlan.js";
|
|
16
|
+
import { checkDeliveredMatchesPlan, checkMaintenanceDelivered } from "../recommendation/verifiers/deliveredMatchesPlan.js";
|
|
17
17
|
import { checkReportedCategoryMatchesPlan, checkRequirementConflictReported, } from "../recommendation/verifiers/reportedCategory.js";
|
|
18
18
|
import { checkDefectsReported, checkIssueTraceability } from "../recommendation/verifiers/issueTraceability.js";
|
|
19
19
|
import { checkExpectedOutcomeAfterExecution, } from "../recommendation/verifiers/expectedOutcome.js";
|
|
@@ -21,7 +21,7 @@ import { findInvalidSourceCitations, findUnchangedFileClaims, listChangedFiles,
|
|
|
21
21
|
import { isPlanOnlyMode } from "../utils/planOnlyMode.js";
|
|
22
22
|
import { getReportLanguage, isEnforcedReportLanguage, findLanguageViolations, findLanguageNearMisses, reportLanguageDisplayName, } from "../utils/reportLanguage.js";
|
|
23
23
|
import { canonicalTestPath, findAssertionRecordByFileName, rederiveAssertionOutcome, } from "./code-refactor/assertion-state.js";
|
|
24
|
-
import {
|
|
24
|
+
import { rederiveReuse, REUSE_SUBMIT_MAX_REFUSALS, reuseChainSkipped, samePath, } from "./code-refactor/reuse-state.js";
|
|
25
25
|
import { retrofitGate, } from "./code-refactor/retrofit-state.js";
|
|
26
26
|
// Mirrors the tools wired to planGuard: UI and E2E are not gated at generation
|
|
27
27
|
// time, so gating them here would be a report-time-only surprise.
|
|
@@ -51,6 +51,22 @@ const MAINTENANCE_CHANGE_ACTIONS = new Set([
|
|
|
51
51
|
DriftAction.Regenerate,
|
|
52
52
|
DriftAction.Delete,
|
|
53
53
|
]);
|
|
54
|
+
// Narrower than the set above, and for a different question. That one asks whether a
|
|
55
|
+
// row touched a file; this asks whether the row left a test standing that a `maintains`
|
|
56
|
+
// claim can rest on. DELETE touched the file and removed the coverage with it, VERIFY
|
|
57
|
+
// and IGNORE never edited anything — none of the three corroborate a claim that an
|
|
58
|
+
// existing test covers a change.
|
|
59
|
+
const MAINTENANCE_COVERAGE_ACTIONS = new Set([
|
|
60
|
+
DriftAction.Update,
|
|
61
|
+
DriftAction.Regenerate,
|
|
62
|
+
]);
|
|
63
|
+
/** The maintenance rows a `maintains` claim may be checked against: an action that leaves
|
|
64
|
+
* a test standing, and an edit this run actually made. `reportOnly` is the second half —
|
|
65
|
+
* an external test's REGENERATE keeps its real action and touches no file, so counting it
|
|
66
|
+
* would credit a claim nothing backs. */
|
|
67
|
+
function rowsLeavingCoverage(rows) {
|
|
68
|
+
return (rows ?? []).filter((row) => row?.action !== undefined && MAINTENANCE_COVERAGE_ACTIONS.has(row.action) && row?.reportOnly !== true);
|
|
69
|
+
}
|
|
54
70
|
/** Objection ids for an untested declared change, `coverage:change:<id>`. The
|
|
55
71
|
* change table reads the agent's answer back off them. */
|
|
56
72
|
const CHANGE_OBJECTION_PREFIX = "coverage:change:";
|
|
@@ -493,7 +509,6 @@ export const additionalRecommendationSchema = z.preprocess(stripNullGroundingFie
|
|
|
493
509
|
// `targetElements: null` is its normal state rather than a lost capture.
|
|
494
510
|
}
|
|
495
511
|
}));
|
|
496
|
-
// TODO(multi-repo maintenance): no `repository` field yet — see readData() TODO below.
|
|
497
512
|
const testMaintenanceSchema = z.object({
|
|
498
513
|
testType: z.nativeEnum(TestType).describe("Type of test."),
|
|
499
514
|
endpoint: z
|
|
@@ -523,6 +538,10 @@ const testMaintenanceSchema = z.object({
|
|
|
523
538
|
"For failing runs: failure name and one-line root cause, e.g. " +
|
|
524
539
|
"'FAILED test_foo — check_schema fails, order_id=1 has discount from prior PATCH test'. " +
|
|
525
540
|
"Empty string for VERIFY/IGNORE/DELETE entries where no after-execution was run."),
|
|
541
|
+
// Server-populated from the verdict's own state section, falling back to the
|
|
542
|
+
// run's primary repo. Always set: a consumer that has to treat absence as
|
|
543
|
+
// "probably the primary" cannot tell a single-repo row from a mis-stamped one.
|
|
544
|
+
repository: z.string(),
|
|
526
545
|
// Server-populated from stateFile execution records — never supplied by the LLM.
|
|
527
546
|
beforeStatus: z.nativeEnum(TestExecutionStatus),
|
|
528
547
|
afterStatus: z.nativeEnum(TestExecutionStatus),
|
|
@@ -601,7 +620,7 @@ function computeReportMetrics(params) {
|
|
|
601
620
|
*
|
|
602
621
|
* The counts are RE-DERIVED from the delivered spec rather than read back from
|
|
603
622
|
* state: the execution fix-up can restore `<testFile>.raw.bak` over it afterwards
|
|
604
|
-
* by plain `cp`, which this tool never sees. See
|
|
623
|
+
* by plain `cp`, which this tool never sees. See rederiveReuse.
|
|
605
624
|
*
|
|
606
625
|
* Tests the reuse tool never ran for and specs whose outcome cannot be re-derived
|
|
607
626
|
* get nothing — which consumers already treat as "no reuse summary" — with one
|
|
@@ -628,7 +647,12 @@ async function attachAssertionOutcome(test, outcomes, checkouts) {
|
|
|
628
647
|
return test;
|
|
629
648
|
return { ...test, assertions: await rederiveAssertionOutcome(record) };
|
|
630
649
|
}
|
|
631
|
-
async function attachReuseOutcome(test, outcomes, handOffs, retrofits
|
|
650
|
+
async function attachReuseOutcome(test, outcomes, handOffs, retrofits,
|
|
651
|
+
// A blocking verdict measured on the delivered files is collected here rather
|
|
652
|
+
// than written into the row: the caller refuses the report on it. Required, not
|
|
653
|
+
// defaulted — a caller that forgot it would ship the report with no refusal and
|
|
654
|
+
// no compiler error. See the refusal below the row map.
|
|
655
|
+
blocking) {
|
|
632
656
|
// POM records describe browser specs; a basename collision with an API test's
|
|
633
657
|
// fileName must not attach them there. A utils-path record is attachable anywhere.
|
|
634
658
|
const record = outcomes?.[path.basename(test.fileName)];
|
|
@@ -638,7 +662,10 @@ async function attachReuseOutcome(test, outcomes, handOffs, retrofits = []) {
|
|
|
638
662
|
const found = record && typeMatches && (test.testType === TestType.UI || record.utils)
|
|
639
663
|
? record
|
|
640
664
|
: undefined;
|
|
641
|
-
const
|
|
665
|
+
const rederived = found ? await rederiveReuse(found) : undefined;
|
|
666
|
+
const derived = rederived?.outcome;
|
|
667
|
+
if (rederived?.blocking)
|
|
668
|
+
blocking.push(rederived.blocking);
|
|
642
669
|
// Pre-existing generated tests this spec's reuse pass rewired (SKYR-4276 A4): the
|
|
643
670
|
// report names each with its recorded execution, so a reviewer sees that the
|
|
644
671
|
// module became a dependency of code they already owned — and that it still runs.
|
|
@@ -681,6 +708,13 @@ function attachVideoPath(row, videos) {
|
|
|
681
708
|
videoPath: videos?.[path.basename(row.testFilePath)]?.videoPath,
|
|
682
709
|
};
|
|
683
710
|
}
|
|
711
|
+
/** Two-space indent on every line, so a multi-line verdict reads as one bullet's body. */
|
|
712
|
+
function indent(text) {
|
|
713
|
+
return text
|
|
714
|
+
.split("\n")
|
|
715
|
+
.map((line) => ` ${line}`)
|
|
716
|
+
.join("\n");
|
|
717
|
+
}
|
|
684
718
|
function deduplicateById(items) {
|
|
685
719
|
const seen = new Set();
|
|
686
720
|
const result = [];
|
|
@@ -824,7 +858,7 @@ const PLANNABLE_TEST_TYPES = new Set([
|
|
|
824
858
|
function runPostExecutionChecks(plan, delivered, results, issues,
|
|
825
859
|
/** The plan-only lane delivers nothing, so there a planned test is proof
|
|
826
860
|
* enough for a bug. Passed in, not read here, so the checks stay pure. */
|
|
827
|
-
planOnly) {
|
|
861
|
+
planOnly, maintained = []) {
|
|
828
862
|
const shipped = delivered
|
|
829
863
|
// A smoke, fuzz or load entry is filtered out rather than reported as
|
|
830
864
|
// unplanned: a plan cannot hold one, so the objection would name a mistake the
|
|
@@ -835,6 +869,9 @@ planOnly) {
|
|
|
835
869
|
.map((plannedTestId) => ({ plannedTestId }));
|
|
836
870
|
return [
|
|
837
871
|
...runPostExecutionCheck("deliveredMatchesPlan", "deliveredMatchesPlan", () => checkDeliveredMatchesPlan(plan, shipped)),
|
|
872
|
+
// The maintenance mirror: a `maintains` entry credited coverage at plan time,
|
|
873
|
+
// so something has to check the file was edited.
|
|
874
|
+
...runPostExecutionCheck("deliveredMatchesPlan", "deliveredMatchesPlan:maintains", () => checkMaintenanceDelivered(plan, maintained)),
|
|
838
875
|
...runPostExecutionCheck("expectedOutcome", "expectedOutcome:executed", () => checkExpectedOutcomeAfterExecution(plan, collectExecutionOutcomes(delivered, results), issues)),
|
|
839
876
|
...runPostExecutionCheck("reportedCategory", "reportedCategory", () => checkReportedCategoryMatchesPlan(plan, delivered)),
|
|
840
877
|
...runPostExecutionCheck("requirementConflictReported", "requirementConflictReported", () => checkRequirementConflictReported(plan, delivered, issues)),
|
|
@@ -1060,8 +1097,8 @@ export function registerSubmitReportTool(server) {
|
|
|
1060
1097
|
const stateManager = StateManager.fromStatePath(stateFile);
|
|
1061
1098
|
let stateData;
|
|
1062
1099
|
try {
|
|
1063
|
-
//
|
|
1064
|
-
//
|
|
1100
|
+
// Primary section only; related sections are read where testMaintenance
|
|
1101
|
+
// is built, so each row keeps its own repo's executions.
|
|
1065
1102
|
stateData = await stateManager.readData();
|
|
1066
1103
|
}
|
|
1067
1104
|
catch (err) {
|
|
@@ -1127,29 +1164,74 @@ export function registerSubmitReportTool(server) {
|
|
|
1127
1164
|
return errorResult;
|
|
1128
1165
|
}
|
|
1129
1166
|
}
|
|
1130
|
-
//
|
|
1131
|
-
//
|
|
1132
|
-
//
|
|
1133
|
-
|
|
1134
|
-
//
|
|
1135
|
-
|
|
1136
|
-
for
|
|
1137
|
-
|
|
1138
|
-
|
|
1139
|
-
|
|
1167
|
+
// Every section that can hold verdicts: primary, then each related repo.
|
|
1168
|
+
// Read once — the file can carry a large diffText and readRepoData would
|
|
1169
|
+
// re-parse it twice per repo.
|
|
1170
|
+
const fullMaintenanceState = await stateManager.readFullState();
|
|
1171
|
+
// The state file's root names the primary only when a caller declared one
|
|
1172
|
+
// (or wrote first). GITHUB_REPOSITORY is the trigger event's own repo, so
|
|
1173
|
+
// it names the primary even for a run that passed neither repo argument.
|
|
1174
|
+
const primaryRepository = fullMaintenanceState?.metadata?.repository?.trim() ||
|
|
1175
|
+
process.env.GITHUB_REPOSITORY?.trim() ||
|
|
1176
|
+
undefined;
|
|
1177
|
+
const maintenanceSections = [{ data: stateData }];
|
|
1178
|
+
for (const [repo, section] of Object.entries(fullMaintenanceState?.relatedRepos ?? {})) {
|
|
1179
|
+
// Only these two keys resolve back to the primary, already collected
|
|
1180
|
+
// above. Any other key — a case variant, or another name over the same
|
|
1181
|
+
// checkout — is a SEPARATE store, so skipping it would discard its
|
|
1182
|
+
// verdicts; duplicates collapse per row below instead.
|
|
1183
|
+
if (!repo.trim())
|
|
1184
|
+
continue;
|
|
1185
|
+
if (repo === primaryRepository)
|
|
1186
|
+
continue;
|
|
1187
|
+
const data = section?.data;
|
|
1188
|
+
if (data)
|
|
1189
|
+
maintenanceSections.push({ repository: repo, data });
|
|
1190
|
+
}
|
|
1191
|
+
// Stamp every row, primary included and single-repo runs included — as the
|
|
1192
|
+
// prompt already requires of the other report sections. A section without
|
|
1193
|
+
// its own name is the primary's, so it takes the run's primary repo.
|
|
1194
|
+
const repositoryOf = (section) => section.repository ?? primaryRepository ?? "";
|
|
1195
|
+
// Required field, so refuse rather than stamp a blank one: nothing named
|
|
1196
|
+
// the primary and the run is not on a GitHub runner either. Only sections
|
|
1197
|
+
// that actually produce rows need a name — a run with no verdicts at all
|
|
1198
|
+
// has nothing to attribute and must still be able to submit.
|
|
1199
|
+
if (!primaryRepository &&
|
|
1200
|
+
maintenanceSections.some((section) => !section.repository &&
|
|
1201
|
+
(section.data.maintenanceVerdicts?.length ?? 0) > 0)) {
|
|
1202
|
+
errorResult = toolError("Cannot name the repository for the maintenance rows: the stateFile records no primary repository " +
|
|
1203
|
+
"and GITHUB_REPOSITORY is unset. Pass `repository` (and `primaryRepository` in a multi-repo run) " +
|
|
1204
|
+
"to skyramp_analyze_changes, then re-run it.");
|
|
1205
|
+
return errorResult;
|
|
1140
1206
|
}
|
|
1141
|
-
|
|
1142
|
-
|
|
1143
|
-
|
|
1207
|
+
// Reject rather than throw: a non-array here, or a missing testFilePath
|
|
1208
|
+
// below, escapes the handler and loses the analytics event in its `finally`.
|
|
1209
|
+
const malformedSections = maintenanceSections
|
|
1210
|
+
.filter(({ data }) => data.maintenanceVerdicts !== undefined &&
|
|
1211
|
+
!Array.isArray(data.maintenanceVerdicts))
|
|
1212
|
+
.map(({ repository }) => repository ?? "the primary repo");
|
|
1213
|
+
if (malformedSections.length > 0) {
|
|
1214
|
+
errorResult = toolError(`stateFile has a non-array maintenanceVerdicts for: ${malformedSections.join(", ")}. ` +
|
|
1215
|
+
`Re-run skyramp_actions for that repository to rewrite the section.`);
|
|
1144
1216
|
return errorResult;
|
|
1145
1217
|
}
|
|
1146
|
-
|
|
1147
|
-
//
|
|
1148
|
-
//
|
|
1218
|
+
const badVerdicts = [];
|
|
1219
|
+
// action/testType/endpoint/description come entirely from each section's
|
|
1220
|
+
// maintenanceVerdicts (written by skyramp_actions right after drift analysis) —
|
|
1221
|
+
// the LLM never supplies testMaintenance directly, so there's nothing to
|
|
1222
|
+
// re-transcribe here.
|
|
1149
1223
|
// `pomFile` rides along per row and is stripped from the wire format below,
|
|
1150
1224
|
// like testFilePath. It cannot be keyed by spec path: a spec emits one UPDATE
|
|
1151
1225
|
// verdict PER page object, so a spec-keyed map keeps only the last.
|
|
1152
|
-
const rawMaintenance = (
|
|
1226
|
+
const rawMaintenance = maintenanceSections.flatMap((section) => (section.data.maintenanceVerdicts ?? [])
|
|
1227
|
+
.filter((v) => {
|
|
1228
|
+
if (v && typeof v.testFilePath === "string" && v.testFilePath.trim())
|
|
1229
|
+
return true;
|
|
1230
|
+
badVerdicts.push(section.repository ?? "the primary repo");
|
|
1231
|
+
return false;
|
|
1232
|
+
})
|
|
1233
|
+
.map((v) => ({
|
|
1234
|
+
repository: repositoryOf(section),
|
|
1153
1235
|
testFilePath: v.testFilePath,
|
|
1154
1236
|
pomFile: v.pomFile,
|
|
1155
1237
|
testType: v.testType,
|
|
@@ -1162,20 +1244,47 @@ export function registerSubmitReportTool(server) {
|
|
|
1162
1244
|
: v.rationale,
|
|
1163
1245
|
beforeDetails: "",
|
|
1164
1246
|
afterDetails: "",
|
|
1165
|
-
}));
|
|
1247
|
+
})));
|
|
1248
|
+
// GitHub slugs are case-insensitive, so two sections can name one repo and
|
|
1249
|
+
// repeat a file. Keep the first — the primary's, given the order above.
|
|
1250
|
+
// pomFile is part of the key: one spec emits one UPDATE verdict per page
|
|
1251
|
+
// object, and those rows are distinct, not duplicates.
|
|
1252
|
+
const seenRows = new Set();
|
|
1253
|
+
const dedupedMaintenance = rawMaintenance.filter((row) => {
|
|
1254
|
+
const key = `${(row.repository ?? "").trim().toLowerCase()}\u0000${row.testFilePath}\u0000${row.pomFile ?? ""}`;
|
|
1255
|
+
if (seenRows.has(key))
|
|
1256
|
+
return false;
|
|
1257
|
+
seenRows.add(key);
|
|
1258
|
+
return true;
|
|
1259
|
+
});
|
|
1260
|
+
if (badVerdicts.length > 0) {
|
|
1261
|
+
errorResult = toolError(`${badVerdicts.length} maintenance verdict(s) in the stateFile have no testFilePath ` +
|
|
1262
|
+
`(sections: ${[...new Set(badVerdicts)].join(", ")}). ` +
|
|
1263
|
+
`Re-run skyramp_actions for those repositories to rewrite the section.`);
|
|
1264
|
+
return errorResult;
|
|
1265
|
+
}
|
|
1266
|
+
// Keyed by the row's own repo, so executions come from its own section.
|
|
1267
|
+
const sectionByRepository = new Map(maintenanceSections.map((section) => [
|
|
1268
|
+
repositoryOf(section),
|
|
1269
|
+
section.data,
|
|
1270
|
+
]));
|
|
1166
1271
|
let testMaintenance;
|
|
1167
1272
|
// Rows missing a required detail (a real execution happened but the LLM didn't
|
|
1168
1273
|
// draft a summary for it) — collected here and rejected below, rather than
|
|
1169
1274
|
// silently shipping a blank field.
|
|
1170
1275
|
const missingDetails = [];
|
|
1171
|
-
if (
|
|
1276
|
+
if (dedupedMaintenance.length > 0) {
|
|
1172
1277
|
// beforeStatus/afterStatus are always stateFile-authoritative. beforeDetails/
|
|
1173
1278
|
// afterDetails come from the LLM's drafted summary (testMaintenanceDetails) —
|
|
1174
1279
|
// required whenever the stateFile shows a matching execution actually ran.
|
|
1175
|
-
testMaintenance =
|
|
1176
|
-
const recorded =
|
|
1280
|
+
testMaintenance = dedupedMaintenance.map((m) => {
|
|
1281
|
+
const recorded = sectionByRepository
|
|
1282
|
+
.get(m.repository)
|
|
1283
|
+
?.existingTests?.find((t) => testFileMatches(t.testFile, m.testFilePath));
|
|
1177
1284
|
const detail = params.testMaintenanceDetails?.find((d) => d.testFilePath === m.testFilePath);
|
|
1178
|
-
|
|
1285
|
+
// Two repos can hold the same basename; name the repo so the agent
|
|
1286
|
+
// knows which row to draft for.
|
|
1287
|
+
const displayName = `${path.basename(m.testFilePath)} (${m.repository})`;
|
|
1179
1288
|
const defaultBeforeStatus = MAINTENANCE_CHANGE_ACTIONS.has(m.action)
|
|
1180
1289
|
? TestExecutionStatus.Unknown
|
|
1181
1290
|
: TestExecutionStatus.Skipped;
|
|
@@ -1205,7 +1314,8 @@ export function registerSubmitReportTool(server) {
|
|
|
1205
1314
|
});
|
|
1206
1315
|
}
|
|
1207
1316
|
if (missingDetails.length > 0) {
|
|
1208
|
-
|
|
1317
|
+
const uniqueMissing = [...new Set(missingDetails)];
|
|
1318
|
+
errorResult = toolError(`${uniqueMissing.length} maintenance row(s) have a recorded execution but no drafted summary: ${uniqueMissing.join(", ")}. ` +
|
|
1209
1319
|
"Add a testMaintenanceDetails entry with the missing field(s), drafted from the execution output you already saw.");
|
|
1210
1320
|
return errorResult;
|
|
1211
1321
|
}
|
|
@@ -1243,6 +1353,11 @@ export function registerSubmitReportTool(server) {
|
|
|
1243
1353
|
...Object.values(fullState?.relatedRepos ?? {}).map((section) => section.repositoryPath),
|
|
1244
1354
|
])),
|
|
1245
1355
|
];
|
|
1356
|
+
// Primary verdicts only, by design: this checks UPDATEs, and
|
|
1357
|
+
// skyramp_actions downgrades every actionable related-repo verdict to
|
|
1358
|
+
// VERIFY (actionsTool.ts). If that changes, pass each section's
|
|
1359
|
+
// verdicts with its OWN checkout root — isBacked relativizes against
|
|
1360
|
+
// `repoRoot` and would otherwise exempt them all.
|
|
1246
1361
|
const unbacked = findUnchangedFileClaims({
|
|
1247
1362
|
repoRoot,
|
|
1248
1363
|
changedFiles,
|
|
@@ -1331,10 +1446,70 @@ export function registerSubmitReportTool(server) {
|
|
|
1331
1446
|
const assertionCheckouts = await stateManager
|
|
1332
1447
|
.listRepoCheckouts()
|
|
1333
1448
|
.catch(() => []);
|
|
1449
|
+
const reuseBlocking = [];
|
|
1334
1450
|
const sanitizedNewTests = await Promise.all(dedupedNewTests.map(async ({ scenarioFile: _sf, traceFile: _tf, frontendTrace: _ft, ...rest }) => {
|
|
1335
|
-
const row = await attachReuseOutcome(normalizeRepository(rest), stateData.reuseOutcomes, stateData.reuseHandOffs, retrofitViews);
|
|
1451
|
+
const row = await attachReuseOutcome(normalizeRepository(rest), stateData.reuseOutcomes, stateData.reuseHandOffs, retrofitViews, reuseBlocking);
|
|
1336
1452
|
return attachAssertionOutcome(row, stateData.assertionOutcomes ?? {}, assertionCheckouts);
|
|
1337
1453
|
}));
|
|
1454
|
+
// A blocking reuse verdict on the delivered files refuses the report. The
|
|
1455
|
+
// live verify pass is a one-time checkpoint on a file that keeps changing —
|
|
1456
|
+
// the execution fix loop and a post-verification rewrite both edit after it —
|
|
1457
|
+
// and the verdict re-derived here is the only one measured on what ships. It
|
|
1458
|
+
// used to become a row field, so the run committed the fault, detected it,
|
|
1459
|
+
// disclosed it, and delivered it. Reject rather than publish it, exactly like
|
|
1460
|
+
// the maintenanceVerdicts check above: this is a repair loop, not an abort —
|
|
1461
|
+
// the agent fixes the file (or documents the decline the check names) and
|
|
1462
|
+
// calls again. Advisories never reach this list, so a report-time advisory
|
|
1463
|
+
// stays reportable.
|
|
1464
|
+
//
|
|
1465
|
+
// BOUNDED, per spec: after REUSE_SUBMIT_MAX_REFUSALS refusals of one spec the
|
|
1466
|
+
// report is accepted and its row ships `verification: failed` with the
|
|
1467
|
+
// reasons. A loop with no exit at the last step of a run delivers nothing when
|
|
1468
|
+
// the agent cannot reach a pass, and a customer with no pull request is worse
|
|
1469
|
+
// off than one with a disclosed fault (see REUSE_SUBMIT_MAX_REFUSALS). The
|
|
1470
|
+
// count is persisted per spec path; a count that cannot be persisted cannot
|
|
1471
|
+
// bound anything, so a failed write accepts rather than refuses forever.
|
|
1472
|
+
// `reuseOutcomes` is keyed by basename, so two rows sharing a file name find
|
|
1473
|
+
// one record and push one verdict twice — deduplicated by path here, as
|
|
1474
|
+
// pendingReuseVerification skips that case on its side. Sorted by file so the
|
|
1475
|
+
// text is stable across calls.
|
|
1476
|
+
if (reuseBlocking.length > 0) {
|
|
1477
|
+
const verdicts = [
|
|
1478
|
+
...new Map(reuseBlocking.map((v) => [v.file, v])).values(),
|
|
1479
|
+
].sort((a, b) => a.file.localeCompare(b.file));
|
|
1480
|
+
const refusals = { ...(stateData.reuseRefusals ?? {}) };
|
|
1481
|
+
const owed = verdicts.filter((v) => (refusals[v.file] ?? 0) < REUSE_SUBMIT_MAX_REFUSALS);
|
|
1482
|
+
let persisted = owed.length > 0;
|
|
1483
|
+
if (owed.length > 0) {
|
|
1484
|
+
for (const v of owed)
|
|
1485
|
+
refusals[v.file] = (refusals[v.file] ?? 0) + 1;
|
|
1486
|
+
try {
|
|
1487
|
+
await stateManager.writeData({ ...stateData, reuseRefusals: refusals });
|
|
1488
|
+
}
|
|
1489
|
+
catch (err) {
|
|
1490
|
+
persisted = false;
|
|
1491
|
+
logger.warning("Could not persist the reuse refusal count — accepting the report rather than refusing without a bound", { error: String(err) });
|
|
1492
|
+
}
|
|
1493
|
+
}
|
|
1494
|
+
if (owed.length > 0 && persisted) {
|
|
1495
|
+
errorResult = toolError(`${owed.length} delivered test${owed.length === 1 ? "" : "s"} fail${owed.length === 1 ? "s" : ""} a blocking code-reuse check on the file as it stands now — the same check skyramp_reuse_code runs with verify: true, and a report cannot ship work that does not pass it:\n` +
|
|
1496
|
+
owed
|
|
1497
|
+
.map((v) => `\n- ${v.file} · ${v.failures.map((f) => f.kind).join(", ")} · refusal ${refusals[v.file]} of ${REUSE_SUBMIT_MAX_REFUSALS}\n${indent(v.detail)}\n Then call ${v.verifyCall} so the repaired files are re-checked.`)
|
|
1498
|
+
.join("\n") +
|
|
1499
|
+
// The bound is stated as a counter only. What happens when it is
|
|
1500
|
+
// reached is not — an agent told that two more calls ship the report
|
|
1501
|
+
// may wait instead of repair, and the sanctioned exits (repair, or the
|
|
1502
|
+
// documented decline) are already named.
|
|
1503
|
+
`\n\nRepair each file above (or document a deliberate decline with the exact marker the check names), re-verify it, and call skyramp_submit_report again.`);
|
|
1504
|
+
return errorResult;
|
|
1505
|
+
}
|
|
1506
|
+
logger.warning("Accepting a report whose delivered files fail a blocking code-reuse check — the refusal bound is reached, the rows record the failure", {
|
|
1507
|
+
files: verdicts.map((v) => ({
|
|
1508
|
+
file: v.file,
|
|
1509
|
+
kinds: v.failures.map((f) => f.kind),
|
|
1510
|
+
})),
|
|
1511
|
+
});
|
|
1512
|
+
}
|
|
1338
1513
|
// The report enum has no `bug_caught` value, so which failure was on purpose
|
|
1339
1514
|
// is derived from the plan rather than from agent prose. Read defensively:
|
|
1340
1515
|
// the plan comes off disk.
|
|
@@ -1373,7 +1548,13 @@ export function registerSubmitReportTool(server) {
|
|
|
1373
1548
|
let objections;
|
|
1374
1549
|
let changeTable;
|
|
1375
1550
|
{
|
|
1376
|
-
const postExecution = runPostExecutionChecks(stateData.plan ?? { plannedTests: [], registrationNumber: 0, answers: [], openObjections: [], answeredObjections: [] }, dedupedNewTests, params.testResults, params.issuesFound ?? [], isPlanOnlyMode()
|
|
1551
|
+
const postExecution = runPostExecutionChecks(stateData.plan ?? { plannedTests: [], registrationNumber: 0, answers: [], openObjections: [], answeredObjections: [] }, dedupedNewTests, params.testResults, params.issuesFound ?? [], isPlanOnlyMode(),
|
|
1552
|
+
// The triage's own verdicts, not this tool's input, so the join is against
|
|
1553
|
+
// what the run is recorded as having edited — and only the rows that left a
|
|
1554
|
+
// test standing and were actually applied, so a no-op assessment, a deletion
|
|
1555
|
+
// or a report-only recommendation cannot corroborate. The verdicts rather
|
|
1556
|
+
// than the published rows because `reportOnly` lives here.
|
|
1557
|
+
rowsLeavingCoverage(stateData.maintenanceVerdicts));
|
|
1377
1558
|
// Read defensively, like the crash-contained checks above: the plan comes
|
|
1378
1559
|
// off disk, so its declared array type holds only on the validated path.
|
|
1379
1560
|
const stillOpen = Array.isArray(stateData.plan?.openObjections)
|
|
@@ -1517,6 +1698,46 @@ export function registerSubmitReportTool(server) {
|
|
|
1517
1698
|
changesByCandidate.set(id, cited.map((entry) => String(entry ?? "").trim()));
|
|
1518
1699
|
}
|
|
1519
1700
|
}
|
|
1701
|
+
// A change an existing test covers raises no `coverage:change:` objection
|
|
1702
|
+
// and ships no planned test, so neither `testedBy` nor `answer` would say
|
|
1703
|
+
// anything about it. Without this the table reads as untested for a change
|
|
1704
|
+
// `coverage` counts as covered.
|
|
1705
|
+
//
|
|
1706
|
+
// Only what a maintenance row corroborates: publishing a claim on its own
|
|
1707
|
+
// would let the table assert coverage the verifier refused — `coverage`
|
|
1708
|
+
// declines to credit an entry whose file does not resolve, and this row
|
|
1709
|
+
// would still have named it. An uncorroborated claim draws
|
|
1710
|
+
// `deliveredMatchesPlan:maintains:` instead of a table entry.
|
|
1711
|
+
//
|
|
1712
|
+
// `testFileMatches` rather than a basename, so a spec under one directory
|
|
1713
|
+
// cannot lend its coverage to a same-named spec under another; and only the
|
|
1714
|
+
// rows that left a test standing, so DELETE and the no-op actions cannot
|
|
1715
|
+
// corroborate a claim that an existing test covers the change.
|
|
1716
|
+
const editedPaths = [];
|
|
1717
|
+
for (const row of rowsLeavingCoverage(stateData.maintenanceVerdicts)) {
|
|
1718
|
+
for (const full of [row?.testFilePath, row?.pomFile]) {
|
|
1719
|
+
const path = typeof full === "string" ? full.trim() : "";
|
|
1720
|
+
if (path)
|
|
1721
|
+
editedPaths.push(path);
|
|
1722
|
+
}
|
|
1723
|
+
}
|
|
1724
|
+
const maintainedByChange = new Map();
|
|
1725
|
+
for (const entry of Array.isArray(stateData.plan?.maintains) ? stateData.plan.maintains : []) {
|
|
1726
|
+
const file = String(entry?.file ?? "").trim();
|
|
1727
|
+
if (!file || !Array.isArray(entry?.changes))
|
|
1728
|
+
continue;
|
|
1729
|
+
if (!editedPaths.some((full) => testFileMatches(full, file)))
|
|
1730
|
+
continue;
|
|
1731
|
+
for (const cited of entry.changes) {
|
|
1732
|
+
const changeId = String(cited ?? "").trim();
|
|
1733
|
+
if (!changeId)
|
|
1734
|
+
continue;
|
|
1735
|
+
const files = maintainedByChange.get(changeId) ?? [];
|
|
1736
|
+
if (!files.includes(file))
|
|
1737
|
+
files.push(file);
|
|
1738
|
+
maintainedByChange.set(changeId, files);
|
|
1739
|
+
}
|
|
1740
|
+
}
|
|
1520
1741
|
const answerByChange = new Map();
|
|
1521
1742
|
for (const objection of objections) {
|
|
1522
1743
|
if (!objection.objectionId.startsWith(CHANGE_OBJECTION_PREFIX))
|
|
@@ -1531,11 +1752,13 @@ export function registerSubmitReportTool(server) {
|
|
|
1531
1752
|
.map((test) => (test.plannedTestId ?? "").trim())
|
|
1532
1753
|
.filter((plannedTestId) => plannedTestId && (changesByCandidate.get(plannedTestId) ?? []).includes(id));
|
|
1533
1754
|
const answer = answerByChange.get(id);
|
|
1755
|
+
const maintainedBy = maintainedByChange.get(id) ?? [];
|
|
1534
1756
|
return {
|
|
1535
1757
|
id,
|
|
1536
1758
|
text: String(change?.text ?? ""),
|
|
1537
1759
|
source: String(change?.source ?? ""),
|
|
1538
1760
|
testedBy,
|
|
1761
|
+
...(maintainedBy.length > 0 ? { maintainedBy } : {}),
|
|
1539
1762
|
...(answer ? { answer } : {}),
|
|
1540
1763
|
};
|
|
1541
1764
|
});
|
|
@@ -1581,8 +1804,6 @@ export function registerSubmitReportTool(server) {
|
|
|
1581
1804
|
additionalRecommendations: dedupedRecommendations.map(normalizeRepository),
|
|
1582
1805
|
// Report wire format keeps main's original `fileName` (basename) — testFilePath is
|
|
1583
1806
|
// an internal-only field, needed for matching but never meant to reach the report.
|
|
1584
|
-
// TODO(multi-repo maintenance): map(normalizeRepository) once testMaintenanceSchema
|
|
1585
|
-
// has a repository field (see TODO above).
|
|
1586
1807
|
testMaintenance: testMaintenance
|
|
1587
1808
|
? await Promise.all(testMaintenance.map(async ({ testFilePath, pomFile, ...row }) => {
|
|
1588
1809
|
// `pomFile` is the file skyramp_actions told the agent to edit;
|
|
@@ -1598,7 +1819,7 @@ export function registerSubmitReportTool(server) {
|
|
|
1598
1819
|
const assertions = record
|
|
1599
1820
|
? await rederiveAssertionOutcome(record)
|
|
1600
1821
|
: undefined;
|
|
1601
|
-
return {
|
|
1822
|
+
return normalizeRepository({
|
|
1602
1823
|
...row,
|
|
1603
1824
|
fileName: path.basename(testFilePath),
|
|
1604
1825
|
...(pomFile &&
|
|
@@ -1606,7 +1827,7 @@ export function registerSubmitReportTool(server) {
|
|
|
1606
1827
|
? { editedFileName: path.basename(pomFile) }
|
|
1607
1828
|
: {}),
|
|
1608
1829
|
...(assertions ? { assertions } : {}),
|
|
1609
|
-
};
|
|
1830
|
+
});
|
|
1610
1831
|
}))
|
|
1611
1832
|
: undefined,
|
|
1612
1833
|
// videoPath is filled from the run's execution records; testFilePath is the
|
|
@@ -410,6 +410,11 @@ export function registerActionsTool(server) {
|
|
|
410
410
|
...(r.rebaselineSnapshots?.length
|
|
411
411
|
? { rebaselineSnapshots: r.rebaselineSnapshots, rebaselineOnly: r.rebaselineOnly === true }
|
|
412
412
|
: {}),
|
|
413
|
+
// Persisted so maintenance coverage can tell a REGENERATE that rewrote a
|
|
414
|
+
// file from one that only advised the developer to: an external test's
|
|
415
|
+
// REGENERATE/DELETE keeps its real action and touches nothing, so without
|
|
416
|
+
// this the report credits a `maintains` claim no edit backs.
|
|
417
|
+
...(r.reportOnly ? { reportOnly: true } : {}),
|
|
413
418
|
})),
|
|
414
419
|
}, { repo: args.repository, repositoryPath, step: "actions" });
|
|
415
420
|
}
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js";
|
|
2
2
|
import { z } from "zod";
|
|
3
|
+
import { AuthFinding } from "../../utils/workspaceAuth.js";
|
|
3
4
|
import { CallToolResult } from "@modelcontextprotocol/sdk/types.js";
|
|
4
5
|
import { TraceFile } from "../../types/RepositoryAnalysis.js";
|
|
5
6
|
import { ChangedFileName } from "../../utils/branchDiff.js";
|
|
@@ -58,6 +59,10 @@ export interface AnalyzeChangesData {
|
|
|
58
59
|
openApiSpecPath?: string;
|
|
59
60
|
/** Whether the spec loaded and carried usable path entries. */
|
|
60
61
|
openApiSpecLoaded: boolean;
|
|
62
|
+
/** What the agent has to decide about this repository's auth, when the
|
|
63
|
+
* workspace and the code disagree. Each carries the sentence the agent acts
|
|
64
|
+
* on; the server changes no value on the strength of one. */
|
|
65
|
+
authFindings?: AuthFinding[];
|
|
61
66
|
};
|
|
62
67
|
/**
|
|
63
68
|
* Identifying `data-*` attributes the diff removed from files that survive it
|
|
@@ -116,18 +121,66 @@ export declare const analyzeChangesOutputSchema: {
|
|
|
116
121
|
authHeader: z.ZodOptional<z.ZodString>;
|
|
117
122
|
openApiSpecPath: z.ZodOptional<z.ZodString>;
|
|
118
123
|
openApiSpecLoaded: z.ZodBoolean;
|
|
124
|
+
authFindings: z.ZodOptional<z.ZodArray<z.ZodObject<{
|
|
125
|
+
kind: z.ZodLiteral<"workspace-disagreement">;
|
|
126
|
+
message: z.ZodString;
|
|
127
|
+
serviceName: z.ZodString;
|
|
128
|
+
field: z.ZodEnum<["authType", "authHeader", "authScheme"]>;
|
|
129
|
+
primary: z.ZodString;
|
|
130
|
+
analysed: z.ZodString;
|
|
131
|
+
primaryWorkspaceFile: z.ZodString;
|
|
132
|
+
analysedWorkspaceFile: z.ZodString;
|
|
133
|
+
}, "strip", z.ZodTypeAny, {
|
|
134
|
+
message: string;
|
|
135
|
+
kind: "workspace-disagreement";
|
|
136
|
+
serviceName: string;
|
|
137
|
+
field: "authHeader" | "authScheme" | "authType";
|
|
138
|
+
primaryWorkspaceFile: string;
|
|
139
|
+
analysedWorkspaceFile: string;
|
|
140
|
+
primary: string;
|
|
141
|
+
analysed: string;
|
|
142
|
+
}, {
|
|
143
|
+
message: string;
|
|
144
|
+
kind: "workspace-disagreement";
|
|
145
|
+
serviceName: string;
|
|
146
|
+
field: "authHeader" | "authScheme" | "authType";
|
|
147
|
+
primaryWorkspaceFile: string;
|
|
148
|
+
analysedWorkspaceFile: string;
|
|
149
|
+
primary: string;
|
|
150
|
+
analysed: string;
|
|
151
|
+
}>, "many">>;
|
|
119
152
|
}, "strip", z.ZodTypeAny, {
|
|
120
153
|
baseUrl: string;
|
|
121
154
|
authMethod: string;
|
|
122
155
|
openApiSpecLoaded: boolean;
|
|
123
156
|
authHeader?: string | undefined;
|
|
124
157
|
openApiSpecPath?: string | undefined;
|
|
158
|
+
authFindings?: {
|
|
159
|
+
message: string;
|
|
160
|
+
kind: "workspace-disagreement";
|
|
161
|
+
serviceName: string;
|
|
162
|
+
field: "authHeader" | "authScheme" | "authType";
|
|
163
|
+
primaryWorkspaceFile: string;
|
|
164
|
+
analysedWorkspaceFile: string;
|
|
165
|
+
primary: string;
|
|
166
|
+
analysed: string;
|
|
167
|
+
}[] | undefined;
|
|
125
168
|
}, {
|
|
126
169
|
baseUrl: string;
|
|
127
170
|
authMethod: string;
|
|
128
171
|
openApiSpecLoaded: boolean;
|
|
129
172
|
authHeader?: string | undefined;
|
|
130
173
|
openApiSpecPath?: string | undefined;
|
|
174
|
+
authFindings?: {
|
|
175
|
+
message: string;
|
|
176
|
+
kind: "workspace-disagreement";
|
|
177
|
+
serviceName: string;
|
|
178
|
+
field: "authHeader" | "authScheme" | "authType";
|
|
179
|
+
primaryWorkspaceFile: string;
|
|
180
|
+
analysedWorkspaceFile: string;
|
|
181
|
+
primary: string;
|
|
182
|
+
analysed: string;
|
|
183
|
+
}[] | undefined;
|
|
131
184
|
}>>;
|
|
132
185
|
uiContext: z.ZodOptional<z.ZodObject<{
|
|
133
186
|
removedElements: z.ZodOptional<z.ZodArray<z.ZodObject<{
|