@skyramp/mcp 0.4.1-rc.1 → 0.4.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (67) hide show
  1. package/build/prompts/enhance-assertions/contractProviderAssertionsPrompt.js +2 -1
  2. package/build/prompts/enhance-assertions/integrationAssertionsPrompt.js +2 -1
  3. package/build/prompts/enhance-assertions/sharedAssertionRules.d.ts +1 -1
  4. package/build/prompts/enhance-assertions/sharedAssertionRules.js +41 -22
  5. package/build/prompts/enhance-assertions/uiAssertionsPrompt.js +17 -9
  6. package/build/prompts/test-recommendation/diffExecutionPlan.js +0 -2
  7. package/build/prompts/test-recommendation/test-recommendation-prompt.js +8 -2
  8. package/build/prompts/testbot/testbot-prompts.js +12 -7
  9. package/build/recommendation/registerPlan.d.ts +5 -1
  10. package/build/recommendation/registerPlan.js +5 -0
  11. package/build/recommendation/types.d.ts +25 -2
  12. package/build/recommendation/verifierContracts.d.ts +16 -4
  13. package/build/recommendation/verifierContracts.js +20 -4
  14. package/build/recommendation/verifiers/coverage.d.ts +10 -0
  15. package/build/recommendation/verifiers/coverage.js +144 -22
  16. package/build/recommendation/verifiers/deliveredMatchesPlan.d.ts +22 -0
  17. package/build/recommendation/verifiers/deliveredMatchesPlan.js +43 -0
  18. package/build/recommendation/verifiers/existingCoverage.js +53 -0
  19. package/build/recommendation/verifiers/expectedValueSourced.js +111 -14
  20. package/build/services/TestGenerationService.js +3 -1
  21. package/build/tools/code-refactor/codeReuseTool.js +1 -1
  22. package/build/tools/code-refactor/reuse-outcome.d.ts +1 -1
  23. package/build/tools/code-refactor/reuse-state.d.ts +85 -7
  24. package/build/tools/code-refactor/reuse-state.js +239 -34
  25. package/build/tools/code-refactor/utils-verify-gates.d.ts +5 -0
  26. package/build/tools/code-refactor/utils-verify-gates.js +103 -11
  27. package/build/tools/generate-tests/generateBatchScenarioRestTool.js +2 -1
  28. package/build/tools/submitReportTool.js +259 -38
  29. package/build/tools/test-management/actionsTool.js +5 -0
  30. package/build/tools/test-management/analyzeChangesTool.d.ts +53 -0
  31. package/build/tools/test-management/analyzeChangesTool.js +55 -2
  32. package/build/tools/test-management/registerTestPlanTool.d.ts +2 -1
  33. package/build/tools/test-management/registerTestPlanTool.js +29 -14
  34. package/build/types/ReuseOutcome.d.ts +73 -7
  35. package/build/types/TestAnalysis.d.ts +6 -0
  36. package/build/types/TestbotReport.d.ts +15 -1
  37. package/build/utils/AnalysisStateManager.d.ts +7 -1
  38. package/build/utils/AnalysisStateManager.js +5 -1
  39. package/build/utils/assertion-verify/api-shared-lints.js +71 -34
  40. package/build/utils/assertion-verify/format.js +2 -2
  41. package/build/utils/assertion-verify/helper-imports.d.ts +7 -0
  42. package/build/utils/assertion-verify/helper-imports.js +119 -27
  43. package/build/utils/assertion-verify/lint-types.d.ts +31 -2
  44. package/build/utils/assertion-verify/lint-types.js +66 -0
  45. package/build/utils/assertion-verify/metrics.d.ts +13 -0
  46. package/build/utils/assertion-verify/metrics.js +16 -0
  47. package/build/utils/assertion-verify/verify.d.ts +11 -6
  48. package/build/utils/assertion-verify/verify.js +56 -15
  49. package/build/utils/canonicalJson.d.ts +11 -0
  50. package/build/utils/canonicalJson.js +17 -0
  51. package/build/utils/utils-verify/action-key.d.ts +27 -0
  52. package/build/utils/utils-verify/action-key.js +292 -0
  53. package/build/utils/utils-verify/allow.d.ts +8 -1
  54. package/build/utils/utils-verify/allow.js +14 -1
  55. package/build/utils/utils-verify/call-sites.d.ts +76 -8
  56. package/build/utils/utils-verify/call-sites.js +256 -70
  57. package/build/utils/utils-verify/language-spec.d.ts +3 -2
  58. package/build/utils/utils-verify/parse.d.ts +22 -3
  59. package/build/utils/utils-verify/parse.js +123 -52
  60. package/build/utils/utils-verify/verify.d.ts +33 -3
  61. package/build/utils/utils-verify/verify.js +126 -12
  62. package/build/utils/workspaceAuth.d.ts +59 -19
  63. package/build/utils/workspaceAuth.js +228 -31
  64. package/package.json +1 -1
  65. package/plugin/prompts/generate-tests/execution-plan.md +1 -1
  66. package/plugin/prompts/generate-tests/generation.md +1 -0
  67. package/plugin/prompts/plan-tests.md +33 -16
@@ -13,7 +13,7 @@ import { StateManager, runArtifactDir, getTestsRepoDir, resolveRunStatePath, } f
13
13
  import { toolError, testFileMatches } from "../utils/utils.js";
14
14
  import { isTestbotEnabled, } from "../utils/featureFlags.js";
15
15
  import { answerFor, unknownAnswerObjections } from "../recommendation/answers.js";
16
- import { checkDeliveredMatchesPlan } from "../recommendation/verifiers/deliveredMatchesPlan.js";
16
+ import { checkDeliveredMatchesPlan, checkMaintenanceDelivered } from "../recommendation/verifiers/deliveredMatchesPlan.js";
17
17
  import { checkReportedCategoryMatchesPlan, checkRequirementConflictReported, } from "../recommendation/verifiers/reportedCategory.js";
18
18
  import { checkDefectsReported, checkIssueTraceability } from "../recommendation/verifiers/issueTraceability.js";
19
19
  import { checkExpectedOutcomeAfterExecution, } from "../recommendation/verifiers/expectedOutcome.js";
@@ -21,7 +21,7 @@ import { findInvalidSourceCitations, findUnchangedFileClaims, listChangedFiles,
21
21
  import { isPlanOnlyMode } from "../utils/planOnlyMode.js";
22
22
  import { getReportLanguage, isEnforcedReportLanguage, findLanguageViolations, findLanguageNearMisses, reportLanguageDisplayName, } from "../utils/reportLanguage.js";
23
23
  import { canonicalTestPath, findAssertionRecordByFileName, rederiveAssertionOutcome, } from "./code-refactor/assertion-state.js";
24
- import { rederiveReuseOutcome, reuseChainSkipped, samePath, } from "./code-refactor/reuse-state.js";
24
+ import { rederiveReuse, REUSE_SUBMIT_MAX_REFUSALS, reuseChainSkipped, samePath, } from "./code-refactor/reuse-state.js";
25
25
  import { retrofitGate, } from "./code-refactor/retrofit-state.js";
26
26
  // Mirrors the tools wired to planGuard: UI and E2E are not gated at generation
27
27
  // time, so gating them here would be a report-time-only surprise.
@@ -51,6 +51,22 @@ const MAINTENANCE_CHANGE_ACTIONS = new Set([
51
51
  DriftAction.Regenerate,
52
52
  DriftAction.Delete,
53
53
  ]);
54
+ // Narrower than the set above, and for a different question. That one asks whether a
55
+ // row touched a file; this asks whether the row left a test standing that a `maintains`
56
+ // claim can rest on. DELETE touched the file and removed the coverage with it, VERIFY
57
+ // and IGNORE never edited anything — none of the three corroborate a claim that an
58
+ // existing test covers a change.
59
+ const MAINTENANCE_COVERAGE_ACTIONS = new Set([
60
+ DriftAction.Update,
61
+ DriftAction.Regenerate,
62
+ ]);
63
+ /** The maintenance rows a `maintains` claim may be checked against: an action that leaves
64
+ * a test standing, and an edit this run actually made. `reportOnly` is the second half —
65
+ * an external test's REGENERATE keeps its real action and touches no file, so counting it
66
+ * would credit a claim nothing backs. */
67
+ function rowsLeavingCoverage(rows) {
68
+ return (rows ?? []).filter((row) => row?.action !== undefined && MAINTENANCE_COVERAGE_ACTIONS.has(row.action) && row?.reportOnly !== true);
69
+ }
54
70
  /** Objection ids for an untested declared change, `coverage:change:<id>`. The
55
71
  * change table reads the agent's answer back off them. */
56
72
  const CHANGE_OBJECTION_PREFIX = "coverage:change:";
@@ -493,7 +509,6 @@ export const additionalRecommendationSchema = z.preprocess(stripNullGroundingFie
493
509
  // `targetElements: null` is its normal state rather than a lost capture.
494
510
  }
495
511
  }));
496
- // TODO(multi-repo maintenance): no `repository` field yet — see readData() TODO below.
497
512
  const testMaintenanceSchema = z.object({
498
513
  testType: z.nativeEnum(TestType).describe("Type of test."),
499
514
  endpoint: z
@@ -523,6 +538,10 @@ const testMaintenanceSchema = z.object({
523
538
  "For failing runs: failure name and one-line root cause, e.g. " +
524
539
  "'FAILED test_foo — check_schema fails, order_id=1 has discount from prior PATCH test'. " +
525
540
  "Empty string for VERIFY/IGNORE/DELETE entries where no after-execution was run."),
541
+ // Server-populated from the verdict's own state section, falling back to the
542
+ // run's primary repo. Always set: a consumer that has to treat absence as
543
+ // "probably the primary" cannot tell a single-repo row from a mis-stamped one.
544
+ repository: z.string(),
526
545
  // Server-populated from stateFile execution records — never supplied by the LLM.
527
546
  beforeStatus: z.nativeEnum(TestExecutionStatus),
528
547
  afterStatus: z.nativeEnum(TestExecutionStatus),
@@ -601,7 +620,7 @@ function computeReportMetrics(params) {
601
620
  *
602
621
  * The counts are RE-DERIVED from the delivered spec rather than read back from
603
622
  * state: the execution fix-up can restore `<testFile>.raw.bak` over it afterwards
604
- * by plain `cp`, which this tool never sees. See rederiveReuseOutcome.
623
+ * by plain `cp`, which this tool never sees. See rederiveReuse.
605
624
  *
606
625
  * Tests the reuse tool never ran for and specs whose outcome cannot be re-derived
607
626
  * get nothing — which consumers already treat as "no reuse summary" — with one
@@ -628,7 +647,12 @@ async function attachAssertionOutcome(test, outcomes, checkouts) {
628
647
  return test;
629
648
  return { ...test, assertions: await rederiveAssertionOutcome(record) };
630
649
  }
631
- async function attachReuseOutcome(test, outcomes, handOffs, retrofits = []) {
650
+ async function attachReuseOutcome(test, outcomes, handOffs, retrofits,
651
+ // A blocking verdict measured on the delivered files is collected here rather
652
+ // than written into the row: the caller refuses the report on it. Required, not
653
+ // defaulted — a caller that forgot it would ship the report with no refusal and
654
+ // no compiler error. See the refusal below the row map.
655
+ blocking) {
632
656
  // POM records describe browser specs; a basename collision with an API test's
633
657
  // fileName must not attach them there. A utils-path record is attachable anywhere.
634
658
  const record = outcomes?.[path.basename(test.fileName)];
@@ -638,7 +662,10 @@ async function attachReuseOutcome(test, outcomes, handOffs, retrofits = []) {
638
662
  const found = record && typeMatches && (test.testType === TestType.UI || record.utils)
639
663
  ? record
640
664
  : undefined;
641
- const derived = found ? await rederiveReuseOutcome(found) : undefined;
665
+ const rederived = found ? await rederiveReuse(found) : undefined;
666
+ const derived = rederived?.outcome;
667
+ if (rederived?.blocking)
668
+ blocking.push(rederived.blocking);
642
669
  // Pre-existing generated tests this spec's reuse pass rewired (SKYR-4276 A4): the
643
670
  // report names each with its recorded execution, so a reviewer sees that the
644
671
  // module became a dependency of code they already owned — and that it still runs.
@@ -681,6 +708,13 @@ function attachVideoPath(row, videos) {
681
708
  videoPath: videos?.[path.basename(row.testFilePath)]?.videoPath,
682
709
  };
683
710
  }
711
+ /** Two-space indent on every line, so a multi-line verdict reads as one bullet's body. */
712
+ function indent(text) {
713
+ return text
714
+ .split("\n")
715
+ .map((line) => ` ${line}`)
716
+ .join("\n");
717
+ }
684
718
  function deduplicateById(items) {
685
719
  const seen = new Set();
686
720
  const result = [];
@@ -824,7 +858,7 @@ const PLANNABLE_TEST_TYPES = new Set([
824
858
  function runPostExecutionChecks(plan, delivered, results, issues,
825
859
  /** The plan-only lane delivers nothing, so there a planned test is proof
826
860
  * enough for a bug. Passed in, not read here, so the checks stay pure. */
827
- planOnly) {
861
+ planOnly, maintained = []) {
828
862
  const shipped = delivered
829
863
  // A smoke, fuzz or load entry is filtered out rather than reported as
830
864
  // unplanned: a plan cannot hold one, so the objection would name a mistake the
@@ -835,6 +869,9 @@ planOnly) {
835
869
  .map((plannedTestId) => ({ plannedTestId }));
836
870
  return [
837
871
  ...runPostExecutionCheck("deliveredMatchesPlan", "deliveredMatchesPlan", () => checkDeliveredMatchesPlan(plan, shipped)),
872
+ // The maintenance mirror: a `maintains` entry credited coverage at plan time,
873
+ // so something has to check the file was edited.
874
+ ...runPostExecutionCheck("deliveredMatchesPlan", "deliveredMatchesPlan:maintains", () => checkMaintenanceDelivered(plan, maintained)),
838
875
  ...runPostExecutionCheck("expectedOutcome", "expectedOutcome:executed", () => checkExpectedOutcomeAfterExecution(plan, collectExecutionOutcomes(delivered, results), issues)),
839
876
  ...runPostExecutionCheck("reportedCategory", "reportedCategory", () => checkReportedCategoryMatchesPlan(plan, delivered)),
840
877
  ...runPostExecutionCheck("requirementConflictReported", "requirementConflictReported", () => checkRequirementConflictReported(plan, delivered, issues)),
@@ -1060,8 +1097,8 @@ export function registerSubmitReportTool(server) {
1060
1097
  const stateManager = StateManager.fromStatePath(stateFile);
1061
1098
  let stateData;
1062
1099
  try {
1063
- // Only reads the primary repo — see the related-repo check below for why related
1064
- // repos' maintenanceVerdicts are never folded in here.
1100
+ // Primary section only; related sections are read where testMaintenance
1101
+ // is built, so each row keeps its own repo's executions.
1065
1102
  stateData = await stateManager.readData();
1066
1103
  }
1067
1104
  catch (err) {
@@ -1127,29 +1164,74 @@ export function registerSubmitReportTool(server) {
1127
1164
  return errorResult;
1128
1165
  }
1129
1166
  }
1130
- // TestbotReport.testMaintenance has no repository field, so a related repo's
1131
- // reconciled verdicts have nowhere to go in this report. Fail loud instead of
1132
- // silently omitting real maintenance work skyramp_actions already did and persisted.
1133
- // Multi-repo maintenance reporting is a separate, unscoped feature (would also need
1134
- // testbot prompt / drift-analysis / test-bot.git renderer changes) — not done here.
1135
- const relatedReposWithVerdicts = [];
1136
- for (const repo of await stateManager.listRelatedRepos()) {
1137
- const relatedData = await stateManager.readRepoData(repo);
1138
- if ((relatedData?.maintenanceVerdicts?.length ?? 0) > 0)
1139
- relatedReposWithVerdicts.push(repo);
1167
+ // Every section that can hold verdicts: primary, then each related repo.
1168
+ // Read once — the file can carry a large diffText and readRepoData would
1169
+ // re-parse it twice per repo.
1170
+ const fullMaintenanceState = await stateManager.readFullState();
1171
+ // The state file's root names the primary only when a caller declared one
1172
+ // (or wrote first). GITHUB_REPOSITORY is the trigger event's own repo, so
1173
+ // it names the primary even for a run that passed neither repo argument.
1174
+ const primaryRepository = fullMaintenanceState?.metadata?.repository?.trim() ||
1175
+ process.env.GITHUB_REPOSITORY?.trim() ||
1176
+ undefined;
1177
+ const maintenanceSections = [{ data: stateData }];
1178
+ for (const [repo, section] of Object.entries(fullMaintenanceState?.relatedRepos ?? {})) {
1179
+ // Only these two keys resolve back to the primary, already collected
1180
+ // above. Any other key — a case variant, or another name over the same
1181
+ // checkout — is a SEPARATE store, so skipping it would discard its
1182
+ // verdicts; duplicates collapse per row below instead.
1183
+ if (!repo.trim())
1184
+ continue;
1185
+ if (repo === primaryRepository)
1186
+ continue;
1187
+ const data = section?.data;
1188
+ if (data)
1189
+ maintenanceSections.push({ repository: repo, data });
1190
+ }
1191
+ // Stamp every row, primary included and single-repo runs included — as the
1192
+ // prompt already requires of the other report sections. A section without
1193
+ // its own name is the primary's, so it takes the run's primary repo.
1194
+ const repositoryOf = (section) => section.repository ?? primaryRepository ?? "";
1195
+ // Required field, so refuse rather than stamp a blank one: nothing named
1196
+ // the primary and the run is not on a GitHub runner either. Only sections
1197
+ // that actually produce rows need a name — a run with no verdicts at all
1198
+ // has nothing to attribute and must still be able to submit.
1199
+ if (!primaryRepository &&
1200
+ maintenanceSections.some((section) => !section.repository &&
1201
+ (section.data.maintenanceVerdicts?.length ?? 0) > 0)) {
1202
+ errorResult = toolError("Cannot name the repository for the maintenance rows: the stateFile records no primary repository " +
1203
+ "and GITHUB_REPOSITORY is unset. Pass `repository` (and `primaryRepository` in a multi-repo run) " +
1204
+ "to skyramp_analyze_changes, then re-run it.");
1205
+ return errorResult;
1140
1206
  }
1141
- if (relatedReposWithVerdicts.length > 0) {
1142
- errorResult = toolError(`Related repo(s) have reconciled maintenance verdicts that cannot be included in this report yet ` +
1143
- `(multi-repo maintenance reporting is not supported — testMaintenance has no repository field): ${relatedReposWithVerdicts.join(", ")}.`);
1207
+ // Reject rather than throw: a non-array here, or a missing testFilePath
1208
+ // below, escapes the handler and loses the analytics event in its `finally`.
1209
+ const malformedSections = maintenanceSections
1210
+ .filter(({ data }) => data.maintenanceVerdicts !== undefined &&
1211
+ !Array.isArray(data.maintenanceVerdicts))
1212
+ .map(({ repository }) => repository ?? "the primary repo");
1213
+ if (malformedSections.length > 0) {
1214
+ errorResult = toolError(`stateFile has a non-array maintenanceVerdicts for: ${malformedSections.join(", ")}. ` +
1215
+ `Re-run skyramp_actions for that repository to rewrite the section.`);
1144
1216
  return errorResult;
1145
1217
  }
1146
- // action/testType/endpoint/description come entirely from stateData.maintenanceVerdicts
1147
- // (written by skyramp_actions right after drift analysis) — the LLM never supplies
1148
- // testMaintenance directly, so there's nothing to re-transcribe here.
1218
+ const badVerdicts = [];
1219
+ // action/testType/endpoint/description come entirely from each section's
1220
+ // maintenanceVerdicts (written by skyramp_actions right after drift analysis) —
1221
+ // the LLM never supplies testMaintenance directly, so there's nothing to
1222
+ // re-transcribe here.
1149
1223
  // `pomFile` rides along per row and is stripped from the wire format below,
1150
1224
  // like testFilePath. It cannot be keyed by spec path: a spec emits one UPDATE
1151
1225
  // verdict PER page object, so a spec-keyed map keeps only the last.
1152
- const rawMaintenance = (stateData.maintenanceVerdicts ?? []).map((v) => ({
1226
+ const rawMaintenance = maintenanceSections.flatMap((section) => (section.data.maintenanceVerdicts ?? [])
1227
+ .filter((v) => {
1228
+ if (v && typeof v.testFilePath === "string" && v.testFilePath.trim())
1229
+ return true;
1230
+ badVerdicts.push(section.repository ?? "the primary repo");
1231
+ return false;
1232
+ })
1233
+ .map((v) => ({
1234
+ repository: repositoryOf(section),
1153
1235
  testFilePath: v.testFilePath,
1154
1236
  pomFile: v.pomFile,
1155
1237
  testType: v.testType,
@@ -1162,20 +1244,47 @@ export function registerSubmitReportTool(server) {
1162
1244
  : v.rationale,
1163
1245
  beforeDetails: "",
1164
1246
  afterDetails: "",
1165
- }));
1247
+ })));
1248
+ // GitHub slugs are case-insensitive, so two sections can name one repo and
1249
+ // repeat a file. Keep the first — the primary's, given the order above.
1250
+ // pomFile is part of the key: one spec emits one UPDATE verdict per page
1251
+ // object, and those rows are distinct, not duplicates.
1252
+ const seenRows = new Set();
1253
+ const dedupedMaintenance = rawMaintenance.filter((row) => {
1254
+ const key = `${(row.repository ?? "").trim().toLowerCase()}\u0000${row.testFilePath}\u0000${row.pomFile ?? ""}`;
1255
+ if (seenRows.has(key))
1256
+ return false;
1257
+ seenRows.add(key);
1258
+ return true;
1259
+ });
1260
+ if (badVerdicts.length > 0) {
1261
+ errorResult = toolError(`${badVerdicts.length} maintenance verdict(s) in the stateFile have no testFilePath ` +
1262
+ `(sections: ${[...new Set(badVerdicts)].join(", ")}). ` +
1263
+ `Re-run skyramp_actions for those repositories to rewrite the section.`);
1264
+ return errorResult;
1265
+ }
1266
+ // Keyed by the row's own repo, so executions come from its own section.
1267
+ const sectionByRepository = new Map(maintenanceSections.map((section) => [
1268
+ repositoryOf(section),
1269
+ section.data,
1270
+ ]));
1166
1271
  let testMaintenance;
1167
1272
  // Rows missing a required detail (a real execution happened but the LLM didn't
1168
1273
  // draft a summary for it) — collected here and rejected below, rather than
1169
1274
  // silently shipping a blank field.
1170
1275
  const missingDetails = [];
1171
- if (rawMaintenance.length > 0) {
1276
+ if (dedupedMaintenance.length > 0) {
1172
1277
  // beforeStatus/afterStatus are always stateFile-authoritative. beforeDetails/
1173
1278
  // afterDetails come from the LLM's drafted summary (testMaintenanceDetails) —
1174
1279
  // required whenever the stateFile shows a matching execution actually ran.
1175
- testMaintenance = rawMaintenance.map((m) => {
1176
- const recorded = stateData.existingTests?.find((t) => testFileMatches(t.testFile, m.testFilePath));
1280
+ testMaintenance = dedupedMaintenance.map((m) => {
1281
+ const recorded = sectionByRepository
1282
+ .get(m.repository)
1283
+ ?.existingTests?.find((t) => testFileMatches(t.testFile, m.testFilePath));
1177
1284
  const detail = params.testMaintenanceDetails?.find((d) => d.testFilePath === m.testFilePath);
1178
- const displayName = path.basename(m.testFilePath);
1285
+ // Two repos can hold the same basename; name the repo so the agent
1286
+ // knows which row to draft for.
1287
+ const displayName = `${path.basename(m.testFilePath)} (${m.repository})`;
1179
1288
  const defaultBeforeStatus = MAINTENANCE_CHANGE_ACTIONS.has(m.action)
1180
1289
  ? TestExecutionStatus.Unknown
1181
1290
  : TestExecutionStatus.Skipped;
@@ -1205,7 +1314,8 @@ export function registerSubmitReportTool(server) {
1205
1314
  });
1206
1315
  }
1207
1316
  if (missingDetails.length > 0) {
1208
- errorResult = toolError(`${missingDetails.length} maintenance row(s) have a recorded execution but no drafted summary: ${missingDetails.join(", ")}. ` +
1317
+ const uniqueMissing = [...new Set(missingDetails)];
1318
+ errorResult = toolError(`${uniqueMissing.length} maintenance row(s) have a recorded execution but no drafted summary: ${uniqueMissing.join(", ")}. ` +
1209
1319
  "Add a testMaintenanceDetails entry with the missing field(s), drafted from the execution output you already saw.");
1210
1320
  return errorResult;
1211
1321
  }
@@ -1243,6 +1353,11 @@ export function registerSubmitReportTool(server) {
1243
1353
  ...Object.values(fullState?.relatedRepos ?? {}).map((section) => section.repositoryPath),
1244
1354
  ])),
1245
1355
  ];
1356
+ // Primary verdicts only, by design: this checks UPDATEs, and
1357
+ // skyramp_actions downgrades every actionable related-repo verdict to
1358
+ // VERIFY (actionsTool.ts). If that changes, pass each section's
1359
+ // verdicts with its OWN checkout root — isBacked relativizes against
1360
+ // `repoRoot` and would otherwise exempt them all.
1246
1361
  const unbacked = findUnchangedFileClaims({
1247
1362
  repoRoot,
1248
1363
  changedFiles,
@@ -1331,10 +1446,70 @@ export function registerSubmitReportTool(server) {
1331
1446
  const assertionCheckouts = await stateManager
1332
1447
  .listRepoCheckouts()
1333
1448
  .catch(() => []);
1449
+ const reuseBlocking = [];
1334
1450
  const sanitizedNewTests = await Promise.all(dedupedNewTests.map(async ({ scenarioFile: _sf, traceFile: _tf, frontendTrace: _ft, ...rest }) => {
1335
- const row = await attachReuseOutcome(normalizeRepository(rest), stateData.reuseOutcomes, stateData.reuseHandOffs, retrofitViews);
1451
+ const row = await attachReuseOutcome(normalizeRepository(rest), stateData.reuseOutcomes, stateData.reuseHandOffs, retrofitViews, reuseBlocking);
1336
1452
  return attachAssertionOutcome(row, stateData.assertionOutcomes ?? {}, assertionCheckouts);
1337
1453
  }));
1454
+ // A blocking reuse verdict on the delivered files refuses the report. The
1455
+ // live verify pass is a one-time checkpoint on a file that keeps changing —
1456
+ // the execution fix loop and a post-verification rewrite both edit after it —
1457
+ // and the verdict re-derived here is the only one measured on what ships. It
1458
+ // used to become a row field, so the run committed the fault, detected it,
1459
+ // disclosed it, and delivered it. Reject rather than publish it, exactly like
1460
+ // the maintenanceVerdicts check above: this is a repair loop, not an abort —
1461
+ // the agent fixes the file (or documents the decline the check names) and
1462
+ // calls again. Advisories never reach this list, so a report-time advisory
1463
+ // stays reportable.
1464
+ //
1465
+ // BOUNDED, per spec: after REUSE_SUBMIT_MAX_REFUSALS refusals of one spec the
1466
+ // report is accepted and its row ships `verification: failed` with the
1467
+ // reasons. A loop with no exit at the last step of a run delivers nothing when
1468
+ // the agent cannot reach a pass, and a customer with no pull request is worse
1469
+ // off than one with a disclosed fault (see REUSE_SUBMIT_MAX_REFUSALS). The
1470
+ // count is persisted per spec path; a count that cannot be persisted cannot
1471
+ // bound anything, so a failed write accepts rather than refuses forever.
1472
+ // `reuseOutcomes` is keyed by basename, so two rows sharing a file name find
1473
+ // one record and push one verdict twice — deduplicated by path here, as
1474
+ // pendingReuseVerification skips that case on its side. Sorted by file so the
1475
+ // text is stable across calls.
1476
+ if (reuseBlocking.length > 0) {
1477
+ const verdicts = [
1478
+ ...new Map(reuseBlocking.map((v) => [v.file, v])).values(),
1479
+ ].sort((a, b) => a.file.localeCompare(b.file));
1480
+ const refusals = { ...(stateData.reuseRefusals ?? {}) };
1481
+ const owed = verdicts.filter((v) => (refusals[v.file] ?? 0) < REUSE_SUBMIT_MAX_REFUSALS);
1482
+ let persisted = owed.length > 0;
1483
+ if (owed.length > 0) {
1484
+ for (const v of owed)
1485
+ refusals[v.file] = (refusals[v.file] ?? 0) + 1;
1486
+ try {
1487
+ await stateManager.writeData({ ...stateData, reuseRefusals: refusals });
1488
+ }
1489
+ catch (err) {
1490
+ persisted = false;
1491
+ logger.warning("Could not persist the reuse refusal count — accepting the report rather than refusing without a bound", { error: String(err) });
1492
+ }
1493
+ }
1494
+ if (owed.length > 0 && persisted) {
1495
+ errorResult = toolError(`${owed.length} delivered test${owed.length === 1 ? "" : "s"} fail${owed.length === 1 ? "s" : ""} a blocking code-reuse check on the file as it stands now — the same check skyramp_reuse_code runs with verify: true, and a report cannot ship work that does not pass it:\n` +
1496
+ owed
1497
+ .map((v) => `\n- ${v.file} · ${v.failures.map((f) => f.kind).join(", ")} · refusal ${refusals[v.file]} of ${REUSE_SUBMIT_MAX_REFUSALS}\n${indent(v.detail)}\n Then call ${v.verifyCall} so the repaired files are re-checked.`)
1498
+ .join("\n") +
1499
+ // The bound is stated as a counter only. What happens when it is
1500
+ // reached is not — an agent told that two more calls ship the report
1501
+ // may wait instead of repair, and the sanctioned exits (repair, or the
1502
+ // documented decline) are already named.
1503
+ `\n\nRepair each file above (or document a deliberate decline with the exact marker the check names), re-verify it, and call skyramp_submit_report again.`);
1504
+ return errorResult;
1505
+ }
1506
+ logger.warning("Accepting a report whose delivered files fail a blocking code-reuse check — the refusal bound is reached, the rows record the failure", {
1507
+ files: verdicts.map((v) => ({
1508
+ file: v.file,
1509
+ kinds: v.failures.map((f) => f.kind),
1510
+ })),
1511
+ });
1512
+ }
1338
1513
  // The report enum has no `bug_caught` value, so which failure was on purpose
1339
1514
  // is derived from the plan rather than from agent prose. Read defensively:
1340
1515
  // the plan comes off disk.
@@ -1373,7 +1548,13 @@ export function registerSubmitReportTool(server) {
1373
1548
  let objections;
1374
1549
  let changeTable;
1375
1550
  {
1376
- const postExecution = runPostExecutionChecks(stateData.plan ?? { plannedTests: [], registrationNumber: 0, answers: [], openObjections: [], answeredObjections: [] }, dedupedNewTests, params.testResults, params.issuesFound ?? [], isPlanOnlyMode());
1551
+ const postExecution = runPostExecutionChecks(stateData.plan ?? { plannedTests: [], registrationNumber: 0, answers: [], openObjections: [], answeredObjections: [] }, dedupedNewTests, params.testResults, params.issuesFound ?? [], isPlanOnlyMode(),
1552
+ // The triage's own verdicts, not this tool's input, so the join is against
1553
+ // what the run is recorded as having edited — and only the rows that left a
1554
+ // test standing and were actually applied, so a no-op assessment, a deletion
1555
+ // or a report-only recommendation cannot corroborate. The verdicts rather
1556
+ // than the published rows because `reportOnly` lives here.
1557
+ rowsLeavingCoverage(stateData.maintenanceVerdicts));
1377
1558
  // Read defensively, like the crash-contained checks above: the plan comes
1378
1559
  // off disk, so its declared array type holds only on the validated path.
1379
1560
  const stillOpen = Array.isArray(stateData.plan?.openObjections)
@@ -1517,6 +1698,46 @@ export function registerSubmitReportTool(server) {
1517
1698
  changesByCandidate.set(id, cited.map((entry) => String(entry ?? "").trim()));
1518
1699
  }
1519
1700
  }
1701
+ // A change an existing test covers raises no `coverage:change:` objection
1702
+ // and ships no planned test, so neither `testedBy` nor `answer` would say
1703
+ // anything about it. Without this the table reads as untested for a change
1704
+ // `coverage` counts as covered.
1705
+ //
1706
+ // Only what a maintenance row corroborates: publishing a claim on its own
1707
+ // would let the table assert coverage the verifier refused — `coverage`
1708
+ // declines to credit an entry whose file does not resolve, and this row
1709
+ // would still have named it. An uncorroborated claim draws
1710
+ // `deliveredMatchesPlan:maintains:` instead of a table entry.
1711
+ //
1712
+ // `testFileMatches` rather than a basename, so a spec under one directory
1713
+ // cannot lend its coverage to a same-named spec under another; and only the
1714
+ // rows that left a test standing, so DELETE and the no-op actions cannot
1715
+ // corroborate a claim that an existing test covers the change.
1716
+ const editedPaths = [];
1717
+ for (const row of rowsLeavingCoverage(stateData.maintenanceVerdicts)) {
1718
+ for (const full of [row?.testFilePath, row?.pomFile]) {
1719
+ const path = typeof full === "string" ? full.trim() : "";
1720
+ if (path)
1721
+ editedPaths.push(path);
1722
+ }
1723
+ }
1724
+ const maintainedByChange = new Map();
1725
+ for (const entry of Array.isArray(stateData.plan?.maintains) ? stateData.plan.maintains : []) {
1726
+ const file = String(entry?.file ?? "").trim();
1727
+ if (!file || !Array.isArray(entry?.changes))
1728
+ continue;
1729
+ if (!editedPaths.some((full) => testFileMatches(full, file)))
1730
+ continue;
1731
+ for (const cited of entry.changes) {
1732
+ const changeId = String(cited ?? "").trim();
1733
+ if (!changeId)
1734
+ continue;
1735
+ const files = maintainedByChange.get(changeId) ?? [];
1736
+ if (!files.includes(file))
1737
+ files.push(file);
1738
+ maintainedByChange.set(changeId, files);
1739
+ }
1740
+ }
1520
1741
  const answerByChange = new Map();
1521
1742
  for (const objection of objections) {
1522
1743
  if (!objection.objectionId.startsWith(CHANGE_OBJECTION_PREFIX))
@@ -1531,11 +1752,13 @@ export function registerSubmitReportTool(server) {
1531
1752
  .map((test) => (test.plannedTestId ?? "").trim())
1532
1753
  .filter((plannedTestId) => plannedTestId && (changesByCandidate.get(plannedTestId) ?? []).includes(id));
1533
1754
  const answer = answerByChange.get(id);
1755
+ const maintainedBy = maintainedByChange.get(id) ?? [];
1534
1756
  return {
1535
1757
  id,
1536
1758
  text: String(change?.text ?? ""),
1537
1759
  source: String(change?.source ?? ""),
1538
1760
  testedBy,
1761
+ ...(maintainedBy.length > 0 ? { maintainedBy } : {}),
1539
1762
  ...(answer ? { answer } : {}),
1540
1763
  };
1541
1764
  });
@@ -1581,8 +1804,6 @@ export function registerSubmitReportTool(server) {
1581
1804
  additionalRecommendations: dedupedRecommendations.map(normalizeRepository),
1582
1805
  // Report wire format keeps main's original `fileName` (basename) — testFilePath is
1583
1806
  // an internal-only field, needed for matching but never meant to reach the report.
1584
- // TODO(multi-repo maintenance): map(normalizeRepository) once testMaintenanceSchema
1585
- // has a repository field (see TODO above).
1586
1807
  testMaintenance: testMaintenance
1587
1808
  ? await Promise.all(testMaintenance.map(async ({ testFilePath, pomFile, ...row }) => {
1588
1809
  // `pomFile` is the file skyramp_actions told the agent to edit;
@@ -1598,7 +1819,7 @@ export function registerSubmitReportTool(server) {
1598
1819
  const assertions = record
1599
1820
  ? await rederiveAssertionOutcome(record)
1600
1821
  : undefined;
1601
- return {
1822
+ return normalizeRepository({
1602
1823
  ...row,
1603
1824
  fileName: path.basename(testFilePath),
1604
1825
  ...(pomFile &&
@@ -1606,7 +1827,7 @@ export function registerSubmitReportTool(server) {
1606
1827
  ? { editedFileName: path.basename(pomFile) }
1607
1828
  : {}),
1608
1829
  ...(assertions ? { assertions } : {}),
1609
- };
1830
+ });
1610
1831
  }))
1611
1832
  : undefined,
1612
1833
  // videoPath is filled from the run's execution records; testFilePath is the
@@ -410,6 +410,11 @@ export function registerActionsTool(server) {
410
410
  ...(r.rebaselineSnapshots?.length
411
411
  ? { rebaselineSnapshots: r.rebaselineSnapshots, rebaselineOnly: r.rebaselineOnly === true }
412
412
  : {}),
413
+ // Persisted so maintenance coverage can tell a REGENERATE that rewrote a
414
+ // file from one that only advised the developer to: an external test's
415
+ // REGENERATE/DELETE keeps its real action and touches nothing, so without
416
+ // this the report credits a `maintains` claim no edit backs.
417
+ ...(r.reportOnly ? { reportOnly: true } : {}),
413
418
  })),
414
419
  }, { repo: args.repository, repositoryPath, step: "actions" });
415
420
  }
@@ -1,5 +1,6 @@
1
1
  import { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js";
2
2
  import { z } from "zod";
3
+ import { AuthFinding } from "../../utils/workspaceAuth.js";
3
4
  import { CallToolResult } from "@modelcontextprotocol/sdk/types.js";
4
5
  import { TraceFile } from "../../types/RepositoryAnalysis.js";
5
6
  import { ChangedFileName } from "../../utils/branchDiff.js";
@@ -58,6 +59,10 @@ export interface AnalyzeChangesData {
58
59
  openApiSpecPath?: string;
59
60
  /** Whether the spec loaded and carried usable path entries. */
60
61
  openApiSpecLoaded: boolean;
62
+ /** What the agent has to decide about this repository's auth, when the
63
+ * workspace and the code disagree. Each carries the sentence the agent acts
64
+ * on; the server changes no value on the strength of one. */
65
+ authFindings?: AuthFinding[];
61
66
  };
62
67
  /**
63
68
  * Identifying `data-*` attributes the diff removed from files that survive it
@@ -116,18 +121,66 @@ export declare const analyzeChangesOutputSchema: {
116
121
  authHeader: z.ZodOptional<z.ZodString>;
117
122
  openApiSpecPath: z.ZodOptional<z.ZodString>;
118
123
  openApiSpecLoaded: z.ZodBoolean;
124
+ authFindings: z.ZodOptional<z.ZodArray<z.ZodObject<{
125
+ kind: z.ZodLiteral<"workspace-disagreement">;
126
+ message: z.ZodString;
127
+ serviceName: z.ZodString;
128
+ field: z.ZodEnum<["authType", "authHeader", "authScheme"]>;
129
+ primary: z.ZodString;
130
+ analysed: z.ZodString;
131
+ primaryWorkspaceFile: z.ZodString;
132
+ analysedWorkspaceFile: z.ZodString;
133
+ }, "strip", z.ZodTypeAny, {
134
+ message: string;
135
+ kind: "workspace-disagreement";
136
+ serviceName: string;
137
+ field: "authHeader" | "authScheme" | "authType";
138
+ primaryWorkspaceFile: string;
139
+ analysedWorkspaceFile: string;
140
+ primary: string;
141
+ analysed: string;
142
+ }, {
143
+ message: string;
144
+ kind: "workspace-disagreement";
145
+ serviceName: string;
146
+ field: "authHeader" | "authScheme" | "authType";
147
+ primaryWorkspaceFile: string;
148
+ analysedWorkspaceFile: string;
149
+ primary: string;
150
+ analysed: string;
151
+ }>, "many">>;
119
152
  }, "strip", z.ZodTypeAny, {
120
153
  baseUrl: string;
121
154
  authMethod: string;
122
155
  openApiSpecLoaded: boolean;
123
156
  authHeader?: string | undefined;
124
157
  openApiSpecPath?: string | undefined;
158
+ authFindings?: {
159
+ message: string;
160
+ kind: "workspace-disagreement";
161
+ serviceName: string;
162
+ field: "authHeader" | "authScheme" | "authType";
163
+ primaryWorkspaceFile: string;
164
+ analysedWorkspaceFile: string;
165
+ primary: string;
166
+ analysed: string;
167
+ }[] | undefined;
125
168
  }, {
126
169
  baseUrl: string;
127
170
  authMethod: string;
128
171
  openApiSpecLoaded: boolean;
129
172
  authHeader?: string | undefined;
130
173
  openApiSpecPath?: string | undefined;
174
+ authFindings?: {
175
+ message: string;
176
+ kind: "workspace-disagreement";
177
+ serviceName: string;
178
+ field: "authHeader" | "authScheme" | "authType";
179
+ primaryWorkspaceFile: string;
180
+ analysedWorkspaceFile: string;
181
+ primary: string;
182
+ analysed: string;
183
+ }[] | undefined;
131
184
  }>>;
132
185
  uiContext: z.ZodOptional<z.ZodObject<{
133
186
  removedElements: z.ZodOptional<z.ZodArray<z.ZodObject<{