@skyramp/mcp 0.4.1-rc.1 → 0.4.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/build/prompts/enhance-assertions/contractProviderAssertionsPrompt.js +2 -1
- package/build/prompts/enhance-assertions/integrationAssertionsPrompt.js +2 -1
- package/build/prompts/enhance-assertions/sharedAssertionRules.d.ts +1 -1
- package/build/prompts/enhance-assertions/sharedAssertionRules.js +41 -22
- package/build/prompts/enhance-assertions/uiAssertionsPrompt.js +17 -9
- package/build/prompts/test-recommendation/diffExecutionPlan.js +0 -2
- package/build/prompts/test-recommendation/test-recommendation-prompt.js +8 -2
- package/build/prompts/testbot/testbot-prompts.js +12 -7
- package/build/recommendation/registerPlan.d.ts +5 -1
- package/build/recommendation/registerPlan.js +5 -0
- package/build/recommendation/types.d.ts +25 -2
- package/build/recommendation/verifierContracts.d.ts +16 -4
- package/build/recommendation/verifierContracts.js +20 -4
- package/build/recommendation/verifiers/coverage.d.ts +10 -0
- package/build/recommendation/verifiers/coverage.js +144 -22
- package/build/recommendation/verifiers/deliveredMatchesPlan.d.ts +22 -0
- package/build/recommendation/verifiers/deliveredMatchesPlan.js +43 -0
- package/build/recommendation/verifiers/existingCoverage.js +53 -0
- package/build/recommendation/verifiers/expectedValueSourced.js +111 -14
- package/build/services/TestGenerationService.js +3 -1
- package/build/tools/code-refactor/codeReuseTool.js +1 -1
- package/build/tools/code-refactor/reuse-outcome.d.ts +1 -1
- package/build/tools/code-refactor/reuse-state.d.ts +85 -7
- package/build/tools/code-refactor/reuse-state.js +239 -34
- package/build/tools/code-refactor/utils-verify-gates.d.ts +5 -0
- package/build/tools/code-refactor/utils-verify-gates.js +103 -11
- package/build/tools/generate-tests/generateBatchScenarioRestTool.js +2 -1
- package/build/tools/submitReportTool.js +259 -38
- package/build/tools/test-management/actionsTool.js +5 -0
- package/build/tools/test-management/analyzeChangesTool.d.ts +53 -0
- package/build/tools/test-management/analyzeChangesTool.js +55 -2
- package/build/tools/test-management/registerTestPlanTool.d.ts +2 -1
- package/build/tools/test-management/registerTestPlanTool.js +29 -14
- package/build/types/ReuseOutcome.d.ts +73 -7
- package/build/types/TestAnalysis.d.ts +6 -0
- package/build/types/TestbotReport.d.ts +15 -1
- package/build/utils/AnalysisStateManager.d.ts +7 -1
- package/build/utils/AnalysisStateManager.js +5 -1
- package/build/utils/assertion-verify/api-shared-lints.js +71 -34
- package/build/utils/assertion-verify/format.js +2 -2
- package/build/utils/assertion-verify/helper-imports.d.ts +7 -0
- package/build/utils/assertion-verify/helper-imports.js +119 -27
- package/build/utils/assertion-verify/lint-types.d.ts +31 -2
- package/build/utils/assertion-verify/lint-types.js +66 -0
- package/build/utils/assertion-verify/metrics.d.ts +13 -0
- package/build/utils/assertion-verify/metrics.js +16 -0
- package/build/utils/assertion-verify/verify.d.ts +11 -6
- package/build/utils/assertion-verify/verify.js +56 -15
- package/build/utils/canonicalJson.d.ts +11 -0
- package/build/utils/canonicalJson.js +17 -0
- package/build/utils/utils-verify/action-key.d.ts +27 -0
- package/build/utils/utils-verify/action-key.js +292 -0
- package/build/utils/utils-verify/allow.d.ts +8 -1
- package/build/utils/utils-verify/allow.js +14 -1
- package/build/utils/utils-verify/call-sites.d.ts +76 -8
- package/build/utils/utils-verify/call-sites.js +256 -70
- package/build/utils/utils-verify/language-spec.d.ts +3 -2
- package/build/utils/utils-verify/parse.d.ts +22 -3
- package/build/utils/utils-verify/parse.js +123 -52
- package/build/utils/utils-verify/verify.d.ts +33 -3
- package/build/utils/utils-verify/verify.js +126 -12
- package/build/utils/workspaceAuth.d.ts +59 -19
- package/build/utils/workspaceAuth.js +228 -31
- package/package.json +1 -1
- package/plugin/prompts/generate-tests/execution-plan.md +1 -1
- package/plugin/prompts/generate-tests/generation.md +1 -0
- package/plugin/prompts/plan-tests.md +33 -16
|
@@ -4,7 +4,7 @@ import * as fs from "fs";
|
|
|
4
4
|
import * as path from "path";
|
|
5
5
|
import yaml from "js-yaml";
|
|
6
6
|
import { logger } from "../../utils/logger.js";
|
|
7
|
-
import { parseWorkspaceAuthType, getDefaultAuthHeader, WorkspaceAuthType, readWorkspaceConfigRaw } from "../../utils/workspaceAuth.js";
|
|
7
|
+
import { parseWorkspaceAuthType, getDefaultAuthHeader, WorkspaceAuthType, readWorkspaceConfigRaw, resolveWorkspaceApiService, findWorkspaceAuthDisagreement } from "../../utils/workspaceAuth.js";
|
|
8
8
|
import { AnalyticsService } from "../../services/AnalyticsService.js";
|
|
9
9
|
import { makeProgressReporter } from "../../utils/progress.js";
|
|
10
10
|
import { StateManager, registerSession, storeSessionData, rememberTestsRepoDir, getActiveRunStatePath, setActiveRunStatePath, } from "../../utils/AnalysisStateManager.js";
|
|
@@ -101,6 +101,20 @@ export const analyzeChangesInputSchema = {
|
|
|
101
101
|
.default(false)
|
|
102
102
|
.describe("When true, diffs the base ref against the working tree (captures uncommitted and unstaged changes). Use for local-dev workflows. Defaults to false (CI mode — committed changes only)."),
|
|
103
103
|
};
|
|
104
|
+
/** The shape an auth finding takes. Declared here beside the other output shapes;
|
|
105
|
+
* the assignment below fails to compile if it drifts from `AuthFinding`. */
|
|
106
|
+
const authFindingShape = z.object({
|
|
107
|
+
kind: z.literal("workspace-disagreement"),
|
|
108
|
+
message: z.string(),
|
|
109
|
+
serviceName: z.string(),
|
|
110
|
+
field: z.enum(["authType", "authHeader", "authScheme"]),
|
|
111
|
+
primary: z.string(),
|
|
112
|
+
analysed: z.string(),
|
|
113
|
+
primaryWorkspaceFile: z.string(),
|
|
114
|
+
analysedWorkspaceFile: z.string(),
|
|
115
|
+
});
|
|
116
|
+
const _authFindingShapeMatchesType = {};
|
|
117
|
+
void _authFindingShapeMatchesType;
|
|
104
118
|
/** Both halves of `uiContext` carry the same record, so the shape is declared once.
|
|
105
119
|
* It mirrors `RemovedUiElement`; the assignment below fails to compile if either drifts. */
|
|
106
120
|
const uiIdentifierShape = z.object({ attribute: z.string(), value: z.string(), file: z.string() });
|
|
@@ -135,6 +149,7 @@ export const analyzeChangesOutputSchema = {
|
|
|
135
149
|
authHeader: z.string().optional(),
|
|
136
150
|
openApiSpecPath: z.string().optional(),
|
|
137
151
|
openApiSpecLoaded: z.boolean(),
|
|
152
|
+
authFindings: z.array(authFindingShape).optional(),
|
|
138
153
|
})
|
|
139
154
|
.optional(),
|
|
140
155
|
uiContext: z
|
|
@@ -285,9 +300,12 @@ export function registerAnalyzeChangesTool(server) {
|
|
|
285
300
|
let wsAuthMethod = "none";
|
|
286
301
|
let wsAuthType;
|
|
287
302
|
let wsAuthScheme;
|
|
303
|
+
// What the agent has to decide about this repository's auth. Empty on a
|
|
304
|
+
// workspace whose config and code agree, which is every clean run.
|
|
305
|
+
const authFindings = [];
|
|
288
306
|
const rawConfig = await readWorkspaceConfigRaw(params.repositoryPath);
|
|
289
307
|
if (rawConfig) {
|
|
290
|
-
const svc = rawConfig
|
|
308
|
+
const svc = resolveWorkspaceApiService(rawConfig);
|
|
291
309
|
if (svc?.api?.baseUrl)
|
|
292
310
|
wsBaseUrl = svc.api.baseUrl;
|
|
293
311
|
if (svc?.api?.authHeader !== undefined)
|
|
@@ -626,6 +644,40 @@ export function registerAnalyzeChangesTool(server) {
|
|
|
626
644
|
if (params.repository)
|
|
627
645
|
setActiveRunStatePath(stateFile);
|
|
628
646
|
registerSession(sessionId, stateFile);
|
|
647
|
+
// A related repo reads its OWN .skyramp/workspace.yml, so it can disagree
|
|
648
|
+
// with the primary's unified entry for the same service and nothing would
|
|
649
|
+
// say so.
|
|
650
|
+
//
|
|
651
|
+
// Runs AFTER the state write, because both checkout paths come off the run's
|
|
652
|
+
// state file and this call's own path is not in it until then. Write order is
|
|
653
|
+
// not fixed: given a declared `primaryRepository`, a related repo may write
|
|
654
|
+
// first and the root is then created empty (AnalysisStateManager.writeRepoData).
|
|
655
|
+
// So whichever call arrives second holds both sides and is the one that
|
|
656
|
+
// reports — the primary against the related repos already recorded, a related
|
|
657
|
+
// repo against the primary.
|
|
658
|
+
if (rawConfig && params.repository) {
|
|
659
|
+
try {
|
|
660
|
+
const checkouts = await stateManager.listRepoCheckouts();
|
|
661
|
+
const analysingPrimary = checkouts.some((checkout) => checkout.primary && checkout.root === params.repositoryPath);
|
|
662
|
+
const others = checkouts.filter((checkout) => checkout.primary !== analysingPrimary &&
|
|
663
|
+
checkout.root !== params.repositoryPath);
|
|
664
|
+
for (const other of others) {
|
|
665
|
+
const otherConfig = await readWorkspaceConfigRaw(other.root);
|
|
666
|
+
const disagreement = analysingPrimary
|
|
667
|
+
? findWorkspaceAuthDisagreement(rawConfig, otherConfig, params.repositoryPath, other.root, other.repository)
|
|
668
|
+
: findWorkspaceAuthDisagreement(otherConfig, rawConfig, other.root, params.repositoryPath, params.repository);
|
|
669
|
+
if (disagreement) {
|
|
670
|
+
authFindings.push(disagreement);
|
|
671
|
+
break;
|
|
672
|
+
}
|
|
673
|
+
}
|
|
674
|
+
}
|
|
675
|
+
catch (error) {
|
|
676
|
+
logger.debug("Could not compare this repository's workspace against the run's others", {
|
|
677
|
+
error: error instanceof Error ? error.message : String(error),
|
|
678
|
+
});
|
|
679
|
+
}
|
|
680
|
+
}
|
|
629
681
|
try {
|
|
630
682
|
await server.server.sendResourceListChanged();
|
|
631
683
|
}
|
|
@@ -672,6 +724,7 @@ export function registerAnalyzeChangesTool(server) {
|
|
|
672
724
|
...(wsAuthHeader ? { authHeader: wsAuthHeader } : {}),
|
|
673
725
|
...(wsSchemaPath ? { openApiSpecPath: wsSchemaPath } : {}),
|
|
674
726
|
openApiSpecLoaded: specFetchSucceeded,
|
|
727
|
+
...(authFindings.length ? { authFindings } : {}),
|
|
675
728
|
},
|
|
676
729
|
// Removed half only. No prompt describes `retiredElements`, and beside the guard
|
|
677
730
|
// list it reads as more guards to record; the plan checks read the state file.
|
|
@@ -3,7 +3,7 @@ import { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js";
|
|
|
3
3
|
import { StateManager, UnifiedAnalysisState } from "../../utils/AnalysisStateManager.js";
|
|
4
4
|
import { TestType } from "../../types/TestTypes.js";
|
|
5
5
|
import { PlanResult } from "../../recommendation/registerPlan.js";
|
|
6
|
-
import { ObjectionAnswer, PlanChange, PlanDefect, PlanInput, VerifyContext } from "../../recommendation/types.js";
|
|
6
|
+
import { ObjectionAnswer, PlanChange, PlanDefect, PlanInput, PlanMaintenance, VerifyContext } from "../../recommendation/types.js";
|
|
7
7
|
/** The whole run's state file: the primary repository at the root, every related
|
|
8
8
|
* repository under `relatedRepos`. */
|
|
9
9
|
type RunState = Awaited<ReturnType<StateManager<UnifiedAnalysisState>["readFullState"]>>;
|
|
@@ -794,6 +794,7 @@ export interface RegisterTestPlanParams {
|
|
|
794
794
|
changes?: PlanChange[];
|
|
795
795
|
defects?: PlanDefect[];
|
|
796
796
|
plannedTests?: RegisterPlannedTestInput[];
|
|
797
|
+
maintains?: PlanMaintenance[];
|
|
797
798
|
answers?: ObjectionAnswer[];
|
|
798
799
|
}
|
|
799
800
|
/** Map the schema-validated input onto a `PlanInput`. `subjectEndpoints` is
|
|
@@ -3,6 +3,7 @@ import * as fs from "fs";
|
|
|
3
3
|
import * as path from "path";
|
|
4
4
|
import { TOOL_REGISTER_TEST_PLAN } from "../../toolNames.js";
|
|
5
5
|
import { logger } from "../../utils/logger.js";
|
|
6
|
+
import { canonicalJson, isPlainObject } from "../../utils/canonicalJson.js";
|
|
6
7
|
import { StateManager, resolveRunStatePath } from "../../utils/AnalysisStateManager.js";
|
|
7
8
|
import { toolError } from "../../utils/utils.js";
|
|
8
9
|
import { TestType } from "../../types/TestTypes.js";
|
|
@@ -80,7 +81,7 @@ export const declarationFieldsSchema = z.object({
|
|
|
80
81
|
.object({
|
|
81
82
|
outcome: z
|
|
82
83
|
.enum(["fail", "pass"])
|
|
83
|
-
.describe("Whether this test should fail or pass against the app as it stands.
|
|
84
|
+
.describe("Whether this test should fail or pass against the app as it stands. Every case a `fail` test sends carries `expectedValue` or `derived`, never both."),
|
|
84
85
|
why: z
|
|
85
86
|
.string()
|
|
86
87
|
.describe("Why it has that outcome. A test that documents a defect asserts what the FIXED code returns — the rejection, the preserved value, the correct status — so it is red today and green after the fix, never today's defective response as correct."),
|
|
@@ -176,17 +177,6 @@ function namesAScreen(url) {
|
|
|
176
177
|
* missing `stepUnderTest`, so a stray one stayed at planned test level and zod
|
|
177
178
|
* dropped it. */
|
|
178
179
|
const DECLARATION_FIELDS = Object.keys(declarationFieldsSchema.shape);
|
|
179
|
-
function isPlainObject(value) {
|
|
180
|
-
return typeof value === "object" && value !== null && !Array.isArray(value);
|
|
181
|
-
}
|
|
182
|
-
/** One value as text with object keys sorted, so the same declaration written
|
|
183
|
-
* twice by hand is not a conflict merely because its keys are ordered
|
|
184
|
-
* differently. List order still counts: two orders are two different lists. */
|
|
185
|
-
function canonicalJson(value) {
|
|
186
|
-
return JSON.stringify(value, (_key, entry) => isPlainObject(entry)
|
|
187
|
-
? Object.fromEntries(Object.keys(entry).sort().map((key) => [key, entry[key]]))
|
|
188
|
-
: entry);
|
|
189
|
-
}
|
|
190
180
|
/** Move a declaration field written one level too high into `declarations`.
|
|
191
181
|
* MEASURED: about one planned test in five did this across 141 eval planned tests, and
|
|
192
182
|
* every stray planned test-level key was a declaration field. A field in BOTH places
|
|
@@ -328,18 +318,28 @@ const registerTestPlanSchema = {
|
|
|
328
318
|
expectedValue: z
|
|
329
319
|
.union([z.string(), z.number(), z.boolean(), z.null()])
|
|
330
320
|
.optional()
|
|
331
|
-
.describe("What the response must carry when this case is sent — the value itself, not a description of it. Declare it here so the test asserts what the change is supposed to do; a value decided while the test is being written can only be the value the running app returned, which is the one value a defect cannot contradict. Leave it out when no source states the value, and the test will assert the shape instead. A test that expects to fail declares
|
|
321
|
+
.describe("What the response must carry when this case is sent — the value itself, not a description of it. Declare it here so the test asserts what the change is supposed to do; a value decided while the test is being written can only be the value the running app returned, which is the one value a defect cannot contradict. Leave it out when no source states the value, and the test will assert the shape instead. `expectedValue` and `derived` are alternatives: a case declares one of them, never both. A test that expects to fail declares one of them on every case it sends."),
|
|
322
|
+
derived: z
|
|
323
|
+
.string()
|
|
324
|
+
.optional()
|
|
325
|
+
.describe("The rule the value follows from, in the source's words, when the source states a rule and not the value: `10 percent of the subtotal`. A test computes the value from what it sent. Use this instead of `expectedValue` whenever the pull request states the rule but never writes the number."),
|
|
332
326
|
expectedFrom: z
|
|
333
327
|
.string()
|
|
334
328
|
.refine((value) => /^(pr-title|pr-description|code|spec:\S.*|convention:\S.*)$/i.test(value.trim()), {
|
|
335
329
|
message: "expectedFrom must be `pr-title`, `pr-description`, `code`, `spec:<path>`, or `convention:<file:line>`",
|
|
336
330
|
})
|
|
337
331
|
.optional()
|
|
338
|
-
.describe("Where you read `expectedValue`: `pr-title`, `pr-description`, `spec:<path>` for a requirements file the description names, `convention:<file:line>` for a rule the application already follows in code this pull request did not change, or `code` for the code under test. `code` draws an objection — it is the one source that agrees with a defect — so name it only when nothing else states the value. A value read from the pull request
|
|
332
|
+
.describe("Where you read `expectedValue` or `derived`: `pr-title`, `pr-description`, `spec:<path>` for a requirements file the description names, `convention:<file:line>` for a rule the application already follows in code this pull request did not change, or `code` for the code under test. `code` draws an objection — it is the one source that agrees with a defect — so name it only when nothing else states the value. A value read from the pull request has to be WRITTEN there — in the quoted sentence or elsewhere in the title or description. If the pull request states the rule and never writes the number, leave `expectedValue` out and put the rule in `derived`. A case that declares `expectedValue` or `derived` and names no source here draws an objection."),
|
|
339
333
|
})
|
|
340
334
|
.strict()
|
|
341
335
|
.refine((entry) => (entry.value !== undefined) !== (entry.absent === true), {
|
|
342
336
|
message: "a case states either `value` or `absent`, never both and never neither",
|
|
337
|
+
})
|
|
338
|
+
// The description called them alternatives and the schema took both,
|
|
339
|
+
// which left generation told to pin a number AND to compute it.
|
|
340
|
+
.refine((entry) => entry.expectedValue === undefined || String(entry.derived ?? "").trim().length === 0, {
|
|
341
|
+
message: "a case states either `expectedValue` or `derived`, never both: `derived` is the case saying no source writes the value",
|
|
342
|
+
path: ["derived"],
|
|
343
343
|
}))
|
|
344
344
|
.min(1, { message: "a `cases` list with no entry states no input — leave it out instead" })
|
|
345
345
|
.optional()
|
|
@@ -434,6 +434,20 @@ const registerTestPlanSchema = {
|
|
|
434
434
|
});
|
|
435
435
|
})
|
|
436
436
|
.describe("Your complete plan, unordered. Nothing ranks, dedups, reorders or trims it, and the server adds no planned test of its own."),
|
|
437
|
+
maintains: z
|
|
438
|
+
.array(z
|
|
439
|
+
.object({
|
|
440
|
+
file: z
|
|
441
|
+
.string()
|
|
442
|
+
.describe("Repository-relative path of the existing test this run edits. It has to resolve in the checkout: a change is credited as covered only when it does, and `existingCoverage` objects when it does not."),
|
|
443
|
+
differsBy: z.string().describe("What your edit changes in that test, in one sentence."),
|
|
444
|
+
changes: z
|
|
445
|
+
.array(z.string())
|
|
446
|
+
.describe("The ids of the declared changes this maintenance covers, spelled as your own `changes` list gives them."),
|
|
447
|
+
})
|
|
448
|
+
.strict())
|
|
449
|
+
.optional()
|
|
450
|
+
.describe("Existing tests this run edits instead of planning new ones. Coverage counts the changes these cite, so a change already covered by a spec you are updating needs no planned test — and a duplicate spec beside the one you edited is the mistake. Declare maintenance here even when the plan holds no planned tests at all: a maintenance-only plan is a complete plan."),
|
|
437
451
|
answers: z
|
|
438
452
|
.array(z.object({
|
|
439
453
|
objectionId: z.string().describe("The objection's id, exactly as this tool reported it. Answers are matched by id, never by position."),
|
|
@@ -456,6 +470,7 @@ export function buildRegistration(params, registrationNumber) {
|
|
|
456
470
|
changes: params.changes ?? [],
|
|
457
471
|
defects: params.defects ?? [],
|
|
458
472
|
answers: params.answers ?? [],
|
|
473
|
+
...(params.maintains?.length ? { maintains: params.maintains } : {}),
|
|
459
474
|
// The tool counts the run's registrations; whatever the caller sent is ignored.
|
|
460
475
|
registrationNumber,
|
|
461
476
|
plannedTests: (params.plannedTests ?? []).map((input) => {
|
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import type { TestExecutionStatus } from "./TestExecution.js";
|
|
2
|
+
import type { UtilsViolationKind } from "../utils/utils-verify/allow.js";
|
|
2
3
|
/**
|
|
3
4
|
* The code-reuse observability shape carried in a TestbotReport — the wire contract
|
|
4
5
|
* between this server and the consumers that render it (test-bot.git). It covers
|
|
@@ -86,6 +87,26 @@ export declare enum HelperVerificationOutcome {
|
|
|
86
87
|
/** The test neither wrote nor imports a shared utils file. */
|
|
87
88
|
NoUtilsFile = "no-utils-file"
|
|
88
89
|
}
|
|
90
|
+
/** Every blocking kind a report-time reuse verification can find. Shared-helper
|
|
91
|
+
* path: the `UtilsViolationKind`s, plus `malformed-allow-marker` for an allow
|
|
92
|
+
* marker the grammar cannot read (which fails a verify on its own). Page-object
|
|
93
|
+
* path: a POM member the spec calls that its source does not define, a spec that
|
|
94
|
+
* does not parse, a spec whose Skyramp runtime imports were stripped, and a spec
|
|
95
|
+
* with no page-object reuse whose selectors match detected page objects and carry
|
|
96
|
+
* no `// kept inline:` decline. A union, so a misspelled kind is a compile error. */
|
|
97
|
+
export type ReuseVerificationFailureKind = UtilsViolationKind | "malformed-allow-marker" | "pom-member-missing" | "parse-error" | "skyramp-infra-missing" | "zero-reuse";
|
|
98
|
+
/** One blocking kind a failed reuse verification found in the delivered files,
|
|
99
|
+
* aggregated: the verifier reports a finding per helper, import line or POM call,
|
|
100
|
+
* and the row states the kind once with its count. `detail` is the findings' own
|
|
101
|
+
* detail text (joined with `; ` when there are several) — for an `assertion-loss`
|
|
102
|
+
* that is the sentence `verify.ts` builds with the counts behind the verdict
|
|
103
|
+
* (baseline, delivered, in the test, in the helpers); it is NOT the gate's
|
|
104
|
+
* PASSED/FAILED prose. */
|
|
105
|
+
export interface ReuseVerificationFailure {
|
|
106
|
+
kind: ReuseVerificationFailureKind;
|
|
107
|
+
count: number;
|
|
108
|
+
detail: string;
|
|
109
|
+
}
|
|
89
110
|
/** A pre-existing Skyramp-generated test the reuse pass edited to import from the
|
|
90
111
|
* shared utils file. `execution` is the recorded post-edit run; a retrofit without
|
|
91
112
|
* one cannot be reported (skyramp_submit_report refuses). */
|
|
@@ -99,13 +120,24 @@ export interface RetrofitOutcome {
|
|
|
99
120
|
* Absent altogether when no utils file exists for the test (the same omission rule
|
|
100
121
|
* as the POM fields at zero candidates: a zero says nothing "no file" does not). */
|
|
101
122
|
export interface HelperReuseOutcome {
|
|
102
|
-
/**
|
|
103
|
-
* found by its header, never by a conventional name (the
|
|
104
|
-
* new file with a different name" once the conventional
|
|
105
|
-
*
|
|
123
|
+
/** REPOSITORY-RELATIVE PATH (posix) of the shared utils file the test imports
|
|
124
|
+
* from or this run wrote — found by its header, never by a conventional name (the
|
|
125
|
+
* reuse prompt allows "a new file with a different name" once the conventional
|
|
126
|
+
* one is large). A path, not a basename, so two same-named modules in one
|
|
127
|
+
* delivery (an API and a UI `skyrampUtils.ts`) stay distinguishable; never
|
|
128
|
+
* absolute, which would leak the runner's temp directory. Unique within ONE
|
|
129
|
+
* repository only: in a multi-repo report the same path can occur in two
|
|
130
|
+
* checkouts, so a renderer must key on the row's `repository` plus this field.
|
|
131
|
+
* Several files are joined with `, `. */
|
|
106
132
|
utilsFile: string;
|
|
107
133
|
/** Helpers the delivered test imports from that file that the file defines. */
|
|
108
134
|
helpersImported?: number;
|
|
135
|
+
/** Those helpers by name, sorted; empty when nothing is imported. Always equals
|
|
136
|
+
* `helpersImported` in length — both come from the same set.
|
|
137
|
+
*
|
|
138
|
+
* TODO: remove `helpersImported` once the testbot renderer reads this field. The
|
|
139
|
+
* two carry one fact; the count stays only because testbot#349 reads it today. */
|
|
140
|
+
helperNames: string[];
|
|
109
141
|
/** Inline request calls in OTHER Skyramp-generated tests beside this one that a
|
|
110
142
|
* helper in `utilsFile` already wraps — reuse that was available and not taken.
|
|
111
143
|
* Omitted at zero. Informational, like {@link ReuseOutcome.missedReuse}. */
|
|
@@ -114,13 +146,35 @@ export interface HelperReuseOutcome {
|
|
|
114
146
|
* per method+path, status-code-only assertions, method+resource names — with
|
|
115
147
|
* documented declines (`reuse-verify: allow …`) counted as holding. */
|
|
116
148
|
verification?: HelperVerificationOutcome;
|
|
149
|
+
/** Why `verification` is `failed`: one entry per blocking kind found, with how
|
|
150
|
+
* many findings of it and the finding's own detail text — see
|
|
151
|
+
* {@link ReuseVerificationFailure}. Present only on a `failed` row. A bare
|
|
152
|
+
* `failed` cannot be acted on and its cause cannot be reconstructed from the
|
|
153
|
+
* artifacts; this is the statement of what failed.
|
|
154
|
+
*
|
|
155
|
+
* How it reaches a report: skyramp_submit_report refuses a report carrying a
|
|
156
|
+
* blocking verdict and names these entries to the agent, up to a bound per file
|
|
157
|
+
* (`REUSE_SUBMIT_MAX_REFUSALS`); past the bound the report is accepted and the
|
|
158
|
+
* row ships with this field set, so the fault is disclosed rather than delivered
|
|
159
|
+
* in silence — or not delivered at all. */
|
|
160
|
+
verificationFailures?: ReuseVerificationFailure[];
|
|
117
161
|
/** Pre-existing generated tests this spec's reuse pass rewired onto the utils
|
|
118
162
|
* file, each with its recorded execution. Omitted when none. */
|
|
119
163
|
retrofits?: RetrofitOutcome[];
|
|
120
|
-
/** Whether the delivered test plus the helpers it imports still hold
|
|
121
|
-
*
|
|
122
|
-
*
|
|
164
|
+
/** Whether the delivered test plus the helpers it imports and calls still hold at
|
|
165
|
+
* least as many assertions as when modularization was handed out. A measured
|
|
166
|
+
* count: a documented decline (`reuse-verify: allow assertion-loss …`) clears
|
|
167
|
+
* `verification` and leaves this `false`. Present only when that baseline exists.
|
|
168
|
+
* Says nothing about the page-error guard — that is
|
|
169
|
+
* {@link pageErrorGuardPreserved}; folding the two into one flag once sent an
|
|
170
|
+
* investigation after a lost assertion when a guard had been deleted. */
|
|
123
171
|
assertionsPreserved?: boolean;
|
|
172
|
+
/** Whether the `page.on('pageerror', …)` guard present at the modularization
|
|
173
|
+
* hand-out is still registered, in the test or in a shared helper it imports and
|
|
174
|
+
* calls. `true` when the hand-out had no guard to lose. Present whenever the
|
|
175
|
+
* baseline exists, absent otherwise — like every optional member here, absent
|
|
176
|
+
* means "not measured", never "false". */
|
|
177
|
+
pageErrorGuardPreserved?: boolean;
|
|
124
178
|
}
|
|
125
179
|
/**
|
|
126
180
|
* What lands in a report's `reuse` field for a generated test. Every member is
|
|
@@ -161,7 +215,19 @@ export interface ReuseOutcome {
|
|
|
161
215
|
/** The subset of those raw locators that a catalogued POM member covers.
|
|
162
216
|
* Informational; see {@link ReuseMissedEntry}. */
|
|
163
217
|
missedReuse?: ReuseMissedEntry[];
|
|
218
|
+
/** Copies of operations written in two or more Skyramp-generated tests beside (and
|
|
219
|
+
* including) this one that NO shared helper wraps — extraction that was due and
|
|
220
|
+
* not performed, the counterpart of `helpers.siblingInlineCallSites` (use not
|
|
221
|
+
* taken). Top level, not on the helpers row: it describes the spec's directory,
|
|
222
|
+
* not the module, and is most telling when no module exists at all — the case
|
|
223
|
+
* where `helpers` is rightly absent. Omitted at zero. Informational. */
|
|
224
|
+
unextractedDuplicateSites?: number;
|
|
164
225
|
verification?: ReuseVerificationOutcome;
|
|
226
|
+
/** Why `verification` is `failed` on the page-object path, on the same terms as
|
|
227
|
+
* {@link HelperReuseOutcome.verificationFailures}: present only when the
|
|
228
|
+
* delivered spec fails the live POM check at report time and the report was
|
|
229
|
+
* accepted past the refusal bound. */
|
|
230
|
+
verificationFailures?: ReuseVerificationFailure[];
|
|
165
231
|
skipped?: ReuseSkippedEntry[];
|
|
166
232
|
/** Shared-helper (SkyrampUtils) reuse. See {@link HelperReuseOutcome}. */
|
|
167
233
|
helpers?: HelperReuseOutcome;
|
|
@@ -53,6 +53,12 @@ export interface MaintenanceActionCore {
|
|
|
53
53
|
* the rewritten PNG — a listed baseline must never exempt the selector edit the same
|
|
54
54
|
* verdict claimed (SKYR-3883 stays in force). */
|
|
55
55
|
rebaselineOnly?: boolean;
|
|
56
|
+
/** True when the action is a RECOMMENDATION the developer applies by hand and this run
|
|
57
|
+
* touched nothing: a REGENERATE or DELETE on an external (user-authored) test. Part of
|
|
58
|
+
* the core shape because the action stays REAL rather than flattening to VERIFY, so
|
|
59
|
+
* without this a downstream stage cannot tell a regeneration that rewrote a file from
|
|
60
|
+
* one that only advised it — and maintenance coverage would credit the advice. */
|
|
61
|
+
reportOnly?: boolean;
|
|
56
62
|
}
|
|
57
63
|
/** Normalized internal recommendation built from LLM-supplied args.recommendations. */
|
|
58
64
|
export interface DriftRecommendation extends MaintenanceActionCore {
|
|
@@ -72,6 +72,10 @@ export interface TestbotReport {
|
|
|
72
72
|
editedFileName?: string;
|
|
73
73
|
description: string;
|
|
74
74
|
action: DriftAction;
|
|
75
|
+
/** `owner/repo` the maintained test lives in. Always set, unlike the other
|
|
76
|
+
* sections: `fileName` is a basename, so in a multi-repo run this is the
|
|
77
|
+
* only thing separating two same-named test files (SKYR-4233). */
|
|
78
|
+
repository: string;
|
|
75
79
|
beforeStatus: TestExecutionStatus;
|
|
76
80
|
beforeDetails: string;
|
|
77
81
|
afterStatus: TestExecutionStatus;
|
|
@@ -154,8 +158,18 @@ export interface TestbotReport {
|
|
|
154
158
|
text: string;
|
|
155
159
|
source: string;
|
|
156
160
|
/** `plannedTestId` of each delivered test citing this change. Empty where the
|
|
157
|
-
* agent answered instead of planning one
|
|
161
|
+
* agent answered instead of planning one, and where the change is covered by
|
|
162
|
+
* an existing test the run maintained — `maintainedBy` carries those. */
|
|
158
163
|
testedBy: string[];
|
|
164
|
+
/** Path of each existing test the run edited that cites this change: a
|
|
165
|
+
* `maintains` entry from the plan, kept only where a `testMaintenance` row
|
|
166
|
+
* names the same file. Checked at name level, which is the level v2 allows
|
|
167
|
+
* for a file, so this says the run reported editing it and not that the edit
|
|
168
|
+
* is known to be right. A claim no maintenance row corroborates is left out
|
|
169
|
+
* and draws `deliveredMatchesPlan:maintains:` instead. A change covered this
|
|
170
|
+
* way has no planned test and needs no answer, so without this field the
|
|
171
|
+
* change table would read as untested. */
|
|
172
|
+
maintainedBy?: string[];
|
|
159
173
|
/** The agent's answer to `coverage:change:<id>`, where it gave one. */
|
|
160
174
|
answer?: string;
|
|
161
175
|
}[];
|
|
@@ -133,7 +133,7 @@ export interface UiAnalysisContext {
|
|
|
133
133
|
* PATH. Recorded in-process where staging finds the edit; skyramp_submit_report
|
|
134
134
|
* refuses a report that leaves one undisclosed or unexecuted. */
|
|
135
135
|
export interface ReuseRetrofit {
|
|
136
|
-
/**
|
|
136
|
+
/** Repository-relative path(s) of the utils file(s) it now imports from. */
|
|
137
137
|
utilsFile: string;
|
|
138
138
|
/** Absolute path of the generated test whose reuse pass caused the edit. */
|
|
139
139
|
testFile: string;
|
|
@@ -184,6 +184,12 @@ export interface UnifiedAnalysisState {
|
|
|
184
184
|
reuseHandOffs?: Record<string, ReuseHandOff>;
|
|
185
185
|
/** See {@link ReuseRetrofit}. Never supplied by the LLM. */
|
|
186
186
|
reuseRetrofits?: Record<string, ReuseRetrofit>;
|
|
187
|
+
/** How many times skyramp_submit_report has refused the report over a blocking
|
|
188
|
+
* reuse verdict on each spec, keyed by the spec's ABSOLUTE PATH (the verdict's
|
|
189
|
+
* `file`). The refusal is bounded (`REUSE_SUBMIT_MAX_REFUSALS`): past the bound
|
|
190
|
+
* the report is accepted with the fault recorded in the row. Written by
|
|
191
|
+
* skyramp_submit_report only; never supplied by the LLM. */
|
|
192
|
+
reuseRefusals?: Record<string, number>;
|
|
187
193
|
/**
|
|
188
194
|
* SKYR-4156. Recorded video per executed browser test, keyed by test-file
|
|
189
195
|
* BASENAME (the same key `reuseOutcomes` uses, so matching needs no path
|
|
@@ -295,7 +295,11 @@ async function writeFileAtomically(target, contents) {
|
|
|
295
295
|
throw error;
|
|
296
296
|
}
|
|
297
297
|
}
|
|
298
|
-
const ROOT_FIELDS_OWNED_ELSEWHERE = [
|
|
298
|
+
const ROOT_FIELDS_OWNED_ELSEWHERE = [
|
|
299
|
+
"plan",
|
|
300
|
+
"registrationCount",
|
|
301
|
+
"reportObjections",
|
|
302
|
+
];
|
|
299
303
|
function carryForwardRootFields(data, existing) {
|
|
300
304
|
if (!existing)
|
|
301
305
|
return {};
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { escapeRegExp } from "../regex.js";
|
|
2
2
|
import { strippedSources } from "./strip-dispatch.js";
|
|
3
|
-
import { balancedCloseIndex, identifierRe, inScope, interpolationRe, lineOfOffset, } from "./lint-types.js";
|
|
3
|
+
import { balancedCloseIndex, identifierRe, inScope, interpolationRe, lineOfOffset, testBlockIndex, testStartOffsets, } from "./lint-types.js";
|
|
4
4
|
/**
|
|
5
5
|
* Deterministic checks shared by integration and provider-contract tests — the
|
|
6
6
|
* machine-checkable slice of the D5 rubric's echo_computed and array_depth
|
|
@@ -12,6 +12,12 @@ const RESPONSE_BINDING_RES = {
|
|
|
12
12
|
python: /^[ \t]*([A-Za-z_]\w*)\s*=\s*[\w.]*send_request\s*\(/gm,
|
|
13
13
|
};
|
|
14
14
|
const STATUS_USE_RE = /^\s*\.\s*(?:statusCode|status_code|status)\b/;
|
|
15
|
+
/** The SDK body-read call for a binding, spelled for the language. */
|
|
16
|
+
function helperCall(language, varName) {
|
|
17
|
+
return language === "python"
|
|
18
|
+
? `skyramp.get_response_value(${varName}, "json.path")`
|
|
19
|
+
: `getValue(${varName}, "json.path")`;
|
|
20
|
+
}
|
|
15
21
|
/** All send_request/sendRequest bindings in the stripped source. */
|
|
16
22
|
function responseBindings(stripped, language) {
|
|
17
23
|
const bindings = [];
|
|
@@ -148,11 +154,17 @@ function lintStatusOnlyResponses(stripped, commentless, language, bindings) {
|
|
|
148
154
|
message: statusUses > 0
|
|
149
155
|
? severity === "hard"
|
|
150
156
|
? `Error response \`${varName}\` is only checked for its status code — the error body (code/message/detail) must be asserted.`
|
|
151
|
-
: `Response \`${varName}\` is only checked for its status code
|
|
157
|
+
: `Response \`${varName}\` is only checked for its status code.`
|
|
152
158
|
: `Response \`${varName}\` is captured but never asserted.`,
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
159
|
+
// The advisory case is phrased as a question, not an instruction. A
|
|
160
|
+
// response no behaviour in the file depends on is CORRECT with its
|
|
161
|
+
// status code and nothing more (SKYR-4310) — telling the agent to
|
|
162
|
+
// assert the body regardless is how a 136-assertion test gets written
|
|
163
|
+
// to prove a default page size. The hard case is unchanged: an error
|
|
164
|
+
// body carries the evidence, so it is never optional.
|
|
165
|
+
remediation: severity === "warn"
|
|
166
|
+
? `If a behaviour this file tests depends on this response's body, assert the fields it depends on via \`${helperCall(language, varName)}\`. If no behaviour depends on it, the status code alone is the right answer — leave it.`
|
|
167
|
+
: `Assert the error body's fields via \`${helperCall(language, varName)}\` after the status assertion.`,
|
|
156
168
|
});
|
|
157
169
|
}
|
|
158
170
|
return findings;
|
|
@@ -303,23 +315,6 @@ const ERROR_ABSENT_RE = /(?:getValue|getResponseValue|get_response_value)\s*\(\s
|
|
|
303
315
|
// expect(getValue(r, "items").length).toBeGreaterThan(0)
|
|
304
316
|
// assert len(skyramp.get_response_value(r, "items")) > 0
|
|
305
317
|
const NON_EMPTY_RE = /(?:getValue|getResponseValue)\s*\(\s*([\w$.]+)\s*,\s*['"][^'"]+['"]\s*\)\s*\.\s*length\s*(?:,\s*['"][^'"]*['"]\s*)?\)\s*\.\s*toBeGreaterThan\s*\(\s*0\s*\)|len\s*\(\s*[\w.]*get_response_value\s*\(\s*([\w$.]+)\s*,\s*['"][^'"]+['"]\s*\)\s*\)\s*>\s*0/g;
|
|
306
|
-
// Starts of test bodies: JS/TS `it(` / `test(` (with `.only`/`.skip`/`.each`
|
|
307
|
-
// modifiers) and Python `def test_`. Two assertions belong to the same test
|
|
308
|
-
// only when no such start lies between them — a `const r = …` re-bound in the
|
|
309
|
-
// next `it` block is a different response, not the same one.
|
|
310
|
-
const TEST_START_RE = /(?<![\w$.])(?:it|test)(?:\s*\.\s*\w+)*\s*\(|^[ \t]*(?:async\s+)?def\s+test_/gm;
|
|
311
|
-
function testBlockIndex(starts, offset) {
|
|
312
|
-
let lo = 0;
|
|
313
|
-
let hi = starts.length;
|
|
314
|
-
while (lo < hi) {
|
|
315
|
-
const mid = (lo + hi) >> 1;
|
|
316
|
-
if (starts[mid] <= offset)
|
|
317
|
-
lo = mid + 1;
|
|
318
|
-
else
|
|
319
|
-
hi = mid;
|
|
320
|
-
}
|
|
321
|
-
return lo;
|
|
322
|
-
}
|
|
323
318
|
/** A success shape asserted together with the absence of an error field on the
|
|
324
319
|
* same response binding, inside the same test body. Both are legitimate on a
|
|
325
320
|
* real success body, so this is advisory only: the pattern is exactly what a
|
|
@@ -328,11 +323,8 @@ function testBlockIndex(starts, offset) {
|
|
|
328
323
|
* validation error carried in `error`). The verifier cannot see the response,
|
|
329
324
|
* so it asks the author to confirm the classification rather than blocking. */
|
|
330
325
|
function lintSuccessShapeWithErrorAbsent(commentless) {
|
|
331
|
-
const starts =
|
|
332
|
-
TEST_START_RE.lastIndex = 0;
|
|
326
|
+
const starts = testStartOffsets(commentless);
|
|
333
327
|
let m;
|
|
334
|
-
while ((m = TEST_START_RE.exec(commentless)) !== null)
|
|
335
|
-
starts.push(m.index);
|
|
336
328
|
// key: `<testBlock>:<binding>` → { offset, field }
|
|
337
329
|
const errorAbsent = new Map();
|
|
338
330
|
ERROR_ABSENT_RE.lastIndex = 0;
|
|
@@ -358,7 +350,39 @@ function lintSuccessShapeWithErrorAbsent(commentless) {
|
|
|
358
350
|
severity: "warn",
|
|
359
351
|
line: lineOfOffset(commentless, absent.offset),
|
|
360
352
|
message: `\`${varName}\` is asserted to have no \`${absent.field}\` field and a non-empty collection — a success shape.`,
|
|
361
|
-
remediation:
|
|
353
|
+
remediation: 'Confirm the recorded response for this exact request is a success body. If the endpoint answers this request with an error or no-match body (a validation 4xx, or a 2xx carrying `{"error": ...}`), assert that body\'s exact fields instead and report the gap in issuesFound. Keep the assertion as written only when the plan declared this test `expected.outcome: fail` (a bug_caught or requirement_conflict test that deliberately asserts the intended behaviour).',
|
|
354
|
+
});
|
|
355
|
+
}
|
|
356
|
+
return findings;
|
|
357
|
+
}
|
|
358
|
+
// The existence idiom on an SDK-helper read: `not.toBeNull()` / `is not None`.
|
|
359
|
+
// Weight 1 in the strength model, and the one shape MAINTENANCE_SCOPE_NOTE
|
|
360
|
+
// names by example.
|
|
361
|
+
const WEAK_EXISTENCE_RE = /(?:getValue|getResponseValue)\s*\(\s*[\w$.]+\s*,\s*['"]([^'"]+)['"][^)]*\)\s*(?:,\s*['"][^'"]*['"]\s*)?\)\s*\.\s*not\s*\.\s*toBeNull\s*\(|get_response_value\s*\(\s*[\w.]+\s*,\s*['"]([^'"]+)['"]\s*\)\s*is\s+not\s+None\b/g;
|
|
362
|
+
/** Maintenance only: an existence-tier assertion left standing inside a test
|
|
363
|
+
* function the agent TOUCHED. `MAINTENANCE_SCOPE_NOTE` holds such a function
|
|
364
|
+
* to the same standard as a new test — "an existing `is not None` on a field
|
|
365
|
+
* the diff gives a knowable value becomes that value".
|
|
366
|
+
*
|
|
367
|
+
* Warn, never hard: the verifier has no diff, so it cannot tell which fields
|
|
368
|
+
* the change made knowable. It reports the candidates and lets the agent
|
|
369
|
+
* decide, which is the honest split — the alternative is blocking a spec on a
|
|
370
|
+
* field whose value genuinely still varies. */
|
|
371
|
+
function lintWeakAssertionInTouchedTest(commentless, opts) {
|
|
372
|
+
const findings = [];
|
|
373
|
+
WEAK_EXISTENCE_RE.lastIndex = 0;
|
|
374
|
+
let m;
|
|
375
|
+
while ((m = WEAK_EXISTENCE_RE.exec(commentless)) !== null) {
|
|
376
|
+
const line = lineOfOffset(commentless, m.index);
|
|
377
|
+
if (!inScope(line, opts))
|
|
378
|
+
continue;
|
|
379
|
+
const field = m[1] ?? m[2];
|
|
380
|
+
findings.push({
|
|
381
|
+
rule: "weak-assertion-in-touched-test",
|
|
382
|
+
severity: "warn",
|
|
383
|
+
line,
|
|
384
|
+
message: `\`${field}\` is asserted only to exist, in a test function this change touches.`,
|
|
385
|
+
remediation: `A test function you touch is held to the same standard as a new one. If the diff gives \`${field}\` a knowable value, assert that value exactly. Leave the existence check only when the value genuinely varies across runs.`,
|
|
362
386
|
});
|
|
363
387
|
}
|
|
364
388
|
return findings;
|
|
@@ -371,15 +395,28 @@ export function lintApiShared(raw, language, opts) {
|
|
|
371
395
|
const maintenance = opts?.scopeLines !== undefined;
|
|
372
396
|
const { commentless, stripped } = strippedSources(raw, language);
|
|
373
397
|
const findings = [];
|
|
398
|
+
// Bindings discovered once, shared by both binding-scoped lints.
|
|
399
|
+
const bindings = responseBindings(stripped, language);
|
|
400
|
+
findings.push(...lintStatusOnlyResponses(stripped, commentless, language, bindings));
|
|
401
|
+
findings.push(...lintSdkHelperBypass(stripped, bindings, language));
|
|
402
|
+
findings.push(...lintShapeOnlyChecks(commentless));
|
|
403
|
+
findings.push(...lintSuccessShapeWithErrorAbsent(commentless));
|
|
374
404
|
if (!maintenance) {
|
|
375
|
-
//
|
|
376
|
-
|
|
377
|
-
|
|
378
|
-
findings.push(...lintSdkHelperBypass(stripped, bindings, language));
|
|
379
|
-
findings.push(...lintShapeOnlyChecks(commentless));
|
|
405
|
+
// Array-depth findings anchor to a path, not a line, so there is nothing
|
|
406
|
+
// to scope them by — running them in maintenance would report untouched
|
|
407
|
+
// customer arrays with no way to filter them out.
|
|
380
408
|
findings.push(...lintArrayDepth(commentless));
|
|
381
|
-
|
|
409
|
+
}
|
|
410
|
+
else {
|
|
411
|
+
findings.push(...lintWeakAssertionInTouchedTest(commentless, opts));
|
|
382
412
|
}
|
|
383
413
|
findings.push(...lintPermissiveStatusMatchers(commentless, opts));
|
|
384
|
-
|
|
414
|
+
// Maintenance used to skip every lint above except the status-matcher one,
|
|
415
|
+
// because line-scoping to the ADDED lines would have flagged pre-existing
|
|
416
|
+
// customer code. The scope is now the touched test FUNCTION (SKYR-4310), so
|
|
417
|
+
// these can run: each finding is kept only when it lands inside a function
|
|
418
|
+
// the agent edited. A finding with no line cannot be placed and is dropped.
|
|
419
|
+
return maintenance
|
|
420
|
+
? findings.filter((f) => f.line !== undefined && inScope(f.line, opts))
|
|
421
|
+
: findings;
|
|
385
422
|
}
|
|
@@ -51,8 +51,8 @@ export function formatAssertionVerifyReport(r, testFile) {
|
|
|
51
51
|
if (r.strengthGateFailed) {
|
|
52
52
|
lines.push("- Assertion strength did not increase over the generated baseline: no assertion was added and no weak matcher was strengthened (e.g. `toBeVisible` → `toHaveText`). Apply the enhancement rules. A documented decline (`assertions complete` comment) is accepted only AFTER at least one assertion was added or replaced — a marker-only edit does not pass this gate.");
|
|
53
53
|
}
|
|
54
|
-
if (r.
|
|
55
|
-
lines.push(
|
|
54
|
+
if (r.weakAdditionsDominant) {
|
|
55
|
+
lines.push(`- The assertions added since the baseline are mostly existence/visibility-tier (\`toBeVisible\`, \`not.toBeNull\`, \`is not None\`, ranges)${r.meanAddedWeight !== undefined ? ` — mean strength ${r.meanAddedWeight.toFixed(1)} against a bar of 2.0` : ""}. When the exact text/value/count is knowable from the trace, source, or request, assert it exactly (\`toHaveText\`/\`toHaveCount\`/\`toBe\`/\`==\`). Adding MORE weak assertions lowers this figure — the fix is to strengthen the ones the behaviour under test depends on, and to drop the ones it does not. If only existence is genuinely knowable, document the decline:`, ` ${token} assertions complete: ${spec} — <reason exact values are not knowable>`);
|
|
56
56
|
}
|
|
57
57
|
lines.push(...warningsBlock(r.warnings, finding));
|
|
58
58
|
lines.push("Re-run this verification (verify: true) after every fix pass — skyramp_execute_test runs the same checks automatically and will not execute this test until they pass.");
|
|
@@ -4,3 +4,10 @@
|
|
|
4
4
|
* uncalled helper is skipped, so its subjects don't excuse anything.
|
|
5
5
|
*/
|
|
6
6
|
export declare function importedHelperSubjects(testFile: string, content: string): Promise<Set<string>>;
|
|
7
|
+
/**
|
|
8
|
+
* The same subjects with their CAPACITY — how many removals each may excuse:
|
|
9
|
+
* one per call site of the helper function that asserts it (an assertion the
|
|
10
|
+
* parser cannot place in a function gets one per call through the import).
|
|
11
|
+
* The removal gate consumes this budget per removed baseline subject.
|
|
12
|
+
*/
|
|
13
|
+
export declare function importedHelperSubjectCapacities(testFile: string, content: string): Promise<Map<string, number>>;
|