@autonoma-ai/planner 0.1.23 → 0.1.24

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -76,16 +76,24 @@ var init_debug = __esm({
76
76
 
77
77
  // src/core/posthog.ts
78
78
  function getPostHogConfig() {
79
+ const signature = [
80
+ process.env.AUTONOMA_POSTHOG_KEY,
81
+ process.env.AUTONOMA_POSTHOG_HOST,
82
+ process.env.DONT_TRACK
83
+ ].join("\0");
84
+ if (cached?.signature === signature) return cached.config;
79
85
  const env = readEnv();
80
86
  const key = (env.AUTONOMA_POSTHOG_KEY ?? POSTHOG_PUBLIC_KEY).trim();
81
87
  const optedOut = env.DONT_TRACK === "1" || env.DONT_TRACK === "true";
82
- return {
88
+ const config = {
83
89
  key,
84
90
  host: (env.AUTONOMA_POSTHOG_HOST ?? DEFAULT_HOST).replace(/\/+$/, ""),
85
91
  enabled: !optedOut && key.length > 0
86
92
  };
93
+ cached = { signature, config };
94
+ return config;
87
95
  }
88
- var POSTHOG_PUBLIC_KEY, DEFAULT_HOST;
96
+ var POSTHOG_PUBLIC_KEY, DEFAULT_HOST, cached;
89
97
  var init_posthog = __esm({
90
98
  "src/core/posthog.ts"() {
91
99
  "use strict";
@@ -211,15 +219,6 @@ function nowUnixNano() {
211
219
  }
212
220
  function captureLog(level, message, attributes = {}) {
213
221
  if (!getPostHogConfig().enabled) return;
214
- if (recordsThisRun >= MAX_RECORDS_PER_RUN) {
215
- if (!capReported) {
216
- capReported = true;
217
- enqueue(buildRecord("warn", "Log budget exhausted for this run; further logs are dropped", {}));
218
- flush();
219
- }
220
- return;
221
- }
222
- recordsThisRun++;
223
222
  enqueue(buildRecord(level, message, attributes));
224
223
  }
225
224
  async function flushLogs(timeoutMs = 1500) {
@@ -277,7 +276,7 @@ function buildRecord(level, message, attributes) {
277
276
  timeUnixNano: `${nowUnixNano()}`,
278
277
  severityNumber: SEVERITY_NUMBERS[level],
279
278
  severityText: level.toUpperCase(),
280
- body: { stringValue: truncate(message, MAX_MESSAGE_CHARS) },
279
+ body: { stringValue: message },
281
280
  attributes: toKeyValues(merged)
282
281
  };
283
282
  }
@@ -304,12 +303,9 @@ function toAnyValue(value) {
304
303
  if (typeof value === "number") {
305
304
  return Number.isInteger(value) ? { intValue: String(value) } : { doubleValue: value };
306
305
  }
307
- return { stringValue: truncate(value, MAX_ATTRIBUTE_CHARS) };
308
- }
309
- function truncate(text2, max) {
310
- return text2.length <= max ? text2 : `${text2.slice(0, max)}...`;
306
+ return { stringValue: value };
311
307
  }
312
- var LOGS_PATH, SERVICE_NAME, MAX_BATCH_RECORDS, FLUSH_INTERVAL_MS, MAX_RECORDS_PER_RUN, MAX_MESSAGE_CHARS, MAX_ATTRIBUTE_CHARS, SEVERITY_NUMBERS, CLOCK_ORIGIN_MS, CLOCK_ORIGIN_NS, queue, inFlight, recordsThisRun, capReported, flushTimer;
308
+ var LOGS_PATH, SERVICE_NAME, MAX_BATCH_RECORDS, FLUSH_INTERVAL_MS, SEVERITY_NUMBERS, CLOCK_ORIGIN_MS, CLOCK_ORIGIN_NS, queue, inFlight, flushTimer;
313
309
  var init_logs = __esm({
314
310
  "src/core/logs.ts"() {
315
311
  "use strict";
@@ -321,9 +317,6 @@ var init_logs = __esm({
321
317
  SERVICE_NAME = "autonoma-planner";
322
318
  MAX_BATCH_RECORDS = 50;
323
319
  FLUSH_INTERVAL_MS = 2e3;
324
- MAX_RECORDS_PER_RUN = 5e3;
325
- MAX_MESSAGE_CHARS = 2e3;
326
- MAX_ATTRIBUTE_CHARS = 500;
327
320
  SEVERITY_NUMBERS = {
328
321
  debug: 5,
329
322
  info: 9,
@@ -334,8 +327,6 @@ var init_logs = __esm({
334
327
  CLOCK_ORIGIN_NS = process.hrtime.bigint();
335
328
  queue = [];
336
329
  inFlight = /* @__PURE__ */ new Set();
337
- recordsThisRun = 0;
338
- capReported = false;
339
330
  }
340
331
  });
341
332
 
@@ -2196,7 +2187,7 @@ function analysisVerdictPlane(category) {
2196
2187
  const tier = analysisFindingTier(category);
2197
2188
  return tier === "bug" || tier === "passed" ? "app_health" : "coverage";
2198
2189
  }
2199
- var analysisVerdictSchema, ANALYSIS_VERDICT, VERDICT_TIER, coverageVerdicts, analysisTestOriginSchema, coverageCategoryCountSchema, coverageSummarySchema, analysisClassificationReportSchema, analysisClassificationSummarySchema, analysisFindingViewSchema, analysisReportDataSchema, analysisIssueKindSchema, analysisIssueSeveritySchema, analysisIssueStatusSchema, resolvedPrimaryScreenshotSchema, analysisIssueSummarySchema, analysisIssueFindingInstanceSchema, analysisIssueDetailSchema, analysisSnapshotIssueChangesSchema;
2190
+ var analysisVerdictSchema, ANALYSIS_VERDICT, VERDICT_TIER, coverageVerdicts, analysisTestOriginSchema, coverageCategoryCountSchema, coverageSummarySchema, analysisClassificationReportSchema, analysisClassificationSummarySchema, analysisFindingViewSchema, analysisReportDataSchema, analysisIssueKindSchema, analysisIssueSeveritySchema, analysisIssueStatusSchema, resolvedPrimaryScreenshotSchema, analysisIssueSummarySchema, analysisIssueFindingInstanceSchema, analysisIssueDetailSchema, analysisPrCoveredTestSchema, analysisPrIssueSchema, analysisPrNewerRunSchema, analysisForPrSchema, analysisSnapshotIssueChangesSchema;
2200
2191
  var init_analysis = __esm({
2201
2192
  "../../packages/types/src/schemas/analysis.ts"() {
2202
2193
  "use strict";
@@ -2389,6 +2380,71 @@ var init_analysis = __esm({
2389
2380
  resolvedAt: z8.date().optional(),
2390
2381
  findingInstances: z8.array(analysisIssueFindingInstanceSchema)
2391
2382
  });
2383
+ analysisPrCoveredTestSchema = z8.object({
2384
+ slug: z8.string(),
2385
+ origin: analysisTestOriginSchema.optional(),
2386
+ /** Impact Analysis's reason for selecting this test for the run. */
2387
+ selectionReason: z8.string().optional(),
2388
+ /** The test's terminal verdict in the run that attributed it here. A plain string, so a stored value outside the
2389
+ * current taxonomy still reads as a label instead of failing the payload. */
2390
+ category: z8.string()
2391
+ });
2392
+ analysisPrIssueSchema = z8.object({
2393
+ id: z8.string(),
2394
+ title: z8.string(),
2395
+ /** What kind of failure this is, which decides WHERE the fix lives: a `bug` is fixed in the repo, while
2396
+ * `environment` and `scenario` are fixed in Autonoma (secrets/preview config, and scenario recipes). */
2397
+ kind: analysisIssueKindSchema,
2398
+ severity: analysisIssueSeveritySchema,
2399
+ expectedBehavior: z8.string().optional(),
2400
+ actualBehavior: z8.string(),
2401
+ /** The grounded diagnosis: how the referenced code produces the symptom, with file:line references and the
2402
+ * verbatim lines that were read. A lead to confirm, never a verdict. */
2403
+ suspectedCause: suspectedCauseSchema.optional(),
2404
+ /** Short-lived signed URL of the issue's hero frame. */
2405
+ screenshotUrl: z8.string().optional(),
2406
+ /** Short-lived signed URL of an animated clip of the designated reproduction, when the run captured one. */
2407
+ clipUrl: z8.string().optional(),
2408
+ /** Distinct runs this issue has been attributed to - its recurrence across the branch. */
2409
+ runCount: z8.number().int().nonnegative(),
2410
+ /** The issue's detail page (login required). */
2411
+ issueUrl: z8.string(),
2412
+ /** The run designated as the clearest reproduction (login required). Absent when none was resolved. */
2413
+ replayUrl: z8.string().optional(),
2414
+ coveredTests: z8.array(analysisPrCoveredTestSchema)
2415
+ });
2416
+ analysisPrNewerRunSchema = z8.object({
2417
+ status: z8.enum(["running", "failed"]),
2418
+ failureReason: z8.string().optional()
2419
+ });
2420
+ analysisForPrSchema = z8.discriminatedUnion("status", [
2421
+ z8.object({ status: z8.literal("no_analysis") }),
2422
+ z8.object({ status: z8.literal("in_progress") }),
2423
+ z8.object({ status: z8.literal("failed"), failureReason: z8.string().optional() }),
2424
+ z8.object({
2425
+ status: z8.literal("complete"),
2426
+ /** The app-health verdict: `client_bug` when the branch has an open bug issue, else `passed`. */
2427
+ verdict: analysisVerdictSchema,
2428
+ /** The Reporter's one-paragraph summary of the run. */
2429
+ summary: z8.string().optional(),
2430
+ /** The Reporter's holistic report prose. Its `evidence:` image tokens resolve against `reportEvidence`, and
2431
+ * its `issue:` tokens against the `issues` below. */
2432
+ reportMarkdown: z8.string().optional(),
2433
+ reportEvidence: z8.array(resolvedEvidenceAssetSchema),
2434
+ /** Per-category counts of the run's non-app-health findings. These never block the PR, and a category here
2435
+ * without a matching issue below is one the run could not turn into something actionable. */
2436
+ coverage: coverageSummarySchema.optional(),
2437
+ testCount: z8.number().int().nonnegative(),
2438
+ clientBugCount: z8.number().int().nonnegative(),
2439
+ /** Impact Analysis's account of why the run selected the tests it did. */
2440
+ impactReasoning: z8.string().optional(),
2441
+ /** The PR overview page (login required). */
2442
+ prUrl: z8.string(),
2443
+ /** The branch's open issues, every kind, most actionable first. */
2444
+ issues: z8.array(analysisPrIssueSchema),
2445
+ newerRun: analysisPrNewerRunSchema.optional()
2446
+ })
2447
+ ]);
2392
2448
  analysisSnapshotIssueChangesSchema = z8.object({
2393
2449
  opened: z8.array(analysisIssueSummarySchema),
2394
2450
  carriedForward: z8.array(analysisIssueSummarySchema),
@@ -4048,8 +4104,140 @@ var init_bug_detail = __esm({
4048
4104
  }
4049
4105
  });
4050
4106
 
4051
- // ../../packages/types/src/schemas/index.ts
4107
+ // ../../packages/types/src/schemas/suite-health.ts
4052
4108
  import { z as z29 } from "zod";
4109
+ var SUITE_HEALTH_LEVELS, suiteHealthLevelSchema, suiteHealthDriverSchema, suiteHealthGateSchema, suiteHealthEvidenceSchema, suiteHealthBreakdownSchema, suiteHealthSchema;
4110
+ var init_suite_health = __esm({
4111
+ "../../packages/types/src/schemas/suite-health.ts"() {
4112
+ "use strict";
4113
+ init_esm_shims();
4114
+ SUITE_HEALTH_LEVELS = ["degraded", "at_risk", "calibrating", "steady", "proven"];
4115
+ suiteHealthLevelSchema = z29.enum(SUITE_HEALTH_LEVELS);
4116
+ suiteHealthDriverSchema = z29.enum(["environment", "scenario", "plan", "engine", "balanced", "none"]);
4117
+ suiteHealthGateSchema = z29.enum(["runs", "pull_requests", "age", "stale_issues", "no_triage"]);
4118
+ suiteHealthEvidenceSchema = z29.object({
4119
+ /** Analysis runs in the window. A run that selected no tests is not one. */
4120
+ runs: z29.number().int(),
4121
+ /** Distinct branches those runs covered. */
4122
+ pullRequests: z29.number().int(),
4123
+ /** Tests the agent re-planned and re-ran, that then passed. */
4124
+ selfHeals: z29.number().int(),
4125
+ /** Tests the agent re-planned and re-ran at all - the denominator of the self-heal rate. */
4126
+ selfHealAttempts: z29.number().int(),
4127
+ /** Total findings in the window - the denominator of the trust rate. */
4128
+ findings: z29.number().int(),
4129
+ /** Whole days since the app's first analysis run of any pipeline. */
4130
+ ageDays: z29.number().int(),
4131
+ /** Whole days since its most recent run. Drives the inactivity decay. */
4132
+ daysSinceLastRun: z29.number().int()
4133
+ });
4134
+ suiteHealthBreakdownSchema = z29.object({
4135
+ passed: z29.number().int(),
4136
+ clientBug: z29.number().int(),
4137
+ environmentFailure: z29.number().int(),
4138
+ scenarioIssue: z29.number().int(),
4139
+ planMismatch: z29.number().int(),
4140
+ engineArtifact: z29.number().int(),
4141
+ invalidTest: z29.number().int()
4142
+ });
4143
+ suiteHealthSchema = z29.object({
4144
+ level: suiteHealthLevelSchema,
4145
+ /** 1-5. Always `suiteHealthRank(level)`; sent so clients never re-derive it. */
4146
+ rank: z29.number().int(),
4147
+ /** 0-100, after modifiers. Display-only - the level is what surfaces read. */
4148
+ score: z29.number(),
4149
+ /**
4150
+ * `(passed + client_bug) / findings`, 0-100, before modifiers. The headline fact: of the tests the agent
4151
+ * investigated, how many produced a verdict you can act on.
4152
+ */
4153
+ trust: z29.number(),
4154
+ evidence: suiteHealthEvidenceSchema,
4155
+ breakdown: suiteHealthBreakdownSchema,
4156
+ driver: suiteHealthDriverSchema,
4157
+ /** Open issues older than a week on a branch that is still live. What DEGRADED is actually made of. */
4158
+ staleIssues: z29.number().int(),
4159
+ /** The gate holding the level down, if the score alone would have placed it higher. */
4160
+ gatedBy: suiteHealthGateSchema.optional(),
4161
+ /** False before the first analysis run ever - the meter shows "waiting for your first pull request". */
4162
+ hasEverRun: z29.boolean()
4163
+ });
4164
+ }
4165
+ });
4166
+
4167
+ // ../../packages/types/src/schemas/suite-health-fix-plan.ts
4168
+ import { z as z30 } from "zod";
4169
+ var suiteHealthFixKindSchema, suiteHealthFixSeveritySchema, suiteHealthFixIssueSchema, suiteHealthFixBranchStateSchema, suiteHealthFixBranchSchema, suiteHealthFixClusterSchema, suiteHealthFixPlanSchema;
4170
+ var init_suite_health_fix_plan = __esm({
4171
+ "../../packages/types/src/schemas/suite-health-fix-plan.ts"() {
4172
+ "use strict";
4173
+ init_esm_shims();
4174
+ init_analysis();
4175
+ suiteHealthFixKindSchema = analysisIssueKindSchema;
4176
+ suiteHealthFixSeveritySchema = analysisIssueSeveritySchema;
4177
+ suiteHealthFixIssueSchema = z30.object({
4178
+ id: z30.string(),
4179
+ kind: suiteHealthFixKindSchema,
4180
+ severity: suiteHealthFixSeveritySchema,
4181
+ title: z30.string(),
4182
+ ageDays: z30.number().int()
4183
+ });
4184
+ suiteHealthFixBranchStateSchema = z30.enum(["open", "merged", "closed", "main"]);
4185
+ suiteHealthFixBranchSchema = z30.object({
4186
+ branchId: z30.string(),
4187
+ branchName: z30.string(),
4188
+ state: suiteHealthFixBranchStateSchema,
4189
+ prNumber: z30.number().int().optional(),
4190
+ prTitle: z30.string().optional(),
4191
+ prUrl: z30.string().optional(),
4192
+ /** The findings shown. Capped - `issueCount` is the true total. */
4193
+ issues: z30.array(suiteHealthFixIssueSchema),
4194
+ issueCount: z30.number().int(),
4195
+ /** Kind tally over EVERY finding on the branch, not just the shown ones - those two routinely differ. */
4196
+ byKind: z30.object({
4197
+ bug: z30.number().int(),
4198
+ environment: z30.number().int(),
4199
+ scenario: z30.number().int()
4200
+ }),
4201
+ /** Age of the oldest unresolved finding on this branch, in whole days. */
4202
+ oldestAgeDays: z30.number().int()
4203
+ });
4204
+ suiteHealthFixClusterSchema = z30.object({
4205
+ title: z30.string(),
4206
+ kind: suiteHealthFixKindSchema,
4207
+ /** Distinct branches carrying it. */
4208
+ branches: z30.number().int(),
4209
+ /** How many of those are still-open pull requests - the ones a fix unblocks today. */
4210
+ openBranches: z30.number().int(),
4211
+ findings: z30.number().int()
4212
+ });
4213
+ suiteHealthFixPlanSchema = z30.object({
4214
+ /**
4215
+ * `owner/repo`, which every MCP tool is keyed by. Absent when it cannot be resolved without asking GitHub,
4216
+ * in which case the prompt tells the agent to read the git remote instead - which is what the MCP's own
4217
+ * instructions tell it to do anyway.
4218
+ */
4219
+ repoFullName: z30.string().optional(),
4220
+ /** Unresolved findings across every branch. */
4221
+ totalIssues: z30.number().int(),
4222
+ byKind: z30.object({
4223
+ bug: z30.number().int(),
4224
+ environment: z30.number().int(),
4225
+ scenario: z30.number().int()
4226
+ }),
4227
+ /** Age of the oldest unresolved finding anywhere on the app, in whole days. */
4228
+ oldestAgeDays: z30.number().int(),
4229
+ /** True when the scan hit its cap, so `totalIssues` is a floor rather than the count. Never hide this. */
4230
+ truncated: z30.boolean(),
4231
+ /** Repeated findings, most-shared first. Empty when nothing repeats. */
4232
+ clusters: z30.array(suiteHealthFixClusterSchema),
4233
+ /** The prompt the user copies. Authored from the rows above. */
4234
+ prompt: z30.string()
4235
+ });
4236
+ }
4237
+ });
4238
+
4239
+ // ../../packages/types/src/schemas/index.ts
4240
+ import { z as z31 } from "zod";
4053
4241
  var PlatformSchema, TestStatusSchema;
4054
4242
  var init_schemas = __esm({
4055
4243
  "../../packages/types/src/schemas/index.ts"() {
@@ -4081,8 +4269,10 @@ var init_schemas = __esm({
4081
4269
  init_pr_pipeline_status();
4082
4270
  init_bug_detail();
4083
4271
  init_investigation_report();
4084
- PlatformSchema = z29.enum(["web", "ios", "android"]);
4085
- TestStatusSchema = z29.enum(["pending", "running", "passed", "failed", "cancelled"]);
4272
+ init_suite_health();
4273
+ init_suite_health_fix_plan();
4274
+ PlatformSchema = z31.enum(["web", "ios", "android"]);
4275
+ TestStatusSchema = z31.enum(["pending", "running", "passed", "failed", "cancelled"]);
4086
4276
  }
4087
4277
  });
4088
4278
 
@@ -4117,16 +4307,16 @@ var init_billing = __esm({
4117
4307
  });
4118
4308
 
4119
4309
  // ../../packages/types/src/schemas/billing.ts
4120
- import { z as z30 } from "zod";
4310
+ import { z as z32 } from "zod";
4121
4311
  var BillingTransactionSchema, BillingStatusSchema, CreateCheckoutInputSchema, AutoTopUpInputSchema, RunCompletedSchema;
4122
4312
  var init_billing2 = __esm({
4123
4313
  "../../packages/types/src/schemas/billing.ts"() {
4124
4314
  "use strict";
4125
4315
  init_esm_shims();
4126
4316
  init_billing();
4127
- BillingTransactionSchema = z30.object({
4128
- id: z30.string(),
4129
- type: z30.enum([
4317
+ BillingTransactionSchema = z32.object({
4318
+ id: z32.string(),
4319
+ type: z32.enum([
4130
4320
  "SUBSCRIPTION_GRANT",
4131
4321
  "SUBSCRIPTION_RESET",
4132
4322
  "TOPUP_PURCHASE",
@@ -4135,70 +4325,70 @@ var init_billing2 = __esm({
4135
4325
  "GENERATION_REFUND",
4136
4326
  "RUN_CONSUMPTION"
4137
4327
  ]),
4138
- amount: z30.number(),
4139
- balanceAfter: z30.number(),
4140
- generationId: z30.string().optional(),
4141
- runId: z30.string().optional(),
4142
- stripePaymentIntentId: z30.string().optional(),
4143
- stripeInvoiceId: z30.string().optional(),
4144
- stripeRefundId: z30.string().optional(),
4145
- createdAt: z30.date()
4328
+ amount: z32.number(),
4329
+ balanceAfter: z32.number(),
4330
+ generationId: z32.string().optional(),
4331
+ runId: z32.string().optional(),
4332
+ stripePaymentIntentId: z32.string().optional(),
4333
+ stripeInvoiceId: z32.string().optional(),
4334
+ stripeRefundId: z32.string().optional(),
4335
+ createdAt: z32.date()
4146
4336
  });
4147
- BillingStatusSchema = z30.object({
4148
- creditBalance: z30.number(),
4149
- subscriptionCreditBalance: z30.number(),
4150
- topupCreditBalance: z30.number(),
4151
- subscriptionStatus: z30.enum(["active", "canceled", "past_due", "unpaid", "trialing", "paused", "incomplete", "incomplete_expired"]).optional(),
4152
- currentPeriodEnd: z30.date().optional(),
4153
- cancelAtPeriodEnd: z30.boolean(),
4154
- gracePeriodEndsAt: z30.date().optional(),
4155
- autoTopUpEnabled: z30.boolean(),
4156
- autoTopUpThreshold: z30.number(),
4157
- transactions: z30.array(BillingTransactionSchema)
4337
+ BillingStatusSchema = z32.object({
4338
+ creditBalance: z32.number(),
4339
+ subscriptionCreditBalance: z32.number(),
4340
+ topupCreditBalance: z32.number(),
4341
+ subscriptionStatus: z32.enum(["active", "canceled", "past_due", "unpaid", "trialing", "paused", "incomplete", "incomplete_expired"]).optional(),
4342
+ currentPeriodEnd: z32.date().optional(),
4343
+ cancelAtPeriodEnd: z32.boolean(),
4344
+ gracePeriodEndsAt: z32.date().optional(),
4345
+ autoTopUpEnabled: z32.boolean(),
4346
+ autoTopUpThreshold: z32.number(),
4347
+ transactions: z32.array(BillingTransactionSchema)
4158
4348
  });
4159
- CreateCheckoutInputSchema = z30.object({
4160
- type: z30.enum([BILLING_CHECKOUT_TYPES.SUBSCRIPTION, BILLING_CHECKOUT_TYPES.TOPUP])
4349
+ CreateCheckoutInputSchema = z32.object({
4350
+ type: z32.enum([BILLING_CHECKOUT_TYPES.SUBSCRIPTION, BILLING_CHECKOUT_TYPES.TOPUP])
4161
4351
  });
4162
- AutoTopUpInputSchema = z30.object({
4163
- enabled: z30.boolean(),
4164
- threshold: z30.number().int().min(0)
4352
+ AutoTopUpInputSchema = z32.object({
4353
+ enabled: z32.boolean(),
4354
+ threshold: z32.number().int().min(0)
4165
4355
  });
4166
- RunCompletedSchema = z30.object({
4167
- generationId: z30.string()
4356
+ RunCompletedSchema = z32.object({
4357
+ generationId: z32.string()
4168
4358
  });
4169
4359
  }
4170
4360
  });
4171
4361
 
4172
4362
  // ../../packages/types/src/schemas/github.ts
4173
- import { z as z31 } from "zod";
4363
+ import { z as z33 } from "zod";
4174
4364
  var GitHubInstallationStatusSchema, GithubInstallationSchema;
4175
4365
  var init_github = __esm({
4176
4366
  "../../packages/types/src/schemas/github.ts"() {
4177
4367
  "use strict";
4178
4368
  init_esm_shims();
4179
- GitHubInstallationStatusSchema = z31.enum(["active", "suspended", "deleted"]);
4180
- GithubInstallationSchema = z31.object({
4181
- id: z31.string(),
4182
- installationId: z31.number(),
4183
- organizationId: z31.string(),
4184
- accountLogin: z31.string(),
4185
- accountId: z31.number(),
4186
- accountType: z31.string(),
4369
+ GitHubInstallationStatusSchema = z33.enum(["active", "suspended", "deleted"]);
4370
+ GithubInstallationSchema = z33.object({
4371
+ id: z33.string(),
4372
+ installationId: z33.number(),
4373
+ organizationId: z33.string(),
4374
+ accountLogin: z33.string(),
4375
+ accountId: z33.number(),
4376
+ accountType: z33.string(),
4187
4377
  status: GitHubInstallationStatusSchema,
4188
- createdAt: z31.date(),
4189
- updatedAt: z31.date()
4378
+ createdAt: z33.date(),
4379
+ updatedAt: z33.date()
4190
4380
  });
4191
4381
  }
4192
4382
  });
4193
4383
 
4194
4384
  // ../../packages/types/src/schemas/vercel.ts
4195
- import { z as z32 } from "zod";
4385
+ import { z as z34 } from "zod";
4196
4386
  var VercelInstallationWireStatusSchema;
4197
4387
  var init_vercel = __esm({
4198
4388
  "../../packages/types/src/schemas/vercel.ts"() {
4199
4389
  "use strict";
4200
4390
  init_esm_shims();
4201
- VercelInstallationWireStatusSchema = z32.enum([
4391
+ VercelInstallationWireStatusSchema = z34.enum([
4202
4392
  "ready",
4203
4393
  "pending",
4204
4394
  "onboarding",
@@ -4257,6 +4447,22 @@ var init_preview_url = __esm({
4257
4447
  }
4258
4448
  });
4259
4449
 
4450
+ // ../../packages/types/src/app-links.ts
4451
+ var init_app_links = __esm({
4452
+ "../../packages/types/src/app-links.ts"() {
4453
+ "use strict";
4454
+ init_esm_shims();
4455
+ }
4456
+ });
4457
+
4458
+ // ../../packages/types/src/designated-run.ts
4459
+ var init_designated_run = __esm({
4460
+ "../../packages/types/src/designated-run.ts"() {
4461
+ "use strict";
4462
+ init_esm_shims();
4463
+ }
4464
+ });
4465
+
4260
4466
  // ../../packages/types/src/constants/pipeline-labels.ts
4261
4467
  var init_pipeline_labels = __esm({
4262
4468
  "../../packages/types/src/constants/pipeline-labels.ts"() {
@@ -4302,6 +4508,8 @@ var init_src = __esm({
4302
4508
  init_vercel();
4303
4509
  init_sensitive_detection();
4304
4510
  init_preview_url();
4511
+ init_app_links();
4512
+ init_designated_run();
4305
4513
  init_constants();
4306
4514
  init_architecture();
4307
4515
  init_step_overlay_points();
@@ -4597,47 +4805,6 @@ var init_colors = __esm({
4597
4805
  }
4598
4806
  });
4599
4807
 
4600
- // src/core/context.ts
4601
- import { readFile as readFile4, writeFile as writeFile2 } from "fs/promises";
4602
- import { join as join10 } from "path";
4603
- async function loadContext(outputDir) {
4604
- try {
4605
- const raw = await readFile4(join10(outputDir, CONTEXT_FILE), "utf-8");
4606
- const parsed = JSON.parse(raw);
4607
- return parsed;
4608
- } catch {
4609
- return void 0;
4610
- }
4611
- }
4612
- function formatContext(ctx) {
4613
- let output = `## Project Context (from the user)
4614
-
4615
- **What this project is:** ${ctx.description}
4616
-
4617
- **Why they want testing:** ${ctx.testingGoal}
4618
-
4619
- **Critical flows (user-declared - these MUST be covered):** ${ctx.criticalFlows}
4620
-
4621
- These are flows the user explicitly said cannot break. Treat them as authoritative: every one of them must be represented faithfully in your output - never drop or downplay them. Start with these, then expand to cover the rest of the application.`;
4622
- if (ctx.pages?.length) {
4623
- output += `
4624
-
4625
- ## Discovered Pages (${ctx.pages.length} routes)
4626
-
4627
- `;
4628
- output += ctx.pages.map((p) => `- \`${p.route}\` - ${p.description} (\`${p.path}\`)`).join("\n");
4629
- }
4630
- return output;
4631
- }
4632
- var CONTEXT_FILE;
4633
- var init_context = __esm({
4634
- "src/core/context.ts"() {
4635
- "use strict";
4636
- init_esm_shims();
4637
- CONTEXT_FILE = ".project-context.json";
4638
- }
4639
- });
4640
-
4641
4808
  // src/core/errors.ts
4642
4809
  import { APICallError, RetryError, LoadAPIKeyError, InvalidPromptError, NoSuchModelError } from "ai";
4643
4810
  function sleep(ms) {
@@ -5102,16 +5269,16 @@ var init_notify = __esm({
5102
5269
  });
5103
5270
 
5104
5271
  // src/core/project-map.ts
5105
- import { readFile as readFile6, writeFile as writeFile5 } from "fs/promises";
5106
- import { join as join13 } from "path";
5107
- import { z as z34 } from "zod";
5272
+ import { readFile as readFile5, writeFile as writeFile4 } from "fs/promises";
5273
+ import { join as join12 } from "path";
5274
+ import { z as z36 } from "zod";
5108
5275
  async function saveProjectMap(outputDir, map) {
5109
- await writeFile5(join13(outputDir, PROJECT_MAP_FILE), JSON.stringify(map, null, 2), "utf-8");
5276
+ await writeFile4(join12(outputDir, PROJECT_MAP_FILE), JSON.stringify(map, null, 2), "utf-8");
5110
5277
  }
5111
5278
  async function loadProjectMap(outputDir) {
5112
- const path3 = join13(outputDir, PROJECT_MAP_FILE);
5279
+ const path3 = join12(outputDir, PROJECT_MAP_FILE);
5113
5280
  try {
5114
- const raw = await readFile6(path3, "utf-8");
5281
+ const raw = await readFile5(path3, "utf-8");
5115
5282
  const parsed = ProjectMapSchema.safeParse(JSON.parse(raw));
5116
5283
  if (parsed.success) return parsed.data;
5117
5284
  debugLog("project-map.json failed schema validation, ignoring it", { path: path3, issues: parsed.error.issues });
@@ -5213,34 +5380,34 @@ var init_project_map = __esm({
5213
5380
  "use strict";
5214
5381
  init_esm_shims();
5215
5382
  init_debug();
5216
- FrontendEntry = z34.object({
5217
- path: z34.string().min(1).describe("Repo-relative path to the frontend app/directory (the UI surface)."),
5218
- framework: z34.string().describe("Detected UI framework/stack, or 'unknown' if not obvious (e.g. next, react, vue, svelte)."),
5219
- dependsOn: z34.array(z34.string()).describe(
5383
+ FrontendEntry = z36.object({
5384
+ path: z36.string().min(1).describe("Repo-relative path to the frontend app/directory (the UI surface)."),
5385
+ framework: z36.string().describe("Detected UI framework/stack, or 'unknown' if not obvious (e.g. next, react, vue, svelte)."),
5386
+ dependsOn: z36.array(z36.string()).describe(
5220
5387
  "Repo-relative paths (each matching a backend in this map) that THIS frontend needs in order to function - the API/service(s) it calls and the data layer(s) that own the records it renders. These are pre-selected when the user picks this frontend, so only list backends this frontend actually depends on. Empty if it needs none."
5221
5388
  ),
5222
- why: z34.string().min(1).describe("One line: the evidence that made you classify this as a frontend.")
5223
- });
5224
- BackendEntry = z34.object({
5225
- path: z34.string().min(1).describe("Repo-relative path to the backend/API/service or the package that owns the data layer."),
5226
- language: z34.string().describe("Primary language, or 'unknown' (e.g. typescript, python, go, rust)."),
5227
- framework: z34.string().describe("Web/service framework or ORM stack, or 'unknown' (e.g. express, hono, fastapi, rails)."),
5228
- dataLayer: z34.object({
5229
- kind: z34.string().describe("How models are defined (e.g. prisma, drizzle, sqlalchemy, typeorm, raw-sql, unknown)."),
5230
- schemaPath: z34.string().optional().describe("Repo-relative path to the schema/models definition, if found.")
5389
+ why: z36.string().min(1).describe("One line: the evidence that made you classify this as a frontend.")
5390
+ });
5391
+ BackendEntry = z36.object({
5392
+ path: z36.string().min(1).describe("Repo-relative path to the backend/API/service or the package that owns the data layer."),
5393
+ language: z36.string().describe("Primary language, or 'unknown' (e.g. typescript, python, go, rust)."),
5394
+ framework: z36.string().describe("Web/service framework or ORM stack, or 'unknown' (e.g. express, hono, fastapi, rails)."),
5395
+ dataLayer: z36.object({
5396
+ kind: z36.string().describe("How models are defined (e.g. prisma, drizzle, sqlalchemy, typeorm, raw-sql, unknown)."),
5397
+ schemaPath: z36.string().optional().describe("Repo-relative path to the schema/models definition, if found.")
5231
5398
  }).optional().describe("Where this backend's database models live. Omit if this backend owns no data layer."),
5232
- why: z34.string().min(1).describe("One line: the evidence that made you classify this as a backend.")
5399
+ why: z36.string().min(1).describe("One line: the evidence that made you classify this as a backend.")
5233
5400
  });
5234
- IgnoreEntry = z34.object({
5235
- path: z34.string().min(1).describe("Repo-relative path to a directory that is NOT relevant to testing this app."),
5236
- why: z34.string().min(1).describe("One line: why it is irrelevant (e.g. docs site, infra, tooling, examples, unrelated app).")
5401
+ IgnoreEntry = z36.object({
5402
+ path: z36.string().min(1).describe("Repo-relative path to a directory that is NOT relevant to testing this app."),
5403
+ why: z36.string().min(1).describe("One line: why it is irrelevant (e.g. docs site, infra, tooling, examples, unrelated app).")
5237
5404
  });
5238
- ProjectMapSchema = z34.object({
5239
- frontends: z34.array(FrontendEntry).describe("The UI surface(s) whose pages/flows get tested. Usually 1; may be more."),
5240
- backends: z34.array(BackendEntry).describe(
5405
+ ProjectMapSchema = z36.object({
5406
+ frontends: z36.array(FrontendEntry).describe("The UI surface(s) whose pages/flows get tested. Usually 1; may be more."),
5407
+ backends: z36.array(BackendEntry).describe(
5241
5408
  "The API/service(s) and data layer(s) that own the models we seed. May be 0 (frontend-only), 1, or many."
5242
5409
  ),
5243
- ignore: z34.array(IgnoreEntry).describe("Directories judged irrelevant to this app's tests, so later steps skip them.")
5410
+ ignore: z36.array(IgnoreEntry).describe("Directories judged irrelevant to this app's tests, so later steps skip them.")
5244
5411
  });
5245
5412
  PROJECT_MAP_FILE = "project-map.json";
5246
5413
  }
@@ -5251,11 +5418,11 @@ var parse_entity_audit_exports = {};
5251
5418
  __export(parse_entity_audit_exports, {
5252
5419
  parseEntityNames: () => parseEntityNames
5253
5420
  });
5254
- import { readFile as readFile8 } from "fs/promises";
5255
- import { join as join15 } from "path";
5421
+ import { readFile as readFile7 } from "fs/promises";
5422
+ import { join as join14 } from "path";
5256
5423
  async function parseEntityNames(outputDir) {
5257
5424
  try {
5258
- const content = await readFile8(join15(outputDir, "entity-audit.md"), "utf-8");
5425
+ const content = await readFile7(join14(outputDir, "entity-audit.md"), "utf-8");
5259
5426
  const names = [];
5260
5427
  for (const line of content.split("\n")) {
5261
5428
  const match = line.match(/^\s+-\s+name:\s+(.+)$/);
@@ -5290,7 +5457,7 @@ var init_count_generated_tests = __esm({
5290
5457
  });
5291
5458
 
5292
5459
  // src/core/display.ts
5293
- import { join as join16 } from "path";
5460
+ import { join as join15 } from "path";
5294
5461
  function formatArgs(input, keys) {
5295
5462
  const parts = [];
5296
5463
  for (const key of keys) {
@@ -5382,7 +5549,7 @@ function logToStore(store, agentId, info, stats) {
5382
5549
  const folder = String(tc.input.folder ?? "");
5383
5550
  const filename = String(tc.input.filename ?? "");
5384
5551
  if (folder && filename) {
5385
- store.noteWrite(join16(TESTS_DIR, folder, normalizeTestFilename(filename)));
5552
+ store.noteWrite(join15(TESTS_DIR, folder, normalizeTestFilename(filename)));
5386
5553
  }
5387
5554
  }
5388
5555
  if (tc.name === "bash") {
@@ -5620,10 +5787,8 @@ async function runAgent(config, prompt, extractResult) {
5620
5787
  });
5621
5788
  try {
5622
5789
  const messages = [{ role: "user", content: prompt }];
5623
- let generation = await agent.generate({
5624
- messages,
5625
- timeout: { stepMs: stepTimeout }
5626
- });
5790
+ const callOptions = stepTimeout > 0 ? { timeout: { stepMs: stepTimeout } } : {};
5791
+ let generation = await agent.generate({ messages, ...callOptions });
5627
5792
  let nudges = 0;
5628
5793
  while (extractResult() === void 0 && nudges < MAX_NUDGES) {
5629
5794
  nudges++;
@@ -5638,10 +5803,7 @@ async function runAgent(config, prompt, extractResult) {
5638
5803
  });
5639
5804
  messages.push(...generation.response.messages);
5640
5805
  messages.push({ role: "user", content: NUDGE_PROMPT });
5641
- generation = await agent.generate({
5642
- messages,
5643
- timeout: { stepMs: stepTimeout }
5644
- });
5806
+ generation = await agent.generate({ messages, ...callOptions });
5645
5807
  }
5646
5808
  return extractResult();
5647
5809
  } catch (err) {
@@ -5688,15 +5850,15 @@ var init_agent = __esm({
5688
5850
  init_logs();
5689
5851
  init_model();
5690
5852
  MAX_RETRIES = 3;
5691
- STEP_TIMEOUT_MS = 12e4;
5853
+ STEP_TIMEOUT_MS = 0;
5692
5854
  MAX_NUDGES = 2;
5693
5855
  NUDGE_PROMPT = "You stopped before completing the task. The task is only complete once the finish tool has been called and succeeded. If your last finish call returned an error, fix exactly what it reported. Otherwise, continue where you left off and call the finish tool when done.";
5694
5856
  }
5695
5857
  });
5696
5858
 
5697
5859
  // src/core/gitignore.ts
5698
- import { readFile as readFile9 } from "fs/promises";
5699
- import { join as join17, relative as relative2 } from "path";
5860
+ import { readFile as readFile8 } from "fs/promises";
5861
+ import { join as join16, relative as relative2 } from "path";
5700
5862
  import { glob as glob2 } from "glob";
5701
5863
  async function loadGitignorePatterns(projectRoot) {
5702
5864
  const patterns = [
@@ -5716,10 +5878,10 @@ async function loadGitignorePatterns(projectRoot) {
5716
5878
  ];
5717
5879
  const matches = await glob2("**/.gitignore", { cwd: projectRoot, dot: true });
5718
5880
  for (const match of matches) {
5719
- const fullPath = join17(projectRoot, match);
5881
+ const fullPath = join16(projectRoot, match);
5720
5882
  try {
5721
- const content = await readFile9(fullPath, "utf-8");
5722
- const prefix = relative2(projectRoot, join17(projectRoot, match, ".."));
5883
+ const content = await readFile8(fullPath, "utf-8");
5884
+ const prefix = relative2(projectRoot, join16(projectRoot, match, ".."));
5723
5885
  const parsed = parseGitignore(content, prefix);
5724
5886
  patterns.push(...parsed);
5725
5887
  } catch (err) {
@@ -5782,7 +5944,7 @@ var init_exec_error = __esm({
5782
5944
  import { execFile as execFile3 } from "child_process";
5783
5945
  import { promisify as promisify2 } from "util";
5784
5946
  import { tool } from "ai";
5785
- import { z as z35 } from "zod";
5947
+ import { z as z37 } from "zod";
5786
5948
  function validateCommand(command, allowed) {
5787
5949
  const trimmed = command.trim();
5788
5950
  if (trimmed.length === 0) return "Empty command";
@@ -5848,8 +6010,8 @@ var init_bash = __esm({
5848
6010
  DEFAULT_ALLOWED = /* @__PURE__ */ new Set(["git", "wc", "sort", "head", "tail", "cat", "ls", "find", "diff", "echo"]);
5849
6011
  TIMEOUT_MS = 3e4;
5850
6012
  MAX_OUTPUT_BYTES = 1024 * 512;
5851
- inputSchema = z35.object({
5852
- command: z35.string().describe("Shell command to execute")
6013
+ inputSchema = z37.object({
6014
+ command: z37.string().describe("Shell command to execute")
5853
6015
  });
5854
6016
  }
5855
6017
  });
@@ -5857,7 +6019,7 @@ var init_bash = __esm({
5857
6019
  // src/tools/glob.ts
5858
6020
  import { tool as tool2 } from "ai";
5859
6021
  import { glob as glob3 } from "glob";
5860
- import { z as z36 } from "zod";
6022
+ import { z as z38 } from "zod";
5861
6023
  async function executeGlob(pattern, cwd, ignorePatterns = DEFAULT_IGNORE) {
5862
6024
  try {
5863
6025
  const matches = await glob3(pattern, {
@@ -5891,9 +6053,9 @@ var init_glob = __esm({
5891
6053
  "src/tools/glob.ts"() {
5892
6054
  "use strict";
5893
6055
  init_esm_shims();
5894
- inputSchema2 = z36.object({
5895
- pattern: z36.string().describe("Glob pattern to match files (e.g. '**/*.ts', 'src/**/*.py')"),
5896
- cwd: z36.string().optional().describe("Directory to search in. Defaults to working directory.")
6056
+ inputSchema2 = z38.object({
6057
+ pattern: z38.string().describe("Glob pattern to match files (e.g. '**/*.ts', 'src/**/*.py')"),
6058
+ cwd: z38.string().optional().describe("Directory to search in. Defaults to working directory.")
5897
6059
  });
5898
6060
  DEFAULT_IGNORE = [
5899
6061
  "**/node_modules/**",
@@ -5911,7 +6073,7 @@ var init_glob = __esm({
5911
6073
  import { execFile as execFile4 } from "child_process";
5912
6074
  import { promisify as promisify3 } from "util";
5913
6075
  import { tool as tool3 } from "ai";
5914
- import { z as z37 } from "zod";
6076
+ import { z as z39 } from "zod";
5915
6077
  function buildGrepTool(workingDirectory) {
5916
6078
  return tool3({
5917
6079
  description: "Search file contents with ripgrep. Returns matching lines with file paths and line numbers.",
@@ -5955,10 +6117,10 @@ var init_grep = __esm({
5955
6117
  init_esm_shims();
5956
6118
  init_exec_error();
5957
6119
  execFileAsync3 = promisify3(execFile4);
5958
- inputSchema3 = z37.object({
5959
- pattern: z37.string().describe("Regex pattern to search for in file contents"),
5960
- glob: z37.string().optional().describe("Glob to filter files (e.g. '*.ts')"),
5961
- path: z37.string().optional().describe("File or directory to search in")
6120
+ inputSchema3 = z39.object({
6121
+ pattern: z39.string().describe("Regex pattern to search for in file contents"),
6122
+ glob: z39.string().optional().describe("Glob to filter files (e.g. '*.ts')"),
6123
+ path: z39.string().optional().describe("File or directory to search in")
5962
6124
  });
5963
6125
  }
5964
6126
  });
@@ -5966,10 +6128,10 @@ var init_grep = __esm({
5966
6128
  // src/tools/list-directory.ts
5967
6129
  import { readdir } from "fs/promises";
5968
6130
  import { stat } from "fs/promises";
5969
- import { join as join18, relative as relative3 } from "path";
6131
+ import { join as join17, relative as relative3 } from "path";
5970
6132
  import { tool as tool4 } from "ai";
5971
6133
  import { minimatch } from "minimatch";
5972
- import { z as z38 } from "zod";
6134
+ import { z as z40 } from "zod";
5973
6135
  function buildMatcher(patterns) {
5974
6136
  const positive = patterns.filter((p) => !p.startsWith("!"));
5975
6137
  const negative = patterns.filter((p) => p.startsWith("!")).map((p) => p.slice(1));
@@ -5991,7 +6153,7 @@ async function buildTree(dirPath, maxDepth, currentDepth, isIgnored, relativeBas
5991
6153
  const withTypes = [];
5992
6154
  for (const name of rawEntries) {
5993
6155
  try {
5994
- const s = await stat(join18(dirPath, name));
6156
+ const s = await stat(join17(dirPath, name));
5995
6157
  withTypes.push({ name, isDir: s.isDirectory() });
5996
6158
  } catch {
5997
6159
  withTypes.push({ name, isDir: false });
@@ -6011,7 +6173,7 @@ async function buildTree(dirPath, maxDepth, currentDepth, isIgnored, relativeBas
6011
6173
  }
6012
6174
  if (entry.isDir) {
6013
6175
  const children = await buildTree(
6014
- join18(dirPath, entry.name),
6176
+ join17(dirPath, entry.name),
6015
6177
  maxDepth,
6016
6178
  currentDepth + 1,
6017
6179
  isIgnored,
@@ -6049,10 +6211,10 @@ async function buildListDirectoryTool(workingDirectory) {
6049
6211
  const isIgnored = buildMatcher(patterns);
6050
6212
  return tool4({
6051
6213
  description: "List directory structure as a tree. Use this for an overview of the project layout. Start at the root (path='.') with depth 3, then increase depth or narrow path if needed. Do NOT call this on every subdirectory - use glob to find specific files instead. Returns cached result if the same path+depth was already requested.",
6052
- inputSchema: z38.object({
6053
- path: z38.string().default(".").describe("Directory path relative to project root. Defaults to root."),
6054
- depth: z38.number().min(1).max(15).default(10).describe("Max depth to traverse (1-15). Default 10."),
6055
- gitignore: z38.boolean().describe(
6214
+ inputSchema: z40.object({
6215
+ path: z40.string().default(".").describe("Directory path relative to project root. Defaults to root."),
6216
+ depth: z40.number().min(1).max(15).default(10).describe("Max depth to traverse (1-15). Default 10."),
6217
+ gitignore: z40.boolean().describe(
6056
6218
  "Whether to respect the gitignore or to ignore it. true will respect it. false will ignore it. Default true"
6057
6219
  ).default(true)
6058
6220
  }),
@@ -6064,7 +6226,7 @@ async function buildListDirectoryTool(workingDirectory) {
6064
6226
  };
6065
6227
  }
6066
6228
  seen.add(cacheKey);
6067
- const targetDir = input.path === "." ? workingDirectory : join18(workingDirectory, input.path);
6229
+ const targetDir = input.path === "." ? workingDirectory : join17(workingDirectory, input.path);
6068
6230
  try {
6069
6231
  const s = await stat(targetDir);
6070
6232
  if (!s.isDirectory()) {
@@ -6100,10 +6262,10 @@ var init_list_directory = __esm({
6100
6262
  });
6101
6263
 
6102
6264
  // src/tools/read-file.ts
6103
- import { readFile as readFile10 } from "fs/promises";
6265
+ import { readFile as readFile9 } from "fs/promises";
6104
6266
  import { relative as relative4, resolve as resolve2 } from "path";
6105
6267
  import { tool as tool5 } from "ai";
6106
- import { z as z39 } from "zod";
6268
+ import { z as z41 } from "zod";
6107
6269
  function resolveSandboxedPath(workingDirectory, filePath) {
6108
6270
  const absolutePath = resolve2(workingDirectory, filePath);
6109
6271
  const relativePath = relative4(workingDirectory, absolutePath);
@@ -6128,7 +6290,7 @@ async function executeReadFile(workingDirectory, filePath, offset, limit) {
6128
6290
  const resolved = resolveSandboxedPath(workingDirectory, filePath);
6129
6291
  if ("error" in resolved) return resolved;
6130
6292
  try {
6131
- const content = await readFile10(resolved.absolutePath, "utf-8");
6293
+ const content = await readFile9(resolved.absolutePath, "utf-8");
6132
6294
  const sliced = sliceLines(content, offset ?? 0, limit ?? MAX_LINES);
6133
6295
  if (sliced.numbered.length > MAX_OUTPUT_BYTES2) {
6134
6296
  sliced.numbered = sliced.numbered.slice(0, MAX_OUTPUT_BYTES2) + `
@@ -6161,10 +6323,10 @@ var init_read_file = __esm({
6161
6323
  init_esm_shims();
6162
6324
  MAX_LINES = 2e3;
6163
6325
  MAX_OUTPUT_BYTES2 = 256 * 1024;
6164
- inputSchema4 = z39.object({
6165
- filePath: z39.string().describe("Path to the file (absolute or relative to working directory)"),
6166
- offset: z39.number().int().min(0).optional().describe("Line number to start reading from (0-based)"),
6167
- limit: z39.number().int().min(1).optional().describe("Maximum number of lines to read")
6326
+ inputSchema4 = z41.object({
6327
+ filePath: z41.string().describe("Path to the file (absolute or relative to working directory)"),
6328
+ offset: z41.number().int().min(0).optional().describe("Line number to start reading from (0-based)"),
6329
+ limit: z41.number().int().min(1).optional().describe("Maximum number of lines to read")
6168
6330
  });
6169
6331
  }
6170
6332
  });
@@ -6187,7 +6349,7 @@ var init_pick_string = __esm({
6187
6349
 
6188
6350
  // src/tools/subagent.ts
6189
6351
  import { ToolLoopAgent as ToolLoopAgent2, hasToolCall as hasToolCall2, stepCountIs as stepCountIs2, tool as tool6 } from "ai";
6190
- import { z as z40 } from "zod";
6352
+ import { z as z42 } from "zod";
6191
6353
  function buildSubagentTools(workingDirectory, onFileRead) {
6192
6354
  const baseReadFile = buildReadFileTool(workingDirectory);
6193
6355
  const readFile25 = onFileRead ? tool6({
@@ -6260,11 +6422,11 @@ var init_subagent = __esm({
6260
6422
  init_glob();
6261
6423
  init_grep();
6262
6424
  init_read_file();
6263
- inputSchema5 = z40.object({
6264
- instruction: z40.string().describe("Focused task for the subagent. Be specific about files and patterns to investigate.")
6425
+ inputSchema5 = z42.object({
6426
+ instruction: z42.string().describe("Focused task for the subagent. Be specific about files and patterns to investigate.")
6265
6427
  });
6266
- resultSchema = z40.object({
6267
- findings: z40.string().describe("Summary of what was found")
6428
+ resultSchema = z42.object({
6429
+ findings: z42.string().describe("Summary of what was found")
6268
6430
  });
6269
6431
  SYSTEM_PROMPT = `You are a code research assistant. You have tools to explore a codebase: bash (shell commands, mainly git), glob (find files), grep (search content), and read_file (read files).
6270
6432
 
@@ -6275,10 +6437,10 @@ Be thorough but focused - only investigate what's relevant to your instruction.`
6275
6437
  });
6276
6438
 
6277
6439
  // src/tools/write-file.ts
6278
- import { writeFile as writeFile6, mkdir as mkdir2 } from "fs/promises";
6440
+ import { writeFile as writeFile5, mkdir as mkdir2 } from "fs/promises";
6279
6441
  import { dirname as dirname2, relative as relative5, resolve as resolve3 } from "path";
6280
6442
  import { tool as tool7 } from "ai";
6281
- import { z as z41 } from "zod";
6443
+ import { z as z43 } from "zod";
6282
6444
  async function executeWriteFile(outputDirectory, filePath, content) {
6283
6445
  const cleaned = filePath.replace(/^autonoma\//, "");
6284
6446
  const absolutePath = resolve3(outputDirectory, cleaned);
@@ -6288,7 +6450,7 @@ async function executeWriteFile(outputDirectory, filePath, content) {
6288
6450
  }
6289
6451
  try {
6290
6452
  await mkdir2(dirname2(absolutePath), { recursive: true });
6291
- await writeFile6(absolutePath, content, "utf-8");
6453
+ await writeFile5(absolutePath, content, "utf-8");
6292
6454
  return { path: relativePath, bytesWritten: content.length };
6293
6455
  } catch (err) {
6294
6456
  const message = err instanceof Error ? err.message : String(err);
@@ -6307,21 +6469,21 @@ var init_write_file = __esm({
6307
6469
  "src/tools/write-file.ts"() {
6308
6470
  "use strict";
6309
6471
  init_esm_shims();
6310
- inputSchema6 = z41.object({
6311
- filePath: z41.string().describe("Path to write (absolute or relative to output directory)"),
6312
- content: z41.string().describe("File content to write")
6472
+ inputSchema6 = z43.object({
6473
+ filePath: z43.string().describe("Path to write (absolute or relative to output directory)"),
6474
+ content: z43.string().describe("File content to write")
6313
6475
  });
6314
6476
  }
6315
6477
  });
6316
6478
 
6317
6479
  // src/tools/ask-user.ts
6318
6480
  import { tool as tool8 } from "ai";
6319
- import { z as z42 } from "zod";
6481
+ import { z as z44 } from "zod";
6320
6482
  function buildAskUserTool() {
6321
6483
  return tool8({
6322
6484
  description: "Ask the user a question ONLY when the answer is truly unknowable from the codebase. Valid reasons: untyped JSON/JSONB field schemas, business rules not in code, config values not in source. NEVER ask about: field names (read the schema), field types (read the ORM model), enum values (read the code), relationships (read foreign keys), numeric values (read the seed data or defaults). If you can find it by reading a file, DO NOT ask - read the file instead.",
6323
- inputSchema: z42.object({
6324
- question: z42.string().describe(
6485
+ inputSchema: z44.object({
6486
+ question: z44.string().describe(
6325
6487
  "A clear, plain-language question. State exactly what you need to know and why you can't find it in code. BAD: 'What are the decimal values for checking_balance?' - GOOD: 'Your Account model has a metadata JSON column with no type definition. What fields go inside it?'"
6326
6488
  )
6327
6489
  }),
@@ -6381,7 +6543,7 @@ var init_tools = __esm({
6381
6543
 
6382
6544
  // src/agents/00-project-mapper/coverage.ts
6383
6545
  import { readdir as readdir2 } from "fs/promises";
6384
- import { join as join19 } from "path";
6546
+ import { join as join18 } from "path";
6385
6547
  async function findUncoveredDirs(projectRoot, map) {
6386
6548
  const entries = [...map.frontends, ...map.backends, ...map.ignore].map(
6387
6549
  (e) => e.path.replace(/^\.\//, "").replace(/\/+$/, "")
@@ -6405,7 +6567,7 @@ async function findUncoveredDirs(projectRoot, map) {
6405
6567
  const childRel = rel === "" ? child.name : `${rel}/${child.name}`;
6406
6568
  if (underEntry(childRel)) continue;
6407
6569
  if (hasEntryWithin(childRel)) {
6408
- if (depth + 1 < MAX_DEPTH) await walk(join19(abs, child.name), childRel, depth + 1);
6570
+ if (depth + 1 < MAX_DEPTH) await walk(join18(abs, child.name), childRel, depth + 1);
6409
6571
  } else {
6410
6572
  out.push(childRel);
6411
6573
  }
@@ -6432,7 +6594,7 @@ __export(project_mapper_exports, {
6432
6594
  runProjectMapper: () => runProjectMapper
6433
6595
  });
6434
6596
  import { tool as tool9 } from "ai";
6435
- import { z as z43 } from "zod";
6597
+ import { z as z45 } from "zod";
6436
6598
  async function runProjectMapper(input) {
6437
6599
  const model = getModel(input.modelId);
6438
6600
  const { logger, onStepFinish } = buildDefaultStepLogger("project-map", MAX_STEPS);
@@ -6466,7 +6628,7 @@ First build the full inventory (workspace/package manifests + top-level director
6466
6628
  }),
6467
6629
  finish: tool9({
6468
6630
  description: "End this step. Call once set_project_map has recorded a complete map.",
6469
- inputSchema: z43.object({}),
6631
+ inputSchema: z45.object({}),
6470
6632
  execute: () => {
6471
6633
  if (captured == null) {
6472
6634
  return { error: "Cannot finish: no map recorded - call set_project_map first." };
@@ -6510,7 +6672,7 @@ __export(pages_finder_exports, {
6510
6672
  import { existsSync } from "fs";
6511
6673
  import * as path2 from "path";
6512
6674
  import { tool as tool10 } from "ai";
6513
- import { z as z44 } from "zod";
6675
+ import { z as z46 } from "zod";
6514
6676
  async function runPageFinder(input) {
6515
6677
  const model = getModel(input.modelId);
6516
6678
  const pageCollector = new PageCollector();
@@ -6545,12 +6707,12 @@ ${input.extraMessage}`;
6545
6707
  }),
6546
6708
  view_pages: tool10({
6547
6709
  description: "use this tool to view all the pages that you already added",
6548
- inputSchema: z44.object(),
6710
+ inputSchema: z46.object(),
6549
6711
  execute: () => pageCollector.viewPages()
6550
6712
  }),
6551
6713
  finish: tool10({
6552
6714
  description: "End this step. Call once every page in the codebase has been added via add_page.",
6553
- inputSchema: z44.object({}),
6715
+ inputSchema: z46.object({}),
6554
6716
  execute: () => {
6555
6717
  if (pageCollector.pages.size === 0) {
6556
6718
  return {
@@ -6577,10 +6739,10 @@ var init_pages_finder = __esm({
6577
6739
  init_agent();
6578
6740
  init_model();
6579
6741
  init_tools();
6580
- Page = z44.object({
6581
- route: z44.string().min(1),
6582
- path: z44.string().min(1),
6583
- description: z44.string().min(10)
6742
+ Page = z46.object({
6743
+ route: z46.string().min(1),
6744
+ path: z46.string().min(1),
6745
+ description: z46.string().min(10)
6584
6746
  });
6585
6747
  PageCollector = class {
6586
6748
  // the key is the path
@@ -6770,15 +6932,15 @@ var kb_generator_exports = {};
6770
6932
  __export(kb_generator_exports, {
6771
6933
  runKBGenerator: () => runKBGenerator
6772
6934
  });
6773
- import { readFile as readFile11 } from "fs/promises";
6774
- import { join as join20, resolve as resolve5 } from "path";
6935
+ import { readFile as readFile10 } from "fs/promises";
6936
+ import { join as join19, resolve as resolve5 } from "path";
6775
6937
  import { tool as tool11 } from "ai";
6776
- import { z as z45 } from "zod";
6938
+ import { z as z47 } from "zod";
6777
6939
  function buildRegisterPagesTool(tracker) {
6778
6940
  return tool11({
6779
6941
  description: "Register ALL page/route files discovered via glob. Call this ONCE after globbing for page files. The system will track which ones you've read and block finish until all are covered.",
6780
- inputSchema: z45.object({
6781
- pages: z45.array(z45.string()).describe("All page file paths found by glob")
6942
+ inputSchema: z47.object({
6943
+ pages: z47.array(z47.string()).describe("All page file paths found by glob")
6782
6944
  }),
6783
6945
  execute: async (input) => {
6784
6946
  tracker.register(input.pages);
@@ -6792,7 +6954,7 @@ function buildRegisterPagesTool(tracker) {
6792
6954
  function buildPageCoverageTool(tracker) {
6793
6955
  return tool11({
6794
6956
  description: "Check how many registered pages you've read vs how many remain.",
6795
- inputSchema: z45.object({}),
6957
+ inputSchema: z47.object({}),
6796
6958
  execute: async () => tracker.coverage()
6797
6959
  });
6798
6960
  }
@@ -6803,9 +6965,9 @@ function requiredReads(total) {
6803
6965
  function buildFinishTool(tracker, onFinish) {
6804
6966
  return tool11({
6805
6967
  description: "Call when you have finished generating the knowledge base. BLOCKED until you have read enough of the registered routes (every route on a small app; a strong majority on a large one) - call page_coverage first to check how many remain.",
6806
- inputSchema: z45.object({
6807
- summary: z45.string().describe("Summary of what was generated"),
6808
- artifacts: z45.array(z45.string()).describe("List of files written")
6968
+ inputSchema: z47.object({
6969
+ summary: z47.string().describe("Summary of what was generated"),
6970
+ artifacts: z47.array(z47.string()).describe("List of files written")
6809
6971
  }),
6810
6972
  execute: async (input) => {
6811
6973
  const cov = tracker.coverage();
@@ -6866,24 +7028,9 @@ async function runKBGenerator(input) {
6866
7028
  result = r;
6867
7029
  };
6868
7030
  const { logger, onStepFinish } = buildDefaultStepLogger("kb", 150);
6869
- const contextBlock = (input.projectContext ? "\n" + formatContext(input.projectContext) + "\n" : "") + formatRetryGuidance(input.retryGuidance);
6870
- const pages = input.projectContext?.pages;
7031
+ const contextBlock = formatRetryGuidance(input.retryGuidance);
6871
7032
  const tracker = new PageTracker(input.projectRoot);
6872
- if (pages?.length) {
6873
- tracker.register(pages.map((p) => p.path));
6874
- }
6875
- const prompt = pages?.length ? `Analyze the codebase at the working directory and generate a complete knowledge base.
6876
- ${contextBlock}
6877
- MANDATORY PROCESS:
6878
- Pages have already been discovered (${pages.length} routes pre-registered). You do NOT need to glob for them.
6879
- 1. Use list_directory at root to understand the project structure
6880
- 2. Read EVERY registered page file with read_file - the system tracks this
6881
- 3. Write AUTONOMA.md progressively as you go (update it after each major area)
6882
- 4. Call page_coverage to verify you've read all pages
6883
- 5. Call finish - it will REJECT if you have not read enough of the registered routes
6884
-
6885
- Output files:
6886
- 1. AUTONOMA.md - with YAML frontmatter (app_name, app_description, core_flows, feature_count)` : `Analyze the codebase at the working directory and generate a complete knowledge base.
7033
+ const prompt = `Analyze the codebase at the working directory and generate a complete knowledge base.
6887
7034
  ${contextBlock}
6888
7035
  MANDATORY PROCESS:
6889
7036
  1. Use list_directory at root to understand the project structure
@@ -6899,8 +7046,8 @@ Output files:
6899
7046
  const agentConfig = buildKbAgentConfig(tracker, model, input, onStepFinish, setResult);
6900
7047
  await runAgent(agentConfig, prompt, () => result);
6901
7048
  logger.summary();
6902
- const autonomaPath = join20(input.outputDir, "AUTONOMA.md");
6903
- const autonomaExists = await readFile11(autonomaPath, "utf-8").then(() => true).catch((err) => {
7049
+ const autonomaPath = join19(input.outputDir, "AUTONOMA.md");
7050
+ const autonomaExists = await readFile10(autonomaPath, "utf-8").then(() => true).catch((err) => {
6904
7051
  debugLog("AUTONOMA.md not found while checking step completion", { err });
6905
7052
  return false;
6906
7053
  });
@@ -6911,32 +7058,6 @@ Output files:
6911
7058
  summary: "Knowledge base generated."
6912
7059
  };
6913
7060
  }
6914
- const finalTracker = new PageTracker(input.projectRoot);
6915
- if (pages?.length) {
6916
- const paths = pages.map((p) => p.path);
6917
- finalTracker.register(paths);
6918
- for (const path3 of paths) finalTracker.markRead(path3);
6919
- }
6920
- const finalConfig = buildKbAgentConfig(finalTracker, model, input, onStepFinish, setResult);
6921
- const declaredCriticalFlows = input.projectContext?.criticalFlows?.trim();
6922
- if (result?.success && declaredCriticalFlows) {
6923
- const beforeSelfReview = result;
6924
- result = void 0;
6925
- const selfReviewPrompt = `Before this knowledge base is shown to the user, verify it honors the critical flows they explicitly declared.
6926
-
6927
- The user said these flows are critical and cannot break:
6928
- "${declaredCriticalFlows}"
6929
-
6930
- Read your AUTONOMA.md output. For EACH critical flow the user named:
6931
- - Confirm it appears as a feature in core_flows (map the user's wording to the matching feature).
6932
- - Confirm that feature is marked core: true with a coreReason.
6933
-
6934
- If any declared critical flow is missing, mismatched, or left core: false, FIX AUTONOMA.md now - add the feature if it is genuinely absent, or flip core to true with a coreReason. Do not downgrade or drop anything the user declared critical.
6935
-
6936
- When AUTONOMA.md correctly reflects every declared critical flow, call finish.`;
6937
- await runAgent(finalConfig, selfReviewPrompt, () => result);
6938
- if (!result) result = beforeSelfReview;
6939
- }
6940
7061
  const reviewed = result;
6941
7062
  return reviewed ?? {
6942
7063
  success: false,
@@ -6950,7 +7071,6 @@ var init_kb_generator = __esm({
6950
7071
  "use strict";
6951
7072
  init_esm_shims();
6952
7073
  init_agent();
6953
- init_context();
6954
7074
  init_debug();
6955
7075
  init_model();
6956
7076
  init_pick_string();
@@ -7113,17 +7233,17 @@ var entity_audit_exports = {};
7113
7233
  __export(entity_audit_exports, {
7114
7234
  runEntityAudit: () => runEntityAudit
7115
7235
  });
7116
- import { readFile as readFile12, writeFile as writeFile7 } from "fs/promises";
7117
- import { join as join21 } from "path";
7236
+ import { readFile as readFile11, writeFile as writeFile6 } from "fs/promises";
7237
+ import { join as join20 } from "path";
7118
7238
  import { tool as tool12 } from "ai";
7119
7239
  import { glob as glob4 } from "glob";
7120
- import { z as z46 } from "zod";
7240
+ import { z as z48 } from "zod";
7121
7241
  function buildRegisterModelsTool(tracker) {
7122
7242
  return tool12({
7123
7243
  description: "Register ALL database models discovered via grep. Call this ONCE after grepping for model definitions. After registering, use next_model to process them one at a time.",
7124
- inputSchema: z46.object({
7125
- models: z46.array(z46.string()).describe("All model/table names found by grep"),
7126
- framework: z46.string().describe("Database framework detected (e.g. 'sqlalchemy', 'prisma', 'drizzle')")
7244
+ inputSchema: z48.object({
7245
+ models: z48.array(z48.string()).describe("All model/table names found by grep"),
7246
+ framework: z48.string().describe("Database framework detected (e.g. 'sqlalchemy', 'prisma', 'drizzle')")
7127
7247
  }),
7128
7248
  execute: async (input) => {
7129
7249
  tracker.register(input.models);
@@ -7139,7 +7259,7 @@ function buildRegisterModelsTool(tracker) {
7139
7259
  function buildNextModelTool(tracker) {
7140
7260
  return tool12({
7141
7261
  description: "Get the next model to audit from the queue. If you called next_model before without calling mark_model_audited, the previous model is auto-skipped (marked as no creation path found). Returns done:true when all models are processed.",
7142
- inputSchema: z46.object({}),
7262
+ inputSchema: z48.object({}),
7143
7263
  execute: async () => {
7144
7264
  const next = tracker.nextModel();
7145
7265
  if (!next) {
@@ -7159,17 +7279,17 @@ function buildNextModelTool(tracker) {
7159
7279
  function buildMarkModelAuditedTool(tracker) {
7160
7280
  return tool12({
7161
7281
  description: "Mark a model as audited after you have determined its creation paths. Call this for EACH model after reading its creation code and determining independently_created + created_by. Include creation_function (e.g. 'UserService.create'), side_effects (list of things the creation does beyond the model itself), and for each created_by entry include owner, via (function name), and why (one sentence explaining the relationship).",
7162
- inputSchema: z46.object({
7163
- model: z46.string().describe("Model name"),
7164
- independently_created: z46.boolean(),
7165
- creation_file: z46.string().optional().describe("File containing the creation function"),
7166
- creation_function: z46.string().optional().describe("Function/method name (e.g. 'UserService.create' or 'create_user')"),
7167
- side_effects: z46.array(z46.string()).optional().describe("Side effects of creation (e.g. 'creates default Settings row', 'hashes password')"),
7168
- created_by: z46.array(
7169
- z46.object({
7170
- owner: z46.string().describe("Owner model name"),
7171
- via: z46.string().optional().describe("Function that creates this model (e.g. 'OrganizationService.create')"),
7172
- why: z46.string().optional().describe("One sentence explaining why this model is created as a side effect")
7282
+ inputSchema: z48.object({
7283
+ model: z48.string().describe("Model name"),
7284
+ independently_created: z48.boolean(),
7285
+ creation_file: z48.string().optional().describe("File containing the creation function"),
7286
+ creation_function: z48.string().optional().describe("Function/method name (e.g. 'UserService.create' or 'create_user')"),
7287
+ side_effects: z48.array(z48.string()).optional().describe("Side effects of creation (e.g. 'creates default Settings row', 'hashes password')"),
7288
+ created_by: z48.array(
7289
+ z48.object({
7290
+ owner: z48.string().describe("Owner model name"),
7291
+ via: z48.string().optional().describe("Function that creates this model (e.g. 'OrganizationService.create')"),
7292
+ why: z48.string().optional().describe("One sentence explaining why this model is created as a side effect")
7173
7293
  })
7174
7294
  ).describe("List of owner models that create this as a side effect, empty array if none")
7175
7295
  }),
@@ -7208,16 +7328,16 @@ function buildMarkModelAuditedTool(tracker) {
7208
7328
  function buildModelCoverageTool(tracker) {
7209
7329
  return tool12({
7210
7330
  description: "Check how many registered models you've audited vs how many remain.",
7211
- inputSchema: z46.object({}),
7331
+ inputSchema: z48.object({}),
7212
7332
  execute: async () => tracker.coverage()
7213
7333
  });
7214
7334
  }
7215
7335
  function buildFinishTool2(tracker, onFinish) {
7216
7336
  return tool12({
7217
7337
  description: "Call when entity audit is complete. BLOCKED if there are unaudited models - call model_coverage first to check.",
7218
- inputSchema: z46.object({
7219
- summary: z46.string().describe("Summary of the audit"),
7220
- artifacts: z46.array(z46.string()).describe("Files written")
7338
+ inputSchema: z48.object({
7339
+ summary: z48.string().describe("Summary of the audit"),
7340
+ artifacts: z48.array(z48.string()).describe("Files written")
7221
7341
  }),
7222
7342
  execute: async (input) => {
7223
7343
  const cov = tracker.coverage();
@@ -7245,7 +7365,7 @@ async function findPrismaSchema(projectRoot) {
7245
7365
  return candidates[0] ?? void 0;
7246
7366
  }
7247
7367
  async function extractPrismaModels(schemaPath) {
7248
- const content = await readFile12(schemaPath, "utf-8");
7368
+ const content = await readFile11(schemaPath, "utf-8");
7249
7369
  return content.split("\n").filter((line) => line.startsWith("model ")).map((line) => line.split(/\s+/)[1]).filter((name) => name != null);
7250
7370
  }
7251
7371
  async function detectFrameworkAndModels(projectRoot) {
@@ -7272,7 +7392,7 @@ async function runEntityAudit(input) {
7272
7392
  const scopeBlock = input.scopeHint != null ? `
7273
7393
  ${input.scopeHint}
7274
7394
  ` : "";
7275
- const contextBlock = (input.projectContext ? "\n" + formatContext(input.projectContext) + "\n" : "") + scopeBlock + formatRetryGuidance(input.retryGuidance);
7395
+ const contextBlock = scopeBlock + formatRetryGuidance(input.retryGuidance);
7276
7396
  const preRegBlock = preRegisteredCount > 0 ? `
7277
7397
  ## Pre-registered models (${preRegisteredCount} found via ${detection.framework} schema at ${detection.schemaFile})
7278
7398
 
@@ -7332,8 +7452,8 @@ ${formatException(err)}`);
7332
7452
  logger.summary();
7333
7453
  const writeCanonicalAudit = async () => {
7334
7454
  if (tracker.auditedModels.size === 0) return void 0;
7335
- const auditPath = join21(input.outputDir, "entity-audit.md");
7336
- await writeFile7(auditPath, tracker.generateAuditMarkdown(), "utf-8");
7455
+ const auditPath = join20(input.outputDir, "entity-audit.md");
7456
+ await writeFile6(auditPath, tracker.generateAuditMarkdown(), "utf-8");
7337
7457
  return auditPath;
7338
7458
  };
7339
7459
  const canonicalPath = await writeCanonicalAudit();
@@ -7347,9 +7467,9 @@ ${formatException(err)}`);
7347
7467
  }
7348
7468
  const reviewed = result;
7349
7469
  if (!reviewed) {
7350
- const auditPath = join21(input.outputDir, "entity-audit.md");
7470
+ const auditPath = join20(input.outputDir, "entity-audit.md");
7351
7471
  try {
7352
- await readFile12(auditPath, "utf-8");
7472
+ await readFile11(auditPath, "utf-8");
7353
7473
  return {
7354
7474
  success: true,
7355
7475
  artifacts: ["entity-audit.md"],
@@ -7371,7 +7491,6 @@ var init_entity_audit = __esm({
7371
7491
  "use strict";
7372
7492
  init_esm_shims();
7373
7493
  init_agent();
7374
- init_context();
7375
7494
  init_debug();
7376
7495
  init_errors();
7377
7496
  init_model();
@@ -7560,8 +7679,8 @@ on a unique constraint. Everything else stays concrete and static, and no other
7560
7679
  });
7561
7680
 
7562
7681
  // src/agents/03-scenario-recipe/scenario-table.ts
7563
- import { readFile as readFile13 } from "fs/promises";
7564
- import { join as join22 } from "path";
7682
+ import { readFile as readFile12 } from "fs/promises";
7683
+ import { join as join21 } from "path";
7565
7684
  import matter2 from "gray-matter";
7566
7685
  function validateScenarioIsConcrete(content) {
7567
7686
  const errors = findUnknownRecipeTokens(content).map(
@@ -7596,22 +7715,22 @@ __export(scenario_recipe_exports, {
7596
7715
  feedbackToScenario: () => feedbackToScenario,
7597
7716
  runScenarioRecipe: () => runScenarioRecipe
7598
7717
  });
7599
- import { readFile as readFile14 } from "fs/promises";
7600
- import { join as join23 } from "path";
7718
+ import { readFile as readFile13 } from "fs/promises";
7719
+ import { join as join22 } from "path";
7601
7720
  import { tool as tool13 } from "ai";
7602
- import { z as z47 } from "zod";
7721
+ import { z as z49 } from "zod";
7603
7722
  function buildFinishTool3(requiredEntities, outputDir, onFinish) {
7604
7723
  return tool13({
7605
7724
  description: "Call when scenario design is complete and scenarios.md is written. BLOCKED if any required entities are missing from the scenario.",
7606
- inputSchema: z47.object({
7607
- summary: z47.string().describe("Summary of the scenario"),
7608
- entityCount: z47.number().describe("Number of entity types in the scenario"),
7609
- artifacts: z47.array(z47.string()).describe("Files written")
7725
+ inputSchema: z49.object({
7726
+ summary: z49.string().describe("Summary of the scenario"),
7727
+ entityCount: z49.number().describe("Number of entity types in the scenario"),
7728
+ artifacts: z49.array(z49.string()).describe("Files written")
7610
7729
  }),
7611
7730
  execute: async (input) => {
7612
7731
  let content;
7613
7732
  try {
7614
- content = await readFile14(join23(outputDir, "scenarios.md"), "utf-8");
7733
+ content = await readFile13(join22(outputDir, "scenarios.md"), "utf-8");
7615
7734
  } catch {
7616
7735
  return { error: "Cannot finish: scenarios.md not found. Write it first." };
7617
7736
  }
@@ -7649,7 +7768,7 @@ async function runScenarioRecipe(input) {
7649
7768
  const scopeBlock = input.scopeHint != null ? `
7650
7769
  ${input.scopeHint}
7651
7770
  ` : "";
7652
- const contextBlock = (input.projectContext ? "\n" + formatContext(input.projectContext) + "\n" : "") + scopeBlock + formatRetryGuidance(input.retryGuidance);
7771
+ const contextBlock = scopeBlock + formatRetryGuidance(input.retryGuidance);
7653
7772
  const requiredEntities = await parseEntityNames(input.outputDir);
7654
7773
  const entityListBlock = requiredEntities.length > 0 ? `
7655
7774
  ## Required entities (${requiredEntities.length} total - ALL must appear in the scenario)
@@ -7690,9 +7809,9 @@ When done, call finish.`;
7690
7809
  logger.summary();
7691
7810
  const reviewed = result;
7692
7811
  if (!reviewed) {
7693
- const scenariosPath = join23(input.outputDir, "scenarios.md");
7812
+ const scenariosPath = join22(input.outputDir, "scenarios.md");
7694
7813
  try {
7695
- await readFile14(scenariosPath, "utf-8");
7814
+ await readFile13(scenariosPath, "utf-8");
7696
7815
  return {
7697
7816
  success: true,
7698
7817
  artifacts: ["scenarios.md"],
@@ -7740,7 +7859,6 @@ var init_scenario_recipe = __esm({
7740
7859
  init_esm_shims();
7741
7860
  init_src();
7742
7861
  init_agent();
7743
- init_context();
7744
7862
  init_debug();
7745
7863
  init_model();
7746
7864
  init_parse_entity_audit();
@@ -7752,12 +7870,12 @@ var init_scenario_recipe = __esm({
7752
7870
  });
7753
7871
 
7754
7872
  // src/agents/04-recipe-builder/entity-order.ts
7755
- import { readFile as readFile15 } from "fs/promises";
7756
- import { join as join24 } from "path";
7873
+ import { readFile as readFile14 } from "fs/promises";
7874
+ import { join as join23 } from "path";
7757
7875
  import matter3 from "gray-matter";
7758
- import { z as z48 } from "zod";
7876
+ import { z as z50 } from "zod";
7759
7877
  async function parseEntityAudit(outputDir) {
7760
- const raw = await readFile15(join24(outputDir, "entity-audit.md"), "utf-8");
7878
+ const raw = await readFile14(join23(outputDir, "entity-audit.md"), "utf-8");
7761
7879
  try {
7762
7880
  const parsed = frontmatterSchema.safeParse(matter3(raw).data);
7763
7881
  if (parsed.success && parsed.data.models.length > 0) {
@@ -7918,34 +8036,34 @@ var init_entity_order = __esm({
7918
8036
  "src/agents/04-recipe-builder/entity-order.ts"() {
7919
8037
  "use strict";
7920
8038
  init_esm_shims();
7921
- createdBySchema = z48.object({
7922
- owner: z48.string(),
7923
- via: z48.string().optional(),
7924
- why: z48.string().optional()
8039
+ createdBySchema = z50.object({
8040
+ owner: z50.string(),
8041
+ via: z50.string().optional(),
8042
+ why: z50.string().optional()
7925
8043
  });
7926
- auditedModelSchema = z48.object({
7927
- name: z48.string(),
7928
- independently_created: z48.coerce.boolean().default(false),
7929
- creation_file: z48.string().optional(),
7930
- creation_function: z48.string().optional(),
7931
- side_effects: z48.array(z48.string()).optional(),
8044
+ auditedModelSchema = z50.object({
8045
+ name: z50.string(),
8046
+ independently_created: z50.coerce.boolean().default(false),
8047
+ creation_file: z50.string().optional(),
8048
+ creation_function: z50.string().optional(),
8049
+ side_effects: z50.array(z50.string()).optional(),
7932
8050
  // Tolerate a stray `created_by:` with no entries (parsed as null by YAML).
7933
- created_by: z48.array(createdBySchema).nullish().transform((v) => v ?? [])
8051
+ created_by: z50.array(createdBySchema).nullish().transform((v) => v ?? [])
7934
8052
  });
7935
- frontmatterSchema = z48.object({
7936
- models: z48.array(auditedModelSchema).nullish().transform((v) => v ?? [])
8053
+ frontmatterSchema = z50.object({
8054
+ models: z50.array(auditedModelSchema).nullish().transform((v) => v ?? [])
7937
8055
  });
7938
8056
  }
7939
8057
  });
7940
8058
 
7941
8059
  // src/agents/04-recipe-builder/completion.ts
7942
- import { readFile as readFile16 } from "fs/promises";
7943
- import { join as join25 } from "path";
7944
- import { z as z49 } from "zod";
8060
+ import { readFile as readFile15 } from "fs/promises";
8061
+ import { join as join24 } from "path";
8062
+ import { z as z51 } from "zod";
7945
8063
  async function readCompletion(outputDir) {
7946
8064
  let raw;
7947
8065
  try {
7948
- raw = await readFile16(join25(outputDir, COMPLETION_MARKER_FILE), "utf-8");
8066
+ raw = await readFile15(join24(outputDir, COMPLETION_MARKER_FILE), "utf-8");
7949
8067
  } catch (err) {
7950
8068
  debugLog("No completion marker yet", { err });
7951
8069
  return false;
@@ -7971,7 +8089,7 @@ var init_completion = __esm({
7971
8089
  init_esm_shims();
7972
8090
  init_debug();
7973
8091
  COMPLETION_MARKER_FILE = ".sdk-integration-complete";
7974
- completionSchema = z49.object({ complete: z49.literal(true) });
8092
+ completionSchema = z51.object({ complete: z51.literal(true) });
7975
8093
  }
7976
8094
  });
7977
8095
 
@@ -8160,8 +8278,8 @@ var init_launcher = __esm({
8160
8278
  });
8161
8279
 
8162
8280
  // src/agents/04-recipe-builder/integration-prompt.ts
8163
- import { writeFile as writeFile8 } from "fs/promises";
8164
- import { join as join26 } from "path";
8281
+ import { writeFile as writeFile7 } from "fs/promises";
8282
+ import { join as join25 } from "path";
8165
8283
  function renderIntegrationPrompt(params) {
8166
8284
  const priorFailureSection = params.priorFailure != null ? `
8167
8285
  \u2550\u2550\u2550 A PRIOR SESSION DID NOT COMPLETE - READ THIS FIRST \u2550\u2550\u2550
@@ -8183,6 +8301,22 @@ Work without asking questions. Make reasonable, codebase-grounded decisions. Onl
8183
8301
  stop for missing secrets, credentials, or external services that genuinely cannot
8184
8302
  be mocked or run locally - and when you do, say exactly what you need and why.
8185
8303
 
8304
+ \u2550\u2550\u2550 BRANCH FIRST - BEFORE YOU CHANGE A SINGLE FILE \u2550\u2550\u2550
8305
+ Your work ships to the developer as a pull request, so cut the branch before your
8306
+ first edit - never commit onto the default branch. Do not assume it is called "main":
8307
+ read it from the remote (\`git symbolic-ref refs/remotes/origin/HEAD\`, or
8308
+ \`git remote show origin\`), then:
8309
+ git fetch origin
8310
+ git switch --create ${INTEGRATION_BRANCH} origin/<default-branch>
8311
+ Two cases to handle before you run that:
8312
+ \u2022 The working tree already has uncommitted changes: they are the developer's, so do
8313
+ NOT stash, discard, or commit them. Branch off the CURRENT HEAD instead, and keep
8314
+ them out of your commit later.
8315
+ \u2022 A branch from a prior session already exists (or you are already on one): stay on
8316
+ it and continue there rather than cutting a second one.
8317
+ If this checkout is not a git repository at all, skip the branch and say so at the end -
8318
+ everything else in this prompt still applies.
8319
+
8186
8320
  \u2550\u2550\u2550 DISCOVER FIRST (never assume) \u2550\u2550\u2550
8187
8321
  Before writing anything, investigate this specific app with your tools. Determine,
8188
8322
  from the actual source, every one of:
@@ -8314,8 +8448,8 @@ Before implementing, write a checklist file inside the app (e.g. IMPLEMENTATION.
8314
8448
  and keep it updated. It must enumerate, as explicit checkboxes: EVERY entity the
8315
8449
  entity audit says needs a factory (by name, copied from the audit), plus the
8316
8450
  endpoint, teardown, the auth callback, the maintenance note, the full-recipe pass,
8317
- and the two-concurrent-instances proof. Check items off only when actually done and
8318
- verified. The single most common failure is stopping with entities left uncovered.
8451
+ the two-concurrent-instances proof, and the pushed branch + opened pull request.
8452
+ Check items off only when actually done and verified. The single most common failure is stopping with entities left uncovered.
8319
8453
 
8320
8454
  \u2550\u2550\u2550 VALIDATE - ENTITY BY ENTITY, THEN THE WHOLE RECIPE \u2550\u2550\u2550
8321
8455
  You validate your own work by driving the endpoint through THIS CLI's signed client
@@ -8375,43 +8509,71 @@ concurrently because the app's own schema forces a global singleton, write that
8375
8509
  IMPLEMENTATION.md - name the table, the constraint, and why it cannot be made per-run -
8376
8510
  and then proceed. Never leave this step silently unfinished.
8377
8511
 
8512
+ \u2550\u2550\u2550 SHIP IT - COMMIT, PUSH, OPEN A PULL REQUEST \u2550\u2550\u2550
8513
+ Everything green means nothing until the developer can review it. Once the full recipe
8514
+ and the concurrency proof pass, put the work up as a pull request - do not leave it
8515
+ sitting uncommitted in the working tree:
8516
+ 1. Read \`git status\` and \`git diff\` and stage ONLY your integration: the endpoint,
8517
+ the factories, the SDK dependency and its lockfile change, and the maintenance
8518
+ note. Leave the developer's pre-existing uncommitted changes out of it, and NEVER
8519
+ commit secrets, .env files, credentials, or local scratch output.
8520
+ 2. Commit in the style the repo already uses (read \`git log\`) - one commit is fine.
8521
+ 3. Push the branch and set its upstream:
8522
+ git push --set-upstream origin ${INTEGRATION_BRANCH}
8523
+ 4. Open a pull request against the DEFAULT branch you cut from, using the repo's own
8524
+ tooling if it is installed and authenticated (e.g. \`gh pr create --base <default>\`).
8525
+ Describe what you added: the endpoint path, which entities got factories, how
8526
+ teardown is scoped, and anything you documented as a limitation. If no PR tool is
8527
+ available or authenticated, the push is still mandatory - then print the compare
8528
+ URL the push prints back so the developer can open the PR in one click.
8529
+ 5. If pushing or opening the PR genuinely cannot be done (no remote, no write access,
8530
+ no auth, not a git repo), leave the work COMMITTED on the branch and write which
8531
+ step failed and why in IMPLEMENTATION.md. Committing is the one part you can always
8532
+ do - never stop at an uncommitted working tree.
8533
+
8378
8534
  \u2550\u2550\u2550 FINISH - THE LAST THING YOU DO \u2550\u2550\u2550
8379
- Only after every entity, the full-recipe pass, and the two-concurrent-instances proof
8380
- are green (or the blocking constraint is documented in IMPLEMENTATION.md), and
8381
- ${params.recipePath} holds the recipe you validated, write the completion marker so the
8382
- CLI knows the session is done and can upload the recipe:
8383
- ${join26(params.outputDir, COMPLETION_MARKER_FILE)}
8535
+ Write the completion marker once ALL of these hold:
8536
+ \u2022 every entity, the full-recipe pass, and the two-concurrent-instances proof are green
8537
+ - or the blocking constraint is documented in IMPLEMENTATION.md
8538
+ \u2022 ${params.recipePath} holds the recipe you validated
8539
+ \u2022 your work is committed, and pushed with a pull request open - or the reason you could
8540
+ not push / open one is documented in IMPLEMENTATION.md
8541
+ A step you documented as genuinely blocked NEVER justifies withholding the marker; a
8542
+ checklist item you simply have not finished always does. The marker is how the CLI knows
8543
+ the session is done and can upload the recipe:
8544
+ ${join25(params.outputDir, COMPLETION_MARKER_FILE)}
8384
8545
  Its contents MUST be exactly:
8385
8546
  { "complete": true }
8386
- Do NOT write this marker while any checklist item is unfinished. It is how control
8387
- returns to the CLI - it is not optional. The planner watches for this marker and
8388
- takes the terminal back shortly after it appears. After writing it, end with ONE
8389
- short closing message telling the developer:
8547
+ Writing it is not optional: the planner watches for this marker and takes the terminal
8548
+ back shortly after it appears. After writing it, end with ONE short closing message that
8549
+ names the pull request you opened - or, if you couldn't open one, the branch you pushed,
8550
+ or the commit you left behind and what blocked the push - and then says:
8390
8551
  "The integration is done. The Autonoma planner takes this terminal back in a
8391
8552
  few seconds to continue the setup - or exit now to continue immediately."
8392
8553
  Nothing after that message - no further questions, summaries, or work.`;
8393
8554
  }
8394
8555
  async function writeIntegrationPrompt(params) {
8395
- const path3 = join26(params.outputDir, INTEGRATION_PROMPT_FILE);
8396
- await writeFile8(path3, renderIntegrationPrompt(params), "utf-8");
8556
+ const path3 = join25(params.outputDir, INTEGRATION_PROMPT_FILE);
8557
+ await writeFile7(path3, renderIntegrationPrompt(params), "utf-8");
8397
8558
  return path3;
8398
8559
  }
8399
- var INTEGRATION_PROMPT_VERSION, INTEGRATION_PROMPT_FILE, DEFAULT_ENDPOINT_PATH;
8560
+ var INTEGRATION_PROMPT_VERSION, INTEGRATION_PROMPT_FILE, DEFAULT_ENDPOINT_PATH, INTEGRATION_BRANCH;
8400
8561
  var init_integration_prompt = __esm({
8401
8562
  "src/agents/04-recipe-builder/integration-prompt.ts"() {
8402
8563
  "use strict";
8403
8564
  init_esm_shims();
8404
8565
  init_src();
8405
8566
  init_completion();
8406
- INTEGRATION_PROMPT_VERSION = 7;
8567
+ INTEGRATION_PROMPT_VERSION = 9;
8407
8568
  INTEGRATION_PROMPT_FILE = "integration-prompt.md";
8408
8569
  DEFAULT_ENDPOINT_PATH = "/api/autonoma";
8570
+ INTEGRATION_BRANCH = "autonoma-integration";
8409
8571
  }
8410
8572
  });
8411
8573
 
8412
8574
  // src/agents/04-recipe-builder/state.ts
8413
- import { readFile as readFile17, writeFile as writeFile9 } from "fs/promises";
8414
- import { join as join27 } from "path";
8575
+ import { readFile as readFile16, writeFile as writeFile8 } from "fs/promises";
8576
+ import { join as join26 } from "path";
8415
8577
  function initialRecipeState() {
8416
8578
  return {
8417
8579
  phase: "tech-detect",
@@ -8421,7 +8583,7 @@ function initialRecipeState() {
8421
8583
  }
8422
8584
  async function loadRecipeState(outputDir) {
8423
8585
  try {
8424
- const raw = await readFile17(join27(outputDir, STATE_FILE2), "utf-8");
8586
+ const raw = await readFile16(join26(outputDir, STATE_FILE2), "utf-8");
8425
8587
  const parsed = JSON.parse(raw);
8426
8588
  return parsed;
8427
8589
  } catch {
@@ -8429,7 +8591,7 @@ async function loadRecipeState(outputDir) {
8429
8591
  }
8430
8592
  }
8431
8593
  async function saveRecipeState(outputDir, state) {
8432
- await writeFile9(join27(outputDir, STATE_FILE2), JSON.stringify(state, null, 2), "utf-8");
8594
+ await writeFile8(join26(outputDir, STATE_FILE2), JSON.stringify(state, null, 2), "utf-8");
8433
8595
  }
8434
8596
  var STATE_FILE2;
8435
8597
  var init_state2 = __esm({
@@ -8442,9 +8604,9 @@ var init_state2 = __esm({
8442
8604
 
8443
8605
  // src/agents/04-recipe-builder/phases/handoff.ts
8444
8606
  import { rm as rm2 } from "fs/promises";
8445
- import { join as join28 } from "path";
8607
+ import { join as join27 } from "path";
8446
8608
  async function runHandoffPhase(state, deps, outputDir) {
8447
- const recipePath = state.recipePath ?? join28(outputDir, RECIPE_FILE);
8609
+ const recipePath = state.recipePath ?? join27(outputDir, RECIPE_FILE);
8448
8610
  state.recipePath = recipePath;
8449
8611
  const launcher = await selectLauncher(deps.launchers, state.agentId ?? deps.presetAgentId, deps.interactive);
8450
8612
  if (launcher == null) {
@@ -8471,7 +8633,7 @@ async function runHandoffPhase(state, deps, outputDir) {
8471
8633
  return { kind: "advance" };
8472
8634
  }
8473
8635
  async function runCompletionPhase(state, deps, outputDir) {
8474
- const recipePath = state.recipePath ?? join28(outputDir, RECIPE_FILE);
8636
+ const recipePath = state.recipePath ?? join27(outputDir, RECIPE_FILE);
8475
8637
  while (true) {
8476
8638
  const [complete, read] = await Promise.all([readCompletion(outputDir), loadRecipe(outputDir)]);
8477
8639
  const uploadProblems = read.status === "ok" ? findRecipeUploadProblems(read.recipe) : [];
@@ -8504,7 +8666,7 @@ async function runCompletionPhase(state, deps, outputDir) {
8504
8666
  function handbackSummary(failure, outputDir) {
8505
8667
  return `${failure}
8506
8668
 
8507
- Finish the integration (see the prompt at ${join28(outputDir, INTEGRATION_PROMPT_FILE)}), then re-run with --resume.`;
8669
+ Finish the integration (see the prompt at ${join27(outputDir, INTEGRATION_PROMPT_FILE)}), then re-run with --resume.`;
8508
8670
  }
8509
8671
  async function launchAgent(launcher, permissionMode, target) {
8510
8672
  const promptFile = await writeIntegrationPrompt({
@@ -8513,7 +8675,7 @@ async function launchAgent(launcher, permissionMode, target) {
8513
8675
  cliCommand: target.cliCommand,
8514
8676
  priorFailure: target.priorFailure
8515
8677
  });
8516
- await rm2(join28(target.outputDir, COMPLETION_MARKER_FILE), { force: true }).catch((err) => {
8678
+ await rm2(join27(target.outputDir, COMPLETION_MARKER_FILE), { force: true }).catch((err) => {
8517
8679
  debugLog("Could not clear a stale completion marker", { err });
8518
8680
  });
8519
8681
  log.info(`Launching ${launcher.label}. It will implement the integration - watch and steer it as it works.`);
@@ -8522,7 +8684,7 @@ async function launchAgent(launcher, permissionMode, target) {
8522
8684
  await countdown({
8523
8685
  title: `Handing off to ${launcher.label}`,
8524
8686
  lines: [
8525
- `Your terminal is about to switch to ${launcher.label}. It will implement the Autonoma SDK integration inside your repo: install the SDK, wire the endpoint, and write a real factory for every entity in the audit, validating each one against your locally running app.`,
8687
+ `Your terminal is about to switch to ${launcher.label}. It will implement the Autonoma SDK integration inside your repo: install the SDK, wire the endpoint, and write a real factory for every entity in the audit, validating each one against your locally running app. It works on its own branch and opens a pull request with the result.`,
8526
8688
  `This dashboard disappears while it works - that's expected. Watch and steer it like any ${launcher.label} session; this usually takes a while.`,
8527
8689
  `When it finishes and exits, you come straight back here and the planner continues where it left off: submitting the validated recipe, then generating your test suite.`
8528
8690
  ],
@@ -8687,465 +8849,37 @@ var init_recipe_builder = __esm({
8687
8849
  }
8688
8850
  });
8689
8851
 
8690
- // src/agents/05-test-generator/restore-deleted-test.ts
8691
- import { access, mkdir as mkdir3, writeFile as writeFile10 } from "fs/promises";
8692
- import { dirname as dirname4 } from "path";
8693
- async function restoreDeletedTest(testPath, originalContent) {
8694
- try {
8695
- await access(testPath);
8696
- return false;
8697
- } catch (err) {
8698
- debugLog("Test absent after its fix pass - restoring the pre-review content", { testPath, err });
8699
- }
8700
- try {
8701
- await mkdir3(dirname4(testPath), { recursive: true });
8702
- await writeFile10(testPath, originalContent, "utf-8");
8703
- return true;
8704
- } catch (err) {
8705
- console.warn(` [fix] Could not restore ${testPath}: ${err instanceof Error ? err.message : String(err)}`);
8706
- captureLog("error", "Failed to restore a reviewed test - it stays missing from the suite", {
8707
- source: "test-generator",
8708
- step: "review-fix",
8709
- path: testPath
8710
- });
8711
- return false;
8712
- }
8852
+ // src/agents/00b-feature-discovery/index.ts
8853
+ import { readFile as readFile17, writeFile as writeFile9 } from "fs/promises";
8854
+ import { join as join28 } from "path";
8855
+ import { tool as tool14 } from "ai";
8856
+ import { z as z52 } from "zod";
8857
+ async function saveFeatures(outputDir, features) {
8858
+ const obj = Object.fromEntries(features);
8859
+ await writeFile9(join28(outputDir, FEATURES_FILE), JSON.stringify(obj, null, 2), "utf-8");
8713
8860
  }
8714
- var init_restore_deleted_test = __esm({
8715
- "src/agents/05-test-generator/restore-deleted-test.ts"() {
8716
- "use strict";
8717
- init_esm_shims();
8718
- init_debug();
8719
- init_logs();
8861
+ async function loadFeatures(outputDir) {
8862
+ try {
8863
+ const raw = await readFile17(join28(outputDir, FEATURES_FILE), "utf-8");
8864
+ const obj = JSON.parse(raw);
8865
+ return new Map(Object.entries(obj));
8866
+ } catch {
8867
+ return void 0;
8720
8868
  }
8721
- });
8722
-
8723
- // src/agents/05-test-generator/rubrics.ts
8724
- import { z as z50 } from "zod";
8725
- function reviewResultSchema(shape) {
8726
- return z50.object(shape);
8727
8869
  }
8728
- var dimensionResultSchema, reviewResultRecordSchema, structuralIntentRubric, flowCompletenessRubric, uiTextRubric, dataAccuracyRubric, ALL_RUBRICS;
8729
- var init_rubrics = __esm({
8730
- "src/agents/05-test-generator/rubrics.ts"() {
8731
- "use strict";
8732
- init_esm_shims();
8733
- dimensionResultSchema = z50.object({
8734
- pass: z50.boolean(),
8735
- evidence: z50.string().describe("What you checked and found - cite file paths, line content, or specific strings"),
8736
- suggestion: z50.string().optional().describe("What the planner agent should fix, if failing")
8737
- });
8738
- reviewResultRecordSchema = z50.record(z50.string(), dimensionResultSchema);
8739
- structuralIntentRubric = {
8740
- name: "structural-intent",
8741
- maxSteps: 8,
8742
- dimensions: ["structuralValidity", "intentQuality", "missionAlignment"],
8743
- resultSchema: reviewResultSchema({
8744
- structuralValidity: dimensionResultSchema.describe(
8745
- "Are all step verbs valid (click/type/scroll/assert/hover/drag/read/refresh)? Are asserts visual-only (no URLs, network, console)? No code selectors? No login steps?"
8746
- ),
8747
- intentQuality: dimensionResultSchema.describe(
8748
- "Is the intent a specific, falsifiable behavioral claim - not just 'verify X is visible'?"
8749
- ),
8750
- missionAlignment: dimensionResultSchema.describe(
8751
- "Does the test's intent + steps verify the feature's core purpose? Not just UI appearance."
8752
- )
8753
- }),
8754
- systemPrompt: `You are a structural reviewer for E2E test plans. Each test will be executed by a VISUAL agent that sees the screen like a human user - it cannot inspect code, network, URLs, or any non-visual state.
8755
-
8756
- Your job is to EVALUATE tests against a rubric, NOT to rewrite them. You have tools to read source code if needed.
8757
-
8758
- ## Rubric dimensions
8870
+ async function runFeatureDiscovery(input) {
8871
+ const model = getModel(input.modelId);
8872
+ const collector = new FeatureCollector();
8873
+ let result;
8874
+ const { logger, onStepFinish } = buildDefaultStepLogger("feature-discovery", 300);
8875
+ const pagesDescription = Array.from(input.pages.entries()).map(([path3, page]) => `- ${page.route} \u2192 ${path3}
8876
+ ${page.description}`).join("\n");
8877
+ const prompt = `Discover sub-features for all pages in this project.
8759
8878
 
8760
- ### 1. Structural validity
8761
- - All step verbs must be one of: click, type, scroll, assert, hover, drag, read, refresh
8762
- - assert: can ONLY verify what a human sees on screen (no URLs, network, console, localStorage)
8763
- - No code selectors (data-testid, aria-label, CSS classes, HTML element types)
8764
- - No login/authentication instructions - user is always already authenticated
8765
- - No internal/meta steps like "(Internal: simulate X)" or "(Note: this assumes Y)"
8766
- - No "or" in assertions - test data is deterministic
8767
- - Assertions must reference specific visible text, not vague descriptions ("success indicator", "results are displayed")
8879
+ Project root: ${input.projectRoot}
8768
8880
 
8769
- ### 2. Intent quality
8770
- Is the intent a specific, falsifiable behavioral claim?
8771
- FAIL: "When a user clicks the clock icon, the Wait modal should open" (just UI mechanics)
8772
- PASS: "Adding a 5-second wait step should insert a Wait action into the step list with the configured duration"
8773
-
8774
- ### 3. Mission alignment
8775
- Does the test's intent + steps actually verify the feature's core purpose?
8776
- FAIL if the intent just describes UI appearance when the feature is about functionality.
8777
-
8778
- When done reviewing, call finish with your structured evaluation.`
8779
- };
8780
- flowCompletenessRubric = {
8781
- name: "flow-completeness",
8782
- maxSteps: 12,
8783
- dimensions: ["actionCompletion", "mutationVerification"],
8784
- resultSchema: reviewResultSchema({
8785
- actionCompletion: dimensionResultSchema.describe(
8786
- "Does the test complete a core action and reach an OUTCOME? Not just opening a modal or clicking a tab."
8787
- ),
8788
- mutationVerification: dimensionResultSchema.describe(
8789
- "Does the test verify its mutation at the source of truth - not just a toast or inline indicator?"
8790
- )
8791
- }),
8792
- systemPrompt: `You are a flow completeness reviewer for E2E test plans. Each test will be executed by a VISUAL agent that sees the screen like a human user.
8793
-
8794
- Your job is to EVALUATE whether the test completes a meaningful action and verifies the result properly. You have tools to read the project's source code to understand what the feature actually does.
8795
-
8796
- ## Rubric dimensions
8797
-
8798
- ### 1. Action completion
8799
- Does the test complete a core action and reach an OUTCOME?
8800
- FAIL if the last meaningful step is just opening a modal, clicking a tab, or viewing a page.
8801
- PASS if the test creates, saves, deletes, configures, or otherwise produces a verifiable result.
8802
-
8803
- Read the source files to understand what the feature's complete workflow looks like. Does the test cover the full cycle?
8804
-
8805
- ### 2. Mutation verification
8806
- Does the test verify its mutation at the source of truth?
8807
- FAIL if the test ends at the point of action - checking a toast, a modal closing, or an inline success indicator.
8808
- PASS if the test navigates to where the mutation's effect should be visible and asserts it there.
8809
-
8810
- For example: after creating a record, does the test navigate back to the list and verify the record appears? After toggling a setting, does it refresh and verify the toggle persists?
8811
-
8812
- Read the source code to understand where the "source of truth" view is for each mutation.
8813
-
8814
- When done reviewing, call finish with your structured evaluation.`
8815
- };
8816
- uiTextRubric = {
8817
- name: "ui-text",
8818
- maxSteps: 20,
8819
- dimensions: ["uiTextAuthenticity"],
8820
- resultSchema: reviewResultSchema({
8821
- uiTextAuthenticity: dimensionResultSchema.describe(
8822
- "Do all quoted strings in steps reference text a human would actually see on screen? Not translation keys, config paths, component names, enum identifiers, or CSS classes."
8823
- )
8824
- }),
8825
- systemPrompt: `You are a UI text authenticity reviewer for E2E test plans. Your ONLY job is verifying that every piece of quoted text in the test steps matches what a human user would actually see on screen.
8826
-
8827
- You have tools to read source code. USE THEM AGGRESSIVELY. Do not guess - verify.
8828
-
8829
- ## Your process for EVERY quoted string in the test:
8830
-
8831
- 1. Grep for the exact string in the project source code
8832
- 2. Check WHERE it appears:
8833
- - If it appears as rendered text in the template/markup \u2192 PASS (it's real visible text)
8834
- - If it appears inside a translation/i18n function call \u2192 it's a TRANSLATION KEY, not visible text. FAIL.
8835
- - If it looks like a code identifier (camelCase, dot.notation, SCREAMING_CASE, PascalCase names) \u2192 FAIL
8836
- 3. If the string is a translation key, trace it to the actual rendered value:
8837
- - Find the translation/i18n file or dictionary
8838
- - Look up the key to find what text actually appears on screen
8839
- - Report both the key used and the correct visible text in your evidence
8840
-
8841
- ## Common patterns to catch:
8842
- - Translation keys used as labels: "aiBackoffice.tabPipeline" instead of "Pipeline"
8843
- - Dot-notation config paths: "settings.general.title"
8844
- - **Icon component names used as button descriptions**: if a quoted string in a test step refers to a button or clickable element, grep for that string in the source code. If it's imported as a component and renders an icon (SVG, image), it's a code identifier - NOT what the user sees. The test must describe the icon visually instead. To verify: find the icon's source file or infer from its name what it depicts, and check whether the test uses a visual description or the code name.
8845
- - Enum values: "QUOTE_REQUEST_RECEIVED", "IN_REVIEW"
8846
- - CSS class names or HTML attributes used as visible text
8847
-
8848
- ## Important:
8849
- - Check EVERY quoted string, not just suspicious ones
8850
- - A string existing in source code is NOT enough - it must be the RENDERED text
8851
- - When in doubt, read more files. You have 20 steps - use them all if needed.
8852
-
8853
- When done reviewing, call finish with your structured evaluation.`
8854
- };
8855
- dataAccuracyRubric = {
8856
- name: "data-accuracy",
8857
- maxSteps: 20,
8858
- dimensions: ["dataAccuracy"],
8859
- resultSchema: reviewResultSchema({
8860
- dataAccuracy: dimensionResultSchema.describe(
8861
- "Do the referenced UI elements (buttons, labels, fields, headings, toasts) actually exist in the source code for this page? Are default states correct? Does all test data (names, values, entities) come from the scenario data - NOT from other tests?"
8862
- )
8863
- }),
8864
- systemPrompt: `You are a data accuracy reviewer for E2E test plans. Your ONLY job is verifying that every UI element referenced in the test actually exists in the source code and behaves as the test expects.
8865
-
8866
- You have tools to read source code. USE THEM AGGRESSIVELY. Do not guess - verify.
8867
-
8868
- ## Your process:
8869
-
8870
- ### 1. Identify the page/component
8871
- Read the test's starting page and find the corresponding source file. Read it.
8872
-
8873
- ### 2. For each UI element referenced in the test:
8874
- - **Buttons**: grep for the button label. Verify it exists as a rendered string (not just a variable name).
8875
- - **Tab names**: find the tab component, read the tab definitions, verify the names match.
8876
- - **Field labels**: find the form component, verify field labels match.
8877
- - **Headings**: verify section/modal headings exist in the JSX.
8878
- - **Toast messages**: find where toasts are triggered, verify the message text.
8879
- - **Dropdown options**: find the select/dropdown component, verify the options.
8880
-
8881
- ### 3. Check default states:
8882
- - Toggle/switch default positions (is it on or off by default?)
8883
- - Default selected tabs (which tab is active on load?)
8884
- - Default form values (what are the initial values?)
8885
- - Conditional rendering (does the element actually show given the default state?)
8886
-
8887
- ### 4. Check preconditions and scenario data grounding:
8888
- - Does the test assume data exists that might not be seeded? (e.g., "click on the first item" when the list might be empty)
8889
- - CRITICAL: If the prompt includes scenario data, every data value the test references (entity names, folder names, app names, URLs, email addresses, etc.) MUST appear in that scenario data. If the test uses a value that only exists because another test created it, that is a FAIL - tests must be independent.
8890
- - Cross-reference every specific name/value in the test steps against the scenario data provided.
8891
-
8892
- ## Important:
8893
- - READ the actual component source files - don't just grep for strings
8894
- - Check conditional rendering - an element might exist in code but only show under certain conditions
8895
- - Verify the FLOW makes sense - after a page refresh, what state resets?
8896
- - Tests MUST be independent - they cannot depend on data created by other tests
8897
-
8898
- When done reviewing, call finish with your structured evaluation.`
8899
- };
8900
- ALL_RUBRICS = [
8901
- structuralIntentRubric,
8902
- flowCompletenessRubric,
8903
- uiTextRubric,
8904
- dataAccuracyRubric
8905
- ];
8906
- }
8907
- });
8908
-
8909
- // src/agents/05-test-generator/review-pass.ts
8910
- import { basename as basename3 } from "path";
8911
- import "ai";
8912
- import { tool as tool14 } from "ai";
8913
- async function runReviewPass(testContent, testPath, rubric, projectRoot, model, scenarioData) {
8914
- let result;
8915
- const agentLabel = `review:${rubric.name}:${basename3(testPath)}`;
8916
- const { onStepFinish } = buildDefaultStepLogger(agentLabel, rubric.maxSteps);
8917
- const finishTool = tool14({
8918
- description: "Submit your structured review. Every dimension must have evidence from your investigation.",
8919
- inputSchema: rubric.resultSchema,
8920
- execute: async (input) => {
8921
- result = reviewResultRecordSchema.parse(input);
8922
- }
8923
- });
8924
- const agentConfig = {
8925
- id: agentLabel,
8926
- systemPrompt: rubric.systemPrompt,
8927
- model,
8928
- maxSteps: rubric.maxSteps,
8929
- stepTimeoutMs: REVIEW_STEP_TIMEOUT_MS,
8930
- maxRetries: REVIEW_MAX_RETRIES,
8931
- tools: (_heartbeat) => ({
8932
- read_file: buildReadFileTool(projectRoot),
8933
- grep: buildGrepTool(projectRoot),
8934
- glob: buildGlobTool(projectRoot),
8935
- bash: buildBashTool(projectRoot),
8936
- finish: finishTool
8937
- }),
8938
- onStepFinish
8939
- };
8940
- const scenarioContext = scenarioData && rubric.name === "data-accuracy" ? `
8941
- ## Scenario data (the ONLY test data that exists in the database)
8942
- \`\`\`
8943
- ${scenarioData}
8944
- \`\`\`
8945
-
8946
- IMPORTANT: Every piece of data the test references (names, titles, URLs, folder names, etc.) MUST exist in the scenario data above. If the test uses a value that doesn't appear in scenarios, it FAILS the dataAccuracy dimension.
8947
- ` : "";
8948
- const prompt = `Review this E2E test plan:
8949
-
8950
- ## Test file: ${testPath}
8951
- \`\`\`
8952
- ${testContent}
8953
- \`\`\`
8954
- ${scenarioContext}
8955
- Evaluate EVERY dimension in your rubric: ${rubric.dimensions.join(", ")}
8956
-
8957
- For each one:
8958
- 1. Investigate using your tools (read source files, grep for strings referenced in the test)
8959
- 2. Provide specific evidence of what you found
8960
- 3. Pass or fail with a clear reason
8961
-
8962
- When done, call finish with your structured evaluation.`;
8963
- await runAgent(agentConfig, prompt, () => result);
8964
- return result ?? void 0;
8965
- }
8966
- var REVIEW_STEP_TIMEOUT_MS, REVIEW_MAX_RETRIES;
8967
- var init_review_pass = __esm({
8968
- "src/agents/05-test-generator/review-pass.ts"() {
8969
- "use strict";
8970
- init_esm_shims();
8971
- init_agent();
8972
- init_tools();
8973
- init_rubrics();
8974
- REVIEW_STEP_TIMEOUT_MS = 3e5;
8975
- REVIEW_MAX_RETRIES = 1;
8976
- }
8977
- });
8978
-
8979
- // src/agents/05-test-generator/review.ts
8980
- import { readFile as readFile18 } from "fs/promises";
8981
- import { join as join29, relative as relative7, basename as basename4 } from "path";
8982
- import "ai";
8983
- import { glob as glob5 } from "glob";
8984
- async function reviewSingleTest(testContent, testPath, projectRoot, model, scenarioData) {
8985
- const passes = await Promise.all(
8986
- ALL_RUBRICS.map(
8987
- (rubric) => runReviewPass(testContent, testPath, rubric, projectRoot, model, scenarioData).catch((err) => {
8988
- console.warn(
8989
- ` [review] ${rubric.name} error on ${basename4(testPath)}: ${err instanceof Error ? err.message : String(err)}`
8990
- );
8991
- return void 0;
8992
- })
8993
- )
8994
- );
8995
- const merged = {};
8996
- for (let i = 0; i < ALL_RUBRICS.length; i++) {
8997
- const rubric = ALL_RUBRICS[i];
8998
- const passResult = passes[i];
8999
- for (const dim of rubric.dimensions) {
9000
- if (passResult && dim in passResult) {
9001
- merged[dim] = passResult[dim];
9002
- } else {
9003
- merged[dim] = { pass: true, evidence: "Rubric pass did not return result - fail-open" };
9004
- }
9005
- }
9006
- }
9007
- return merged;
9008
- }
9009
- async function runConsolidatedReview(outputDir, projectRoot, model, deadline) {
9010
- const testsDir = join29(outputDir, TESTS_DIR);
9011
- const logger = createStepLogger("review", 5);
9012
- let scenarioData;
9013
- try {
9014
- scenarioData = await readFile18(join29(outputDir, "scenarios.md"), "utf-8");
9015
- } catch {
9016
- }
9017
- const testFiles = await glob5(join29(testsDir, TEST_FILE_GLOB));
9018
- const tests = [];
9019
- for (const testPath of testFiles) {
9020
- if (!isTestFile(testPath)) continue;
9021
- if (testPath.includes("/_invalid/")) continue;
9022
- const content = await readFile18(testPath, "utf-8");
9023
- const flowMatch = content.match(/^---\n[\s\S]*?flow:\s*["']?([^"'\n]+)["']?\s*\n[\s\S]*?---/m);
9024
- tests.push({
9025
- path: testPath,
9026
- relativePath: relative7(testsDir, testPath),
9027
- content,
9028
- flow: flowMatch?.[1]?.trim() ?? "unknown"
9029
- });
9030
- }
9031
- const totalAgents = tests.length * ALL_RUBRICS.length;
9032
- logger.log({
9033
- stepNumber: 1,
9034
- maxSteps: 2,
9035
- text: `Reviewing ${tests.length} tests \xD7 ${ALL_RUBRICS.length} rubrics = ${totalAgents} agents (${MAX_CONCURRENT_TESTS} tests concurrent)`,
9036
- toolCalls: [],
9037
- toolErrors: [],
9038
- writtenFiles: []
9039
- });
9040
- let passed = 0;
9041
- let failed = 0;
9042
- let ranOutOfTime = false;
9043
- const feedback = [];
9044
- for (let i = 0; i < tests.length; i += MAX_CONCURRENT_TESTS) {
9045
- if (Date.now() > deadline) {
9046
- ranOutOfTime = true;
9047
- const unreviewed = tests.length - i;
9048
- console.log(` [review] Out of time - ${unreviewed} of ${tests.length} tests left unreviewed`);
9049
- captureLog("warn", `Review out of time - ${unreviewed} of ${tests.length} tests left unreviewed`, {
9050
- source: "review",
9051
- unreviewed,
9052
- total: tests.length
9053
- });
9054
- break;
9055
- }
9056
- const batch = tests.slice(i, i + MAX_CONCURRENT_TESTS);
9057
- const results = await Promise.all(
9058
- batch.map(async (test) => {
9059
- const result = await reviewSingleTest(
9060
- test.content,
9061
- test.relativePath,
9062
- projectRoot,
9063
- model,
9064
- scenarioData
9065
- );
9066
- return { test, result };
9067
- })
9068
- );
9069
- for (const { test, result } of results) {
9070
- const failedDimensions = [];
9071
- for (const [key, dim] of Object.entries(result)) {
9072
- if (!dim.pass) failedDimensions.push(key);
9073
- }
9074
- if (failedDimensions.length === 0) {
9075
- passed++;
9076
- } else {
9077
- failed++;
9078
- feedback.push({
9079
- testPath: test.path,
9080
- relativePath: test.relativePath,
9081
- content: test.content,
9082
- flow: test.flow,
9083
- passed: false,
9084
- dimensions: result,
9085
- failedDimensions
9086
- });
9087
- }
9088
- }
9089
- console.log(
9090
- ` [review] Progress: ${Math.min(i + MAX_CONCURRENT_TESTS, tests.length)}/${tests.length} reviewed, ${passed} passed, ${failed} failed`
9091
- );
9092
- }
9093
- logger.log({
9094
- stepNumber: 2,
9095
- maxSteps: 2,
9096
- text: `Review complete: ${passed} passed, ${failed} failed`,
9097
- toolCalls: [],
9098
- toolErrors: [],
9099
- writtenFiles: []
9100
- });
9101
- logger.summary();
9102
- return { passed, failed, feedback, ranOutOfTime };
9103
- }
9104
- var MAX_CONCURRENT_TESTS;
9105
- var init_review = __esm({
9106
- "src/agents/05-test-generator/review.ts"() {
9107
- "use strict";
9108
- init_esm_shims();
9109
- init_display();
9110
- init_logs();
9111
- init_test_files();
9112
- init_review_pass();
9113
- init_rubrics();
9114
- MAX_CONCURRENT_TESTS = 4;
9115
- }
9116
- });
9117
-
9118
- // src/agents/00b-feature-discovery/index.ts
9119
- import { readFile as readFile19, writeFile as writeFile11 } from "fs/promises";
9120
- import { join as join30 } from "path";
9121
- import { tool as tool15 } from "ai";
9122
- import { z as z51 } from "zod";
9123
- async function saveFeatures(outputDir, features) {
9124
- const obj = Object.fromEntries(features);
9125
- await writeFile11(join30(outputDir, FEATURES_FILE), JSON.stringify(obj, null, 2), "utf-8");
9126
- }
9127
- async function loadFeatures(outputDir) {
9128
- try {
9129
- const raw = await readFile19(join30(outputDir, FEATURES_FILE), "utf-8");
9130
- const obj = JSON.parse(raw);
9131
- return new Map(Object.entries(obj));
9132
- } catch {
9133
- return void 0;
9134
- }
9135
- }
9136
- async function runFeatureDiscovery(input) {
9137
- const model = getModel(input.modelId);
9138
- const collector = new FeatureCollector();
9139
- let result;
9140
- const { logger, onStepFinish } = buildDefaultStepLogger("feature-discovery", 300);
9141
- const pagesDescription = Array.from(input.pages.entries()).map(([path3, page]) => `- ${page.route} \u2192 ${path3}
9142
- ${page.description}`).join("\n");
9143
- const prompt = `Discover sub-features for all pages in this project.
9144
-
9145
- Project root: ${input.projectRoot}
9146
-
9147
- ## Pages to analyze
9148
- ${pagesDescription}
8881
+ ## Pages to analyze
8882
+ ${pagesDescription}
9149
8883
 
9150
8884
  Process every page. Call add_feature for each sub-feature you discover. When done, call finish.`;
9151
8885
  const agentConfig = {
@@ -9157,10 +8891,10 @@ Process every page. Call add_feature for each sub-feature you discover. When don
9157
8891
  const tools = await buildCodebaseTools(model, input.projectRoot, input.outputDir, heartbeat);
9158
8892
  return {
9159
8893
  ...tools,
9160
- add_feature: tool15({
8894
+ add_feature: tool14({
9161
8895
  description: "Add a discovered sub-feature",
9162
8896
  inputSchema: Feature.extend({
9163
- id: z51.string().min(1).describe("Unique kebab-case ID (e.g. 'settings-notifications-tab')")
8897
+ id: z52.string().min(1).describe("Unique kebab-case ID (e.g. 'settings-notifications-tab')")
9164
8898
  }),
9165
8899
  execute: (featureInput) => {
9166
8900
  const { id, ...rest } = featureInput;
@@ -9172,19 +8906,19 @@ Process every page. Call add_feature for each sub-feature you discover. When don
9172
8906
  return `Feature "${id}" added (${collector.features.size} total)`;
9173
8907
  }
9174
8908
  }),
9175
- view_features: tool15({
8909
+ view_features: tool14({
9176
8910
  description: "View all discovered features so far",
9177
- inputSchema: z51.object({}),
8911
+ inputSchema: z52.object({}),
9178
8912
  execute: () => collector.viewFeatures()
9179
8913
  }),
9180
- view_pages: tool15({
8914
+ view_pages: tool14({
9181
8915
  description: "View the pages list to know what to analyze",
9182
- inputSchema: z51.object({}),
8916
+ inputSchema: z52.object({}),
9183
8917
  execute: () => pagesDescription
9184
8918
  }),
9185
- finish: tool15({
8919
+ finish: tool14({
9186
8920
  description: "Signal that feature discovery is complete",
9187
- inputSchema: z51.object({ summary: z51.string() }),
8921
+ inputSchema: z52.object({ summary: z52.string() }),
9188
8922
  execute: async (finishInput) => {
9189
8923
  result = {
9190
8924
  success: true,
@@ -9215,13 +8949,13 @@ var init_b_feature_discovery = __esm({
9215
8949
  init_model();
9216
8950
  init_tools();
9217
8951
  FEATURES_FILE = "features.json";
9218
- Feature = z51.object({
9219
- name: z51.string().min(1).describe("Human-readable name (e.g. 'Settings > Notifications Tab', 'Create Project Modal')"),
9220
- type: z51.enum(["tab", "modal", "form", "table", "wizard", "nested-route", "complex-component"]),
9221
- parentPagePath: z51.string().min(1).describe("The page path this feature belongs to (from the pages list)"),
9222
- sourceFiles: z51.array(z51.string()).min(1).describe("Relative paths to the source files for this sub-feature"),
9223
- interactiveElements: z51.number().int().min(0).describe("Count of interactive elements found (buttons, inputs, toggles, etc.)"),
9224
- description: z51.string().min(10).describe("What this sub-feature does")
8952
+ Feature = z52.object({
8953
+ name: z52.string().min(1).describe("Human-readable name (e.g. 'Settings > Notifications Tab', 'Create Project Modal')"),
8954
+ type: z52.enum(["tab", "modal", "form", "table", "wizard", "nested-route", "complex-component"]),
8955
+ parentPagePath: z52.string().min(1).describe("The page path this feature belongs to (from the pages list)"),
8956
+ sourceFiles: z52.array(z52.string()).min(1).describe("Relative paths to the source files for this sub-feature"),
8957
+ interactiveElements: z52.number().int().min(0).describe("Count of interactive elements found (buttons, inputs, toggles, etc.)"),
8958
+ description: z52.string().min(10).describe("What this sub-feature does")
9225
8959
  });
9226
8960
  FeatureCollector = class {
9227
8961
  features = /* @__PURE__ */ new Map();
@@ -9304,21 +9038,21 @@ Use kebab-case IDs that indicate the parent page and feature type:
9304
9038
  });
9305
9039
 
9306
9040
  // src/agents/05-test-generator/graph.ts
9307
- import { readFile as readFile20, writeFile as writeFile12 } from "fs/promises";
9308
- import { join as join31 } from "path";
9041
+ import { readFile as readFile18, writeFile as writeFile10 } from "fs/promises";
9042
+ import { join as join29 } from "path";
9309
9043
  function estimateExpectedTests(written, processed, totalNodes) {
9310
9044
  if (totalNodes <= 0) return written;
9311
9045
  const rate = written > 0 && processed >= LIVE_RATE_MIN_PROCESSED ? written / processed : TESTS_PER_NODE_PRIOR;
9312
9046
  return Math.max(written, Math.round(rate * totalNodes));
9313
9047
  }
9314
9048
  async function saveBfsState(outputDir, state) {
9315
- const path3 = join31(outputDir, state.stateFile);
9316
- await writeFile12(path3, JSON.stringify(state.serialize(), null, 2), "utf-8");
9049
+ const path3 = join29(outputDir, state.stateFile);
9050
+ await writeFile10(path3, JSON.stringify(state.serialize(), null, 2), "utf-8");
9317
9051
  }
9318
9052
  async function loadBfsState(outputDir) {
9319
- const path3 = join31(outputDir, BFS_STATE_FILE);
9053
+ const path3 = join29(outputDir, BFS_STATE_FILE);
9320
9054
  try {
9321
- const raw = await readFile20(path3, "utf-8");
9055
+ const raw = await readFile18(path3, "utf-8");
9322
9056
  return CoverageState.deserialize(JSON.parse(raw));
9323
9057
  } catch {
9324
9058
  return void 0;
@@ -9522,7 +9256,7 @@ If a node has no testable behavior (utility, redirect), call next_node to skip i
9522
9256
  - get_progress: Check how many nodes tested vs remaining
9523
9257
 
9524
9258
  ### Writing
9525
- - write_test: Write a test file with validated frontmatter
9259
+ - write_test: Write one test. You give it the test's parts (title, intent, steps, ...); it renders and validates the file.
9526
9260
  - create_folder: Create a folder under qa-tests/
9527
9261
 
9528
9262
  ### Research
@@ -9531,6 +9265,17 @@ If a node has no testable behavior (utility, redirect), call next_node to skip i
9531
9265
  ### Completion
9532
9266
  - finish: Signal you're done with a coverage report
9533
9267
 
9268
+ ## Two sources of truth - know which governs what (CRITICAL)
9269
+
9270
+ You have exactly two authorities, and they never overlap:
9271
+
9272
+ - **The application source code** governs STRUCTURE and BEHAVIOUR: what elements exist, their exact labels, default states, conditional rendering, and what each action does.
9273
+ - **The test data section of this prompt** governs DATA: every balance, name, email, count, status and date your assertions reference.
9274
+
9275
+ Never take a data value from the application's source. Files named seed, fixture, factory, mock, demo or sample hold the app's own local development data - Autonoma does NOT run them, and the values in them will NOT be on screen. A test asserting a balance found in a seed file fails on every run.
9276
+
9277
+ If the source and the test data disagree about a value, the test data wins. Always.
9278
+
9534
9279
  ## Source code grounding (CRITICAL)
9535
9280
 
9536
9281
  Before writing tests for any node, READ the source files for that page using read_file or spawn_researcher. Your tests MUST only reference elements you found in the actual source code.
@@ -9548,14 +9293,14 @@ If you can't find an element in the source, don't assert it. Read more files or
9548
9293
 
9549
9294
  Each feature has a "mission" - the ONE thing it must do correctly. Your tests MUST verify the mission. Before writing tests for any node:
9550
9295
 
9551
- 1. Find the mission for this feature from the Feature Missions section below
9296
+ 1. Read the \`mission\` field that next_node returned for this node. That is the mission. If it is absent, derive one from the core_flows in the Knowledge Base.
9552
9297
  2. Ask: "Does my planned test verify the mission, or just UI mechanics?"
9553
9298
  3. At least ONE test per feature must directly assert the mission outcome
9554
- 4. Core features (core: true) also have a coreReason explaining the blast radius of failure - allocate more depth to these
9299
+ 4. next_node also returns \`interactiveElements\` - the number of interactive elements discovery counted on this node. Use it to size your test count (see "Test depth"), and treat a high count as a signal to explore harder.
9555
9300
 
9556
9301
  Example: If the mission is "Show correct execution counts and success rates for the selected time range":
9557
9302
  - BAD: assert: the "Executions" tab heading is visible (UI mechanics - proves nothing about data)
9558
- - GOOD: assert: text "12" is visible in the executions count (verifies actual data from scenarios)
9303
+ - GOOD: assert: text "12" is visible in the executions count (verifies actual data from the test data)
9559
9304
 
9560
9305
  If you find yourself writing a test that only opens/closes UI elements without verifying the mission outcome, STOP and redesign the test.
9561
9306
 
@@ -9628,79 +9373,87 @@ Use NESTED folders to mirror the app hierarchy. Use create_folder with "/" separ
9628
9373
 
9629
9374
  Group related areas under parent folders.
9630
9375
 
9631
- ## Test file format
9632
-
9633
- Every test file must start with YAML frontmatter:
9634
-
9635
- \`\`\`yaml
9636
- ---
9637
- title: "Toggle recording stops active session"
9638
- description: "Verify toggling recording OFF stops the session"
9639
- intent: "When the recording toggle is ON (default), clicking it should stop recording and show a confirmation toast"
9640
- criticality: critical
9641
- scenario: standard
9642
- flow: "User Settings"
9643
- verification: "Refresh the Settings page, assert the recording toggle is in the OFF position"
9644
- ---
9645
- \`\`\`
9646
-
9647
- ### Frontmatter rules
9648
- - title: Short, descriptive test name
9649
- - description: One sentence explaining what the test verifies
9650
- - intent: A specific, falsifiable claim derived from the feature's MISSION - what the user does, what the feature produces, and why it matters. Focus on OUTCOMES, not UI mechanics.
9651
- GOOD: "Toggling recording from ON to OFF stops the active session and shows a confirmation toast"
9652
- BAD: "Click the recording toggle" (that's a step, not an intent)
9653
- BAD: "Verify the page displays correctly" (visibility check, not a behavior)
9654
- Derive from the mission: if the mission is "Generate valid config files", every test's intent must be about generating, previewing, or copying config - not about UI elements appearing.
9655
- - criticality: One of: critical, high, mid, low
9656
- - scenario: Which scenario this test uses (usually "standard")
9657
- - flow: Which feature/flow this belongs to (must match a feature from AUTONOMA.md)
9658
- - verification: (REQUIRED - write tool rejects without it) WHERE to navigate and WHAT to assert to prove the mutation worked. Every test performs a mutation - render-only tests should be folded into another test's flow. Must describe the source of truth, not a UI acknowledgment.
9376
+ ## Writing a test
9377
+
9378
+ You do NOT write markdown. Call write_test with the test's parts and the file is rendered
9379
+ for you - identically every time, so two tests that say the same thing look the same. The
9380
+ tool's schema rejects anything malformed, so read the field descriptions and answer them.
9381
+
9382
+ The fields, and what makes each one good:
9383
+
9384
+ - **title** - short and descriptive.
9385
+ - **description** - one sentence on what the test verifies.
9386
+ - **intent** - a specific, falsifiable claim derived from the node's MISSION: the expected
9387
+ INITIAL STATE, the ACTION the user takes, the EXPECTED OUTCOME, and why it matters. Write
9388
+ it before the steps - it is the north star the execution agent uses when the screen does
9389
+ not match a step. If the steps conflict with the intent, fix the STEPS.
9390
+ GOOD: "Toggling recording from ON to OFF stops the active session, and the setting persists across a refresh"
9391
+ BAD: "Click the recording toggle" (a step, not an intent)
9392
+ BAD: "Verify the page displays correctly" (a visibility check, not a behaviour)
9393
+ - **criticality** - critical, high, mid or low.
9394
+ - **scenario** - usually "standard".
9395
+ - **flow** - must match a flow from AUTONOMA.md.
9396
+ - **verification** - WHERE to navigate and WHAT to assert to prove the mutation worked. Name
9397
+ the source of truth. A toast, a confirmation dialog or an inline success indicator is an
9398
+ acknowledgment, not proof.
9659
9399
  GOOD: "Navigate to the test list, assert 'Login Flow' is visible in the table"
9660
9400
  GOOD: "Refresh the page, assert the toggle retained its OFF state"
9661
- BAD: "Assert toast 'Deleted' appears" (UI acknowledgment, not verification)
9662
-
9663
- ### Test body format
9664
-
9665
- After frontmatter:
9666
-
9667
- **Setup**: Which page the user starts on. Describe the clicks to reach the page. Read the app's layout/navigation code to determine the correct path. Look for sidebar, tab navigation, and route definitions. NEVER invent navigation paths - if you can't find how to reach a page, use spawn_researcher to investigate.
9668
-
9669
- NEVER write "Login as..." or "Log in" in Setup. The user is ALWAYS already authenticated. Setup only describes WHERE the user is, not authentication.
9670
- NEVER write tests that require navigating to invalid URLs, 404 pages, or error states.
9671
-
9672
- **Intent**: A specific, falsifiable claim derived from the feature's MISSION. States what the user does, what should happen, and WHY it matters. Write this BEFORE writing steps - it's the "north star" that the execution agent uses to adapt if steps don't match reality.
9673
-
9674
- The intent is NOT "what UI appears" - it's "what the feature DOES".
9675
-
9676
- Include in your intent:
9677
- - The expected INITIAL STATE of relevant elements
9678
- - The ACTION the user takes
9679
- - The EXPECTED OUTCOME (what the feature produces, not what UI appears)
9680
- - Why this matters to the user
9681
-
9682
- The intent is the source of truth. If your steps conflict with the intent, fix the STEPS.
9683
-
9684
- **Steps**: Numbered list using ONLY these actions (any other verb is INVALID):
9401
+ BAD: "Assert toast 'Deleted' appears"
9402
+ - **setup** - which page the user starts on and how they got there. Read the app's
9403
+ layout/navigation code for the real path; never invent one - use spawn_researcher if you
9404
+ cannot find it. NEVER authentication: no "Login as...", "Log in", "Sign in", no typing
9405
+ credentials to reach the page. Signing in happens outside the test. This holds even when
9406
+ the test's own subject is the login screen: start on it, do not authenticate your way to
9407
+ it. Never a route that 404s or errors.
9408
+ - **steps** - the action sequence. Each step is a verb, a description, and (for assert) a
9409
+ location.
9410
+ - **verificationSteps** - the steps that navigate to the source of truth and assert the
9411
+ mutation landed. This is what the \`verification\` field describes, as steps.
9412
+ - **expectedResult** - what is true when the test passes.
9413
+ - **notes** - OPTIONAL free text. Use it when something was genuinely unresolvable: a default
9414
+ state you could not determine, an element you could not find in the source, an assumption
9415
+ you had to make. It is recorded for a human and never written into the test file, so it
9416
+ cannot affect execution. Prefer flagging an uncertainty here over guessing silently.
9417
+
9418
+ ### Step verbs
9419
+
9420
+ Only these exist, and the schema rejects anything else:
9685
9421
  - click: Click a button, link, or element
9686
9422
  - type: Type text into an input field
9687
9423
  - scroll: Scroll to an element or position
9688
- - assert: Verify something VISUALLY visible on screen - text, headings, buttons, labels, images. MUST include location context when the same text could appear in multiple places. Use visual landmarks: "in the side panel", "in the modal", "in the table header", "below the form", "in the toast notification". CANNOT assert URLs, network requests, console logs, cookies, localStorage, or any non-visual state.
9424
+ - assert: Verify something VISUALLY visible - text, headings, buttons, labels, images. CANNOT
9425
+ assert URLs, network requests, console logs, cookies, localStorage, or any non-visual state.
9689
9426
  - hover: Hover over an element
9690
9427
  - drag: Drag an element
9691
- - read: Read text from an element into a variable
9692
9428
  - refresh: Refresh the page
9693
9429
 
9694
- BANNED actions (NEVER use these):
9695
- - wait: - INVALID. Do not write "wait:" steps.
9696
- - verify: - INVALID. Use "assert:" instead.
9697
- - navigate: - INVALID. Put navigation in Setup, not in steps.
9698
- - select: - INVALID. Use "click:" to select dropdown items.
9699
- - check: - INVALID. Use "click:" to check checkboxes, "assert:" to verify state.
9430
+ There is no wait (do not simulate delays), no verify (use assert), no navigate (navigation
9431
+ belongs in setup), no select (click the trigger, then click the option), no check (click it),
9432
+ and no read - nothing can capture a value for a later step to compare against.
9433
+
9434
+ Because nothing carries a value between steps, an assertion can never be RELATIVE to an
9435
+ earlier state. "the balance is $500 lower than before" is INVALID - nothing remembers
9436
+ "before". You know the starting value from the test data and what the action does, so assert
9437
+ the exact resulting value: description \`text "$11,950.50"\`, location \`in the Checking Account card\`.
9438
+
9439
+ ### The location field
9700
9440
 
9701
- **Verification**: Steps that usually navigate AWAY from the action screen to the source of truth and assert the mutation's effect. This section implements what the frontmatter 'verification' field describes.
9441
+ Every click, type and assert step needs a **location** saying where on screen its target is -
9442
+ "in the modal", "in the toast notification", "on the Sony WH-1000XM5 product card", "in the
9443
+ dashboard header", "as a page heading". It is a separate field so it cannot be forgotten in
9444
+ prose. scroll and refresh act on the page, so they do not need one.
9702
9445
 
9703
- **Expected Result**: What should be true when the test passes
9446
+ It matters as much for an action as for a check. \`click: the "Add Funds" button\` reads as
9447
+ unambiguous right up to the moment you notice the same label sits in the page header AND
9448
+ inside the modal that first click opened - and clicking the wrong one fails later, somewhere
9449
+ that looks unrelated.
9450
+
9451
+ Write it by default. Omit it only when you have READ the page source and confirmed the string
9452
+ renders in exactly one place. Ask: "could this text plausibly appear anywhere else on this
9453
+ screen?" A label that is also a column header, a button text repeated per row ("Buy", "Edit",
9454
+ "Delete" in a list), a heading matching a nav item, a value in both a summary and a detail
9455
+ row - all need a location. The cost of an unnecessary one is nothing; the cost of a missing
9456
+ one is an assertion that matches the wrong element.
9704
9457
 
9705
9458
  ### Interaction requirements (CRITICAL)
9706
9459
  - Every test MUST include at least 2 meaningful interactions (click, type, drag). Tests that ONLY assert visibility of elements are REJECTED.
@@ -9717,169 +9470,841 @@ BAD PATTERN (open/close cycle - tests nothing):
9717
9470
  4. assert: "Import component" is no longer visible
9718
9471
  This only proves the modal opens and closes. It does NOT test importing a component.
9719
9472
 
9720
- GOOD PATTERN (completing the action - tests the feature):
9721
- 1. click: the "Import component" button
9722
- 2. assert: "Import component" is visible in the modal header
9723
- 3. click: "Login Component" in the component list
9724
- 4. click: the "Import" button
9725
- 5. assert: "Login Component" is visible in the step list
9473
+ GOOD PATTERN (completing the action - tests the feature):
9474
+ 1. click: the "Import component" button
9475
+ 2. assert: "Import component" is visible in the modal header
9476
+ 3. click: "Login Component" in the component list
9477
+ 4. click: the "Import" button
9478
+ 5. assert: "Login Component" is visible in the step list
9479
+
9480
+ If your last assertion is about a modal being open, a heading being visible, or an element disappearing, you probably haven't tested anything. Ask: "What is the OUTCOME of this action?" Your test must prove the feature's MISSION is fulfilled.
9481
+
9482
+ ### Variation coverage
9483
+ When you find branching patterns in the source, decide whether each variant produces DIFFERENT BEHAVIOR (different code path, different UI, different output) or just passes different data through the SAME code path.
9484
+
9485
+ The question to ask: "In the source code, is there a conditional that renders different components or runs different logic based on this variant?" If yes \u2192 separate tests. If no \u2192 one test is enough.
9486
+
9487
+ Why this matters: tests that only vary in what string is passed through identical UI don't catch different bugs - they just inflate the test count.
9488
+
9489
+ GOOD variations (code branches differently):
9490
+ - Different providers rendering different templates/forms per provider
9491
+ - Different platform types showing different upload/input components
9492
+ - Status states rendering different visual components per state
9493
+
9494
+ BAD variations (same code path, different data):
9495
+ - Switching between different items that use the same component
9496
+ - Deleting different records through the same confirmation flow
9497
+ - Filtering by different values through the same dropdown
9498
+
9499
+ ### Default state awareness (CRITICAL)
9500
+ Read the source code to find default states for toggles, checkboxes, and dropdowns. BEFORE writing a step that interacts with a stateful element:
9501
+ - Check the source code for the element's initial value
9502
+ - If a toggle defaults to ON: clicking it turns it OFF (stops/disables)
9503
+ - ALWAYS state the expected state transition: "click: the 'Recording' toggle to switch it from ON to OFF" - not just "click: the 'Recording' toggle"
9504
+ - Assert the initial state BEFORE interacting
9505
+
9506
+ ### Test writing rules
9507
+ - Each test follows ONE deterministic path - no conditionals, no "e.g.", no "(mocked or ...)"
9508
+ - A step names ONE target, never a choice. "click the sign in or onboarding button" is invalid: those two controls go to different places, so nothing about the step is falsifiable - a run cannot say what it did, and a reader cannot say whether it was right. The same for an expectation: "assert the total or subtotal" passes either way and proves nothing. You have read the source and you have the test data, so say which one. For a button whose label you are unsure of, read the source again; for an icon button, describe the icon visually (see "Icon buttons").
9509
+ This is about the target, not the word: text INSIDE quotes is the application's own wording, so \`assert: text "Invalid email or password"\` is correct and expected.
9510
+ - Assertions must specify EXACT text, element, or visual state - never "or similar", never "e.g."
9511
+ - Be specific: use exact button text, field names, toast messages FROM THE CODE
9512
+ - One test per file
9513
+ - Never write meta-tests that "audit" scenario/fixture contents
9514
+ - Reference the test data when a flow needs real values, using its exact values
9515
+ - Do NOT write tests that verify the test infrastructure itself
9516
+ - Every step must be concrete and reproducible. "assert: text 'Deal Created' is visible in the toast notification" is GOOD as a STEP. "assert: success indicator appears" is BAD - it names no exact text and no location.
9517
+
9518
+ ### Visual-only rules (CRITICAL)
9519
+ Tests are executed by a VISUAL agent that sees the screen like a human. It can ONLY see what's rendered on screen.
9520
+
9521
+ The agent CANNOT access:
9522
+ - URLs or the browser address bar
9523
+ - Network requests or API calls
9524
+ - Console logs or errors
9525
+ - localStorage, cookies, or session data
9526
+ - HTML source, DOM structure, or element attributes
9527
+
9528
+ Therefore:
9529
+ - NEVER assert URLs: "assert: URL contains /creation" is INVALID. Instead assert visible page content.
9530
+ - NEVER assert network: "assert: API call was made" is INVALID
9531
+ - NEVER assert non-visual state: "assert: form state is valid" is INVALID
9532
+ - NEVER reference HTML elements: no "div", "span", "section", "input", "button" as element types
9533
+ - NEVER reference data attributes: no "data-testid", "data-cy", "data-test"
9534
+ - NEVER reference aria attributes: no "[aria-label]", "[role=dialog]"
9535
+ - NEVER reference CSS selectors: no "#id", ".class-name", "[attribute=value]"
9536
+ - NEVER use meta-steps: no "(Internal: ...)", "(Note: ...)", or parenthetical commentary
9537
+ - Instead, describe what the user SEES: button text, label text, placeholder text, heading text, visible icons, tab names
9538
+
9539
+ ### Test data references (CRITICAL)
9540
+ The "Test data" section of this prompt lists EXACTLY what Autonoma writes to the database before your tests run. Since WE control that data, assertions reference its exact values.
9541
+ - When a test needs to verify data is displayed, use the EXACT values from the test data (names, emails, titles, counts)
9542
+ - A value shown as \`<generated per run>\` differs every run - NEVER assert it literally. Assert a stable field on the same row instead.
9543
+ - Never take a data value from the app's own seed/fixture/mock files (see "Two sources of truth")
9544
+ - Do NOT assert on values that are auto-generated or vary at runtime (like database IDs); assert on the stable scenario values instead
9545
+ - NEVER use "Dynamic:", "{variableName}", "{{token}}", or "e.g." in steps or assertions. You have exact data - use it.
9546
+ - NEVER assume facts not stated in the scenario data.
9547
+
9548
+ ## Test generation ordering (for consistency)
9549
+ When generating tests for a node, follow this deterministic order:
9550
+ 1. First: CRUD operations for the primary entity (Create, Read/View, Update, Delete)
9551
+ 2. Second: State transitions (toggle, enable/disable, activate/deactivate)
9552
+ 3. Third: Validation (required fields, invalid input, boundary values)
9553
+ 4. Fourth: Navigation and linking (links to detail pages, breadcrumbs, back navigation)
9554
+ 5. Fifth: Edge cases (empty states, maximum values, permission boundaries)
9555
+
9556
+ ## Test depth - proportional to complexity (ENFORCED)
9557
+
9558
+ You determine feature complexity by READING THE SOURCE CODE, not by counting files. Before writing tests for a node:
9559
+ 1. Read the page file and explore all related source files
9560
+ 2. Count the interactive elements you find: forms, buttons, toggles, modals, tables, tabs
9561
+ 3. Write tests proportional to what you found - more interactive elements = more tests
9562
+
9563
+ A complex multi-step wizard with many forms needs 8-15 tests. A simple settings page with one toggle needs 2-3 tests. Use your judgment based on what you actually read in the source.
9564
+
9565
+ ## CRUD completeness (MANDATORY - zero tolerance)
9566
+
9567
+ If the source code for a feature supports Create, Read, Edit, and Delete for ANY entity, you MUST write tests for ALL of them. Missing even ONE CRUD operation is a critical failure.
9568
+
9569
+ **How to detect CRUD support:** Look for:
9570
+ - Create: "New", "Add", "Create" buttons; form submission handlers; modals with input fields
9571
+ - Read/View: Detail pages, list pages, tables, cards displaying entity data
9572
+ - Edit: "Edit", "Rename", "Update" buttons; pre-filled forms
9573
+ - Delete: "Delete", "Remove", "Trash" buttons; confirmation dialogs
9574
+
9575
+ If you find yourself writing only 1-2 tests for a CRUD page, STOP. Re-read the source. Find ALL the entity operations. Write tests for each.
9576
+
9577
+ ## Outcome verification - STRUCTURALLY ENFORCED
9578
+
9579
+ The write_test schema REJECTS any test without a \`verification\` field. This is not advisory - it is a hard gate.
9580
+
9581
+ What does NOT count as verification (these are UI acknowledgments, not proof):
9582
+ - Toast messages
9583
+ - Confirmation dialogs
9584
+ - Inline success indicators
9585
+ - The action button changing state
9586
+
9587
+ Verification destinations:
9588
+ - After CREATE \u2192 verify in list/table
9589
+ - After EDIT \u2192 verify changed field in detail/list view
9590
+ - After DELETE \u2192 verify absence in list, refresh, verify still absent
9591
+ - After TOGGLE \u2192 refresh, verify retained state
9592
+
9593
+ ## CRUD test templates (for any page with forms/CRUD):
9594
+ 1. **Create**: fill all fields, submit, verify the item appears
9595
+ 2. **Validation**: submit with empty required fields, verify error messages
9596
+ 3. **Edit**: modify existing item, save, verify change reflected
9597
+ 4. **Delete**: remove item, verify disappears, refresh, verify stays gone
9598
+ 5. **Boundary**: extremely long strings, special characters
9599
+
9600
+ **For pages with dropdowns/filters:**
9601
+ - You MUST click the dropdown trigger first, THEN click an option.
9602
+
9603
+ **For elements revealed by hover:**
9604
+ - Include a hover step before clicking elements that only appear on hover.
9605
+
9606
+ **After every action (create/edit/delete), verify the OUTCOME:**
9607
+ - BAD: click "Save" and move on
9608
+ - GOOD: click "Save" \u2192 assert the saved data appears in the list/detail view
9609
+
9610
+ ## Excluded routes
9611
+ - **Admin/backoffice pages**: routes under /admin/ are excluded from test generation. These require special auth, affect all users globally, and are not part of the standard user experience.
9612
+ - **Authentication itself** is never the subject of a test and never a step: signing the user in is handled outside the test (see Setup). If a queued node is a login/signup screen, test only what it does BESIDES authenticating - field validation, error messaging, navigation to password reset - and never a successful-login flow.
9613
+
9614
+ ## Test distribution guidelines
9615
+ - Core flows (from AUTONOMA.md where core: true): spend MOST of your time here. These features break \u2192 users leave.
9616
+ - Supporting flows: adequate coverage - happy path plus important variations.
9617
+ - Simple display/config pages: basic coverage.
9618
+
9619
+ ## Coverage dimensions
9620
+
9621
+ You track THREE kinds of coverage:
9622
+ 1. Route/file coverage: which routes explored, which source files visited
9623
+ 2. Entity coverage: which entity types and variations (enum values, states) appear in tests
9624
+ 3. Behavioral variant coverage: which code-branching variants have dedicated tests. If a switch/map dispatches to N different renderers, you should have tests for the most important variants.
9625
+
9626
+ When you finish, all dimensions are reported so the user knows what's covered and what gaps remain.`;
9627
+ }
9628
+ });
9629
+
9630
+ // src/agents/05-test-generator/recipe-context.ts
9631
+ import { readFile as readFile19 } from "fs/promises";
9632
+ import { join as join30 } from "path";
9633
+ import { z as z53 } from "zod";
9634
+ async function loadRecipeContext(outputDir) {
9635
+ let raw;
9636
+ try {
9637
+ raw = await readFile19(join30(outputDir, "recipe.json"), "utf-8");
9638
+ } catch (err) {
9639
+ debugLog("recipe.json not present; test generation falls back to scenarios.md alone", { err });
9640
+ return "";
9641
+ }
9642
+ const parsed = recipeFileSchema.safeParse(JSON.parse(raw));
9643
+ if (!parsed.success) {
9644
+ debugLog("recipe.json did not match the expected shape; skipping it", { issues: parsed.error.issues });
9645
+ return "";
9646
+ }
9647
+ const sections = parsed.data.recipes.map(renderRecipe);
9648
+ if (sections.length === 0) return "";
9649
+ return `
9650
+ ## Test data (from recipe.json - THE source of truth)
9651
+
9652
+ These are the exact rows Autonoma writes to the database before your tests run. Assert against
9653
+ these values, never against seed/fixture/mock data you find in the application's own source.
9654
+
9655
+ A value shown as \`<generated per run>\` is templated and differs on every run - never assert it
9656
+ literally. Assert a stable neighbouring field instead.
9657
+ ${sections.join("\n")}`;
9658
+ }
9659
+ function renderRecipe(recipe) {
9660
+ const lines = [`
9661
+ ### Scenario "${recipe.name}"`];
9662
+ if (recipe.description != null) lines.push(`
9663
+ ${recipe.description}`);
9664
+ for (const [model, rows] of Object.entries(recipe.create ?? {})) {
9665
+ lines.push(`
9666
+ **${model}** (${rows.length} row${rows.length === 1 ? "" : "s"})`);
9667
+ for (const row of rows) {
9668
+ const fields = Object.entries(row).filter(([key]) => !STRUCTURAL_KEYS.has(key)).map(([key, value]) => `${key}=${renderValue(value)}`);
9669
+ if (fields.length === 0) continue;
9670
+ const alias = row["_alias"];
9671
+ const label = typeof alias === "string" ? `${alias}: ` : "";
9672
+ lines.push(`- ${label}${fields.join(", ")}`);
9673
+ }
9674
+ }
9675
+ return lines.join("\n");
9676
+ }
9677
+ function renderValue(value) {
9678
+ if (typeof value === "string") {
9679
+ return TEMPLATE_TOKEN.test(value) ? "<generated per run>" : JSON.stringify(value);
9680
+ }
9681
+ if (typeof value === "object" && value !== null && "_ref" in value) {
9682
+ const ref = Reflect.get(value, "_ref");
9683
+ return typeof ref === "string" ? `(the row aliased ${ref})` : JSON.stringify(value);
9684
+ }
9685
+ return JSON.stringify(value);
9686
+ }
9687
+ var STRUCTURAL_KEYS, TEMPLATE_TOKEN, recipeRowSchema, recipeFileSchema;
9688
+ var init_recipe_context = __esm({
9689
+ "src/agents/05-test-generator/recipe-context.ts"() {
9690
+ "use strict";
9691
+ init_esm_shims();
9692
+ init_debug();
9693
+ STRUCTURAL_KEYS = /* @__PURE__ */ new Set(["_alias", "_ref"]);
9694
+ TEMPLATE_TOKEN = /\{\{\s*\w+\s*\}\}/;
9695
+ recipeRowSchema = z53.record(z53.string(), z53.unknown());
9696
+ recipeFileSchema = z53.object({
9697
+ recipes: z53.array(
9698
+ z53.object({
9699
+ name: z53.string(),
9700
+ description: z53.string().optional(),
9701
+ create: z53.record(z53.string(), z53.array(recipeRowSchema)).optional()
9702
+ })
9703
+ )
9704
+ });
9705
+ }
9706
+ });
9707
+
9708
+ // src/agents/05-test-generator/restore-deleted-test.ts
9709
+ import { access, mkdir as mkdir3, writeFile as writeFile11 } from "fs/promises";
9710
+ import { dirname as dirname4 } from "path";
9711
+ async function restoreDeletedTest(testPath, originalContent) {
9712
+ try {
9713
+ await access(testPath);
9714
+ return false;
9715
+ } catch (err) {
9716
+ debugLog("Test absent after its fix pass - restoring the pre-review content", { testPath, err });
9717
+ }
9718
+ try {
9719
+ await mkdir3(dirname4(testPath), { recursive: true });
9720
+ await writeFile11(testPath, originalContent, "utf-8");
9721
+ return true;
9722
+ } catch (err) {
9723
+ console.warn(` [fix] Could not restore ${testPath}: ${err instanceof Error ? err.message : String(err)}`);
9724
+ captureLog("error", "Failed to restore a reviewed test - it stays missing from the suite", {
9725
+ source: "test-generator",
9726
+ step: "review-fix",
9727
+ path: testPath
9728
+ });
9729
+ return false;
9730
+ }
9731
+ }
9732
+ var init_restore_deleted_test = __esm({
9733
+ "src/agents/05-test-generator/restore-deleted-test.ts"() {
9734
+ "use strict";
9735
+ init_esm_shims();
9736
+ init_debug();
9737
+ init_logs();
9738
+ }
9739
+ });
9740
+
9741
+ // src/core/pool.ts
9742
+ async function runPool(items, { limit, shouldContinue }, work) {
9743
+ const completed = [];
9744
+ const failed = [];
9745
+ const skipped = [];
9746
+ const inFlight2 = /* @__PURE__ */ new Set();
9747
+ let next = 0;
9748
+ const dispatch = (item, index) => {
9749
+ const promise = work(item, index).then((result) => {
9750
+ completed.push({ item, index, result });
9751
+ }).catch((error) => {
9752
+ failed.push({ item, index, error });
9753
+ }).finally(() => {
9754
+ inFlight2.delete(promise);
9755
+ });
9756
+ inFlight2.add(promise);
9757
+ };
9758
+ const cap = Math.max(1, limit);
9759
+ while (next < items.length) {
9760
+ if (shouldContinue != null && !shouldContinue()) {
9761
+ skipped.push(...items.slice(next));
9762
+ break;
9763
+ }
9764
+ while (inFlight2.size < cap && next < items.length) {
9765
+ dispatch(items[next], next);
9766
+ next++;
9767
+ }
9768
+ if (inFlight2.size > 0) await Promise.race(inFlight2);
9769
+ }
9770
+ await Promise.all(inFlight2);
9771
+ return { completed, failed, skipped };
9772
+ }
9773
+ var init_pool = __esm({
9774
+ "src/core/pool.ts"() {
9775
+ "use strict";
9776
+ init_esm_shims();
9777
+ }
9778
+ });
9779
+
9780
+ // src/agents/05-test-generator/rubrics.ts
9781
+ import { z as z54 } from "zod";
9782
+ function reviewResultSchema(shape) {
9783
+ return z54.object(shape);
9784
+ }
9785
+ var dimensionResultSchema, reviewResultRecordSchema, structuralIntentRubric, flowCompletenessRubric, uiTextRubric, dataAccuracyRubric, ALL_RUBRICS;
9786
+ var init_rubrics = __esm({
9787
+ "src/agents/05-test-generator/rubrics.ts"() {
9788
+ "use strict";
9789
+ init_esm_shims();
9790
+ dimensionResultSchema = z54.object({
9791
+ pass: z54.boolean(),
9792
+ evidence: z54.string().describe("What you checked and found - cite file paths, line content, or specific strings"),
9793
+ suggestion: z54.string().optional().describe("What the planner agent should fix, if failing")
9794
+ });
9795
+ reviewResultRecordSchema = z54.record(z54.string(), dimensionResultSchema);
9796
+ structuralIntentRubric = {
9797
+ name: "structural-intent",
9798
+ maxSteps: 8,
9799
+ dimensions: ["structuralValidity", "intentQuality", "missionAlignment"],
9800
+ resultSchema: reviewResultSchema({
9801
+ structuralValidity: dimensionResultSchema.describe(
9802
+ 'Are all step verbs valid (click/type/scroll/assert/hover/drag/refresh)? Are asserts visual-only (no URLs, network, console)? Does every assert name WHERE on screen it looks? No code selectors? No login steps? No "or" anywhere?'
9803
+ ),
9804
+ intentQuality: dimensionResultSchema.describe(
9805
+ "Is the intent a specific, falsifiable behavioral claim - not just 'verify X is visible'?"
9806
+ ),
9807
+ missionAlignment: dimensionResultSchema.describe(
9808
+ "Does the test's intent + steps verify the feature's core purpose? Not just UI appearance."
9809
+ )
9810
+ }),
9811
+ systemPrompt: `You are a structural reviewer for E2E test plans. Each test will be executed by a VISUAL agent that sees the screen like a human user - it cannot inspect code, network, URLs, or any non-visual state.
9812
+
9813
+ Your job is to EVALUATE tests against a rubric, NOT to rewrite them. You have tools to read source code if needed.
9726
9814
 
9727
- If your last assertion is about a modal being open, a heading being visible, or an element disappearing, you probably haven't tested anything. Ask: "What is the OUTCOME of this action?" Your test must prove the feature's MISSION is fulfilled.
9815
+ ## Rubric dimensions
9728
9816
 
9729
- ### Variation coverage
9730
- When you find branching patterns in the source, decide whether each variant produces DIFFERENT BEHAVIOR (different code path, different UI, different output) or just passes different data through the SAME code path.
9817
+ ### 1. Structural validity
9818
+ - All step verbs must be one of: click, type, scroll, assert, hover, drag, refresh
9819
+ - assert: can ONLY verify what a human sees on screen (no URLs, network, console, localStorage)
9820
+ - No code selectors (data-testid, aria-label, CSS classes, HTML element types)
9821
+ - The setup must not sign the user in - the run arrives authenticated, so "log in as X" or "after logging in" is a FAIL. Naming a login or sign-in PAGE as where the user starts is fine, and a test OF the login screen must be able to say so; judge what the sentence instructs, not whether it contains the word.
9822
+ - No internal/meta steps like "(Internal: simulate X)" or "(Note: this assumes Y)"
9823
+ - No step may offer a CHOICE of target: "click the sign in or onboarding button" names two controls that go to different places, and "assert the total or subtotal" passes either way, so neither is falsifiable. Judge the target, not the word - "or" inside a quoted string is the application's own message ("Invalid email or password") and is exactly what a test should assert.
9824
+ - Every click, type and assert names WHERE on screen its target is (in the modal, in the dashboard header, on the card, as a page heading, ...). A bare "click the Save button" or "assert text X is visible" fails unless that label provably appears exactly once on the screen at that point - the same label in a header and in the modal it opens is the common case
9825
+ - No assertion relative to an earlier state ("$500 less than before") - nothing remembers "before"
9826
+ - Assertions must reference specific visible text, not vague descriptions ("success indicator", "results are displayed")
9731
9827
 
9732
- The question to ask: "In the source code, is there a conditional that renders different components or runs different logic based on this variant?" If yes \u2192 separate tests. If no \u2192 one test is enough.
9828
+ ### 2. Intent quality
9829
+ Is the intent a specific, falsifiable behavioral claim?
9830
+ FAIL: "When a user clicks the clock icon, the Wait modal should open" (just UI mechanics)
9831
+ PASS: "Adding a 5-second wait step should insert a Wait action into the step list with the configured duration"
9733
9832
 
9734
- Why this matters: tests that only vary in what string is passed through identical UI don't catch different bugs - they just inflate the test count.
9833
+ ### 3. Mission alignment
9834
+ Does the test's intent + steps actually verify the feature's core purpose?
9835
+ FAIL if the intent just describes UI appearance when the feature is about functionality.
9735
9836
 
9736
- GOOD variations (code branches differently):
9737
- - Different providers rendering different templates/forms per provider
9738
- - Different platform types showing different upload/input components
9739
- - Status states rendering different visual components per state
9837
+ When done reviewing, call finish with your structured evaluation.`
9838
+ };
9839
+ flowCompletenessRubric = {
9840
+ name: "flow-completeness",
9841
+ maxSteps: 12,
9842
+ dimensions: ["actionCompletion", "mutationVerification"],
9843
+ resultSchema: reviewResultSchema({
9844
+ actionCompletion: dimensionResultSchema.describe(
9845
+ "Does the test complete a core action and reach an OUTCOME? Not just opening a modal or clicking a tab."
9846
+ ),
9847
+ mutationVerification: dimensionResultSchema.describe(
9848
+ "Does the test verify its mutation at the source of truth - not just a toast or inline indicator?"
9849
+ )
9850
+ }),
9851
+ systemPrompt: `You are a flow completeness reviewer for E2E test plans. Each test will be executed by a VISUAL agent that sees the screen like a human user.
9740
9852
 
9741
- BAD variations (same code path, different data):
9742
- - Switching between different items that use the same component
9743
- - Deleting different records through the same confirmation flow
9744
- - Filtering by different values through the same dropdown
9853
+ Your job is to EVALUATE whether the test completes a meaningful action and verifies the result properly. You have tools to read the project's source code to understand what the feature actually does.
9745
9854
 
9746
- ### Default state awareness (CRITICAL)
9747
- Read the source code to find default states for toggles, checkboxes, and dropdowns. BEFORE writing a step that interacts with a stateful element:
9748
- - Check the source code for the element's initial value
9749
- - If a toggle defaults to ON: clicking it turns it OFF (stops/disables)
9750
- - ALWAYS state the expected state transition: "click: the 'Recording' toggle to switch it from ON to OFF" - not just "click: the 'Recording' toggle"
9751
- - Assert the initial state BEFORE interacting
9855
+ ## Rubric dimensions
9752
9856
 
9753
- ### Assertion location context (CRITICAL)
9754
- EVERY assertion MUST include location context - where on the page the element appears. Never write a bare "assert: text X is visible". Always specify: in the modal, in the sidebar, in the table, in the header, in the toast notification, in the dropdown, on the card, in the form, in the dialog, in the panel, as a page heading, as a button label, etc.
9755
- - GOOD: assert: text "Run preview" is visible in the side panel
9756
- - BAD: assert: text "Status" is visible (WHERE? column header? form label? sidebar?)
9857
+ ### 1. Action completion
9858
+ Does the test complete a core action and reach an OUTCOME?
9859
+ FAIL if the last meaningful step is just opening a modal, clicking a tab, or viewing a page.
9860
+ PASS if the test creates, saves, deletes, configures, or otherwise produces a verifiable result.
9757
9861
 
9758
- ### Test writing rules
9759
- - Each test follows ONE deterministic path - no conditionals, no "e.g.", no "(mocked or ...)"
9760
- - "or" in click/type steps is OK for naming the same element (click: the "Edit" or "Pencil" icon) - these are visual synonyms
9761
- - "or" in assert steps is NEVER OK - since scenarios define the exact data, you always know what to expect
9762
- - Assertions must specify EXACT text, element, or visual state - never "or similar", never "e.g."
9763
- - Be specific: use exact button text, field names, toast messages FROM THE CODE
9764
- - One test per file
9765
- - Never write meta-tests that "audit" scenario/fixture contents
9766
- - Reference scenario data when needed for real user flows, using the exact values from the scenario
9767
- - Do NOT write tests that verify the test infrastructure itself
9768
- - Every step must be concrete and reproducible. "assert: text 'Deal Created' is visible in toast" is GOOD. "assert: success indicator appears" is BAD.
9862
+ Read the source files to understand what the feature's complete workflow looks like. Does the test cover the full cycle?
9769
9863
 
9770
- ### Visual-only rules (CRITICAL)
9771
- Tests are executed by a VISUAL agent that sees the screen like a human. It can ONLY see what's rendered on screen.
9864
+ ### 2. Mutation verification
9865
+ Does the test verify its mutation at the source of truth?
9866
+ FAIL if the test ends at the point of action - checking a toast, a modal closing, or an inline success indicator.
9867
+ PASS if the test navigates to where the mutation's effect should be visible and asserts it there.
9772
9868
 
9773
- The agent CANNOT access:
9774
- - URLs or the browser address bar
9775
- - Network requests or API calls
9776
- - Console logs or errors
9777
- - localStorage, cookies, or session data
9778
- - HTML source, DOM structure, or element attributes
9869
+ For example: after creating a record, does the test navigate back to the list and verify the record appears? After toggling a setting, does it refresh and verify the toggle persists?
9779
9870
 
9780
- Therefore:
9781
- - NEVER assert URLs: "assert: URL contains /creation" is INVALID. Instead assert visible page content.
9782
- - NEVER assert network: "assert: API call was made" is INVALID
9783
- - NEVER assert non-visual state: "assert: form state is valid" is INVALID
9784
- - NEVER reference HTML elements: no "div", "span", "section", "input", "button" as element types
9785
- - NEVER reference data attributes: no "data-testid", "data-cy", "data-test"
9786
- - NEVER reference aria attributes: no "[aria-label]", "[role=dialog]"
9787
- - NEVER reference CSS selectors: no "#id", ".class-name", "[attribute=value]"
9788
- - NEVER use meta-steps: no "(Internal: ...)", "(Note: ...)", or parenthetical commentary
9789
- - Instead, describe what the user SEES: button text, label text, placeholder text, heading text, visible icons, tab names
9871
+ Read the source code to understand where the "source of truth" view is for each mutation.
9790
9872
 
9791
- ### Scenario data references (CRITICAL)
9792
- The scenarios define EXACTLY what data exists in the database. Since WE control the test data, assertions should reference EXACT values from the scenario.
9793
- - When a test needs to verify data is displayed, use the EXACT values from the scenario (names, emails, titles, counts)
9794
- - Read scenarios.md carefully and use the exact entity names, counts, and field values in your assertions
9795
- - Do NOT assert on values that are auto-generated or vary at runtime (like database IDs); assert on the stable scenario values instead
9796
- - NEVER use "Dynamic:", "{variableName}", "{{token}}", or "e.g." in steps or assertions. You have exact data - use it.
9797
- - NEVER assume facts not stated in the scenario data.
9873
+ When done reviewing, call finish with your structured evaluation.`
9874
+ };
9875
+ uiTextRubric = {
9876
+ name: "ui-text",
9877
+ maxSteps: 20,
9878
+ dimensions: ["uiTextAuthenticity"],
9879
+ resultSchema: reviewResultSchema({
9880
+ uiTextAuthenticity: dimensionResultSchema.describe(
9881
+ "Do all quoted strings in steps reference text a human would actually see on screen? Not translation keys, config paths, component names, enum identifiers, or CSS classes."
9882
+ )
9883
+ }),
9884
+ systemPrompt: `You are a UI text authenticity reviewer for E2E test plans. Your ONLY job is verifying that every piece of quoted text in the test steps matches what a human user would actually see on screen.
9798
9885
 
9799
- ## Test generation ordering (for consistency)
9800
- When generating tests for a node, follow this deterministic order:
9801
- 1. First: CRUD operations for the primary entity (Create, Read/View, Update, Delete)
9802
- 2. Second: State transitions (toggle, enable/disable, activate/deactivate)
9803
- 3. Third: Validation (required fields, invalid input, boundary values)
9804
- 4. Fourth: Navigation and linking (links to detail pages, breadcrumbs, back navigation)
9805
- 5. Fifth: Edge cases (empty states, maximum values, permission boundaries)
9886
+ You have tools to read source code. USE THEM AGGRESSIVELY. Do not guess - verify.
9806
9887
 
9807
- ## Test depth - proportional to complexity (ENFORCED)
9888
+ ## Your process for EVERY quoted string in the test:
9808
9889
 
9809
- You determine feature complexity by READING THE SOURCE CODE, not by counting files. Before writing tests for a node:
9810
- 1. Read the page file and explore all related source files
9811
- 2. Count the interactive elements you find: forms, buttons, toggles, modals, tables, tabs
9812
- 3. Write tests proportional to what you found - more interactive elements = more tests
9890
+ 1. Grep for the exact string in the project source code
9891
+ 2. Check WHERE it appears:
9892
+ - If it appears as rendered text in the template/markup \u2192 PASS (it's real visible text)
9893
+ - If it appears inside a translation/i18n function call \u2192 it's a TRANSLATION KEY, not visible text. FAIL.
9894
+ - If it looks like a code identifier (camelCase, dot.notation, SCREAMING_CASE, PascalCase names) \u2192 FAIL
9895
+ 3. If the string is a translation key, trace it to the actual rendered value:
9896
+ - Find the translation/i18n file or dictionary
9897
+ - Look up the key to find what text actually appears on screen
9898
+ - Report both the key used and the correct visible text in your evidence
9813
9899
 
9814
- A complex multi-step wizard with many forms needs 8-15 tests. A simple settings page with one toggle needs 2-3 tests. Use your judgment based on what you actually read in the source.
9900
+ ## Common patterns to catch:
9901
+ - Translation keys used as labels: "aiBackoffice.tabPipeline" instead of "Pipeline"
9902
+ - Dot-notation config paths: "settings.general.title"
9903
+ - **Icon component names used as button descriptions**: if a quoted string in a test step refers to a button or clickable element, grep for that string in the source code. If it's imported as a component and renders an icon (SVG, image), it's a code identifier - NOT what the user sees. The test must describe the icon visually instead. To verify: find the icon's source file or infer from its name what it depicts, and check whether the test uses a visual description or the code name.
9904
+ - Enum values: "QUOTE_REQUEST_RECEIVED", "IN_REVIEW"
9905
+ - CSS class names or HTML attributes used as visible text
9815
9906
 
9816
- ## CRUD completeness (MANDATORY - zero tolerance)
9907
+ ## Important:
9908
+ - Check EVERY quoted string, not just suspicious ones
9909
+ - A string existing in source code is NOT enough - it must be the RENDERED text
9910
+ - When in doubt, read more files. You have 20 steps - use them all if needed.
9817
9911
 
9818
- If the source code for a feature supports Create, Read, Edit, and Delete for ANY entity, you MUST write tests for ALL of them. Missing even ONE CRUD operation is a critical failure.
9912
+ When done reviewing, call finish with your structured evaluation.`
9913
+ };
9914
+ dataAccuracyRubric = {
9915
+ name: "data-accuracy",
9916
+ maxSteps: 20,
9917
+ dimensions: ["dataAccuracy"],
9918
+ resultSchema: reviewResultSchema({
9919
+ dataAccuracy: dimensionResultSchema.describe(
9920
+ "Do the referenced UI elements (buttons, labels, fields, headings, toasts) actually exist in the source code for this page? Are default states correct? Does all test data (names, values, entities) come from the provided test data - NOT from the app's own seed/fixture/mock files, and NOT from other tests?"
9921
+ )
9922
+ }),
9923
+ systemPrompt: `You are a data accuracy reviewer for E2E test plans. Your ONLY job is verifying that every UI element referenced in the test actually exists in the source code and behaves as the test expects.
9819
9924
 
9820
- **How to detect CRUD support:** Look for:
9821
- - Create: "New", "Add", "Create" buttons; form submission handlers; modals with input fields
9822
- - Read/View: Detail pages, list pages, tables, cards displaying entity data
9823
- - Edit: "Edit", "Rename", "Update" buttons; pre-filled forms
9824
- - Delete: "Delete", "Remove", "Trash" buttons; confirmation dialogs
9925
+ You have tools to read source code. USE THEM AGGRESSIVELY. Do not guess - verify.
9825
9926
 
9826
- If you find yourself writing only 1-2 tests for a CRUD page, STOP. Re-read the source. Find ALL the entity operations. Write tests for each.
9927
+ ## Your process:
9827
9928
 
9828
- ## Outcome verification - STRUCTURALLY ENFORCED
9929
+ ### 1. Identify the page/component
9930
+ Read the test's starting page and find the corresponding source file. Read it.
9829
9931
 
9830
- The write_test tool REJECTS any test without a \`verification\` frontmatter field. This is not advisory - it's a hard gate.
9932
+ ### 2. For each UI element referenced in the test:
9933
+ - **Buttons**: grep for the button label. Verify it exists as a rendered string (not just a variable name).
9934
+ - **Tab names**: find the tab component, read the tab definitions, verify the names match.
9935
+ - **Field labels**: find the form component, verify field labels match.
9936
+ - **Headings**: verify section/modal headings exist in the JSX.
9937
+ - **Toast messages**: find where toasts are triggered, verify the message text.
9938
+ - **Dropdown options**: find the select/dropdown component, verify the options.
9831
9939
 
9832
- What does NOT count as verification (these are UI acknowledgments, not proof):
9833
- - Toast messages
9834
- - Confirmation dialogs
9835
- - Inline success indicators
9836
- - The action button changing state
9940
+ ### 3. Check default states:
9941
+ - Toggle/switch default positions (is it on or off by default?)
9942
+ - Default selected tabs (which tab is active on load?)
9943
+ - Default form values (what are the initial values?)
9944
+ - Conditional rendering (does the element actually show given the default state?)
9837
9945
 
9838
- Verification destinations:
9839
- - After CREATE \u2192 verify in list/table
9840
- - After EDIT \u2192 verify changed field in detail/list view
9841
- - After DELETE \u2192 verify absence in list, refresh, verify still absent
9842
- - After TOGGLE \u2192 refresh, verify retained state
9946
+ ### 4. Check preconditions and scenario data grounding:
9947
+ - Does the test assume data exists that might not be seeded? (e.g., "click on the first item" when the list might be empty)
9948
+ - CRITICAL: If the prompt includes test data, every data value the test references (entity names, folder names, app names, URLs, email addresses, etc.) MUST appear in that test data. If the test uses a value that only exists because another test created it, that is a FAIL - tests must be independent.
9949
+ - CRITICAL: a value that appears ONLY in the application's own seed/fixture/factory/mock/demo files is a FAIL, even though you found it in the source. Autonoma does not run those files; the provided test data is what will be on screen.
9950
+ - A value the test data marks as "<generated per run>" must never be asserted literally - that is a FAIL.
9951
+ - Cross-reference every specific name/value in the test steps against the test data provided.
9843
9952
 
9844
- ## CRUD test templates (for any page with forms/CRUD):
9845
- 1. **Create**: fill all fields, submit, verify the item appears
9846
- 2. **Validation**: submit with empty required fields, verify error messages
9847
- 3. **Edit**: modify existing item, save, verify change reflected
9848
- 4. **Delete**: remove item, verify disappears, refresh, verify stays gone
9849
- 5. **Boundary**: extremely long strings, special characters
9953
+ ## Important:
9954
+ - READ the actual component source files - don't just grep for strings
9955
+ - Check conditional rendering - an element might exist in code but only show under certain conditions
9956
+ - Verify the FLOW makes sense - after a page refresh, what state resets?
9957
+ - Tests MUST be independent - they cannot depend on data created by other tests
9850
9958
 
9851
- **For pages with dropdowns/filters:**
9852
- - You MUST click the dropdown trigger first, THEN click an option.
9959
+ When done reviewing, call finish with your structured evaluation.`
9960
+ };
9961
+ ALL_RUBRICS = [
9962
+ structuralIntentRubric,
9963
+ flowCompletenessRubric,
9964
+ uiTextRubric,
9965
+ dataAccuracyRubric
9966
+ ];
9967
+ }
9968
+ });
9853
9969
 
9854
- **For elements revealed by hover:**
9855
- - Include a hover step before clicking elements that only appear on hover.
9970
+ // src/agents/05-test-generator/review-pass.ts
9971
+ import { basename as basename3 } from "path";
9972
+ import "ai";
9973
+ import { tool as tool15 } from "ai";
9974
+ async function runReviewPass(testContent, testPath, rubric, projectRoot, model, scenarioData) {
9975
+ let result;
9976
+ const agentLabel = `review:${rubric.name}:${basename3(testPath)}`;
9977
+ const { onStepFinish } = buildDefaultStepLogger(agentLabel, rubric.maxSteps);
9978
+ const finishTool = tool15({
9979
+ description: "Submit your structured review. Every dimension must have evidence from your investigation.",
9980
+ inputSchema: rubric.resultSchema,
9981
+ execute: async (input) => {
9982
+ result = reviewResultRecordSchema.parse(input);
9983
+ }
9984
+ });
9985
+ const agentConfig = {
9986
+ id: agentLabel,
9987
+ systemPrompt: rubric.systemPrompt,
9988
+ model,
9989
+ maxSteps: rubric.maxSteps,
9990
+ stepTimeoutMs: REVIEW_STEP_TIMEOUT_MS,
9991
+ maxRetries: REVIEW_MAX_RETRIES,
9992
+ tools: (_heartbeat) => ({
9993
+ read_file: buildReadFileTool(projectRoot),
9994
+ grep: buildGrepTool(projectRoot),
9995
+ glob: buildGlobTool(projectRoot),
9996
+ bash: buildBashTool(projectRoot),
9997
+ finish: finishTool
9998
+ }),
9999
+ onStepFinish
10000
+ };
10001
+ const scenarioContext = scenarioData && rubric.name === "data-accuracy" ? `
10002
+ ## Scenario data (the ONLY test data that exists in the database)
10003
+ \`\`\`
10004
+ ${scenarioData}
10005
+ \`\`\`
9856
10006
 
9857
- **After every action (create/edit/delete), verify the OUTCOME:**
9858
- - BAD: click "Save" and move on
9859
- - GOOD: click "Save" \u2192 assert the saved data appears in the list/detail view
10007
+ IMPORTANT: Every piece of data the test references (names, titles, URLs, folder names, etc.) MUST exist in the scenario data above. If the test uses a value that doesn't appear in scenarios, it FAILS the dataAccuracy dimension.
10008
+ ` : "";
10009
+ const prompt = `Review this E2E test plan:
9860
10010
 
9861
- ## Excluded routes
9862
- - **Admin/backoffice pages**: routes under /admin/ are excluded from test generation. These require special auth, affect all users globally, and are not part of the standard user experience.
9863
- - **Auth/login pages**: never test authentication flows - the user is always already logged in.
10011
+ ## Test file: ${testPath}
10012
+ \`\`\`
10013
+ ${testContent}
10014
+ \`\`\`
10015
+ ${scenarioContext}
10016
+ Evaluate EVERY dimension in your rubric: ${rubric.dimensions.join(", ")}
9864
10017
 
9865
- ## Test distribution guidelines
9866
- - Core flows (from AUTONOMA.md where core: true): spend MOST of your time here. These features break \u2192 users leave.
9867
- - Supporting flows: adequate coverage - happy path plus important variations.
9868
- - Simple display/config pages: basic coverage.
10018
+ For each one:
10019
+ 1. Investigate using your tools (read source files, grep for strings referenced in the test)
10020
+ 2. Provide specific evidence of what you found
10021
+ 3. Pass or fail with a clear reason
9869
10022
 
9870
- ## Coverage dimensions
10023
+ When done, call finish with your structured evaluation.`;
10024
+ await runAgent(agentConfig, prompt, () => result);
10025
+ return result ?? void 0;
10026
+ }
10027
+ var REVIEW_STEP_TIMEOUT_MS, REVIEW_MAX_RETRIES;
10028
+ var init_review_pass = __esm({
10029
+ "src/agents/05-test-generator/review-pass.ts"() {
10030
+ "use strict";
10031
+ init_esm_shims();
10032
+ init_agent();
10033
+ init_tools();
10034
+ init_rubrics();
10035
+ REVIEW_STEP_TIMEOUT_MS = 3e5;
10036
+ REVIEW_MAX_RETRIES = 1;
10037
+ }
10038
+ });
9871
10039
 
9872
- You track THREE kinds of coverage:
9873
- 1. Route/file coverage: which routes explored, which source files visited
9874
- 2. Entity coverage: which entity types and variations (enum values, states) appear in tests
9875
- 3. Behavioral variant coverage: which code-branching variants have dedicated tests. If a switch/map dispatches to N different renderers, you should have tests for the most important variants.
10040
+ // src/agents/05-test-generator/review.ts
10041
+ import { readFile as readFile20 } from "fs/promises";
10042
+ import { join as join31, relative as relative7, basename as basename4 } from "path";
10043
+ import "ai";
10044
+ import { glob as glob5 } from "glob";
10045
+ function mergeRubricResults(reported) {
10046
+ const merged = {};
10047
+ for (const rubric of ALL_RUBRICS) {
10048
+ const result = reported.get(rubric.name);
10049
+ for (const dim of rubric.dimensions) {
10050
+ merged[dim] = result?.[dim] ?? {
10051
+ pass: true,
10052
+ evidence: "Rubric pass did not return result - fail-open"
10053
+ };
10054
+ }
10055
+ }
10056
+ return merged;
10057
+ }
10058
+ async function reviewOneTest({
10059
+ projectRoot,
10060
+ model,
10061
+ test,
10062
+ dataContract
10063
+ }) {
10064
+ const reported = /* @__PURE__ */ new Map();
10065
+ await Promise.all(
10066
+ ALL_RUBRICS.map(async (rubric) => {
10067
+ const result = await runReviewPass(
10068
+ test.content,
10069
+ test.relativePath,
10070
+ rubric,
10071
+ projectRoot,
10072
+ model,
10073
+ dataContract
10074
+ ).catch((err) => {
10075
+ console.warn(
10076
+ ` [review] ${rubric.name} error on ${basename4(test.relativePath)}: ${err instanceof Error ? err.message : String(err)}`
10077
+ );
10078
+ return void 0;
10079
+ });
10080
+ reported.set(rubric.name, result ?? {});
10081
+ })
10082
+ );
10083
+ const dimensions = mergeRubricResults(reported);
10084
+ return {
10085
+ relativePath: test.relativePath,
10086
+ content: test.content,
10087
+ dimensions,
10088
+ failedDimensions: Object.entries(dimensions).filter(([, dim]) => !dim.pass).map(([key]) => key)
10089
+ };
10090
+ }
10091
+ async function readDataContract(outputDir) {
10092
+ const recipe = await loadRecipeContext(outputDir);
10093
+ if (recipe !== "") return recipe;
10094
+ try {
10095
+ return await readFile20(join31(outputDir, "scenarios.md"), "utf-8");
10096
+ } catch (err) {
10097
+ debugLog("No data contract available for review", { err });
10098
+ return void 0;
10099
+ }
10100
+ }
10101
+ async function runConsolidatedReview(outputDir, projectRoot, model, deadline, settled = /* @__PURE__ */ new Set(), precomputed = []) {
10102
+ const testsDir = join31(outputDir, TESTS_DIR);
10103
+ const logger = createStepLogger("review", 5);
10104
+ const scenarioData = await readDataContract(outputDir);
10105
+ const testFiles = await glob5(join31(testsDir, TEST_FILE_GLOB));
10106
+ const tests = [];
10107
+ for (const testPath of testFiles) {
10108
+ if (!isTestFile(testPath)) continue;
10109
+ if (testPath.includes("/_invalid/")) continue;
10110
+ const relativePath = relative7(testsDir, testPath);
10111
+ if (settled.has(relativePath)) continue;
10112
+ const content = await readFile20(testPath, "utf-8");
10113
+ const flowMatch = content.match(/^---\n[\s\S]*?flow:\s*["']?([^"'\n]+)["']?\s*\n[\s\S]*?---/m);
10114
+ tests.push({ path: testPath, relativePath, content, flow: flowMatch?.[1]?.trim() ?? "unknown" });
10115
+ }
10116
+ const reusable = /* @__PURE__ */ new Map();
10117
+ for (const review of precomputed) {
10118
+ const current = tests.find((test) => test.relativePath === review.relativePath);
10119
+ if (current != null && current.content === review.content) reusable.set(review.relativePath, review);
10120
+ }
10121
+ const toReview = tests.filter((test) => !reusable.has(test.relativePath));
10122
+ const jobs = toReview.flatMap((test) => ALL_RUBRICS.map((rubric) => ({ test, rubric })));
10123
+ logger.log({
10124
+ stepNumber: 1,
10125
+ maxSteps: 2,
10126
+ text: `Reviewing ${toReview.length} tests \xD7 ${ALL_RUBRICS.length} rubrics = ${jobs.length} agents (${REVIEW_CONCURRENCY} concurrent)` + (reusable.size > 0 ? `; ${reusable.size} already reviewed during generation` : ""),
10127
+ toolCalls: [],
10128
+ toolErrors: [],
10129
+ writtenFiles: []
10130
+ });
10131
+ let passed = 0;
10132
+ let failed = 0;
10133
+ const feedback = [];
10134
+ const passedPaths = [];
10135
+ const byTest = /* @__PURE__ */ new Map();
10136
+ let reviewed = 0;
10137
+ const outcome = await runPool(
10138
+ jobs,
10139
+ { limit: REVIEW_CONCURRENCY, shouldContinue: () => Date.now() <= deadline },
10140
+ (job) => runReviewPass(job.test.content, job.test.relativePath, job.rubric, projectRoot, model, scenarioData)
10141
+ );
10142
+ for (const { item, error } of outcome.failed) {
10143
+ console.warn(
10144
+ ` [review] ${item.rubric.name} error on ${basename4(item.test.path)}: ${error instanceof Error ? error.message : String(error)}`
10145
+ );
10146
+ }
10147
+ const record = (relativePath, rubricName, result) => {
10148
+ const slot = byTest.get(relativePath) ?? /* @__PURE__ */ new Map();
10149
+ slot.set(rubricName, result);
10150
+ byTest.set(relativePath, slot);
10151
+ };
10152
+ for (const { item, result } of outcome.completed) {
10153
+ record(item.test.relativePath, item.rubric.name, result ?? {});
10154
+ }
10155
+ for (const { item } of outcome.failed) {
10156
+ record(item.test.relativePath, item.rubric.name, {});
10157
+ }
10158
+ const skippedTests = new Set(outcome.skipped.map((job) => job.test.relativePath));
10159
+ const ranOutOfTime = outcome.skipped.length > 0;
10160
+ for (const test of tests) {
10161
+ const alreadyReviewed = reusable.get(test.relativePath);
10162
+ const reported = byTest.get(test.relativePath);
10163
+ const complete = reported != null && reported.size === ALL_RUBRICS.length;
10164
+ if (alreadyReviewed == null && (!complete || skippedTests.has(test.relativePath))) continue;
10165
+ reviewed++;
10166
+ const merged = alreadyReviewed?.dimensions ?? mergeRubricResults(reported ?? /* @__PURE__ */ new Map());
10167
+ const failedDimensions = alreadyReviewed?.failedDimensions ?? Object.entries(merged).filter(([, dim]) => !dim.pass).map(([key]) => key);
10168
+ if (failedDimensions.length === 0) {
10169
+ passed++;
10170
+ passedPaths.push(test.relativePath);
10171
+ } else {
10172
+ failed++;
10173
+ feedback.push({
10174
+ testPath: test.path,
10175
+ relativePath: test.relativePath,
10176
+ content: test.content,
10177
+ flow: test.flow,
10178
+ passed: false,
10179
+ dimensions: merged,
10180
+ failedDimensions
10181
+ });
10182
+ }
10183
+ }
10184
+ if (ranOutOfTime) {
10185
+ const unreviewed = tests.length - reviewed;
10186
+ console.log(` [review] Out of time - ${unreviewed} of ${tests.length} tests left unreviewed`);
10187
+ captureLog("warn", `Review out of time - ${unreviewed} of ${tests.length} tests left unreviewed`, {
10188
+ source: "review",
10189
+ unreviewed,
10190
+ total: tests.length
10191
+ });
10192
+ }
10193
+ logger.log({
10194
+ stepNumber: 2,
10195
+ maxSteps: 2,
10196
+ text: `Review complete: ${passed} passed, ${failed} failed`,
10197
+ toolCalls: [],
10198
+ toolErrors: [],
10199
+ writtenFiles: []
10200
+ });
10201
+ logger.summary();
10202
+ return { passed, failed, feedback, ranOutOfTime, passedPaths };
10203
+ }
10204
+ var REVIEW_CONCURRENCY;
10205
+ var init_review = __esm({
10206
+ "src/agents/05-test-generator/review.ts"() {
10207
+ "use strict";
10208
+ init_esm_shims();
10209
+ init_debug();
10210
+ init_display();
10211
+ init_logs();
10212
+ init_pool();
10213
+ init_test_files();
10214
+ init_recipe_context();
10215
+ init_review_pass();
10216
+ init_rubrics();
10217
+ REVIEW_CONCURRENCY = 16;
10218
+ }
10219
+ });
9876
10220
 
9877
- When you finish, all dimensions are reported so the user knows what's covered and what gaps remain.`;
10221
+ // src/agents/05-test-generator/review-pipeline.ts
10222
+ var PIPELINE_CONCURRENCY, ReviewPipeline;
10223
+ var init_review_pipeline = __esm({
10224
+ "src/agents/05-test-generator/review-pipeline.ts"() {
10225
+ "use strict";
10226
+ init_esm_shims();
10227
+ init_logs();
10228
+ init_review();
10229
+ PIPELINE_CONCURRENCY = 4;
10230
+ ReviewPipeline = class {
10231
+ constructor(outputDir, projectRoot, model, deadline) {
10232
+ this.outputDir = outputDir;
10233
+ this.projectRoot = projectRoot;
10234
+ this.model = model;
10235
+ this.deadline = deadline;
10236
+ }
10237
+ inFlight = /* @__PURE__ */ new Set();
10238
+ done = [];
10239
+ queue = [];
10240
+ seen = /* @__PURE__ */ new Set();
10241
+ /** The contract is identical for every test, so it is read and rendered once. */
10242
+ dataContract;
10243
+ closed = false;
10244
+ /**
10245
+ * Hand a freshly written test over for review. Returns immediately - the
10246
+ * generator must never block on a reviewer.
10247
+ */
10248
+ submit(test) {
10249
+ if (this.closed || this.seen.has(test.relativePath)) return;
10250
+ this.seen.add(test.relativePath);
10251
+ this.queue.push(test);
10252
+ this.pump();
10253
+ }
10254
+ pump() {
10255
+ while (this.inFlight.size < PIPELINE_CONCURRENCY && this.queue.length > 0) {
10256
+ if (Date.now() > this.deadline) {
10257
+ this.queue.length = 0;
10258
+ return;
10259
+ }
10260
+ const test = this.queue.shift();
10261
+ this.dataContract ??= readDataContract(this.outputDir);
10262
+ const promise = this.dataContract.then(
10263
+ (dataContract) => reviewOneTest({ projectRoot: this.projectRoot, model: this.model, test, dataContract })
10264
+ ).then((review) => {
10265
+ this.done.push(review);
10266
+ }).catch((err) => {
10267
+ captureLog("warn", `Pipelined review failed; the test will be reviewed in the fix cycles`, {
10268
+ source: "review-pipeline",
10269
+ path: test.relativePath,
10270
+ error: err instanceof Error ? err.message : String(err)
10271
+ });
10272
+ }).finally(() => {
10273
+ this.inFlight.delete(promise);
10274
+ this.pump();
10275
+ });
10276
+ this.inFlight.add(promise);
10277
+ }
10278
+ }
10279
+ /**
10280
+ * Stop accepting work and wait for what is running. Everything queued but not
10281
+ * started is dropped - the cycles that follow will pick those tests up, and
10282
+ * they are cheaper to review there than to hold generation open for.
10283
+ */
10284
+ async drain() {
10285
+ this.closed = true;
10286
+ this.queue.length = 0;
10287
+ await Promise.all(this.inFlight);
10288
+ return [...this.done];
10289
+ }
10290
+ };
9878
10291
  }
9879
10292
  });
9880
10293
 
9881
10294
  // src/agents/05-test-generator/validation.ts
9882
10295
  import matter4 from "gray-matter";
10296
+ function isValidVerb(verb) {
10297
+ return VALID_VERBS.has(verb);
10298
+ }
10299
+ function isInteractionVerb(verb) {
10300
+ return INTERACTION_VERBS.has(verb);
10301
+ }
10302
+ function requiresLocation(verb) {
10303
+ return LOCATED_VERBS.has(verb);
10304
+ }
10305
+ function parseStepVerbs(content) {
10306
+ return [...content.matchAll(STEP_LINE_PATTERN)].map((m) => m[1]);
10307
+ }
9883
10308
  function validateTestContent(content) {
9884
10309
  const errors = [];
9885
10310
  if (!/^---\n[\s\S]*?\n---/.test(content)) {
@@ -9899,17 +10324,13 @@ function validateTestContent(content) {
9899
10324
  if (!/\*\*Intent\*\*:/.test(content)) {
9900
10325
  errors.push("Missing **Intent**: section");
9901
10326
  }
9902
- const stepMatches = content.match(/^\d+\.\s+(click|type|scroll|assert|hover|drag|read|refresh):/gm) || [];
9903
- const interactions = stepMatches.filter((s) => /^\d+\.\s+(click|type|drag):/.test(s));
9904
- if (interactions.length < 2) {
9905
- errors.push(`Only ${interactions.length} interaction(s) (minimum 2)`);
10327
+ const verbs = parseStepVerbs(content);
10328
+ const interactions = verbs.filter(isInteractionVerb);
10329
+ if (interactions.length < MIN_INTERACTIONS) {
10330
+ errors.push(`Only ${interactions.length} interaction(s) (minimum ${MIN_INTERACTIONS})`);
9906
10331
  }
9907
- const allSteps = content.match(/^\d+\.\s+(\w+):/gm) || [];
9908
- for (const step of allSteps) {
9909
- const verbMatch = step.match(/^\d+\.\s+(\w+):/);
9910
- if (verbMatch && !VALID_VERBS.has(verbMatch[1])) {
9911
- errors.push(`Invalid verb: "${verbMatch[1]}"`);
9912
- }
10332
+ for (const verb of verbs) {
10333
+ if (!isValidVerb(verb)) errors.push(`Invalid verb: "${verb}"`);
9913
10334
  }
9914
10335
  const bodyStart = content.indexOf("---", 3);
9915
10336
  const body = bodyStart > -1 ? content.slice(bodyStart + 3) : content;
@@ -9919,90 +10340,136 @@ function validateTestContent(content) {
9919
10340
  }
9920
10341
  return { valid: errors.length === 0, errors };
9921
10342
  }
9922
- var VALID_VERBS, CRITICALITY_LEVELS;
10343
+ var STEP_VERBS, VALID_VERBS, INTERACTION_VERBS, LOCATED_VERBS, STEP_LINE_PATTERN, MIN_INTERACTIONS, CRITICALITY_LEVELS;
9923
10344
  var init_validation = __esm({
9924
10345
  "src/agents/05-test-generator/validation.ts"() {
9925
10346
  "use strict";
9926
10347
  init_esm_shims();
9927
- VALID_VERBS = /* @__PURE__ */ new Set(["click", "type", "scroll", "assert", "hover", "drag", "read", "refresh"]);
10348
+ STEP_VERBS = ["click", "type", "scroll", "assert", "hover", "drag", "refresh"];
10349
+ VALID_VERBS = new Set(STEP_VERBS);
10350
+ INTERACTION_VERBS = /* @__PURE__ */ new Set(["click", "type", "drag"]);
10351
+ LOCATED_VERBS = /* @__PURE__ */ new Set(["assert", "click", "type"]);
10352
+ STEP_LINE_PATTERN = /^\d+\.\s+(\w+):/gm;
10353
+ MIN_INTERACTIONS = 2;
9928
10354
  CRITICALITY_LEVELS = ["critical", "high", "mid", "low"];
9929
10355
  }
9930
10356
  });
9931
10357
 
9932
- // src/agents/05-test-generator/tools.ts
9933
- import { mkdir as mkdir4, writeFile as writeFile13 } from "fs/promises";
9934
- import { dirname as dirname5, join as join32 } from "path";
9935
- import { hasToolCall as hasToolCall3, stepCountIs as stepCountIs3, tool as tool16, ToolLoopAgent as ToolLoopAgent3 } from "ai";
9936
- import matter5 from "gray-matter";
9937
- import { z as z52 } from "zod";
9938
- function findForbiddenPlaceholder(stepsSection) {
9939
- const placeholderPatterns = [
9940
- { pattern: /Dynamic:\s/gi, name: '"Dynamic:" placeholder' },
9941
- { pattern: /\{\{[a-zA-Z0-9_]+\}\}/g, name: "{{token}} placeholder" },
9942
- { pattern: /(?<!\{)\{[a-z][a-zA-Z]*\}(?!\})/g, name: "bare {variable}" },
9943
- { pattern: /\(e\.g\./gi, name: '"(e.g." example' },
9944
- { pattern: /(?:^|\s)e\.g\.,?\s/gim, name: '"e.g." example' }
10358
+ // src/agents/05-test-generator/test-spec.ts
10359
+ import { z as z55 } from "zod";
10360
+ function hasPlaceholder(text2) {
10361
+ if (text2 == null) return false;
10362
+ return PLACEHOLDER_PATTERNS.some(({ pattern }) => pattern.test(text2));
10363
+ }
10364
+ function renderTestMarkdown(spec) {
10365
+ const frontmatter = [
10366
+ "---",
10367
+ `title: ${JSON.stringify(spec.title)}`,
10368
+ `description: ${JSON.stringify(spec.description)}`,
10369
+ `intent: ${JSON.stringify(spec.intent)}`,
10370
+ `criticality: ${spec.criticality}`,
10371
+ `scenario: ${spec.scenario}`,
10372
+ `flow: ${JSON.stringify(spec.flow)}`,
10373
+ `verification: ${JSON.stringify(spec.verification)}`,
10374
+ "---"
10375
+ ].join("\n");
10376
+ const sections = [
10377
+ frontmatter,
10378
+ `**Setup**: ${spec.setup}`,
10379
+ `**Intent**: ${spec.intent}`,
10380
+ `**Steps**:
10381
+ ${renderSteps(spec.steps)}`
9945
10382
  ];
9946
- for (const { pattern, name } of placeholderPatterns) {
9947
- const matches = stepsSection.match(pattern);
9948
- if (matches && matches.length > 0) {
9949
- return { name, match: matches[0] };
9950
- }
10383
+ if (spec.verificationSteps.length > 0) {
10384
+ sections.push(`**Verification**:
10385
+ ${renderSteps(spec.verificationSteps)}`);
9951
10386
  }
9952
- return void 0;
10387
+ sections.push(`**Expected Result**: ${spec.expectedResult}`);
10388
+ return `${sections.join("\n\n")}
10389
+ `;
9953
10390
  }
9954
- function buildWriteTestTool(state, outputDir) {
10391
+ function renderSteps(steps) {
10392
+ return steps.map((step, index) => {
10393
+ const location = step.location?.trim();
10394
+ const text2 = location != null && location !== "" ? `${step.description} ${location}` : step.description;
10395
+ return `${index + 1}. ${step.verb}: ${text2}`;
10396
+ }).join("\n");
10397
+ }
10398
+ var PLACEHOLDER_PATTERNS, stepSchema, testSpecSchema;
10399
+ var init_test_spec = __esm({
10400
+ "src/agents/05-test-generator/test-spec.ts"() {
10401
+ "use strict";
10402
+ init_esm_shims();
10403
+ init_validation();
10404
+ PLACEHOLDER_PATTERNS = [
10405
+ { pattern: /Dynamic:\s/i, name: '"Dynamic:" placeholder' },
10406
+ { pattern: /\{\{[a-zA-Z0-9_]+\}\}/, name: "{{token}} placeholder" },
10407
+ { pattern: /(?<!\{)\{[a-z][a-zA-Z]*\}(?!\})/, name: "bare {variable}" },
10408
+ { pattern: /\be\.g\./i, name: '"e.g." example' }
10409
+ ];
10410
+ stepSchema = z55.object({
10411
+ verb: z55.enum(STEP_VERBS).describe("The action. Only these verbs exist."),
10412
+ description: z55.string().min(1).describe(
10413
+ `What to do or check, naming the exact visible text. For assert, the thing expected on screen - e.g. 'text "Transfer Successful"'.`
10414
+ ),
10415
+ location: z55.string().optional().describe(
10416
+ 'WHERE on screen the target is - "in the modal", "in the toast notification", "on the Sony WH-1000XM5 product card", "in the dashboard header", "as a page heading". REQUIRED for click, type and assert: the same label routinely appears more than once (a header button and the modal button it opens, "Buy" on every card in a list), and acting on the wrong one fails confusingly. Not needed for scroll or refresh, which act on the page.'
10417
+ )
10418
+ }).refine((step) => !requiresLocation(step.verb) || (step.location?.trim() ?? "") !== "", {
10419
+ message: 'This verb requires a "location" - say WHERE on screen its target is. The same label often appears more than once (a header button and the modal button it opens), so naming the element is not enough on its own.',
10420
+ path: ["location"]
10421
+ }).refine((step) => !hasPlaceholder(step.description) && !hasPlaceholder(step.location), {
10422
+ message: "steps carry no variables - use the exact value from the test data, not a placeholder or example",
10423
+ path: ["description"]
10424
+ });
10425
+ testSpecSchema = z55.object({
10426
+ title: z55.string().min(1).describe("Short, descriptive test name."),
10427
+ description: z55.string().min(1).describe("One sentence explaining what the test verifies."),
10428
+ intent: z55.string().min(30).describe(
10429
+ "A specific, falsifiable claim derived from the node's mission: what the user does, what the feature produces, why it matters. Not the steps, not 'the page displays correctly'."
10430
+ ),
10431
+ criticality: z55.enum(CRITICALITY_LEVELS),
10432
+ scenario: z55.string().min(1).describe('Which scenario this test uses (usually "standard").'),
10433
+ flow: z55.string().min(1).describe("Which feature/flow this belongs to (must match a flow from AUTONOMA.md)."),
10434
+ verification: z55.string().min(20).describe(
10435
+ "WHERE to navigate and WHAT to assert to prove the mutation worked. Must name the source of truth - a toast, a confirmation dialog or an inline success indicator is an acknowledgment, not proof."
10436
+ ),
10437
+ setup: z55.string().min(1).describe(
10438
+ "Which page the user starts on and how they got there. Never authentication - the user is already signed in."
10439
+ ),
10440
+ steps: z55.array(stepSchema).min(1).describe("The action sequence, in order."),
10441
+ verificationSteps: z55.array(stepSchema).describe(
10442
+ "Steps that navigate to the source of truth and assert the mutation landed. Implements the `verification` field above."
10443
+ ),
10444
+ expectedResult: z55.string().min(1).describe("What should be true when the test passes."),
10445
+ notes: z55.string().optional().describe(
10446
+ "Optional free text for anything you could not resolve or want to flag - an ambiguous default state, a element you could not find in the source, an assumption you had to make. Recorded for humans and NEVER written into the test file, so it will not affect execution. Use it instead of guessing silently."
10447
+ )
10448
+ }).refine((spec) => spec.steps.filter((step) => isInteractionVerb(step.verb)).length >= MIN_INTERACTIONS, {
10449
+ message: `A test needs at least ${MIN_INTERACTIONS} real interactions (click/type/drag) - a visibility-only test verifies nothing`,
10450
+ path: ["steps"]
10451
+ });
10452
+ }
10453
+ });
10454
+
10455
+ // src/agents/05-test-generator/tools.ts
10456
+ import { mkdir as mkdir4, writeFile as writeFile12 } from "fs/promises";
10457
+ import { dirname as dirname5, join as join32 } from "path";
10458
+ import { hasToolCall as hasToolCall3, stepCountIs as stepCountIs3, tool as tool16, ToolLoopAgent as ToolLoopAgent3 } from "ai";
10459
+ import { z as z56 } from "zod";
10460
+ function buildWriteTestTool(state, outputDir, onWritten) {
9955
10461
  return tool16({
9956
- description: "Write a test file to qa-tests/{folder}/{filename}.md. Validates frontmatter before writing. Returns error if frontmatter is invalid.",
9957
- inputSchema: z52.object({
9958
- folder: z52.string().describe("Subfolder name under qa-tests/"),
9959
- filename: z52.string().describe(`File name ending in ${TEST_FILE_EXT} (e.g. login-valid-credentials${TEST_FILE_EXT})`),
9960
- content: z52.string().describe("Full file content including YAML frontmatter"),
9961
- nodeId: z52.string().describe(
10462
+ description: "Write one test to qa-tests/{folder}/{filename}.md. You supply the test's parts; the file is rendered for you, so do not write markdown or frontmatter yourself.",
10463
+ inputSchema: z56.object({
10464
+ folder: z56.string().describe("Subfolder name under qa-tests/"),
10465
+ filename: z56.string().describe(`File name ending in ${TEST_FILE_EXT} (e.g. login-valid-credentials${TEST_FILE_EXT})`),
10466
+ nodeId: z56.string().describe(
9962
10467
  "The id next_node returned for this feature, copied verbatim. Not a re-slugged version of it, not a folder path, not the test filename."
9963
- )
10468
+ ),
10469
+ test: testSpecSchema
9964
10470
  }),
9965
10471
  execute: async (input) => {
9966
- const frontmatter = extractFrontmatter(input.content);
9967
- if (!frontmatter) {
9968
- return { error: "File must start with YAML frontmatter (--- delimiters)" };
9969
- }
9970
- const parsed = testFrontmatterSchema.safeParse(frontmatter);
9971
- if (!parsed.success) {
9972
- return {
9973
- error: `Invalid frontmatter: ${parsed.error.issues.map((i) => i.message).join(", ")}`
9974
- };
9975
- }
9976
- if (!/\*\*Intent\*\*:/.test(input.content)) {
9977
- return {
9978
- error: "Test must include an **Intent**: section between Setup and Steps describing what behavior is being tested"
9979
- };
9980
- }
9981
- const allSteps = input.content.match(/^\d+\.\s+(\w+):/gm) || [];
9982
- for (const step of allSteps) {
9983
- const verbMatch = step.match(/^\d+\.\s+(\w+):/);
9984
- if (verbMatch && !VALID_VERBS.has(verbMatch[1])) {
9985
- return {
9986
- error: `Invalid step verb "${verbMatch[1]}". Only valid verbs are: ${[...VALID_VERBS].join(", ")}`
9987
- };
9988
- }
9989
- }
9990
- const stepMatches = input.content.match(/^\d+\.\s+(click|type|scroll|assert|hover|drag|read|refresh):/gm) || [];
9991
- const interactions = stepMatches.filter((s) => /^\d+\.\s+(click|type|drag):/.test(s));
9992
- if (interactions.length < 2) {
9993
- return {
9994
- error: `Test has ${interactions.length} interaction(s) (click/type/drag). Minimum is 2. Visibility-only tests are not allowed - what BEHAVIOR does this test verify?`
9995
- };
9996
- }
9997
- const bodyStart = input.content.indexOf("---", 3);
9998
- const body = bodyStart > -1 ? input.content.slice(bodyStart + 3) : input.content;
9999
- const stepsSection = body.slice(body.indexOf("**Steps**") || 0);
10000
- const placeholder = findForbiddenPlaceholder(stepsSection);
10001
- if (placeholder) {
10002
- return {
10003
- error: `Test steps contain ${placeholder.name}: "${placeholder.match}". Use EXACT values from scenarios.md - not placeholders or examples.`
10004
- };
10005
- }
10472
+ const content = renderTestMarkdown(input.test);
10006
10473
  const relPath = join32(TESTS_DIR, input.folder, normalizeTestFilename(input.filename));
10007
10474
  const absPath = join32(outputDir, relPath);
10008
10475
  const nodeId = state.resolveNodeId(input.nodeId, relPath);
@@ -10023,9 +10490,23 @@ function buildWriteTestTool(state, outputDir) {
10023
10490
  }
10024
10491
  try {
10025
10492
  await mkdir4(dirname5(absPath), { recursive: true });
10026
- await writeFile13(absPath, input.content, "utf-8");
10493
+ await writeFile12(absPath, content, "utf-8");
10027
10494
  state.markTested(nodeId, [relPath]);
10028
10495
  await saveBfsState(outputDir, state);
10496
+ const notes = input.test.notes?.trim();
10497
+ if (notes != null && notes !== "") {
10498
+ track("cli_write_test_notes", { node_id: nodeId });
10499
+ captureLog("info", `Test author flagged an uncertainty: ${notes}`, {
10500
+ source: "test-generator",
10501
+ path: relPath,
10502
+ node_id: nodeId
10503
+ });
10504
+ }
10505
+ onWritten?.({
10506
+ relativePath: join32(input.folder, normalizeTestFilename(input.filename)),
10507
+ content,
10508
+ flow: input.test.flow
10509
+ });
10029
10510
  if (nodeId !== input.nodeId) {
10030
10511
  debugLog("write_test received an unknown nodeId; attributed it to the current node", {
10031
10512
  given: input.nodeId,
@@ -10045,11 +10526,11 @@ function buildWriteTestTool(state, outputDir) {
10045
10526
  });
10046
10527
  return {
10047
10528
  path: relPath,
10048
- title: parsed.data.title,
10529
+ title: input.test.title,
10049
10530
  note: `nodeId "${input.nodeId}" is not a known node - recorded under "${nodeId}". Pass the id returned by next_node verbatim.`
10050
10531
  };
10051
10532
  }
10052
- return { path: relPath, title: parsed.data.title };
10533
+ return { path: relPath, title: input.test.title };
10053
10534
  } catch (err) {
10054
10535
  const message = err instanceof Error ? err.message : String(err);
10055
10536
  return { error: `Failed to write test: ${message}` };
@@ -10060,8 +10541,8 @@ function buildWriteTestTool(state, outputDir) {
10060
10541
  function buildCreateFolderTool(outputDir) {
10061
10542
  return tool16({
10062
10543
  description: "Create a folder under qa-tests/ for organizing tests.",
10063
- inputSchema: z52.object({
10064
- folder: z52.string().describe("Folder name (kebab-case)")
10544
+ inputSchema: z56.object({
10545
+ folder: z56.string().describe("Folder name (kebab-case)")
10065
10546
  }),
10066
10547
  execute: async (input) => {
10067
10548
  const absPath = join32(outputDir, TESTS_DIR, input.folder);
@@ -10078,7 +10559,7 @@ function buildCreateFolderTool(outputDir) {
10078
10559
  function buildNextNodeTool(state, outputDir) {
10079
10560
  return tool16({
10080
10561
  description: "Get the next node to write tests for. If you called next_node before without writing any tests (via write_test), the previous node is auto-skipped. Returns done:true when all nodes are processed.",
10081
- inputSchema: z52.object({}),
10562
+ inputSchema: z56.object({}),
10082
10563
  execute: async () => {
10083
10564
  const next = state.nextNode();
10084
10565
  await saveBfsState(outputDir, state);
@@ -10096,10 +10577,15 @@ function buildNextNodeTool(state, outputDir) {
10096
10577
  routePath: next.node.routePath,
10097
10578
  sourceFiles: next.node.sourceFiles,
10098
10579
  parentId: next.node.parentId,
10099
- depth: next.node.depth
10580
+ depth: next.node.depth,
10581
+ // The node's mission and its discovered element count. Every test
10582
+ // for this node has to verify the mission; the element count is
10583
+ // what "test depth proportional to complexity" is measured against.
10584
+ mission: next.node.description,
10585
+ interactiveElements: next.node.interactiveElements
10100
10586
  },
10101
10587
  remaining: next.remaining,
10102
- instruction: `Explore "${next.node.name}": read its source files, find all interactive elements, then write tests with write_test. If no tests are needed after reading the source (e.g. utility route, redirect), call next_node to skip.`
10588
+ instruction: `Explore "${next.node.name}": read its source files, find all interactive elements, then write tests with write_test. ` + (next.node.description != null ? `This node's MISSION is: "${next.node.description}". At least one test must directly assert that mission's outcome. ` : "") + "If no tests are needed after reading the source (e.g. utility route, redirect), call next_node to skip."
10103
10589
  };
10104
10590
  }
10105
10591
  });
@@ -10107,7 +10593,7 @@ function buildNextNodeTool(state, outputDir) {
10107
10593
  function buildGetProgressTool(state) {
10108
10594
  return tool16({
10109
10595
  description: "Check how many nodes have been tested vs how many remain.",
10110
- inputSchema: z52.object({}),
10596
+ inputSchema: z56.object({}),
10111
10597
  execute: async () => {
10112
10598
  const stats = state.summary();
10113
10599
  const nodes = [...state.nodes.values()].map((n) => ({
@@ -10123,12 +10609,12 @@ function buildGetProgressTool(state) {
10123
10609
  function buildSpawnResearcherTool(model, workingDirectory, onHeartbeat) {
10124
10610
  return tool16({
10125
10611
  description: "Spawn a research subagent to read and analyze source files without polluting your context. Use for complex sub-features where you don't want to read 20 files yourself.",
10126
- inputSchema: z52.object({
10127
- instruction: z52.string().describe("What to research - be specific about files and what to look for")
10612
+ inputSchema: z56.object({
10613
+ instruction: z56.string().describe("What to research - be specific about files and what to look for")
10128
10614
  }),
10129
10615
  execute: async (input) => {
10130
- const resultSchema2 = z52.object({
10131
- findings: z52.string().describe("Summary of what was found")
10616
+ const resultSchema2 = z56.object({
10617
+ findings: z56.string().describe("Summary of what was found")
10132
10618
  });
10133
10619
  let result;
10134
10620
  const subagent = new ToolLoopAgent3({
@@ -10165,15 +10651,6 @@ function buildSpawnResearcherTool(model, workingDirectory, onHeartbeat) {
10165
10651
  }
10166
10652
  });
10167
10653
  }
10168
- function extractFrontmatter(content) {
10169
- try {
10170
- const { data } = matter5(content);
10171
- return data && Object.keys(data).length > 0 ? data : void 0;
10172
- } catch {
10173
- return void 0;
10174
- }
10175
- }
10176
- var testFrontmatterSchema;
10177
10654
  var init_tools2 = __esm({
10178
10655
  "src/agents/05-test-generator/tools.ts"() {
10179
10656
  "use strict";
@@ -10185,32 +10662,20 @@ var init_tools2 = __esm({
10185
10662
  init_test_files();
10186
10663
  init_tools();
10187
10664
  init_graph();
10188
- init_validation();
10189
- testFrontmatterSchema = z52.object({
10190
- title: z52.string().min(1),
10191
- description: z52.string().min(1),
10192
- intent: z52.string().min(30, "Intent must be at least 30 characters - describe the BEHAVIOR being tested, not the steps"),
10193
- criticality: z52.enum(CRITICALITY_LEVELS),
10194
- scenario: z52.string().min(1),
10195
- flow: z52.string().min(1),
10196
- verification: z52.string().min(
10197
- 20,
10198
- "Verification must describe WHERE to navigate and WHAT to assert at the source of truth - not UI acknowledgments like toasts"
10199
- )
10200
- });
10665
+ init_test_spec();
10201
10666
  }
10202
10667
  });
10203
10668
 
10204
10669
  // src/agents/05-test-generator/write-index.ts
10205
- import { readFile as readFile21, writeFile as writeFile14 } from "fs/promises";
10670
+ import { readFile as readFile21, writeFile as writeFile13 } from "fs/promises";
10206
10671
  import { join as join33 } from "path";
10207
10672
  async function scanTest(outputDir, path3) {
10208
10673
  try {
10209
10674
  const content = await readFile21(join33(outputDir, path3), "utf-8");
10210
10675
  return {
10211
10676
  criticality: content.match(/criticality:\s*(\w+)/)?.[1] ?? "",
10212
- steps: content.match(STEP_VERBS)?.length ?? 0,
10213
- interactions: content.match(INTERACTION_VERBS)?.length ?? 0
10677
+ steps: content.match(STEP_VERBS2)?.length ?? 0,
10678
+ interactions: content.match(INTERACTION_VERBS2)?.length ?? 0
10214
10679
  };
10215
10680
  } catch (err) {
10216
10681
  debugLog("Skipping a test file the index could not read", { path: path3, err });
@@ -10277,7 +10742,7 @@ ${folders.map((f) => `| ${f.name} | ${f.test_count} |`).join("\n")}
10277
10742
 
10278
10743
  ${[...testsByFolder.entries()].flatMap(([_folder, tests]) => tests.map((t) => `- \`${t}\``)).join("\n")}
10279
10744
  ${renderSection("Features with no tests", "the run walked these and produced no test for them - re-run the planner to cover them", untested)}${renderSection("Tests lost in review", "the review cycle removed these and nothing could put them back - re-run the planner to regenerate them", lost)}`;
10280
- await writeFile14(join33(outputDir, TESTS_DIR, TEST_INDEX_FILE), content, "utf-8");
10745
+ await writeFile13(join33(outputDir, TESTS_DIR, TEST_INDEX_FILE), content, "utf-8");
10281
10746
  }
10282
10747
  function untestedFeatures(state) {
10283
10748
  const withTests = new Set(state.testsWritten.keys());
@@ -10293,7 +10758,7 @@ ${items.length} ${items.length === 1 ? "entry" : "entries"} - ${explanation}.
10293
10758
  ${items.map((item) => `- ${item}`).join("\n")}
10294
10759
  `;
10295
10760
  }
10296
- var STEP_VERBS, INTERACTION_VERBS;
10761
+ var STEP_VERBS2, INTERACTION_VERBS2;
10297
10762
  var init_write_index = __esm({
10298
10763
  "src/agents/05-test-generator/write-index.ts"() {
10299
10764
  "use strict";
@@ -10301,8 +10766,8 @@ var init_write_index = __esm({
10301
10766
  init_debug();
10302
10767
  init_test_files();
10303
10768
  init_validation();
10304
- STEP_VERBS = /^\d+\.\s+(click|type|scroll|assert|hover|drag|read|refresh):/gm;
10305
- INTERACTION_VERBS = /^\d+\.\s+(click|type|drag):/gm;
10769
+ STEP_VERBS2 = /^\d+\.\s+(click|type|scroll|assert|hover|drag|read|refresh):/gm;
10770
+ INTERACTION_VERBS2 = /^\d+\.\s+(click|type|drag):/gm;
10306
10771
  }
10307
10772
  });
10308
10773
 
@@ -10312,11 +10777,11 @@ __export(test_generator_exports, {
10312
10777
  runTestGenerator: () => runTestGenerator
10313
10778
  });
10314
10779
  import { existsSync as existsSync2 } from "fs";
10315
- import { mkdir as mkdir5, readFile as readFile22, rmdir, unlink, writeFile as writeFile15 } from "fs/promises";
10780
+ import { mkdir as mkdir5, readFile as readFile22, rmdir, unlink, writeFile as writeFile14 } from "fs/promises";
10316
10781
  import { basename as basename5, join as join34 } from "path";
10317
10782
  import { tool as tool17 } from "ai";
10318
- import { z as z53 } from "zod";
10319
10783
  import { glob as glob6 } from "glob";
10784
+ import { z as z57 } from "zod";
10320
10785
  async function preseedQueue(state, projectRoot, pages, features) {
10321
10786
  let seeded = 0;
10322
10787
  const pageIdByPath = /* @__PURE__ */ new Map();
@@ -10334,7 +10799,8 @@ async function preseedQueue(state, projectRoot, pages, features) {
10334
10799
  sourceFiles: [relPath],
10335
10800
  parentId: void 0,
10336
10801
  depth: 0,
10337
- status: "queued"
10802
+ status: "queued",
10803
+ description: page.description
10338
10804
  };
10339
10805
  if (state.enqueue(node)) seeded++;
10340
10806
  }
@@ -10349,7 +10815,9 @@ async function preseedQueue(state, projectRoot, pages, features) {
10349
10815
  sourceFiles: feature.sourceFiles,
10350
10816
  parentId,
10351
10817
  depth: 1,
10352
- status: "queued"
10818
+ status: "queued",
10819
+ description: feature.description,
10820
+ interactiveElements: feature.interactiveElements
10353
10821
  };
10354
10822
  if (state.enqueue(node)) seeded++;
10355
10823
  }
@@ -10358,15 +10826,15 @@ async function preseedQueue(state, projectRoot, pages, features) {
10358
10826
  Pre-seeded: ${seeded} nodes (pages + sub-features). Call next_node to start processing them one at a time.` : "";
10359
10827
  }
10360
10828
  async function runTestGenerator(input) {
10361
- const model = getModel(input.modelId);
10829
+ const model = input.model ?? getModel(input.modelId);
10362
10830
  const ignorePatterns = await loadGitignorePatterns(input.projectRoot);
10363
10831
  const existingState = await loadBfsState(input.outputDir);
10364
10832
  const state = existingState ?? new CoverageState();
10365
10833
  let result;
10366
10834
  const finishTool = tool17({
10367
10835
  description: "Call when the BFS queue is empty and all routes have been explored.",
10368
- inputSchema: z53.object({
10369
- summary: z53.string().describe("Coverage summary")
10836
+ inputSchema: z57.object({
10837
+ summary: z57.string().describe("Coverage summary")
10370
10838
  }),
10371
10839
  execute: async (finishInput) => {
10372
10840
  const stats = state.summary();
@@ -10402,14 +10870,20 @@ ${autonomaMd}
10402
10870
  `;
10403
10871
  } catch {
10404
10872
  }
10405
- try {
10406
- const scenariosMd = await readFile22(join34(input.outputDir, "scenarios.md"), "utf-8");
10407
- kbContext += `
10873
+ const recipeContext = await loadRecipeContext(input.outputDir);
10874
+ if (recipeContext !== "") {
10875
+ kbContext += recipeContext;
10876
+ } else {
10877
+ try {
10878
+ const scenariosMd = await readFile22(join34(input.outputDir, "scenarios.md"), "utf-8");
10879
+ kbContext += `
10408
10880
  ## Scenarios
10409
10881
 
10410
10882
  ${scenariosMd}
10411
10883
  `;
10412
- } catch {
10884
+ } catch (err) {
10885
+ debugLog("Neither recipe.json nor scenarios.md is readable; generating without a data contract", { err });
10886
+ }
10413
10887
  }
10414
10888
  let features;
10415
10889
  if (!existingState) {
@@ -10430,7 +10904,7 @@ ${scenariosMd}
10430
10904
  const preseedContext = existingState ? "" : await preseedQueue(state, input.projectRoot, input.pages, features ?? void 0);
10431
10905
  const resumeContext = existingState ? `
10432
10906
  You are RESUMING a previous run. ${existingState.summary().tested} nodes tested, ${existingState.summary().totalTests} tests written. Call next_node to continue.` : "";
10433
- const contextBlock = (input.projectContext ? "\n" + formatContext(input.projectContext) + "\n" : "") + formatRetryGuidance(input.retryGuidance);
10907
+ const contextBlock = formatRetryGuidance(input.retryGuidance);
10434
10908
  let prompt = `Generate E2E test cases by processing every node in the queue.
10435
10909
  ${contextBlock}${kbContext}${resumeContext}${preseedContext}
10436
10910
 
@@ -10454,6 +10928,8 @@ Do NOT try to finish early. Process EVERY node via next_node until it returns do
10454
10928
  const MAX_STALE_CHUNKS = 3;
10455
10929
  let totalSteps = 0;
10456
10930
  const logger = createStepLogger("test-gen", CHUNK_STEPS);
10931
+ const reviewDeadline = Date.now() + REVIEW_BUDGET_MS;
10932
+ const pipeline = new ReviewPipeline(input.outputDir, input.projectRoot, model, reviewDeadline);
10457
10933
  const listDirectoryFn = await buildListDirectoryTool(input.projectRoot);
10458
10934
  const agentConfig = {
10459
10935
  id: "test-generator",
@@ -10468,7 +10944,10 @@ Do NOT try to finish early. Process EVERY node via next_node until it returns do
10468
10944
  grep: buildGrepTool(input.projectRoot),
10469
10945
  bash: buildBashTool(input.projectRoot),
10470
10946
  list_directory: listDirectoryFn,
10471
- write_test: buildWriteTestTool(state, input.outputDir),
10947
+ write_test: buildWriteTestTool(state, input.outputDir, (test) => {
10948
+ consecutiveRejections = 0;
10949
+ pipeline.submit(test);
10950
+ }),
10472
10951
  create_folder: buildCreateFolderTool(input.outputDir),
10473
10952
  next_node: buildNextNodeTool(state, input.outputDir),
10474
10953
  get_progress: buildGetProgressTool(state),
@@ -10477,6 +10956,7 @@ Do NOT try to finish early. Process EVERY node via next_node until it returns do
10477
10956
  }),
10478
10957
  onStepFinish: (info) => {
10479
10958
  logger.log(info);
10959
+ recordToolErrors(info.toolErrors);
10480
10960
  const stats = state.summary();
10481
10961
  if (info.stepNumber > 0 && info.stepNumber % 10 === 0) {
10482
10962
  logger.checkpoint(
@@ -10485,6 +10965,29 @@ Do NOT try to finish early. Process EVERY node via next_node until it returns do
10485
10965
  }
10486
10966
  }
10487
10967
  };
10968
+ let consecutiveRejections = 0;
10969
+ let lastRejection;
10970
+ function recordToolErrors(toolErrors) {
10971
+ const writeErrors = toolErrors.filter((e) => e.name === "write_test");
10972
+ if (writeErrors.length === 0) return;
10973
+ consecutiveRejections += writeErrors.length;
10974
+ lastRejection = String(writeErrors.at(-1)?.error ?? "").slice(0, REJECTION_MESSAGE_CHARS);
10975
+ if (consecutiveRejections === MAX_CONSECUTIVE_REJECTIONS) {
10976
+ track("cli_write_test_rejection_loop", { rejections: consecutiveRejections });
10977
+ captureLog("error", `write_test has rejected ${consecutiveRejections} attempts in a row`, {
10978
+ source: "test-generator",
10979
+ rejections: consecutiveRejections,
10980
+ last_error: lastRejection
10981
+ });
10982
+ console.error(
10983
+ `
10984
+ write_test has rejected ${consecutiveRejections} attempts in a row without a test being written.
10985
+ The model cannot satisfy a validation rule, so retrying will not converge:
10986
+ ${lastRejection}
10987
+ `
10988
+ );
10989
+ }
10990
+ }
10488
10991
  let staleChunks = 0;
10489
10992
  let lastTestCount = state.summary().totalTests;
10490
10993
  while (!result) {
@@ -10496,6 +10999,10 @@ ${formatException(err)}`);
10496
10999
  }
10497
11000
  totalSteps += CHUNK_STEPS;
10498
11001
  if (result) break;
11002
+ if (consecutiveRejections >= MAX_CONSECUTIVE_REJECTIONS) {
11003
+ console.log(` [chunk] write_test is rejecting every attempt - stopping rather than burning the budget.`);
11004
+ break;
11005
+ }
10499
11006
  const stats = state.summary();
10500
11007
  const newTests = stats.totalTests - lastTestCount;
10501
11008
  if (newTests === 0) {
@@ -10540,7 +11047,11 @@ IMPORTANT: Do NOT try to finish early. Process every node via next_node until it
10540
11047
  console.log(` Generated ${journeyCount} journey tests`);
10541
11048
  }
10542
11049
  const lostTests = /* @__PURE__ */ new Set();
10543
- const reviewDeadline = Date.now() + REVIEW_BUDGET_MS;
11050
+ const pipelined = await pipeline.drain();
11051
+ if (pipelined.length > 0) {
11052
+ console.log(` ${pipelined.length} tests were already reviewed during generation`);
11053
+ }
11054
+ const settled = /* @__PURE__ */ new Set();
10544
11055
  for (let cycle = 0; cycle < MAX_REVIEW_CYCLES; cycle++) {
10545
11056
  if (Date.now() > reviewDeadline) {
10546
11057
  console.log(` Review budget spent after ${cycle} cycle${cycle === 1 ? "" : "s"} - moving on`);
@@ -10555,22 +11066,51 @@ IMPORTANT: Do NOT try to finish early. Process every node via next_node until it
10555
11066
  }
10556
11067
  console.log(` Review cycle ${cycle + 1}/${MAX_REVIEW_CYCLES}`);
10557
11068
  state.setPhase(`review cycle ${cycle + 1}/${MAX_REVIEW_CYCLES}`);
10558
- const reviewResult = await runConsolidatedReview(input.outputDir, input.projectRoot, model, reviewDeadline);
10559
- console.log(` Review: ${reviewResult.passed} passed, ${reviewResult.failed} failed`);
11069
+ const scanDeadline = Date.now() + Math.floor((reviewDeadline - Date.now()) * REVIEW_SCAN_SHARE);
11070
+ const reviewResult = await runConsolidatedReview(
11071
+ input.outputDir,
11072
+ input.projectRoot,
11073
+ model,
11074
+ scanDeadline,
11075
+ settled,
11076
+ cycle === 0 ? pipelined : []
11077
+ );
11078
+ for (const path3 of reviewResult.passedPaths) settled.add(path3);
11079
+ console.log(
11080
+ ` Review: ${reviewResult.passed} passed, ${reviewResult.failed} failed (${settled.size} settled overall)`
11081
+ );
10560
11082
  if (reviewResult.feedback.length === 0) {
10561
11083
  console.log(` All tests passed review - done`);
10562
11084
  break;
10563
11085
  }
10564
- if (reviewResult.ranOutOfTime) {
10565
- console.log(` Review budget ran out mid-cycle - moving on`);
10566
- track("cli_review_budget_exhausted", { cycles_completed: cycle });
10567
- captureLog("warn", `Review budget ran out mid-cycle - some tests were left unreviewed`, {
11086
+ const scanCutShort = reviewResult.ranOutOfTime;
11087
+ if (Date.now() > reviewDeadline) {
11088
+ console.log(` Review budget spent - moving on`);
11089
+ track("cli_review_budget_exhausted", {
11090
+ cycles_completed: cycle,
11091
+ findings_discarded: reviewResult.feedback.length
11092
+ });
11093
+ captureLog("warn", `Review budget spent - findings from this cycle were not fixed`, {
10568
11094
  source: "test-generator",
10569
11095
  step: "review",
10570
- cycles_completed: cycle
11096
+ cycles_completed: cycle,
11097
+ findings_discarded: reviewResult.feedback.length
10571
11098
  });
10572
11099
  break;
10573
11100
  }
11101
+ if (scanCutShort) {
11102
+ console.log(` Review scan cut short - fixing the ${reviewResult.feedback.length} it did judge`);
11103
+ track("cli_review_scan_cut_short", {
11104
+ cycles_completed: cycle,
11105
+ findings_carried: reviewResult.feedback.length
11106
+ });
11107
+ captureLog("warn", `Review scan cut short - some tests were left unreviewed`, {
11108
+ source: "test-generator",
11109
+ step: "review",
11110
+ cycles_completed: cycle,
11111
+ findings_carried: reviewResult.feedback.length
11112
+ });
11113
+ }
10574
11114
  for (const fb of reviewResult.feedback) {
10575
11115
  try {
10576
11116
  await unlink(fb.testPath);
@@ -10627,6 +11167,7 @@ IMPORTANT: Do NOT try to finish early. Process every node via next_node until it
10627
11167
  });
10628
11168
  }
10629
11169
  console.log(` Fix pass complete`);
11170
+ if (scanCutShort) break;
10630
11171
  }
10631
11172
  state.setPhase("checking every test");
10632
11173
  const allTestFiles = await glob6(join34(input.outputDir, TESTS_DIR, TEST_FILE_GLOB));
@@ -10642,7 +11183,7 @@ IMPORTANT: Do NOT try to finish early. Process every node via next_node until it
10642
11183
  const dest = join34(invalidDir, basename5(testPath));
10643
11184
  const annotated = `<!-- VALIDATION ERRORS: ${validation.errors.join("; ")} -->
10644
11185
  ${content}`;
10645
- await writeFile15(dest, annotated, "utf-8");
11186
+ await writeFile14(dest, annotated, "utf-8");
10646
11187
  await unlink(testPath);
10647
11188
  markedInvalid++;
10648
11189
  }
@@ -10746,13 +11287,13 @@ Each journey test:
10746
11287
  - Has scenario: standard
10747
11288
  - Includes an **Intent**: section explaining the cross-feature flow being tested
10748
11289
  - Verifies that the OUTPUT of one feature is correctly consumed by the NEXT feature
10749
- - Goes in the "journeys" folder
11290
+ - Goes in the "${JOURNEY_NODE_ID}" folder
10750
11291
 
10751
- Write 5-8 journey tests using the write_test tool with folder "journeys". Then call finish.`;
11292
+ Write 5-8 journey tests using the write_test tool with folder "${JOURNEY_NODE_ID}" and nodeId "${JOURNEY_NODE_ID}". Then call finish.`;
10752
11293
  const ignorePatterns = await loadGitignorePatterns(projectRoot);
10753
11294
  const journeyState = new CoverageState({ stateFile: JOURNEY_STATE_FILE, reportsProgress: false });
10754
11295
  journeyState.enqueue({
10755
- id: "journeys",
11296
+ id: JOURNEY_NODE_ID,
10756
11297
  name: "Journey Tests",
10757
11298
  sourceFiles: [],
10758
11299
  parentId: void 0,
@@ -10763,7 +11304,7 @@ Write 5-8 journey tests using the write_test tool with folder "journeys". Then c
10763
11304
  let journeyResult;
10764
11305
  const journeyFinish = tool17({
10765
11306
  description: "Signal journey generation is complete.",
10766
- inputSchema: z53.object({ summary: z53.string() }),
11307
+ inputSchema: z57.object({ summary: z57.string() }),
10767
11308
  execute: async (finishInput) => {
10768
11309
  journeyResult = {
10769
11310
  success: true,
@@ -10798,33 +11339,38 @@ ${formatException(err)}`);
10798
11339
  logger.summary();
10799
11340
  return journeyState.allTestPaths().length;
10800
11341
  }
10801
- var MAX_CONCURRENCY, MAX_REVIEW_CYCLES, REVIEW_BUDGET_MS;
11342
+ var MAX_CONCURRENCY, MAX_REVIEW_CYCLES, REVIEW_BUDGET_MS, MAX_CONSECUTIVE_REJECTIONS, REJECTION_MESSAGE_CHARS, REVIEW_SCAN_SHARE, JOURNEY_NODE_ID;
10802
11343
  var init_test_generator = __esm({
10803
11344
  "src/agents/05-test-generator/index.ts"() {
10804
11345
  "use strict";
10805
11346
  init_esm_shims();
10806
11347
  init_agent();
10807
11348
  init_analytics();
10808
- init_context();
11349
+ init_debug();
10809
11350
  init_display();
10810
11351
  init_errors();
10811
11352
  init_gitignore();
10812
11353
  init_logs();
10813
11354
  init_model();
10814
- init_restore_deleted_test();
10815
- init_review();
10816
- init_debug();
10817
11355
  init_test_files();
10818
11356
  init_tools();
10819
11357
  init_b_feature_discovery();
10820
11358
  init_graph();
10821
11359
  init_prompt5();
11360
+ init_recipe_context();
11361
+ init_restore_deleted_test();
11362
+ init_review();
11363
+ init_review_pipeline();
10822
11364
  init_tools2();
10823
11365
  init_validation();
10824
11366
  init_write_index();
10825
11367
  MAX_CONCURRENCY = 8;
10826
11368
  MAX_REVIEW_CYCLES = 4;
10827
11369
  REVIEW_BUDGET_MS = 45 * 60 * 1e3;
11370
+ MAX_CONSECUTIVE_REJECTIONS = 25;
11371
+ REJECTION_MESSAGE_CHARS = 300;
11372
+ REVIEW_SCAN_SHARE = 0.6;
11373
+ JOURNEY_NODE_ID = "journeys";
10828
11374
  }
10829
11375
  });
10830
11376
 
@@ -12052,8 +12598,8 @@ var init_useTerminalSize = __esm({
12052
12598
 
12053
12599
  // src/ui/grid.ts
12054
12600
  function rgbOf(hex) {
12055
- const cached = rgbCache.get(hex);
12056
- if (cached != null) return cached;
12601
+ const cached2 = rgbCache.get(hex);
12602
+ if (cached2 != null) return cached2;
12057
12603
  const r = parseInt(hex.slice(1, 3), 16);
12058
12604
  const g = parseInt(hex.slice(3, 5), 16);
12059
12605
  const b = parseInt(hex.slice(5, 7), 16);
@@ -12839,7 +13385,7 @@ ensureSupportedNode();
12839
13385
 
12840
13386
  // src/index.ts
12841
13387
  init_submit();
12842
- import { readFile as readFile24, writeFile as writeFile16 } from "fs/promises";
13388
+ import { readFile as readFile24, writeFile as writeFile15 } from "fs/promises";
12843
13389
  import { join as join35 } from "path";
12844
13390
 
12845
13391
  // src/agents/04-recipe-builder/sdk-command.ts
@@ -13029,7 +13575,7 @@ function shortHash(seed) {
13029
13575
  }
13030
13576
 
13031
13577
  // src/agents/04-recipe-builder/sdk-command.ts
13032
- import { z as z33 } from "zod";
13578
+ import { z as z35 } from "zod";
13033
13579
 
13034
13580
  // src/agents/04-recipe-builder/http-client.ts
13035
13581
  init_esm_shims();
@@ -13078,15 +13624,15 @@ async function down(config, refsToken) {
13078
13624
 
13079
13625
  // src/agents/04-recipe-builder/sdk-command.ts
13080
13626
  var DEFAULT_REQUEST_TIMEOUT_MS = 12e4;
13081
- var createSchema = z33.record(z33.string(), z33.array(z33.unknown()));
13082
- var recipeSliceSchema = z33.object({ create: createSchema, variables: ScenarioRecipeVariablesSchema.optional() });
13083
- var envelopeSchema = z33.object({ recipes: z33.array(recipeSliceSchema).min(1) });
13084
- var flagsSchema = z33.object({
13085
- url: z33.string().min(1).optional(),
13086
- recipe: z33.string().min(1).optional(),
13087
- "refs-token": z33.string().min(1).optional(),
13088
- "test-run-id": z33.string().min(1).optional(),
13089
- timeout: z33.coerce.number().int().positive().optional()
13627
+ var createSchema = z35.record(z35.string(), z35.array(z35.unknown()));
13628
+ var recipeSliceSchema = z35.object({ create: createSchema, variables: ScenarioRecipeVariablesSchema.optional() });
13629
+ var envelopeSchema = z35.object({ recipes: z35.array(recipeSliceSchema).min(1) });
13630
+ var flagsSchema = z35.object({
13631
+ url: z35.string().min(1).optional(),
13632
+ recipe: z35.string().min(1).optional(),
13633
+ "refs-token": z35.string().min(1).optional(),
13634
+ "test-run-id": z35.string().min(1).optional(),
13635
+ timeout: z35.coerce.number().int().positive().optional()
13090
13636
  }).strict();
13091
13637
  async function runSdkCommand(argv, io) {
13092
13638
  const action = argv[0];
@@ -13295,7 +13841,6 @@ function loadConfig(args) {
13295
13841
  // src/index.ts
13296
13842
  init_analytics();
13297
13843
  init_colors();
13298
- init_context();
13299
13844
  init_errors();
13300
13845
 
13301
13846
  // src/core/flush-telemetry.ts
@@ -13315,13 +13860,13 @@ init_model();
13315
13860
  // src/core/output.ts
13316
13861
  init_esm_shims();
13317
13862
  init_debug();
13318
- import { mkdir, rm, writeFile as writeFile3 } from "fs/promises";
13863
+ import { mkdir, rm, writeFile as writeFile2 } from "fs/promises";
13319
13864
  import { homedir as homedir3 } from "os";
13320
- import { join as join11 } from "path";
13321
- var AUTONOMA_HOME3 = join11(homedir3(), ".autonoma");
13865
+ import { join as join10 } from "path";
13866
+ var AUTONOMA_HOME3 = join10(homedir3(), ".autonoma");
13322
13867
  var WRITE_PROBE_FILE = ".write-probe";
13323
13868
  function getOutputDir(projectSlug) {
13324
- return join11(AUTONOMA_HOME3, projectSlug);
13869
+ return join10(AUTONOMA_HOME3, projectSlug);
13325
13870
  }
13326
13871
  async function clearOutputDir(projectSlug) {
13327
13872
  const dir = getOutputDir(projectSlug);
@@ -13336,9 +13881,9 @@ function displayPath(absPath) {
13336
13881
  async function ensureOutputDir(projectSlug) {
13337
13882
  const dir = getOutputDir(projectSlug);
13338
13883
  await mkdir(dir, { recursive: true });
13339
- const probe = join11(dir, WRITE_PROBE_FILE);
13884
+ const probe = join10(dir, WRITE_PROBE_FILE);
13340
13885
  try {
13341
- await writeFile3(probe, "ok", "utf-8");
13886
+ await writeFile2(probe, "ok", "utf-8");
13342
13887
  } catch (err) {
13343
13888
  const reason = err instanceof Error ? err.message : String(err);
13344
13889
  throw new Error(
@@ -13362,8 +13907,8 @@ init_prompts();
13362
13907
  // src/core/git.ts
13363
13908
  init_esm_shims();
13364
13909
  import { execFile } from "child_process";
13365
- import { readFile as readFile5, writeFile as writeFile4 } from "fs/promises";
13366
- import { join as join12 } from "path";
13910
+ import { readFile as readFile4, writeFile as writeFile3 } from "fs/promises";
13911
+ import { join as join11 } from "path";
13367
13912
  import { promisify } from "util";
13368
13913
  var execFileAsync = promisify(execFile);
13369
13914
  var GIT_INFO_FILE = ".git-info.json";
@@ -13387,11 +13932,11 @@ async function readGitInfo(projectRoot) {
13387
13932
  };
13388
13933
  }
13389
13934
  async function saveGitInfo(outputDir, info) {
13390
- await writeFile4(join12(outputDir, GIT_INFO_FILE), JSON.stringify(info, null, 2), "utf-8");
13935
+ await writeFile3(join11(outputDir, GIT_INFO_FILE), JSON.stringify(info, null, 2), "utf-8");
13391
13936
  }
13392
13937
  async function loadGitInfo(outputDir) {
13393
13938
  try {
13394
- const raw = await readFile5(join12(outputDir, GIT_INFO_FILE), "utf-8");
13939
+ const raw = await readFile4(join11(outputDir, GIT_INFO_FILE), "utf-8");
13395
13940
  const parsed = JSON.parse(raw);
13396
13941
  if (typeof parsed === "object" && parsed != null && "sha" in parsed && typeof parsed.sha === "string") {
13397
13942
  const branch = "branch" in parsed && typeof parsed.branch === "string" ? parsed.branch : void 0;
@@ -13413,15 +13958,15 @@ init_state();
13413
13958
  init_esm_shims();
13414
13959
  init_prompts();
13415
13960
  init_debug();
13416
- import { readFile as readFile7 } from "fs/promises";
13417
- import { join as join14 } from "path";
13961
+ import { readFile as readFile6 } from "fs/promises";
13962
+ import { join as join13 } from "path";
13418
13963
  init_test_files();
13419
13964
  var ARTIFACT_FILES = ["AUTONOMA.md", "scenarios.md", "entity-audit.md"];
13420
13965
  async function readArtifacts(outputDir) {
13421
13966
  const files = await Promise.all(
13422
13967
  ARTIFACT_FILES.map(async (name) => {
13423
13968
  try {
13424
- return { name, content: await readFile7(join14(outputDir, name), "utf-8") };
13969
+ return { name, content: await readFile6(join13(outputDir, name), "utf-8") };
13425
13970
  } catch (err) {
13426
13971
  debugLog(`Artifact ${name} not on disk; skipping upload`, { err });
13427
13972
  return void 0;
@@ -13437,7 +13982,7 @@ async function readTestCases(outputDir) {
13437
13982
  const segments = testPath.split("/").slice(1);
13438
13983
  const name = segments[segments.length - 1];
13439
13984
  const folder = segments.slice(0, -1).join("/");
13440
- const content = await readFile7(join14(outputDir, testPath), "utf-8");
13985
+ const content = await readFile6(join13(outputDir, testPath), "utf-8");
13441
13986
  return { name, content, folder: folder.length > 0 ? folder : void 0 };
13442
13987
  })
13443
13988
  );
@@ -13501,7 +14046,7 @@ process.setSourceMapsEnabled(true);
13501
14046
  var PAGES_FILE = "pages.json";
13502
14047
  async function savePages(outputDir, pages) {
13503
14048
  const obj = Object.fromEntries(pages);
13504
- await writeFile16(join35(outputDir, PAGES_FILE), JSON.stringify(obj, null, 2), "utf-8");
14049
+ await writeFile15(join35(outputDir, PAGES_FILE), JSON.stringify(obj, null, 2), "utf-8");
13505
14050
  }
13506
14051
  async function loadPages(outputDir) {
13507
14052
  try {
@@ -13593,7 +14138,7 @@ async function promptScopeSelection(map) {
13593
14138
  return { frontend, backends };
13594
14139
  }
13595
14140
  }
13596
- async function runStep(step, outputDir, state, config, projectContext, nonInteractive, retryGuidance) {
14141
+ async function runStep(step, outputDir, state, config, nonInteractive, retryGuidance) {
13597
14142
  const label = STEP_LABELS[step];
13598
14143
  note(STEP_INTROS[step], `Step: ${label}`);
13599
14144
  const stepStartedAt = Date.now();
@@ -13605,14 +14150,6 @@ async function runStep(step, outputDir, state, config, projectContext, nonIntera
13605
14150
  });
13606
14151
  state = await markStep(outputDir, state, step, "running");
13607
14152
  getActiveStore()?.startStep(step);
13608
- if (step !== "pagesFinder" && projectContext && !projectContext.pages) {
13609
- const pages = await loadPages(outputDir);
13610
- if (pages.size > 0) {
13611
- projectContext = { ...projectContext, pages: [...pages.values()] };
13612
- }
13613
- }
13614
- const knownPages = projectContext?.pages?.length ?? 0;
13615
- if (knownPages > 0) getActiveStore()?.setSizes({ pages: knownPages });
13616
14153
  let stepMetrics = {};
13617
14154
  try {
13618
14155
  let result;
@@ -13676,12 +14213,10 @@ async function runStep(step, outputDir, state, config, projectContext, nonIntera
13676
14213
  }
13677
14214
  case "kb": {
13678
14215
  const { runKBGenerator: runKBGenerator2 } = await Promise.resolve().then(() => (init_kb_generator(), kb_generator_exports));
13679
- stepMetrics = { page_count: projectContext?.pages?.length ?? 0 };
13680
14216
  result = await runKBGenerator2({
13681
14217
  projectRoot: config.projectRoot,
13682
14218
  outputDir,
13683
14219
  modelId: config.modelId,
13684
- projectContext,
13685
14220
  nonInteractive,
13686
14221
  retryGuidance
13687
14222
  });
@@ -13694,7 +14229,6 @@ async function runStep(step, outputDir, state, config, projectContext, nonIntera
13694
14229
  projectRoot: config.projectRoot,
13695
14230
  outputDir,
13696
14231
  modelId: config.modelId,
13697
- projectContext,
13698
14232
  nonInteractive,
13699
14233
  retryGuidance,
13700
14234
  scopeHint: auditMap != null ? formatBackendScope(auditMap) : void 0
@@ -13709,7 +14243,6 @@ async function runStep(step, outputDir, state, config, projectContext, nonIntera
13709
14243
  outputDir,
13710
14244
  modelId: config.modelId,
13711
14245
  config,
13712
- projectContext,
13713
14246
  nonInteractive,
13714
14247
  retryGuidance,
13715
14248
  scopeHint: recipeMap != null ? formatBackendScope(recipeMap) : void 0
@@ -13724,7 +14257,6 @@ async function runStep(step, outputDir, state, config, projectContext, nonIntera
13724
14257
  outputDir,
13725
14258
  modelId: config.modelId,
13726
14259
  config,
13727
- projectContext,
13728
14260
  nonInteractive,
13729
14261
  retryGuidance,
13730
14262
  agent: config.agent,
@@ -13741,7 +14273,6 @@ async function runStep(step, outputDir, state, config, projectContext, nonIntera
13741
14273
  outputDir,
13742
14274
  modelId: config.modelId,
13743
14275
  config,
13744
- projectContext,
13745
14276
  nonInteractive,
13746
14277
  pages,
13747
14278
  retryGuidance
@@ -13831,10 +14362,10 @@ async function promptStepFailure(label) {
13831
14362
  return { kind: "retry", guidance: trimmed || void 0 };
13832
14363
  }
13833
14364
  }
13834
- async function runStepWithRecovery(step, outputDir, state, config, projectContext, nonInteractive) {
14365
+ async function runStepWithRecovery(step, outputDir, state, config, nonInteractive) {
13835
14366
  let guidance;
13836
14367
  while (true) {
13837
- state = await runStep(step, outputDir, state, config, projectContext, nonInteractive, guidance);
14368
+ state = await runStep(step, outputDir, state, config, nonInteractive, guidance);
13838
14369
  if (state.steps[step] !== "failed" || nonInteractive) return state;
13839
14370
  const action = await promptStepFailure(STEP_LABELS[step]);
13840
14371
  if (action.kind === "exit") return state;
@@ -13989,7 +14520,6 @@ async function main() {
13989
14520
  }
13990
14521
  if (!nonInteractive) mountedUi = await mountDashboard(outputDir, config.projectSlug);
13991
14522
  let isResuming = !!(args.resume || args.step);
13992
- let projectContext;
13993
14523
  const hasProgress = Object.values(state.steps).some((s) => s === "done" || s === "running");
13994
14524
  if (!nonInteractive && !isResuming && !hasProgress) {
13995
14525
  await welcome({
@@ -14023,11 +14553,6 @@ async function main() {
14023
14553
  }
14024
14554
  }
14025
14555
  if (isResuming) seedDashboard(mountedUi, state);
14026
- const saved = await loadContext(outputDir);
14027
- if (saved) {
14028
- projectContext = saved;
14029
- log.info(`Loaded project context from previous run`);
14030
- }
14031
14556
  note(
14032
14557
  `${outputDir}
14033
14558
 
@@ -14051,7 +14576,7 @@ or reveal hidden files (macOS: Cmd+Shift+. ) to see it.`,
14051
14576
  log.error("Cannot run test generation yet - the scenario recipe step must complete first.");
14052
14577
  return;
14053
14578
  }
14054
- state = await runStepWithRecovery(targetStep, outputDir, state, config, projectContext, nonInteractive);
14579
+ state = await runStepWithRecovery(targetStep, outputDir, state, config, nonInteractive);
14055
14580
  mountedUi?.unmount();
14056
14581
  mountedUi = void 0;
14057
14582
  if (state.steps[targetStep] === "failed") {
@@ -14076,7 +14601,7 @@ or reveal hidden files (macOS: Cmd+Shift+. ) to see it.`,
14076
14601
  try {
14077
14602
  for (let i = startIdx; i < steps.length; i++) {
14078
14603
  const step = steps[i];
14079
- state = await runStepWithRecovery(step, outputDir, state, config, projectContext, nonInteractive);
14604
+ state = await runStepWithRecovery(step, outputDir, state, config, nonInteractive);
14080
14605
  if (state.steps[step] === "paused") {
14081
14606
  break;
14082
14607
  }