@tea-agent/loop-agent 0.39.0-beta.13 → 0.39.0-beta.15

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -287,9 +287,20 @@ export async function runFrontendReviewContextGate(input) {
287
287
  if (!parsedContract.success) {
288
288
  throw new FrontendReviewContextFailure("review-context-invalid-contract", "frontend review context invalid implementation contract");
289
289
  }
290
+ // Authorized diff surface = deliverable files + every verification-target
291
+ // file the contract references. The writer legitimately writes VT test
292
+ // artifacts (behavior verification), and the reviewer must audit their
293
+ // diffs; scoping this to targets.files alone failed the baseline-overlap
294
+ // check on app.test.js (r19 extreme smoke).
295
+ const authorizedChangedPaths = [
296
+ ...new Set([
297
+ ...parsedContract.data.targets.files,
298
+ ...parsedContract.data.verificationTargets.map((target) => target.file),
299
+ ]),
300
+ ];
290
301
  const diff = await runFrontendWorktreeDiffGate({
291
302
  ...input,
292
- authorizedChangedPaths: parsedContract.data.targets.files,
303
+ authorizedChangedPaths,
293
304
  });
294
305
  const verificationTrace = await readRequiredJson(input.runDir, "contracts/frontend-verification-trace.json");
295
306
  const repairAssessment = await readOptionalRepairAssessment(input.runDir);
@@ -260,7 +260,12 @@ export function restorePlanPatchFromCommittedFacts(records) {
260
260
  if (isRecord(fact.patch))
261
261
  return fact.patch;
262
262
  }
263
- return undefined;
263
+ // No finalize-published snapshot: the receipt fix loop can end an attempt
264
+ // before any finalize_plan succeeds while dozens of record_* facts are
265
+ // already committed. Assemble the patch from those record facts instead of
266
+ // declaring the ledger missing (r19: "ledger missing" discarded a ledger
267
+ // with 92 committed facts).
268
+ return assemblePlanPatchFromCommittedFacts(records);
264
269
  }
265
270
  /**
266
271
  * A+B (AC-004): reverse of `planLedgerFactsFromPatch` for the incremental
@@ -280,12 +285,33 @@ export function assemblePlanPatchFromCommittedFacts(records, contractRequirement
280
285
  PLAN_LEDGER_FACT_KINDS.includes(fact.kind));
281
286
  if (facts.length === 0)
282
287
  return undefined;
283
- for (const fact of facts) {
284
- if (isRecord(fact.patch))
285
- return fact.patch;
286
- }
288
+ // Singleton record facts: the LAST committed fact wins. The model corrects
289
+ // a rejected plan by re-recording the fact (r18: 39 finalize retries never
290
+ // converged because the first state-flow fact kept shadowing the
291
+ // corrections). Requirements / verification targets / evidence gaps stay
292
+ // additive (aggregated below); component choices accumulate; singletons
293
+ // supersede.
294
+ const lastByKind = (kind) => {
295
+ let found;
296
+ for (const fact of facts) {
297
+ if (fact.kind === kind)
298
+ found = fact;
299
+ }
300
+ return found;
301
+ };
302
+ // Patch snapshots (published by earlier finalize calls) carry no routes;
303
+ // exclude them so a snapshot cannot shadow the model's route selection.
304
+ const routeSurface = (() => {
305
+ let found;
306
+ for (const fact of facts) {
307
+ if (fact.kind !== "target-surface" || isRecord(fact.patch))
308
+ continue;
309
+ found = fact;
310
+ }
311
+ return found;
312
+ })();
287
313
  const patch = {};
288
- const targetSurface = facts.find((fact) => fact.kind === "target-surface");
314
+ const targetSurface = routeSurface;
289
315
  if (targetSurface) {
290
316
  const routes = canonicalStringSet(asStringArray(targetSurface.routes));
291
317
  if (routes.length > 0)
@@ -295,14 +321,34 @@ export function assemblePlanPatchFromCommittedFacts(records, contractRequirement
295
321
  const collectedChoices = componentChoices.flatMap((fact) => Array.isArray(fact.uiComponentChoices)
296
322
  ? fact.uiComponentChoices.filter(isRecord)
297
323
  : []);
324
+ // Later corrections supersede earlier submissions per purpose: without
325
+ // this, a re-recorded choice (fixing a wrong specReference) would appear
326
+ // twice and the reviewer would still audit the stale entry (r19).
327
+ const choicesByPurpose = new Map();
328
+ const dedupedChoices = [];
329
+ for (const choice of collectedChoices) {
330
+ const purpose = asString(choice.purpose);
331
+ if (!purpose) {
332
+ dedupedChoices.push(choice);
333
+ continue;
334
+ }
335
+ if (choicesByPurpose.has(purpose)) {
336
+ const at = dedupedChoices.findIndex((existing) => asString(existing.purpose) === purpose);
337
+ dedupedChoices[at] = choice;
338
+ }
339
+ else {
340
+ choicesByPurpose.set(purpose, choice);
341
+ dedupedChoices.push(choice);
342
+ }
343
+ }
298
344
  const stylingStrategy = componentChoices
299
345
  .map((fact) => asString(fact.stylingStrategy))
300
346
  .find((value) => value.length > 0);
301
- if (collectedChoices.length > 0)
302
- patch.uiComponentChoices = collectedChoices;
347
+ if (dedupedChoices.length > 0)
348
+ patch.uiComponentChoices = dedupedChoices;
303
349
  if (stylingStrategy)
304
350
  patch.stylingStrategy = stylingStrategy;
305
- const stateFlow = facts.find((fact) => fact.kind === "state-flow");
351
+ const stateFlow = lastByKind("state-flow");
306
352
  if (stateFlow) {
307
353
  if (Array.isArray(stateFlow.uiStates)) {
308
354
  patch.uiStates = stateFlow.uiStates.filter(isRecord);
@@ -311,17 +357,17 @@ export function assemblePlanPatchFromCommittedFacts(records, contractRequirement
311
357
  patch.interactions = stateFlow.interactions.filter(isRecord);
312
358
  }
313
359
  }
314
- const mockApi = facts.find((fact) => fact.kind === "mock-api");
360
+ const mockApi = lastByKind("mock-api");
315
361
  if (mockApi && isRecord(mockApi.mockApi)) {
316
362
  patch.mockApi = mockApi.mockApi;
317
363
  }
318
- const deviation = facts.find((fact) => fact.kind === "design-deviation");
364
+ const deviation = lastByKind("design-deviation");
319
365
  if (deviation) {
320
366
  const conflicts = canonicalStringSet(asStringArray(deviation.conflicts));
321
367
  if (conflicts.length > 0)
322
368
  patch.designEvidence = { conflicts };
323
369
  }
324
- const dependency = facts.find((fact) => fact.kind === "dependency");
370
+ const dependency = lastByKind("dependency");
325
371
  const dependencyPolicy = asString(dependency?.policy);
326
372
  if (dependencyPolicy)
327
373
  patch.dependencyPolicy = dependencyPolicy;
@@ -3210,6 +3210,7 @@ async function buildFrontendHybridDagFromTask(sources) {
3210
3210
  ...(requiresOpenspecClassification ? ["When a component choice uses an OpenSpec selection, cite that selection; otherwise do not classify unrelated candidates."] : []),
3211
3211
  "Call finalize_plan exactly once after the necessary typed facts. Return no Markdown narrative.",
3212
3212
  "TOOL-ONLY PLAN: Do not read Contract/Scout stdout, task sources, or Scout-confirmed target files. Contract and Scout already own evidence discovery; use the injected upstream facts, record a genuine evidence gap when those facts are insufficient, and start committing record_* facts immediately. For decision=new, pass sourceRequirementIds to record_component_choice; runtime derives the exact PRD citation from the frozen ledger.",
3213
+ "Output budget protocol (hard, max output <=16K per turn): never enumerate-reason the whole requirement list before your first record_* call — that reasoning burns the entire output budget and the attempt dies with zero committed facts. Process requirements in order: think about ONE requirement briefly, immediately emit its record calls (up to 5 per message), then move to the next. If your budget runs low, stop recording and call finalize_plan with what is committed — the retry ladder continues the remainder in a fresh session.",
3213
3214
  fixedVerificationContext,
3214
3215
  scopedOpenspecContext,
3215
3216
  mockContextBlock,
@@ -315,9 +315,11 @@ export const dagFrontendVerificationBundleSchema = z
315
315
  behaviorEvidence: dagShellVerifyEvidenceSchema,
316
316
  lintBaselineNodeId: dagFrontendNodeIdSchema.optional(),
317
317
  writerNodeIds: z.array(dagFrontendNodeIdSchema).optional(),
318
- /** M8: verify-shell only ever runs in initial mode; a failure is terminal
319
- * (routes to recovery) and never selects a same-run repair branch. */
320
- mode: z.literal("initial"),
318
+ /** "initial" = first-pass assess bundle; "repair" = the convergence
319
+ * loop's post-repair reverify, which re-runs the same frozen commands
320
+ * and fails on any failure (still no same-run repair branch inside the
321
+ * node itself). */
322
+ mode: z.enum(["initial", "repair"]),
321
323
  })
322
324
  .superRefine((bundle, context) => {
323
325
  const groups = [
package/harness.json CHANGED
@@ -83,8 +83,8 @@
83
83
  "pi": {
84
84
  "description": "Pi 负责规划、评审、诊断;当 DAG toolProfile=write 时也可做有界写入。模型按复杂度三档配置,支持 provider/model 字符串或带 thinking 的对象;斜杠前为 Pi provider,后为 modelId,勿只写裸 modelId。",
85
85
  "LOW": "wizard-local/minimax-m3",
86
- "MED": "wizard-local/gpt-5.6-sol",
87
- "HIGH": "wizard-local/gpt-5.6-sol"
86
+ "MED": "wizard-local/gpt-5.5",
87
+ "HIGH": "wizard-local/gpt-5.5"
88
88
  }
89
89
  }
90
- }
90
+ }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@tea-agent/loop-agent",
3
- "version": "0.39.0-beta.13",
3
+ "version": "0.39.0-beta.15",
4
4
  "type": "module",
5
5
  "bin": {
6
6
  "loop-agent": "bin/loop-agent.js",
@@ -106,4 +106,4 @@
106
106
  "vitest": "^3.2.4"
107
107
  },
108
108
  "packageManager": "pnpm@10.33.0+sha512.10568bb4a6afb58c9eb3630da90cc9516417abebd3fabbe6739f0ae795728da1491e9db5a544c76ad8eb7570f5c4bb3d6c637b2cb41bfdcdb47fa823c8649319"
109
- }
109
+ }