@tea-agent/loop-agent 0.42.0-next.8 → 0.42.0-next.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -2,6 +2,15 @@
2
2
 
3
3
  ## [Unreleased]
4
4
 
5
+ - 冒烟再暴露 writeSet 授权缺口:冻结验证命令引用的基建文件(如 `--config` 的 vitest 配置)不在 plan verification targets 内时,writer 被具体 writeSet 拦住无法创建、verify-shell 必然失败。现在 admission 生效边界确定性并入冻结命令引用的文件(机械提取、仍受 allowedPaths/forbiddenPaths 约束),writer 聚焦检查失败后可自行补齐基建文件。
6
+ - 修复 /inspect/* 静态资源(operator-chrome.js/css、观测文档)缺少 Cache-Control 的问题:现在与 Console SPA 一律 `no-store`。此前浏览器启发式缓存会让 Console 更新/重启后继续挂旧版导航模块,表现为顶部导航「回退」到只有操作/观测、缺少对话入口的旧版布局。
7
+
8
+ - 冻结验证命令引用的文件(如 `--config` 的 vitest 配置)现在必须在 plan 阶段就有 verification target 覆盖:缺失会作为缺失事实驱动 coverage/compact/finalize 会话当轮补记并进入 admission writeSet,无法补齐则在 plan 阶段确定性失败,不再等到 verify-shell 才以 missing config 失败。
9
+ - 修复 writer 终态失败 run 无法重跑的问题:rerun 反馈的 writer 失败通道把 `attemptSummary.completedPasses` 硬编码为 0,被 `dagRerunFeedbackSchema` 的 positive 校验拒绝,导致 `dag rerun-task` 在解析父 run 反馈时崩溃;现与相邻通道一致取 1。
10
+ - 冒烟再暴露并修复 planner 思考烧尽:compact Plan 首会话因 `stopReason=length` 只思考零提交时,先做一次 tool-first 同 scope 重试,仍失败则机械降级为按需求减半的 coverage 分批会话;耗尽后归类为 `planner-thinking-exhausted` 并给出 thinking=off/换模型建议,不再原样重放同一 prompt 烧完修复预算。
11
+ - 继续加固前端恢复:生产 `runDag` 现在使用共享 Console operation store 执行带反馈的 `dag rerun-task`,避免恢复 descriptor 无法解析;compact Plan 仅在单 target surface 生效,finalize 提示与 concrete writeSet 路径统一规范化。
12
+ - 修复前端冒烟暴露的 P0/P1 流程问题:设计评审 `request_design_changes` 现在 fail-closed 阻断 writer,自动恢复从 Plan 重启并携带 findings/evidence;admission 冻结的 concrete writeSet 同时成为 writer 的实际授权边界,legacy `admitted` 不再放行;长度耗尽且零写入的 provider 结果按 `writer-thinking-exhausted` 归类。生产小型 Plan(≤8 个需求且估算调用量受控)合并为两次 Pi session,并在 session 间刷新 typed ledger 上下文;Scout 重试仅补 unresolved 路径,减少重复侦查与上下文消耗。
13
+ - 修复 backend-test 对 `payload = {...}` 经 `dict(payload)` 和 `**kwargs` wrapper 透传时漏观测请求字段的问题:保守传播 wrapper 参数别名与浅拷贝 payload,并按分支实际赋值观察 scenario 字段;继续保持 payload SAFE、EXACT_1_TO_1、scenario MATCH 与 fail-closed 门禁。计划生成提示词同时明确无有限分区域时省略整个 `## Scenario Partitions` 章节,避免生成 prose-only 空节触发 N3 阻断。
5
14
  - 修复 Operator Chat 打开已丢失会话时把孤儿 JSONL 升级成 503、以及删会话后残留 workspace 选择被误报为跨工作区冲突的问题:缺失会话文件现在自愈为 404 并清掉记录,前端恢复路径安静忘记该会话。
6
15
  - 修复前端 `--from-text` 任务缺少 source-fidelity ledger、DAG 生成后降级为 v1 并在 `frontend-contract-pi` 0ms 失败的问题:文本入口现在持久化可校验的 v2 绑定输入,init-hybrid 复用任务自有需求源;Console 默认模型与 Pi readiness 同时排除无凭据 provider,避免把不可执行模型误报为可用。
7
16
  - 常规回归 `npm test` 默认跳过约 6 分钟的 I/O-heavy 测试池(仍跑 fast + integration);完整三池改为 `npm run test:full` / `LOOP_AGENT_FULL_TEST=1`。本地 `ci.sh`、GitHub PR 与 release/next 发布门禁仍跑全量,不降低交付验证。
@@ -1,3 +1,4 @@
1
+ import path from "node:path";
1
2
  import { createDagEventObserver } from "../../workflows/dag/event-observer.js";
2
3
  import { composeDagRunObservers } from "../../workflows/dag/observer-compose.js";
3
4
  import { createDagCanvasObserver, resolveCanvasPath, } from "../../workflows/dag/canvas-observer.js";
@@ -28,6 +29,40 @@ function buildRunDagNextSteps(runId) {
28
29
  `loop-agent dag report --run-id ${runId} --markdown # advanced forensic`,
29
30
  ];
30
31
  }
32
+ async function buildDefaultFrontendRecoveryDeps(repoRoot) {
33
+ // Keep the application layer free of a static worker→application cycle. The
34
+ // recovery controller is only needed after a frontend terminal failure.
35
+ const [{ openConsoleAppData }, { OperationStore }, { createOperationEventStore }, { LoopAgentClient }, { resolveSiblingLoopAgentBin }] = await Promise.all([
36
+ import("../../infrastructure/console/app-data.js"),
37
+ import("../../infrastructure/console/operation-store.js"),
38
+ import("../../worker/console/operation-sse.js"),
39
+ import("../../worker/loop-agent/loop-agent-client.js"),
40
+ import("../../worker/console/sibling-controller.js"),
41
+ ]);
42
+ const appData = openConsoleAppData({ repoRoot });
43
+ const store = new OperationStore(appData);
44
+ const events = createOperationEventStore({
45
+ persistenceDir: path.join(appData.operations, "events"),
46
+ });
47
+ const client = new LoopAgentClient({
48
+ loopAgentBin: resolveSiblingLoopAgentBin(),
49
+ artifactRoot: path.join(appData.root, "client-artifacts", appData.fingerprint),
50
+ resolveIdentity: true,
51
+ });
52
+ return {
53
+ store,
54
+ operationRunnerDeps: {
55
+ store,
56
+ events,
57
+ client,
58
+ repoRoot,
59
+ env: {
60
+ ...process.env,
61
+ LOOP_CONSOLE_APP_DATA: appData.root,
62
+ },
63
+ },
64
+ };
65
+ }
31
66
  export async function runDagUseCase(input) {
32
67
  const spec = await loadDagSpecFromFile(input.dagPath);
33
68
  assertValidDagSpec(spec);
@@ -80,6 +115,10 @@ export async function runDagUseCase(input) {
80
115
  ...(canvasError ? { canvasError } : {}),
81
116
  };
82
117
  }
118
+ const recovery = input.recovery ??
119
+ (!input.initOnly && spec.tasks.some((task) => task.id === "frontend-implement-pi")
120
+ ? await buildDefaultFrontendRecoveryDeps(input.cwd)
121
+ : undefined);
83
122
  const summary = await runDag(spec, {
84
123
  cwd: input.cwd,
85
124
  initOnly: input.initOnly,
@@ -90,6 +129,7 @@ export async function runDagUseCase(input) {
90
129
  ...(input.workerAssociation
91
130
  ? { workerAssociation: input.workerAssociation }
92
131
  : {}),
132
+ ...(recovery ? { recovery } : {}),
93
133
  });
94
134
  const canvasError = await flushCanvasSafely(canvas);
95
135
  if (resolvedCanvasPath) {
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "schemaVersion": 1,
3
- "version": "0.42.0-next.8",
4
- "gitSha": "1d1ada8ede350e5778d753c04ed72bacd778e43d",
5
- "builtAt": "2026-09-02T23:47:25.236Z"
3
+ "version": "0.42.0-next.9",
4
+ "gitSha": "13597aa31afc1faccb9a5eaca296d57d7ab93c2e",
5
+ "builtAt": "2026-09-03T05:18:16.410Z"
6
6
  }
@@ -31,6 +31,34 @@ import { writeEffectiveContextReceipt } from "../workflows/dag/context-receipt.j
31
31
  * recoverable partial-write-set (incomplete-write-set) upgrade.
32
32
  */
33
33
  export const WRITER_THINKING_EXHAUSTED_CATEGORY = "writer-thinking-exhausted";
34
+ /**
35
+ * Planner classification mirroring writer-thinking-exhausted: a read-only
36
+ * planning session stopped on length, observed thinking, and committed zero
37
+ * typed facts with no assistant text. By the time this survives the segmented
38
+ * ladder the scope has already been degraded, so the durable fix is a
39
+ * thinking-capped or non-thinking model for the tier — not another replay of
40
+ * the same full-scope prompt.
41
+ */
42
+ export const PLANNER_THINKING_EXHAUSTED_CATEGORY = "planner-thinking-exhausted";
43
+ export function isPlannerThinkingExhausted(result, committedAnyFacts) {
44
+ if (result.ok)
45
+ return false;
46
+ const evidence = readWriterThinkingExhaustionEvidence(result);
47
+ if (evidence.stopReason !== "length")
48
+ return false;
49
+ if (evidence.thinkingObserved !== true)
50
+ return false;
51
+ if (committedAnyFacts)
52
+ return false;
53
+ if ((result.assistantText ?? "").trim())
54
+ return false;
55
+ // Gateways sometimes relabel a length-stopped stream as `network` or
56
+ // `nonzero-exit`; provider evidence outweighs the transport label.
57
+ if (result.failureCategory &&
58
+ !["empty-output", "network", "nonzero-exit", "unknown"].includes(result.failureCategory))
59
+ return false;
60
+ return true;
61
+ }
34
62
  /**
35
63
  * The writer session burned an excessive token budget (a read-edit-test loop
36
64
  * that never converged) and still failed. Distinct from empty-output so the
@@ -69,8 +97,6 @@ function readWriterThinkingExhaustionEvidence(result) {
69
97
  export function isWriterThinkingExhausted(result, mapped, changeManifestChangedFiles) {
70
98
  if (mapped.ok)
71
99
  return false;
72
- if (mapped.failureCategory !== "empty-output")
73
- return false;
74
100
  const evidence = readWriterThinkingExhaustionEvidence(result);
75
101
  if (evidence.stopReason !== "length")
76
102
  return false;
@@ -78,6 +104,13 @@ export function isWriterThinkingExhausted(result, mapped, changeManifestChangedF
78
104
  return false;
79
105
  if ((evidence.writeToolCallCount ?? 0) !== 0)
80
106
  return false;
107
+ // Gateways sometimes classify a length-stopped stream as `network` or
108
+ // `nonzero-exit` because the terminal event is carried in stderr. The
109
+ // provider evidence is stronger than that transport label when no write
110
+ // tool was called and the run produced no diff.
111
+ if (mapped.failureCategory &&
112
+ !["empty-output", "network", "nonzero-exit", "unknown"].includes(mapped.failureCategory))
113
+ return false;
81
114
  if (changeManifestChangedFiles === undefined)
82
115
  return false;
83
116
  if (changeManifestChangedFiles.length !== 0)
@@ -2904,6 +2937,8 @@ const FRONTEND_PLAN_UX_LOCAL_MAX_RECORD_CALLS = 6;
2904
2937
  // safety bound, but do not let the old 32-session ceiling skip finalize for a
2905
2938
  // legitimate large plan.
2906
2939
  const FRONTEND_PLAN_BATCH_MAX_SESSIONS = 128;
2940
+ const FRONTEND_PLAN_SMALL_MAX_REQUIREMENTS = 8;
2941
+ const FRONTEND_PLAN_SMALL_MAX_ESTIMATED_CALLS = 24;
2907
2942
  function compactPromptString(value, maxChars) {
2908
2943
  if (typeof value !== "string" || value.trim().length === 0)
2909
2944
  return undefined;
@@ -2920,6 +2955,45 @@ function compactPromptStringArray(value, maxEntries = 12, maxChars = 180) {
2920
2955
  .filter((item) => item !== undefined)
2921
2956
  .slice(0, maxEntries);
2922
2957
  }
2958
+ function countFrontendPlanTargetSurfaces(basePrompt) {
2959
+ const match = /<frontend_plan_input>[\s\S]*?<\/frontend_plan_input>/.exec(basePrompt);
2960
+ if (!match)
2961
+ return undefined;
2962
+ for (const line of match[0].split(/\r?\n/)) {
2963
+ try {
2964
+ const payload = JSON.parse(line);
2965
+ if (Array.isArray(payload.targetSurface))
2966
+ return payload.targetSurface.length;
2967
+ }
2968
+ catch {
2969
+ // surrounding lines are prose
2970
+ }
2971
+ }
2972
+ return undefined;
2973
+ }
2974
+ /**
2975
+ * Deterministically extract repository file paths that frozen verification
2976
+ * commands operate on (`--config <file>`, `node --check <file>`). A frozen
2977
+ * command whose referenced file is outside the planner's verification targets
2978
+ * can never run: the admission writeSet derives from those targets, so the
2979
+ * writer is not authorized to create the file and verify-shell fails ~20
2980
+ * minutes later. Surfacing the gap as plan missing-facts lets the coverage or
2981
+ * compact session record the missing verification target inside the same
2982
+ * attempt instead.
2983
+ */
2984
+ export function collectFrontendVerificationCommandFiles(basePrompt) {
2985
+ const files = new Set();
2986
+ const configRe = /--config\s+([\w@./-]+\.(?:js|mjs|cjs|ts|json))/g;
2987
+ const checkRe = /node\s+--check\s+([\w@./-]+\.(?:js|mjs|cjs))/g;
2988
+ for (const re of [configRe, checkRe]) {
2989
+ for (const match of basePrompt.matchAll(re)) {
2990
+ const file = match[1];
2991
+ if (file && file.includes("/"))
2992
+ files.add(file);
2993
+ }
2994
+ }
2995
+ return [...files].sort();
2996
+ }
2923
2997
  /**
2924
2998
  * Remove the repeated full planner input from a coverage batch. The normal
2925
2999
  * plan prompt already contains a bounded JSON handoff, but repeating all
@@ -3226,6 +3300,7 @@ export async function runFrontendPlanSegmentedSessions(input) {
3226
3300
  const coverageSegment = FRONTEND_PLAN_SEGMENTS.find((segment) => segment.id === "coverage");
3227
3301
  const uxSegment = FRONTEND_PLAN_SEGMENTS.find((segment) => segment.id === "ux-local");
3228
3302
  const globalMockDataSegment = FRONTEND_PLAN_SEGMENTS.find((segment) => segment.id === "global-mock-data");
3303
+ const finalizeSegment = FRONTEND_PLAN_SEGMENTS.find((segment) => segment.id === "finalize");
3229
3304
  const allRequirementIds = input.requirementIds ?? [];
3230
3305
  const buildPhasePrompt = (segment, missing = []) => {
3231
3306
  const compact = allRequirementIds.length > 0
@@ -3295,6 +3370,42 @@ export async function runFrontendPlanSegmentedSessions(input) {
3295
3370
  : []),
3296
3371
  ].filter(Boolean).join("\n\n");
3297
3372
  };
3373
+ const buildCompactLocalPrompt = (missing = []) => {
3374
+ const ledger = input.committedFacts
3375
+ ? compactFrontendPlanLedgerContext({
3376
+ committedFacts: input.committedFacts(),
3377
+ requirementIds: allRequirementIds,
3378
+ kinds: [
3379
+ "plan-requirement",
3380
+ "plan-verification-target",
3381
+ "component-choice",
3382
+ "state-flow",
3383
+ ],
3384
+ })
3385
+ : "";
3386
+ return [
3387
+ compactFrontendPlanPromptForRequirementSlice(input.basePrompt, allRequirementIds),
3388
+ "PLAN PHASE — compact local planning for a small frontend request.",
3389
+ "For every listed requirement, record plan-requirement and verification-target facts, then record the component-choice and state-flow facts needed by the observable UX. Do not read the repository or task source; use only the committed input above. Do not call finalize_plan in this session.",
3390
+ "TOOL-FIRST: your first assistant actions must be record_* tool calls, at most 2-3 facts per message. Do not draft the whole analysis before recording; if a fact is uncertain, record it with an evidence gap instead of reasoning longer.",
3391
+ ledger,
3392
+ ...(missing.length > 0
3393
+ ? [
3394
+ "MISSING-FACT QUEUE: repair ONLY these items, then re-check the compact local phase:",
3395
+ ...missing.map((item) => `- ${item.kind}${item.id ? ` ${item.id}` : ""} for ${item.requirementIds.join(", ")}: ${item.reason}`),
3396
+ ]
3397
+ : []),
3398
+ ].filter(Boolean).join("\n\n");
3399
+ };
3400
+ const compactFinalizeInstruction = "This is a small-request compact pass. Reconcile the committed local facts with route, data-flow, Mock/API, dependency and deviation policy, then call finalize_plan exactly once.";
3401
+ const buildCompactFinalizePrompt = (missing = []) => [buildPhasePrompt(finalizeSegment, missing), compactFinalizeInstruction].join("\n\n");
3402
+ const mapPlannerExhaustion = (r, committedAnyFacts) => isPlannerThinkingExhausted(r, committedAnyFacts)
3403
+ ? {
3404
+ ...r,
3405
+ failureCategory: PLANNER_THINKING_EXHAUSTED_CATEGORY,
3406
+ stderr: `${r.stderr}\n${PLANNER_THINKING_EXHAUSTED_CATEGORY}: stopReason=length, thinking observed, 0 typed facts committed; the batch ladder degraded the scope without converging — set thinking=off for this tier or switch to a non-thinking model`.trim(),
3407
+ }
3408
+ : r;
3298
3409
  // An empty list means "ledger unreadable / unknown" and falls back to one
3299
3410
  // unscoped coverage session; a non-empty list enables deterministic sharding.
3300
3411
  const requirementIdsProvided = input.requirementIds !== undefined && input.requirementIds.length > 0;
@@ -3307,7 +3418,42 @@ export async function runFrontendPlanSegmentedSessions(input) {
3307
3418
  : [];
3308
3419
  const incompleteRequirementIds = new Set(initialMissing.flatMap((item) => item.requirementIds));
3309
3420
  const coverageWorkIds = (input.requirementIds ?? []).filter((id) => pending.includes(id) || incompleteRequirementIds.has(id));
3310
- if (!requirementIdsProvided) {
3421
+ const estimatedCalls = (input.requirementIds ?? []).reduce((total, id) => total + Math.max(1, input.requirementCosts?.get(id) ?? 2), 0);
3422
+ const targetSurfaceCount = countFrontendPlanTargetSurfaces(input.basePrompt);
3423
+ // Small, single-surface requests do not benefit from six isolated Pi
3424
+ // sessions. Keep the typed ledger as the authority, but let one local
3425
+ // session establish requirement/UX facts and one final session establish
3426
+ // cross-cutting policy + finalize. The old sharded ladder remains available
3427
+ // for larger plans and for the unscoped compatibility path.
3428
+ const useCompactSmallPlan = requirementIdsProvided &&
3429
+ input.compactSmallPlan === true &&
3430
+ input.requirementCosts !== undefined &&
3431
+ estimatedCalls > 12 &&
3432
+ (input.requirementIds?.length ?? 0) <= FRONTEND_PLAN_SMALL_MAX_REQUIREMENTS &&
3433
+ estimatedCalls <= FRONTEND_PLAN_SMALL_MAX_ESTIMATED_CALLS &&
3434
+ targetSurfaceCount === 1;
3435
+ if (useCompactSmallPlan) {
3436
+ const compactLocalTools = new Set([
3437
+ "record_plan_requirement",
3438
+ "record_plan_verification_target",
3439
+ "record_plan_evidence_gap",
3440
+ "record_component_choice",
3441
+ "record_state_flow",
3442
+ "adopt_staged_fact",
3443
+ ]);
3444
+ queue.push({
3445
+ id: "compact-local",
3446
+ toolNames: compactLocalTools,
3447
+ requirementSlice: [...allRequirementIds],
3448
+ prompt: buildCompactLocalPrompt(),
3449
+ });
3450
+ queue.push({
3451
+ id: "finalize",
3452
+ toolNames: null,
3453
+ prompt: buildCompactFinalizePrompt(),
3454
+ });
3455
+ }
3456
+ else if (!requirementIdsProvided) {
3311
3457
  queue.push({
3312
3458
  id: "coverage",
3313
3459
  toolNames: coverageSegment.toolNames,
@@ -3329,41 +3475,45 @@ export async function runFrontendPlanSegmentedSessions(input) {
3329
3475
  prompt: buildCoveragePrompt(slice),
3330
3476
  }));
3331
3477
  }
3332
- const uxCosts = new Map((input.requirementIds ?? []).map((id) => [
3333
- id,
3334
- Math.max(2, Math.min(3, input.requirementCosts?.get(id) ?? 2)),
3335
- ]));
3336
- const uxSlices = requirementIdsProvided
3337
- ? batchFrontendPlanRequirements({
3338
- requirementIds: input.requirementIds,
3339
- maxEstimatedRecordCalls: FRONTEND_PLAN_UX_LOCAL_MAX_RECORD_CALLS,
3340
- maxRequirements: 3,
3341
- requirementCosts: uxCosts,
3342
- })
3343
- : [[]];
3344
- uxSlices.forEach((slice, batchIndex) => queue.push({
3345
- id: `ux-local-${batchIndex + 1}`,
3346
- toolNames: uxSegment.toolNames,
3347
- ...(slice.length > 0 ? { requirementSlice: slice } : {}),
3348
- prompt: slice.length > 0
3349
- ? buildUxPrompt(slice)
3350
- : buildPhasePrompt(uxSegment),
3351
- }));
3352
- for (const segment of FRONTEND_PLAN_SEGMENTS) {
3353
- if (["coverage", "ux-local", "finalize"].includes(segment.id))
3354
- continue;
3478
+ if (useCompactSmallPlan) {
3479
+ // Compact mode already queued both sessions above.
3480
+ }
3481
+ else {
3482
+ const uxCosts = new Map((input.requirementIds ?? []).map((id) => [
3483
+ id,
3484
+ Math.max(2, Math.min(3, input.requirementCosts?.get(id) ?? 2)),
3485
+ ]));
3486
+ const uxSlices = requirementIdsProvided
3487
+ ? batchFrontendPlanRequirements({
3488
+ requirementIds: input.requirementIds,
3489
+ maxEstimatedRecordCalls: FRONTEND_PLAN_UX_LOCAL_MAX_RECORD_CALLS,
3490
+ maxRequirements: 3,
3491
+ requirementCosts: uxCosts,
3492
+ })
3493
+ : [[]];
3494
+ uxSlices.forEach((slice, batchIndex) => queue.push({
3495
+ id: `ux-local-${batchIndex + 1}`,
3496
+ toolNames: uxSegment.toolNames,
3497
+ ...(slice.length > 0 ? { requirementSlice: slice } : {}),
3498
+ prompt: slice.length > 0
3499
+ ? buildUxPrompt(slice)
3500
+ : buildPhasePrompt(uxSegment),
3501
+ }));
3502
+ for (const segment of FRONTEND_PLAN_SEGMENTS) {
3503
+ if (["coverage", "ux-local", "finalize"].includes(segment.id))
3504
+ continue;
3505
+ queue.push({
3506
+ id: segment.id,
3507
+ toolNames: segment.toolNames,
3508
+ prompt: buildPhasePrompt(segment),
3509
+ });
3510
+ }
3355
3511
  queue.push({
3356
- id: segment.id,
3357
- toolNames: segment.toolNames,
3358
- prompt: buildPhasePrompt(segment),
3512
+ id: finalizeSegment.id,
3513
+ toolNames: finalizeSegment.toolNames,
3514
+ prompt: buildPhasePrompt(finalizeSegment),
3359
3515
  });
3360
3516
  }
3361
- const finalizeSegment = FRONTEND_PLAN_SEGMENTS.find((segment) => segment.id === "finalize");
3362
- queue.push({
3363
- id: finalizeSegment.id,
3364
- toolNames: finalizeSegment.toolNames,
3365
- prompt: buildPhasePrompt(finalizeSegment),
3366
- });
3367
3517
  let last;
3368
3518
  let index = 0;
3369
3519
  let invocationCount = 0;
@@ -3389,6 +3539,9 @@ export async function runFrontendPlanSegmentedSessions(input) {
3389
3539
  const promptSlice = remaining.length > 0 ? remaining : session.coverageSlice;
3390
3540
  prompt = buildCoveragePrompt(promptSlice, session.missingFacts ?? preexistingMissing);
3391
3541
  }
3542
+ else if (session.id === "compact-local") {
3543
+ prompt = buildCompactLocalPrompt(session.missingFacts);
3544
+ }
3392
3545
  else if (session.id.startsWith("ux-local-")) {
3393
3546
  // Build this at execution time: coverage facts are committed by the
3394
3547
  // preceding sessions and must be visible to the UX-local model.
@@ -3402,11 +3555,14 @@ export async function runFrontendPlanSegmentedSessions(input) {
3402
3555
  // stale queue entry from dropping facts written by earlier phases.
3403
3556
  const segment = FRONTEND_PLAN_SEGMENTS.find((candidate) => candidate.id === session.id);
3404
3557
  if (segment) {
3405
- prompt = buildPhasePrompt(segment, session.missingFacts);
3558
+ prompt =
3559
+ session.id === "finalize" && useCompactSmallPlan
3560
+ ? buildCompactFinalizePrompt(session.missingFacts)
3561
+ : buildPhasePrompt(segment, session.missingFacts);
3406
3562
  }
3407
3563
  }
3408
3564
  const committedBefore = input.committedFactCount();
3409
- input.setActiveRequirementScope?.(session.id.startsWith("ux-local-")
3565
+ input.setActiveRequirementScope?.(session.id === "compact-local" || session.id.startsWith("ux-local-")
3410
3566
  ? session.requirementSlice ?? []
3411
3567
  : []);
3412
3568
  if (invocationCount >= FRONTEND_PLAN_BATCH_MAX_SESSIONS)
@@ -3444,27 +3600,67 @@ export async function runFrontendPlanSegmentedSessions(input) {
3444
3600
  committedFacts: input.committedFacts(),
3445
3601
  })
3446
3602
  : [];
3603
+ const isCompactLocalSession = session.id === "compact-local";
3447
3604
  const isUxLocalSession = session.id.startsWith("ux-local-");
3448
- const missingPhase = isUxLocalSession && session.requirementSlice && input.committedFacts
3449
- ? collectFrontendPlanPhaseMissingFacts({
3450
- phase: "ux-local",
3451
- requirementIds: session.requirementSlice,
3452
- committedFacts: input.committedFacts(),
3453
- behaviorRequiredRequirementIds: input.behaviorRequiredRequirementIds,
3454
- })
3455
- : session.id === "global-mock-data" && allRequirementIds.length > 0 && input.committedFacts
3456
- ? collectFrontendPlanPhaseMissingFacts({
3457
- phase: "global-mock-data",
3605
+ const missingPhase = isCompactLocalSession && input.committedFacts
3606
+ ? [
3607
+ ...collectFrontendPlanMissingFacts({
3458
3608
  requirementIds: allRequirementIds,
3459
3609
  committedFacts: input.committedFacts(),
3610
+ }),
3611
+ ...collectFrontendPlanPhaseMissingFacts({
3612
+ phase: "ux-local",
3613
+ requirementIds: allRequirementIds,
3614
+ committedFacts: input.committedFacts(),
3615
+ behaviorRequiredRequirementIds: input.behaviorRequiredRequirementIds,
3616
+ }),
3617
+ ]
3618
+ : isUxLocalSession && session.requirementSlice && input.committedFacts
3619
+ ? collectFrontendPlanPhaseMissingFacts({
3620
+ phase: "ux-local",
3621
+ requirementIds: session.requirementSlice,
3622
+ committedFacts: input.committedFacts(),
3623
+ behaviorRequiredRequirementIds: input.behaviorRequiredRequirementIds,
3460
3624
  })
3461
- : [];
3625
+ : session.id === "global-mock-data" && allRequirementIds.length > 0 && input.committedFacts
3626
+ ? collectFrontendPlanPhaseMissingFacts({
3627
+ phase: "global-mock-data",
3628
+ requirementIds: allRequirementIds,
3629
+ committedFacts: input.committedFacts(),
3630
+ })
3631
+ : [];
3462
3632
  const missingPhaseFacts = [...missingCoverage, ...missingPhase];
3633
+ // Frozen verification commands must operate on files the planner has
3634
+ // committed verification targets for; otherwise the admission writeSet
3635
+ // cannot authorize the file and verify-shell deterministically fails.
3636
+ const verificationCommandFiles = collectFrontendVerificationCommandFiles(input.basePrompt);
3637
+ if (verificationCommandFiles.length > 0 && allRequirementIds.length > 0 && input.committedFacts) {
3638
+ const coveredVerificationFiles = new Set(input
3639
+ .committedFacts()
3640
+ .map(committedFactFromPlanRecord)
3641
+ .filter((fact) => Boolean(fact && fact.origin === "plan" && fact.kind === "plan-verification-target"))
3642
+ .map((fact) => fact.entry?.file)
3643
+ .filter((file) => typeof file === "string"));
3644
+ for (const file of verificationCommandFiles) {
3645
+ if (coveredVerificationFiles.has(file))
3646
+ continue;
3647
+ missingPhaseFacts.push({
3648
+ kind: "plan-verification-target",
3649
+ id: file,
3650
+ requirementIds: allRequirementIds.slice(0, 1),
3651
+ reason: `frozen verification command references ${file} but no committed verification target covers it; record a static verification target for this file so it joins the writeSet`,
3652
+ });
3653
+ }
3654
+ }
3463
3655
  const promptForMissingPhase = (missing) => session.coverageOnly
3464
3656
  ? buildCoveragePrompt(session.coverageSlice ?? [], missing)
3465
- : isUxLocalSession
3466
- ? buildUxPrompt(session.requirementSlice ?? [], missing)
3467
- : buildPhasePrompt(globalMockDataSegment, missing);
3657
+ : isCompactLocalSession
3658
+ ? buildCompactLocalPrompt(missing)
3659
+ : session.id === "finalize" && useCompactSmallPlan
3660
+ ? buildCompactFinalizePrompt(missing)
3661
+ : isUxLocalSession
3662
+ ? buildUxPrompt(session.requirementSlice ?? [], missing)
3663
+ : buildPhasePrompt(globalMockDataSegment, missing);
3468
3664
  if (result.ok) {
3469
3665
  if (missingPhaseFacts.length === 0) {
3470
3666
  index += 1;
@@ -3508,10 +3704,36 @@ export async function runFrontendPlanSegmentedSessions(input) {
3508
3704
  index += 1;
3509
3705
  continue;
3510
3706
  }
3707
+ // A length-stopped, fact-less compact session is the planner variant of
3708
+ // writer-thinking-exhausted. Give the same scope exactly one tool-first
3709
+ // retry before the split below re-batches the requirements, because one
3710
+ // reinforced full-scope pass is cheaper than re-planning split halves.
3711
+ const plannerThinkingBurn = isCompactLocalSession &&
3712
+ committedAfter === committedBefore &&
3713
+ !(result.assistantText ?? "").trim() &&
3714
+ !result.stderr.trim() &&
3715
+ !result.timedOut &&
3716
+ readWriterThinkingExhaustionEvidence(result).stopReason === "length";
3717
+ if (plannerThinkingBurn && (session.retryCount ?? 0) < 1) {
3718
+ queue[index] = {
3719
+ ...session,
3720
+ retryCount: (session.retryCount ?? 0) + 1,
3721
+ prompt: buildCompactLocalPrompt(),
3722
+ };
3723
+ continue;
3724
+ }
3511
3725
  // Option 5: a multi-requirement coverage batch that failed with ZERO
3512
3726
  // new facts and no provider stderr is the upfront-reasoning burn —
3513
- // halve the slice and retry instead of failing the attempt.
3514
- const coverageSlice = session.coverageOnly ? session.coverageSlice : undefined;
3727
+ // halve the slice and retry instead of failing the attempt. The compact
3728
+ // local session participates through its requirementSlice so a
3729
+ // thinking-burned small plan degrades into smaller batched sessions
3730
+ // instead of replaying one full-scope prompt until the repair budget
3731
+ // runs out.
3732
+ const coverageSlice = session.coverageOnly
3733
+ ? session.coverageSlice
3734
+ : isCompactLocalSession
3735
+ ? session.requirementSlice
3736
+ : undefined;
3515
3737
  const zeroProgressBurn = coverageSlice !== undefined &&
3516
3738
  coverageSlice.length > 1 &&
3517
3739
  committedAfter === committedBefore &&
@@ -3522,6 +3744,26 @@ export async function runFrontendPlanSegmentedSessions(input) {
3522
3744
  const half = Math.ceil(coverageSlice.length / 2);
3523
3745
  const firstSlice = coverageSlice.slice(0, half);
3524
3746
  const secondSlice = coverageSlice.slice(half);
3747
+ if (isCompactLocalSession) {
3748
+ // The split halves leave compact mode: continue them as ordinary
3749
+ // coverage sessions so every downstream ladder branch applies.
3750
+ queue.splice(index, 1, {
3751
+ ...session,
3752
+ id: `coverage-compact-split-1`,
3753
+ coverageOnly: true,
3754
+ coverageSlice: firstSlice,
3755
+ requirementSlice: firstSlice,
3756
+ prompt: buildCoveragePrompt(firstSlice),
3757
+ }, {
3758
+ ...session,
3759
+ id: `coverage-compact-split-2`,
3760
+ coverageOnly: true,
3761
+ coverageSlice: secondSlice,
3762
+ requirementSlice: secondSlice,
3763
+ prompt: buildCoveragePrompt(secondSlice),
3764
+ });
3765
+ continue;
3766
+ }
3525
3767
  queue.splice(index, 1, {
3526
3768
  ...session,
3527
3769
  coverageSlice: firstSlice,
@@ -3546,7 +3788,7 @@ export async function runFrontendPlanSegmentedSessions(input) {
3546
3788
  };
3547
3789
  continue;
3548
3790
  }
3549
- return result;
3791
+ return mapPlannerExhaustion(result, committedAfter > committedBefore);
3550
3792
  }
3551
3793
  if (index < queue.length) {
3552
3794
  return {
@@ -3571,13 +3813,13 @@ export async function runFrontendPlanSegmentedSessions(input) {
3571
3813
  failureCategory: "invalid-output",
3572
3814
  };
3573
3815
  }
3574
- return (last ?? {
3816
+ return mapPlannerExhaustion(last ?? {
3575
3817
  ok: false,
3576
3818
  stdout: "",
3577
3819
  stderr: "frontend plan segmentation produced no session",
3578
3820
  failureCategory: "empty-output",
3579
3821
  durationMs: 0,
3580
- });
3822
+ }, false);
3581
3823
  }
3582
3824
  export async function executeDagPiNode(input, meta, piStepFn = executePiStep, writeGuardDependencies = DEFAULT_DAG_PI_WRITE_GUARD_DEPENDENCIES) {
3583
3825
  const started = Date.now();
@@ -4060,6 +4302,7 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
4060
4302
  committedFactCount: () => planLedgerTools.committedFactCount(),
4061
4303
  requirementIds: planRequirementIds,
4062
4304
  requirementCosts: planRequirementCosts,
4305
+ compactSmallPlan: true,
4063
4306
  committedRequirementIds: () => planLedgerTools.committedRequirementIds(),
4064
4307
  committedFacts: () => planLedgerTools.committedFacts(),
4065
4308
  behaviorRequiredRequirementIds,
@@ -4079,6 +4322,18 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
4079
4322
  ...(writerToolPolicy ? { writerToolPolicy } : {}),
4080
4323
  });
4081
4324
  }
4325
+ try {
4326
+ const verificationCommandFiles = collectFrontendVerificationCommandFiles(input.prompt);
4327
+ const debugPath = path.join(meta.runDir, input.task.id, "verification-command-coverage.json");
4328
+ await import("node:fs/promises").then(({ writeFile, mkdir }) => mkdir(path.dirname(debugPath), { recursive: true }).then(() => writeFile(debugPath, `${JSON.stringify({
4329
+ schemaVersion: 1,
4330
+ basePromptChars: input.prompt.length,
4331
+ verificationCommandFiles,
4332
+ }, null, 2)}\n`)));
4333
+ }
4334
+ catch {
4335
+ // best-effort diagnostic breadcrumb
4336
+ }
4082
4337
  }
4083
4338
  catch (error) {
4084
4339
  if (playwrightToolContext) {
@@ -4860,6 +5115,8 @@ export function mapPiResultToDagNodeResult(result, firstProtocolLine) {
4860
5115
  tokensUsed: result.tokensUsed,
4861
5116
  parsedEvents: result.parsedEvents,
4862
5117
  stopReason: readWriterThinkingExhaustionEvidence(result).stopReason,
5118
+ thinkingObserved: readWriterThinkingExhaustionEvidence(result).thinkingObserved,
5119
+ writeToolCallCount: readWriterThinkingExhaustionEvidence(result).writeToolCallCount,
4863
5120
  };
4864
5121
  }
4865
5122
  function canonicalizeProtocolFirstLine(assistantText, firstProtocolLine) {