@tea-agent/loop-agent 0.42.0-next.8 → 0.42.0-next.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +9 -0
- package/dist/application/dag/run-dag.js +40 -0
- package/dist/build-stamp.json +3 -3
- package/dist/executors/dag-pi-executor.js +313 -56
- package/dist/executors/shell-executor.js +47 -19
- package/dist/worker/observe/routes.js +4 -0
- package/dist/workflows/dag/backend-test-case-coverage-analysis.js +18 -0
- package/dist/workflows/dag/backend-test-scenario-param.js +30 -1
- package/dist/workflows/dag/frontend-recovery-plan.js +2 -1
- package/dist/workflows/dag/frontend-recovery-run.js +33 -5
- package/dist/workflows/dag/frontend-writer-admission.js +44 -10
- package/dist/workflows/dag/init-hybrid.js +3 -2
- package/dist/workflows/dag/node-execution.js +76 -0
- package/dist/workflows/dag/rerun-feedback.js +59 -0
- package/dist/workflows/dag/rerun-task.js +8 -2
- package/dist/workflows/dag/runner.js +18 -11
- package/dist/workflows/dag/scheduler.js +21 -6
- package/docs/templates/backend-test-dag.json +1 -1
- package/package.json +1 -1
package/CHANGELOG.md
CHANGED
|
@@ -2,6 +2,15 @@
|
|
|
2
2
|
|
|
3
3
|
## [Unreleased]
|
|
4
4
|
|
|
5
|
+
- 冒烟再暴露 writeSet 授权缺口:冻结验证命令引用的基建文件(如 `--config` 的 vitest 配置)不在 plan verification targets 内时,writer 被具体 writeSet 拦住无法创建、verify-shell 必然失败。现在 admission 生效边界确定性并入冻结命令引用的文件(机械提取、仍受 allowedPaths/forbiddenPaths 约束),writer 聚焦检查失败后可自行补齐基建文件。
|
|
6
|
+
- 修复 /inspect/* 静态资源(operator-chrome.js/css、观测文档)缺少 Cache-Control 的问题:现在与 Console SPA 一律 `no-store`。此前浏览器启发式缓存会让 Console 更新/重启后继续挂旧版导航模块,表现为顶部导航「回退」到只有操作/观测、缺少对话入口的旧版布局。
|
|
7
|
+
|
|
8
|
+
- 冻结验证命令引用的文件(如 `--config` 的 vitest 配置)现在必须在 plan 阶段就有 verification target 覆盖:缺失会作为缺失事实驱动 coverage/compact/finalize 会话当轮补记并进入 admission writeSet,无法补齐则在 plan 阶段确定性失败,不再等到 verify-shell 才以 missing config 失败。
|
|
9
|
+
- 修复 writer 终态失败 run 无法重跑的问题:rerun 反馈的 writer 失败通道把 `attemptSummary.completedPasses` 硬编码为 0,被 `dagRerunFeedbackSchema` 的 positive 校验拒绝,导致 `dag rerun-task` 在解析父 run 反馈时崩溃;现与相邻通道一致取 1。
|
|
10
|
+
- 冒烟再暴露并修复 planner 思考烧尽:compact Plan 首会话因 `stopReason=length` 只思考零提交时,先做一次 tool-first 同 scope 重试,仍失败则机械降级为按需求减半的 coverage 分批会话;耗尽后归类为 `planner-thinking-exhausted` 并给出 thinking=off/换模型建议,不再原样重放同一 prompt 烧完修复预算。
|
|
11
|
+
- 继续加固前端恢复:生产 `runDag` 现在使用共享 Console operation store 执行带反馈的 `dag rerun-task`,避免恢复 descriptor 无法解析;compact Plan 仅在单 target surface 生效,finalize 提示与 concrete writeSet 路径统一规范化。
|
|
12
|
+
- 修复前端冒烟暴露的 P0/P1 流程问题:设计评审 `request_design_changes` 现在 fail-closed 阻断 writer,自动恢复从 Plan 重启并携带 findings/evidence;admission 冻结的 concrete writeSet 同时成为 writer 的实际授权边界,legacy `admitted` 不再放行;长度耗尽且零写入的 provider 结果按 `writer-thinking-exhausted` 归类。生产小型 Plan(≤8 个需求且估算调用量受控)合并为两次 Pi session,并在 session 间刷新 typed ledger 上下文;Scout 重试仅补 unresolved 路径,减少重复侦查与上下文消耗。
|
|
13
|
+
- 修复 backend-test 对 `payload = {...}` 经 `dict(payload)` 和 `**kwargs` wrapper 透传时漏观测请求字段的问题:保守传播 wrapper 参数别名与浅拷贝 payload,并按分支实际赋值观察 scenario 字段;继续保持 payload SAFE、EXACT_1_TO_1、scenario MATCH 与 fail-closed 门禁。计划生成提示词同时明确无有限分区域时省略整个 `## Scenario Partitions` 章节,避免生成 prose-only 空节触发 N3 阻断。
|
|
5
14
|
- 修复 Operator Chat 打开已丢失会话时把孤儿 JSONL 升级成 503、以及删会话后残留 workspace 选择被误报为跨工作区冲突的问题:缺失会话文件现在自愈为 404 并清掉记录,前端恢复路径安静忘记该会话。
|
|
6
15
|
- 修复前端 `--from-text` 任务缺少 source-fidelity ledger、DAG 生成后降级为 v1 并在 `frontend-contract-pi` 0ms 失败的问题:文本入口现在持久化可校验的 v2 绑定输入,init-hybrid 复用任务自有需求源;Console 默认模型与 Pi readiness 同时排除无凭据 provider,避免把不可执行模型误报为可用。
|
|
7
16
|
- 常规回归 `npm test` 默认跳过约 6 分钟的 I/O-heavy 测试池(仍跑 fast + integration);完整三池改为 `npm run test:full` / `LOOP_AGENT_FULL_TEST=1`。本地 `ci.sh`、GitHub PR 与 release/next 发布门禁仍跑全量,不降低交付验证。
|
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import path from "node:path";
|
|
1
2
|
import { createDagEventObserver } from "../../workflows/dag/event-observer.js";
|
|
2
3
|
import { composeDagRunObservers } from "../../workflows/dag/observer-compose.js";
|
|
3
4
|
import { createDagCanvasObserver, resolveCanvasPath, } from "../../workflows/dag/canvas-observer.js";
|
|
@@ -28,6 +29,40 @@ function buildRunDagNextSteps(runId) {
|
|
|
28
29
|
`loop-agent dag report --run-id ${runId} --markdown # advanced forensic`,
|
|
29
30
|
];
|
|
30
31
|
}
|
|
32
|
+
async function buildDefaultFrontendRecoveryDeps(repoRoot) {
|
|
33
|
+
// Keep the application layer free of a static worker→application cycle. The
|
|
34
|
+
// recovery controller is only needed after a frontend terminal failure.
|
|
35
|
+
const [{ openConsoleAppData }, { OperationStore }, { createOperationEventStore }, { LoopAgentClient }, { resolveSiblingLoopAgentBin }] = await Promise.all([
|
|
36
|
+
import("../../infrastructure/console/app-data.js"),
|
|
37
|
+
import("../../infrastructure/console/operation-store.js"),
|
|
38
|
+
import("../../worker/console/operation-sse.js"),
|
|
39
|
+
import("../../worker/loop-agent/loop-agent-client.js"),
|
|
40
|
+
import("../../worker/console/sibling-controller.js"),
|
|
41
|
+
]);
|
|
42
|
+
const appData = openConsoleAppData({ repoRoot });
|
|
43
|
+
const store = new OperationStore(appData);
|
|
44
|
+
const events = createOperationEventStore({
|
|
45
|
+
persistenceDir: path.join(appData.operations, "events"),
|
|
46
|
+
});
|
|
47
|
+
const client = new LoopAgentClient({
|
|
48
|
+
loopAgentBin: resolveSiblingLoopAgentBin(),
|
|
49
|
+
artifactRoot: path.join(appData.root, "client-artifacts", appData.fingerprint),
|
|
50
|
+
resolveIdentity: true,
|
|
51
|
+
});
|
|
52
|
+
return {
|
|
53
|
+
store,
|
|
54
|
+
operationRunnerDeps: {
|
|
55
|
+
store,
|
|
56
|
+
events,
|
|
57
|
+
client,
|
|
58
|
+
repoRoot,
|
|
59
|
+
env: {
|
|
60
|
+
...process.env,
|
|
61
|
+
LOOP_CONSOLE_APP_DATA: appData.root,
|
|
62
|
+
},
|
|
63
|
+
},
|
|
64
|
+
};
|
|
65
|
+
}
|
|
31
66
|
export async function runDagUseCase(input) {
|
|
32
67
|
const spec = await loadDagSpecFromFile(input.dagPath);
|
|
33
68
|
assertValidDagSpec(spec);
|
|
@@ -80,6 +115,10 @@ export async function runDagUseCase(input) {
|
|
|
80
115
|
...(canvasError ? { canvasError } : {}),
|
|
81
116
|
};
|
|
82
117
|
}
|
|
118
|
+
const recovery = input.recovery ??
|
|
119
|
+
(!input.initOnly && spec.tasks.some((task) => task.id === "frontend-implement-pi")
|
|
120
|
+
? await buildDefaultFrontendRecoveryDeps(input.cwd)
|
|
121
|
+
: undefined);
|
|
83
122
|
const summary = await runDag(spec, {
|
|
84
123
|
cwd: input.cwd,
|
|
85
124
|
initOnly: input.initOnly,
|
|
@@ -90,6 +129,7 @@ export async function runDagUseCase(input) {
|
|
|
90
129
|
...(input.workerAssociation
|
|
91
130
|
? { workerAssociation: input.workerAssociation }
|
|
92
131
|
: {}),
|
|
132
|
+
...(recovery ? { recovery } : {}),
|
|
93
133
|
});
|
|
94
134
|
const canvasError = await flushCanvasSafely(canvas);
|
|
95
135
|
if (resolvedCanvasPath) {
|
package/dist/build-stamp.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"schemaVersion": 1,
|
|
3
|
-
"version": "0.42.0-next.
|
|
4
|
-
"gitSha": "
|
|
5
|
-
"builtAt": "2026-09-
|
|
3
|
+
"version": "0.42.0-next.9",
|
|
4
|
+
"gitSha": "13597aa31afc1faccb9a5eaca296d57d7ab93c2e",
|
|
5
|
+
"builtAt": "2026-09-03T05:18:16.410Z"
|
|
6
6
|
}
|
|
@@ -31,6 +31,34 @@ import { writeEffectiveContextReceipt } from "../workflows/dag/context-receipt.j
|
|
|
31
31
|
* recoverable partial-write-set (incomplete-write-set) upgrade.
|
|
32
32
|
*/
|
|
33
33
|
export const WRITER_THINKING_EXHAUSTED_CATEGORY = "writer-thinking-exhausted";
|
|
34
|
+
/**
|
|
35
|
+
* Planner classification mirroring writer-thinking-exhausted: a read-only
|
|
36
|
+
* planning session stopped on length, observed thinking, and committed zero
|
|
37
|
+
* typed facts with no assistant text. By the time this survives the segmented
|
|
38
|
+
* ladder the scope has already been degraded, so the durable fix is a
|
|
39
|
+
* thinking-capped or non-thinking model for the tier — not another replay of
|
|
40
|
+
* the same full-scope prompt.
|
|
41
|
+
*/
|
|
42
|
+
export const PLANNER_THINKING_EXHAUSTED_CATEGORY = "planner-thinking-exhausted";
|
|
43
|
+
export function isPlannerThinkingExhausted(result, committedAnyFacts) {
|
|
44
|
+
if (result.ok)
|
|
45
|
+
return false;
|
|
46
|
+
const evidence = readWriterThinkingExhaustionEvidence(result);
|
|
47
|
+
if (evidence.stopReason !== "length")
|
|
48
|
+
return false;
|
|
49
|
+
if (evidence.thinkingObserved !== true)
|
|
50
|
+
return false;
|
|
51
|
+
if (committedAnyFacts)
|
|
52
|
+
return false;
|
|
53
|
+
if ((result.assistantText ?? "").trim())
|
|
54
|
+
return false;
|
|
55
|
+
// Gateways sometimes relabel a length-stopped stream as `network` or
|
|
56
|
+
// `nonzero-exit`; provider evidence outweighs the transport label.
|
|
57
|
+
if (result.failureCategory &&
|
|
58
|
+
!["empty-output", "network", "nonzero-exit", "unknown"].includes(result.failureCategory))
|
|
59
|
+
return false;
|
|
60
|
+
return true;
|
|
61
|
+
}
|
|
34
62
|
/**
|
|
35
63
|
* The writer session burned an excessive token budget (a read-edit-test loop
|
|
36
64
|
* that never converged) and still failed. Distinct from empty-output so the
|
|
@@ -69,8 +97,6 @@ function readWriterThinkingExhaustionEvidence(result) {
|
|
|
69
97
|
export function isWriterThinkingExhausted(result, mapped, changeManifestChangedFiles) {
|
|
70
98
|
if (mapped.ok)
|
|
71
99
|
return false;
|
|
72
|
-
if (mapped.failureCategory !== "empty-output")
|
|
73
|
-
return false;
|
|
74
100
|
const evidence = readWriterThinkingExhaustionEvidence(result);
|
|
75
101
|
if (evidence.stopReason !== "length")
|
|
76
102
|
return false;
|
|
@@ -78,6 +104,13 @@ export function isWriterThinkingExhausted(result, mapped, changeManifestChangedF
|
|
|
78
104
|
return false;
|
|
79
105
|
if ((evidence.writeToolCallCount ?? 0) !== 0)
|
|
80
106
|
return false;
|
|
107
|
+
// Gateways sometimes classify a length-stopped stream as `network` or
|
|
108
|
+
// `nonzero-exit` because the terminal event is carried in stderr. The
|
|
109
|
+
// provider evidence is stronger than that transport label when no write
|
|
110
|
+
// tool was called and the run produced no diff.
|
|
111
|
+
if (mapped.failureCategory &&
|
|
112
|
+
!["empty-output", "network", "nonzero-exit", "unknown"].includes(mapped.failureCategory))
|
|
113
|
+
return false;
|
|
81
114
|
if (changeManifestChangedFiles === undefined)
|
|
82
115
|
return false;
|
|
83
116
|
if (changeManifestChangedFiles.length !== 0)
|
|
@@ -2904,6 +2937,8 @@ const FRONTEND_PLAN_UX_LOCAL_MAX_RECORD_CALLS = 6;
|
|
|
2904
2937
|
// safety bound, but do not let the old 32-session ceiling skip finalize for a
|
|
2905
2938
|
// legitimate large plan.
|
|
2906
2939
|
const FRONTEND_PLAN_BATCH_MAX_SESSIONS = 128;
|
|
2940
|
+
const FRONTEND_PLAN_SMALL_MAX_REQUIREMENTS = 8;
|
|
2941
|
+
const FRONTEND_PLAN_SMALL_MAX_ESTIMATED_CALLS = 24;
|
|
2907
2942
|
function compactPromptString(value, maxChars) {
|
|
2908
2943
|
if (typeof value !== "string" || value.trim().length === 0)
|
|
2909
2944
|
return undefined;
|
|
@@ -2920,6 +2955,45 @@ function compactPromptStringArray(value, maxEntries = 12, maxChars = 180) {
|
|
|
2920
2955
|
.filter((item) => item !== undefined)
|
|
2921
2956
|
.slice(0, maxEntries);
|
|
2922
2957
|
}
|
|
2958
|
+
function countFrontendPlanTargetSurfaces(basePrompt) {
|
|
2959
|
+
const match = /<frontend_plan_input>[\s\S]*?<\/frontend_plan_input>/.exec(basePrompt);
|
|
2960
|
+
if (!match)
|
|
2961
|
+
return undefined;
|
|
2962
|
+
for (const line of match[0].split(/\r?\n/)) {
|
|
2963
|
+
try {
|
|
2964
|
+
const payload = JSON.parse(line);
|
|
2965
|
+
if (Array.isArray(payload.targetSurface))
|
|
2966
|
+
return payload.targetSurface.length;
|
|
2967
|
+
}
|
|
2968
|
+
catch {
|
|
2969
|
+
// surrounding lines are prose
|
|
2970
|
+
}
|
|
2971
|
+
}
|
|
2972
|
+
return undefined;
|
|
2973
|
+
}
|
|
2974
|
+
/**
|
|
2975
|
+
* Deterministically extract repository file paths that frozen verification
|
|
2976
|
+
* commands operate on (`--config <file>`, `node --check <file>`). A frozen
|
|
2977
|
+
* command whose referenced file is outside the planner's verification targets
|
|
2978
|
+
* can never run: the admission writeSet derives from those targets, so the
|
|
2979
|
+
* writer is not authorized to create the file and verify-shell fails ~20
|
|
2980
|
+
* minutes later. Surfacing the gap as plan missing-facts lets the coverage or
|
|
2981
|
+
* compact session record the missing verification target inside the same
|
|
2982
|
+
* attempt instead.
|
|
2983
|
+
*/
|
|
2984
|
+
export function collectFrontendVerificationCommandFiles(basePrompt) {
|
|
2985
|
+
const files = new Set();
|
|
2986
|
+
const configRe = /--config\s+([\w@./-]+\.(?:js|mjs|cjs|ts|json))/g;
|
|
2987
|
+
const checkRe = /node\s+--check\s+([\w@./-]+\.(?:js|mjs|cjs))/g;
|
|
2988
|
+
for (const re of [configRe, checkRe]) {
|
|
2989
|
+
for (const match of basePrompt.matchAll(re)) {
|
|
2990
|
+
const file = match[1];
|
|
2991
|
+
if (file && file.includes("/"))
|
|
2992
|
+
files.add(file);
|
|
2993
|
+
}
|
|
2994
|
+
}
|
|
2995
|
+
return [...files].sort();
|
|
2996
|
+
}
|
|
2923
2997
|
/**
|
|
2924
2998
|
* Remove the repeated full planner input from a coverage batch. The normal
|
|
2925
2999
|
* plan prompt already contains a bounded JSON handoff, but repeating all
|
|
@@ -3226,6 +3300,7 @@ export async function runFrontendPlanSegmentedSessions(input) {
|
|
|
3226
3300
|
const coverageSegment = FRONTEND_PLAN_SEGMENTS.find((segment) => segment.id === "coverage");
|
|
3227
3301
|
const uxSegment = FRONTEND_PLAN_SEGMENTS.find((segment) => segment.id === "ux-local");
|
|
3228
3302
|
const globalMockDataSegment = FRONTEND_PLAN_SEGMENTS.find((segment) => segment.id === "global-mock-data");
|
|
3303
|
+
const finalizeSegment = FRONTEND_PLAN_SEGMENTS.find((segment) => segment.id === "finalize");
|
|
3229
3304
|
const allRequirementIds = input.requirementIds ?? [];
|
|
3230
3305
|
const buildPhasePrompt = (segment, missing = []) => {
|
|
3231
3306
|
const compact = allRequirementIds.length > 0
|
|
@@ -3295,6 +3370,42 @@ export async function runFrontendPlanSegmentedSessions(input) {
|
|
|
3295
3370
|
: []),
|
|
3296
3371
|
].filter(Boolean).join("\n\n");
|
|
3297
3372
|
};
|
|
3373
|
+
const buildCompactLocalPrompt = (missing = []) => {
|
|
3374
|
+
const ledger = input.committedFacts
|
|
3375
|
+
? compactFrontendPlanLedgerContext({
|
|
3376
|
+
committedFacts: input.committedFacts(),
|
|
3377
|
+
requirementIds: allRequirementIds,
|
|
3378
|
+
kinds: [
|
|
3379
|
+
"plan-requirement",
|
|
3380
|
+
"plan-verification-target",
|
|
3381
|
+
"component-choice",
|
|
3382
|
+
"state-flow",
|
|
3383
|
+
],
|
|
3384
|
+
})
|
|
3385
|
+
: "";
|
|
3386
|
+
return [
|
|
3387
|
+
compactFrontendPlanPromptForRequirementSlice(input.basePrompt, allRequirementIds),
|
|
3388
|
+
"PLAN PHASE — compact local planning for a small frontend request.",
|
|
3389
|
+
"For every listed requirement, record plan-requirement and verification-target facts, then record the component-choice and state-flow facts needed by the observable UX. Do not read the repository or task source; use only the committed input above. Do not call finalize_plan in this session.",
|
|
3390
|
+
"TOOL-FIRST: your first assistant actions must be record_* tool calls, at most 2-3 facts per message. Do not draft the whole analysis before recording; if a fact is uncertain, record it with an evidence gap instead of reasoning longer.",
|
|
3391
|
+
ledger,
|
|
3392
|
+
...(missing.length > 0
|
|
3393
|
+
? [
|
|
3394
|
+
"MISSING-FACT QUEUE: repair ONLY these items, then re-check the compact local phase:",
|
|
3395
|
+
...missing.map((item) => `- ${item.kind}${item.id ? ` ${item.id}` : ""} for ${item.requirementIds.join(", ")}: ${item.reason}`),
|
|
3396
|
+
]
|
|
3397
|
+
: []),
|
|
3398
|
+
].filter(Boolean).join("\n\n");
|
|
3399
|
+
};
|
|
3400
|
+
const compactFinalizeInstruction = "This is a small-request compact pass. Reconcile the committed local facts with route, data-flow, Mock/API, dependency and deviation policy, then call finalize_plan exactly once.";
|
|
3401
|
+
const buildCompactFinalizePrompt = (missing = []) => [buildPhasePrompt(finalizeSegment, missing), compactFinalizeInstruction].join("\n\n");
|
|
3402
|
+
const mapPlannerExhaustion = (r, committedAnyFacts) => isPlannerThinkingExhausted(r, committedAnyFacts)
|
|
3403
|
+
? {
|
|
3404
|
+
...r,
|
|
3405
|
+
failureCategory: PLANNER_THINKING_EXHAUSTED_CATEGORY,
|
|
3406
|
+
stderr: `${r.stderr}\n${PLANNER_THINKING_EXHAUSTED_CATEGORY}: stopReason=length, thinking observed, 0 typed facts committed; the batch ladder degraded the scope without converging — set thinking=off for this tier or switch to a non-thinking model`.trim(),
|
|
3407
|
+
}
|
|
3408
|
+
: r;
|
|
3298
3409
|
// An empty list means "ledger unreadable / unknown" and falls back to one
|
|
3299
3410
|
// unscoped coverage session; a non-empty list enables deterministic sharding.
|
|
3300
3411
|
const requirementIdsProvided = input.requirementIds !== undefined && input.requirementIds.length > 0;
|
|
@@ -3307,7 +3418,42 @@ export async function runFrontendPlanSegmentedSessions(input) {
|
|
|
3307
3418
|
: [];
|
|
3308
3419
|
const incompleteRequirementIds = new Set(initialMissing.flatMap((item) => item.requirementIds));
|
|
3309
3420
|
const coverageWorkIds = (input.requirementIds ?? []).filter((id) => pending.includes(id) || incompleteRequirementIds.has(id));
|
|
3310
|
-
|
|
3421
|
+
const estimatedCalls = (input.requirementIds ?? []).reduce((total, id) => total + Math.max(1, input.requirementCosts?.get(id) ?? 2), 0);
|
|
3422
|
+
const targetSurfaceCount = countFrontendPlanTargetSurfaces(input.basePrompt);
|
|
3423
|
+
// Small, single-surface requests do not benefit from six isolated Pi
|
|
3424
|
+
// sessions. Keep the typed ledger as the authority, but let one local
|
|
3425
|
+
// session establish requirement/UX facts and one final session establish
|
|
3426
|
+
// cross-cutting policy + finalize. The old sharded ladder remains available
|
|
3427
|
+
// for larger plans and for the unscoped compatibility path.
|
|
3428
|
+
const useCompactSmallPlan = requirementIdsProvided &&
|
|
3429
|
+
input.compactSmallPlan === true &&
|
|
3430
|
+
input.requirementCosts !== undefined &&
|
|
3431
|
+
estimatedCalls > 12 &&
|
|
3432
|
+
(input.requirementIds?.length ?? 0) <= FRONTEND_PLAN_SMALL_MAX_REQUIREMENTS &&
|
|
3433
|
+
estimatedCalls <= FRONTEND_PLAN_SMALL_MAX_ESTIMATED_CALLS &&
|
|
3434
|
+
targetSurfaceCount === 1;
|
|
3435
|
+
if (useCompactSmallPlan) {
|
|
3436
|
+
const compactLocalTools = new Set([
|
|
3437
|
+
"record_plan_requirement",
|
|
3438
|
+
"record_plan_verification_target",
|
|
3439
|
+
"record_plan_evidence_gap",
|
|
3440
|
+
"record_component_choice",
|
|
3441
|
+
"record_state_flow",
|
|
3442
|
+
"adopt_staged_fact",
|
|
3443
|
+
]);
|
|
3444
|
+
queue.push({
|
|
3445
|
+
id: "compact-local",
|
|
3446
|
+
toolNames: compactLocalTools,
|
|
3447
|
+
requirementSlice: [...allRequirementIds],
|
|
3448
|
+
prompt: buildCompactLocalPrompt(),
|
|
3449
|
+
});
|
|
3450
|
+
queue.push({
|
|
3451
|
+
id: "finalize",
|
|
3452
|
+
toolNames: null,
|
|
3453
|
+
prompt: buildCompactFinalizePrompt(),
|
|
3454
|
+
});
|
|
3455
|
+
}
|
|
3456
|
+
else if (!requirementIdsProvided) {
|
|
3311
3457
|
queue.push({
|
|
3312
3458
|
id: "coverage",
|
|
3313
3459
|
toolNames: coverageSegment.toolNames,
|
|
@@ -3329,41 +3475,45 @@ export async function runFrontendPlanSegmentedSessions(input) {
|
|
|
3329
3475
|
prompt: buildCoveragePrompt(slice),
|
|
3330
3476
|
}));
|
|
3331
3477
|
}
|
|
3332
|
-
|
|
3333
|
-
|
|
3334
|
-
|
|
3335
|
-
|
|
3336
|
-
|
|
3337
|
-
|
|
3338
|
-
|
|
3339
|
-
|
|
3340
|
-
|
|
3341
|
-
|
|
3342
|
-
|
|
3343
|
-
|
|
3344
|
-
|
|
3345
|
-
|
|
3346
|
-
|
|
3347
|
-
|
|
3348
|
-
|
|
3349
|
-
|
|
3350
|
-
:
|
|
3351
|
-
|
|
3352
|
-
|
|
3353
|
-
|
|
3354
|
-
|
|
3478
|
+
if (useCompactSmallPlan) {
|
|
3479
|
+
// Compact mode already queued both sessions above.
|
|
3480
|
+
}
|
|
3481
|
+
else {
|
|
3482
|
+
const uxCosts = new Map((input.requirementIds ?? []).map((id) => [
|
|
3483
|
+
id,
|
|
3484
|
+
Math.max(2, Math.min(3, input.requirementCosts?.get(id) ?? 2)),
|
|
3485
|
+
]));
|
|
3486
|
+
const uxSlices = requirementIdsProvided
|
|
3487
|
+
? batchFrontendPlanRequirements({
|
|
3488
|
+
requirementIds: input.requirementIds,
|
|
3489
|
+
maxEstimatedRecordCalls: FRONTEND_PLAN_UX_LOCAL_MAX_RECORD_CALLS,
|
|
3490
|
+
maxRequirements: 3,
|
|
3491
|
+
requirementCosts: uxCosts,
|
|
3492
|
+
})
|
|
3493
|
+
: [[]];
|
|
3494
|
+
uxSlices.forEach((slice, batchIndex) => queue.push({
|
|
3495
|
+
id: `ux-local-${batchIndex + 1}`,
|
|
3496
|
+
toolNames: uxSegment.toolNames,
|
|
3497
|
+
...(slice.length > 0 ? { requirementSlice: slice } : {}),
|
|
3498
|
+
prompt: slice.length > 0
|
|
3499
|
+
? buildUxPrompt(slice)
|
|
3500
|
+
: buildPhasePrompt(uxSegment),
|
|
3501
|
+
}));
|
|
3502
|
+
for (const segment of FRONTEND_PLAN_SEGMENTS) {
|
|
3503
|
+
if (["coverage", "ux-local", "finalize"].includes(segment.id))
|
|
3504
|
+
continue;
|
|
3505
|
+
queue.push({
|
|
3506
|
+
id: segment.id,
|
|
3507
|
+
toolNames: segment.toolNames,
|
|
3508
|
+
prompt: buildPhasePrompt(segment),
|
|
3509
|
+
});
|
|
3510
|
+
}
|
|
3355
3511
|
queue.push({
|
|
3356
|
-
id:
|
|
3357
|
-
toolNames:
|
|
3358
|
-
prompt: buildPhasePrompt(
|
|
3512
|
+
id: finalizeSegment.id,
|
|
3513
|
+
toolNames: finalizeSegment.toolNames,
|
|
3514
|
+
prompt: buildPhasePrompt(finalizeSegment),
|
|
3359
3515
|
});
|
|
3360
3516
|
}
|
|
3361
|
-
const finalizeSegment = FRONTEND_PLAN_SEGMENTS.find((segment) => segment.id === "finalize");
|
|
3362
|
-
queue.push({
|
|
3363
|
-
id: finalizeSegment.id,
|
|
3364
|
-
toolNames: finalizeSegment.toolNames,
|
|
3365
|
-
prompt: buildPhasePrompt(finalizeSegment),
|
|
3366
|
-
});
|
|
3367
3517
|
let last;
|
|
3368
3518
|
let index = 0;
|
|
3369
3519
|
let invocationCount = 0;
|
|
@@ -3389,6 +3539,9 @@ export async function runFrontendPlanSegmentedSessions(input) {
|
|
|
3389
3539
|
const promptSlice = remaining.length > 0 ? remaining : session.coverageSlice;
|
|
3390
3540
|
prompt = buildCoveragePrompt(promptSlice, session.missingFacts ?? preexistingMissing);
|
|
3391
3541
|
}
|
|
3542
|
+
else if (session.id === "compact-local") {
|
|
3543
|
+
prompt = buildCompactLocalPrompt(session.missingFacts);
|
|
3544
|
+
}
|
|
3392
3545
|
else if (session.id.startsWith("ux-local-")) {
|
|
3393
3546
|
// Build this at execution time: coverage facts are committed by the
|
|
3394
3547
|
// preceding sessions and must be visible to the UX-local model.
|
|
@@ -3402,11 +3555,14 @@ export async function runFrontendPlanSegmentedSessions(input) {
|
|
|
3402
3555
|
// stale queue entry from dropping facts written by earlier phases.
|
|
3403
3556
|
const segment = FRONTEND_PLAN_SEGMENTS.find((candidate) => candidate.id === session.id);
|
|
3404
3557
|
if (segment) {
|
|
3405
|
-
prompt =
|
|
3558
|
+
prompt =
|
|
3559
|
+
session.id === "finalize" && useCompactSmallPlan
|
|
3560
|
+
? buildCompactFinalizePrompt(session.missingFacts)
|
|
3561
|
+
: buildPhasePrompt(segment, session.missingFacts);
|
|
3406
3562
|
}
|
|
3407
3563
|
}
|
|
3408
3564
|
const committedBefore = input.committedFactCount();
|
|
3409
|
-
input.setActiveRequirementScope?.(session.id.startsWith("ux-local-")
|
|
3565
|
+
input.setActiveRequirementScope?.(session.id === "compact-local" || session.id.startsWith("ux-local-")
|
|
3410
3566
|
? session.requirementSlice ?? []
|
|
3411
3567
|
: []);
|
|
3412
3568
|
if (invocationCount >= FRONTEND_PLAN_BATCH_MAX_SESSIONS)
|
|
@@ -3444,27 +3600,67 @@ export async function runFrontendPlanSegmentedSessions(input) {
|
|
|
3444
3600
|
committedFacts: input.committedFacts(),
|
|
3445
3601
|
})
|
|
3446
3602
|
: [];
|
|
3603
|
+
const isCompactLocalSession = session.id === "compact-local";
|
|
3447
3604
|
const isUxLocalSession = session.id.startsWith("ux-local-");
|
|
3448
|
-
const missingPhase =
|
|
3449
|
-
?
|
|
3450
|
-
|
|
3451
|
-
requirementIds: session.requirementSlice,
|
|
3452
|
-
committedFacts: input.committedFacts(),
|
|
3453
|
-
behaviorRequiredRequirementIds: input.behaviorRequiredRequirementIds,
|
|
3454
|
-
})
|
|
3455
|
-
: session.id === "global-mock-data" && allRequirementIds.length > 0 && input.committedFacts
|
|
3456
|
-
? collectFrontendPlanPhaseMissingFacts({
|
|
3457
|
-
phase: "global-mock-data",
|
|
3605
|
+
const missingPhase = isCompactLocalSession && input.committedFacts
|
|
3606
|
+
? [
|
|
3607
|
+
...collectFrontendPlanMissingFacts({
|
|
3458
3608
|
requirementIds: allRequirementIds,
|
|
3459
3609
|
committedFacts: input.committedFacts(),
|
|
3610
|
+
}),
|
|
3611
|
+
...collectFrontendPlanPhaseMissingFacts({
|
|
3612
|
+
phase: "ux-local",
|
|
3613
|
+
requirementIds: allRequirementIds,
|
|
3614
|
+
committedFacts: input.committedFacts(),
|
|
3615
|
+
behaviorRequiredRequirementIds: input.behaviorRequiredRequirementIds,
|
|
3616
|
+
}),
|
|
3617
|
+
]
|
|
3618
|
+
: isUxLocalSession && session.requirementSlice && input.committedFacts
|
|
3619
|
+
? collectFrontendPlanPhaseMissingFacts({
|
|
3620
|
+
phase: "ux-local",
|
|
3621
|
+
requirementIds: session.requirementSlice,
|
|
3622
|
+
committedFacts: input.committedFacts(),
|
|
3623
|
+
behaviorRequiredRequirementIds: input.behaviorRequiredRequirementIds,
|
|
3460
3624
|
})
|
|
3461
|
-
:
|
|
3625
|
+
: session.id === "global-mock-data" && allRequirementIds.length > 0 && input.committedFacts
|
|
3626
|
+
? collectFrontendPlanPhaseMissingFacts({
|
|
3627
|
+
phase: "global-mock-data",
|
|
3628
|
+
requirementIds: allRequirementIds,
|
|
3629
|
+
committedFacts: input.committedFacts(),
|
|
3630
|
+
})
|
|
3631
|
+
: [];
|
|
3462
3632
|
const missingPhaseFacts = [...missingCoverage, ...missingPhase];
|
|
3633
|
+
// Frozen verification commands must operate on files the planner has
|
|
3634
|
+
// committed verification targets for; otherwise the admission writeSet
|
|
3635
|
+
// cannot authorize the file and verify-shell deterministically fails.
|
|
3636
|
+
const verificationCommandFiles = collectFrontendVerificationCommandFiles(input.basePrompt);
|
|
3637
|
+
if (verificationCommandFiles.length > 0 && allRequirementIds.length > 0 && input.committedFacts) {
|
|
3638
|
+
const coveredVerificationFiles = new Set(input
|
|
3639
|
+
.committedFacts()
|
|
3640
|
+
.map(committedFactFromPlanRecord)
|
|
3641
|
+
.filter((fact) => Boolean(fact && fact.origin === "plan" && fact.kind === "plan-verification-target"))
|
|
3642
|
+
.map((fact) => fact.entry?.file)
|
|
3643
|
+
.filter((file) => typeof file === "string"));
|
|
3644
|
+
for (const file of verificationCommandFiles) {
|
|
3645
|
+
if (coveredVerificationFiles.has(file))
|
|
3646
|
+
continue;
|
|
3647
|
+
missingPhaseFacts.push({
|
|
3648
|
+
kind: "plan-verification-target",
|
|
3649
|
+
id: file,
|
|
3650
|
+
requirementIds: allRequirementIds.slice(0, 1),
|
|
3651
|
+
reason: `frozen verification command references ${file} but no committed verification target covers it; record a static verification target for this file so it joins the writeSet`,
|
|
3652
|
+
});
|
|
3653
|
+
}
|
|
3654
|
+
}
|
|
3463
3655
|
const promptForMissingPhase = (missing) => session.coverageOnly
|
|
3464
3656
|
? buildCoveragePrompt(session.coverageSlice ?? [], missing)
|
|
3465
|
-
:
|
|
3466
|
-
?
|
|
3467
|
-
:
|
|
3657
|
+
: isCompactLocalSession
|
|
3658
|
+
? buildCompactLocalPrompt(missing)
|
|
3659
|
+
: session.id === "finalize" && useCompactSmallPlan
|
|
3660
|
+
? buildCompactFinalizePrompt(missing)
|
|
3661
|
+
: isUxLocalSession
|
|
3662
|
+
? buildUxPrompt(session.requirementSlice ?? [], missing)
|
|
3663
|
+
: buildPhasePrompt(globalMockDataSegment, missing);
|
|
3468
3664
|
if (result.ok) {
|
|
3469
3665
|
if (missingPhaseFacts.length === 0) {
|
|
3470
3666
|
index += 1;
|
|
@@ -3508,10 +3704,36 @@ export async function runFrontendPlanSegmentedSessions(input) {
|
|
|
3508
3704
|
index += 1;
|
|
3509
3705
|
continue;
|
|
3510
3706
|
}
|
|
3707
|
+
// A length-stopped, fact-less compact session is the planner variant of
|
|
3708
|
+
// writer-thinking-exhausted. Give the same scope exactly one tool-first
|
|
3709
|
+
// retry before the split below re-batches the requirements, because one
|
|
3710
|
+
// reinforced full-scope pass is cheaper than re-planning split halves.
|
|
3711
|
+
const plannerThinkingBurn = isCompactLocalSession &&
|
|
3712
|
+
committedAfter === committedBefore &&
|
|
3713
|
+
!(result.assistantText ?? "").trim() &&
|
|
3714
|
+
!result.stderr.trim() &&
|
|
3715
|
+
!result.timedOut &&
|
|
3716
|
+
readWriterThinkingExhaustionEvidence(result).stopReason === "length";
|
|
3717
|
+
if (plannerThinkingBurn && (session.retryCount ?? 0) < 1) {
|
|
3718
|
+
queue[index] = {
|
|
3719
|
+
...session,
|
|
3720
|
+
retryCount: (session.retryCount ?? 0) + 1,
|
|
3721
|
+
prompt: buildCompactLocalPrompt(),
|
|
3722
|
+
};
|
|
3723
|
+
continue;
|
|
3724
|
+
}
|
|
3511
3725
|
// Option 5: a multi-requirement coverage batch that failed with ZERO
|
|
3512
3726
|
// new facts and no provider stderr is the upfront-reasoning burn —
|
|
3513
|
-
// halve the slice and retry instead of failing the attempt.
|
|
3514
|
-
|
|
3727
|
+
// halve the slice and retry instead of failing the attempt. The compact
|
|
3728
|
+
// local session participates through its requirementSlice so a
|
|
3729
|
+
// thinking-burned small plan degrades into smaller batched sessions
|
|
3730
|
+
// instead of replaying one full-scope prompt until the repair budget
|
|
3731
|
+
// runs out.
|
|
3732
|
+
const coverageSlice = session.coverageOnly
|
|
3733
|
+
? session.coverageSlice
|
|
3734
|
+
: isCompactLocalSession
|
|
3735
|
+
? session.requirementSlice
|
|
3736
|
+
: undefined;
|
|
3515
3737
|
const zeroProgressBurn = coverageSlice !== undefined &&
|
|
3516
3738
|
coverageSlice.length > 1 &&
|
|
3517
3739
|
committedAfter === committedBefore &&
|
|
@@ -3522,6 +3744,26 @@ export async function runFrontendPlanSegmentedSessions(input) {
|
|
|
3522
3744
|
const half = Math.ceil(coverageSlice.length / 2);
|
|
3523
3745
|
const firstSlice = coverageSlice.slice(0, half);
|
|
3524
3746
|
const secondSlice = coverageSlice.slice(half);
|
|
3747
|
+
if (isCompactLocalSession) {
|
|
3748
|
+
// The split halves leave compact mode: continue them as ordinary
|
|
3749
|
+
// coverage sessions so every downstream ladder branch applies.
|
|
3750
|
+
queue.splice(index, 1, {
|
|
3751
|
+
...session,
|
|
3752
|
+
id: `coverage-compact-split-1`,
|
|
3753
|
+
coverageOnly: true,
|
|
3754
|
+
coverageSlice: firstSlice,
|
|
3755
|
+
requirementSlice: firstSlice,
|
|
3756
|
+
prompt: buildCoveragePrompt(firstSlice),
|
|
3757
|
+
}, {
|
|
3758
|
+
...session,
|
|
3759
|
+
id: `coverage-compact-split-2`,
|
|
3760
|
+
coverageOnly: true,
|
|
3761
|
+
coverageSlice: secondSlice,
|
|
3762
|
+
requirementSlice: secondSlice,
|
|
3763
|
+
prompt: buildCoveragePrompt(secondSlice),
|
|
3764
|
+
});
|
|
3765
|
+
continue;
|
|
3766
|
+
}
|
|
3525
3767
|
queue.splice(index, 1, {
|
|
3526
3768
|
...session,
|
|
3527
3769
|
coverageSlice: firstSlice,
|
|
@@ -3546,7 +3788,7 @@ export async function runFrontendPlanSegmentedSessions(input) {
|
|
|
3546
3788
|
};
|
|
3547
3789
|
continue;
|
|
3548
3790
|
}
|
|
3549
|
-
return result;
|
|
3791
|
+
return mapPlannerExhaustion(result, committedAfter > committedBefore);
|
|
3550
3792
|
}
|
|
3551
3793
|
if (index < queue.length) {
|
|
3552
3794
|
return {
|
|
@@ -3571,13 +3813,13 @@ export async function runFrontendPlanSegmentedSessions(input) {
|
|
|
3571
3813
|
failureCategory: "invalid-output",
|
|
3572
3814
|
};
|
|
3573
3815
|
}
|
|
3574
|
-
return (last ?? {
|
|
3816
|
+
return mapPlannerExhaustion(last ?? {
|
|
3575
3817
|
ok: false,
|
|
3576
3818
|
stdout: "",
|
|
3577
3819
|
stderr: "frontend plan segmentation produced no session",
|
|
3578
3820
|
failureCategory: "empty-output",
|
|
3579
3821
|
durationMs: 0,
|
|
3580
|
-
});
|
|
3822
|
+
}, false);
|
|
3581
3823
|
}
|
|
3582
3824
|
export async function executeDagPiNode(input, meta, piStepFn = executePiStep, writeGuardDependencies = DEFAULT_DAG_PI_WRITE_GUARD_DEPENDENCIES) {
|
|
3583
3825
|
const started = Date.now();
|
|
@@ -4060,6 +4302,7 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
|
|
|
4060
4302
|
committedFactCount: () => planLedgerTools.committedFactCount(),
|
|
4061
4303
|
requirementIds: planRequirementIds,
|
|
4062
4304
|
requirementCosts: planRequirementCosts,
|
|
4305
|
+
compactSmallPlan: true,
|
|
4063
4306
|
committedRequirementIds: () => planLedgerTools.committedRequirementIds(),
|
|
4064
4307
|
committedFacts: () => planLedgerTools.committedFacts(),
|
|
4065
4308
|
behaviorRequiredRequirementIds,
|
|
@@ -4079,6 +4322,18 @@ export async function executeDagPiNode(input, meta, piStepFn = executePiStep, wr
|
|
|
4079
4322
|
...(writerToolPolicy ? { writerToolPolicy } : {}),
|
|
4080
4323
|
});
|
|
4081
4324
|
}
|
|
4325
|
+
try {
|
|
4326
|
+
const verificationCommandFiles = collectFrontendVerificationCommandFiles(input.prompt);
|
|
4327
|
+
const debugPath = path.join(meta.runDir, input.task.id, "verification-command-coverage.json");
|
|
4328
|
+
await import("node:fs/promises").then(({ writeFile, mkdir }) => mkdir(path.dirname(debugPath), { recursive: true }).then(() => writeFile(debugPath, `${JSON.stringify({
|
|
4329
|
+
schemaVersion: 1,
|
|
4330
|
+
basePromptChars: input.prompt.length,
|
|
4331
|
+
verificationCommandFiles,
|
|
4332
|
+
}, null, 2)}\n`)));
|
|
4333
|
+
}
|
|
4334
|
+
catch {
|
|
4335
|
+
// best-effort diagnostic breadcrumb
|
|
4336
|
+
}
|
|
4082
4337
|
}
|
|
4083
4338
|
catch (error) {
|
|
4084
4339
|
if (playwrightToolContext) {
|
|
@@ -4860,6 +5115,8 @@ export function mapPiResultToDagNodeResult(result, firstProtocolLine) {
|
|
|
4860
5115
|
tokensUsed: result.tokensUsed,
|
|
4861
5116
|
parsedEvents: result.parsedEvents,
|
|
4862
5117
|
stopReason: readWriterThinkingExhaustionEvidence(result).stopReason,
|
|
5118
|
+
thinkingObserved: readWriterThinkingExhaustionEvidence(result).thinkingObserved,
|
|
5119
|
+
writeToolCallCount: readWriterThinkingExhaustionEvidence(result).writeToolCallCount,
|
|
4863
5120
|
};
|
|
4864
5121
|
}
|
|
4865
5122
|
function canonicalizeProtocolFirstLine(assistantText, firstProtocolLine) {
|