@tea-agent/loop-agent 0.36.1-beta.0 → 0.36.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. package/CHANGELOG.md +11 -12
  2. package/dist/build-stamp.json +3 -3
  3. package/dist/cli/command-definitions.js +7 -0
  4. package/dist/cli/program.js +6 -1
  5. package/dist/commands/dag-request-interrupt.js +20 -0
  6. package/dist/executors/pi-sdk-executor.js +34 -0
  7. package/dist/shared/operator/capabilities.js +76 -3
  8. package/dist/task/source-prepare/semantic-intake.js +144 -23
  9. package/dist/worker/console/chat/semantic-activity.js +6 -0
  10. package/dist/worker/console/chat/workspace-landing.js +1 -0
  11. package/dist/worker/console/operation-runner.js +48 -0
  12. package/dist/worker/console/operation-wait.js +41 -0
  13. package/dist/worker/console/operator-actions.js +235 -26
  14. package/dist/worker/console/operator-user-error.js +4 -0
  15. package/dist/worker/console/recovery-cta.js +50 -3
  16. package/dist/worker/console/recovery-error-copy.js +198 -0
  17. package/dist/worker/console/static/assets/index-D83DYAFG.css +1 -0
  18. package/dist/worker/console/static/assets/index-IXm7oYjL.js +59 -0
  19. package/dist/worker/console/static/index.html +2 -2
  20. package/dist/worker/console/static-src/app/console-types.js +13 -9
  21. package/dist/worker/console/static-src/app/useOperatorActions.js +4 -2
  22. package/dist/worker/console/static-src/app/useRecoveryActions.js +65 -56
  23. package/dist/worker/console/static-src/app/useRecoveryConsole.js +68 -1
  24. package/dist/worker/observability/interrupt-eligibility.js +264 -0
  25. package/dist/worker/observe/static/operator-chrome.d.ts +1 -0
  26. package/dist/worker/observe/static/operator-chrome.js +5 -0
  27. package/dist/worker/observe/static/views/dag.js +12 -0
  28. package/dist/workflows/dag/failure-routing.js +4 -9
  29. package/dist/workflows/dag/init-hybrid.js +2 -4
  30. package/dist/workflows/dag/interrupt-request.js +559 -0
  31. package/dist/workflows/dag/lifecycle.js +0 -4
  32. package/dist/workflows/dag/node-execution.js +6 -16
  33. package/dist/workflows/dag/report.js +0 -6
  34. package/dist/workflows/dag/retry-policy.js +0 -11
  35. package/dist/workflows/dag/runner.js +72 -4
  36. package/docs/templates/frontend-task-constraints.md +7 -13
  37. package/package.json +1 -1
  38. package/skills/loop-agent/references/command-reference.md +1 -0
  39. package/dist/worker/console/static/assets/index-2OeZODxk.js +0 -57
  40. package/dist/worker/console/static/assets/index-DVJlUL8X.css +0 -1
@@ -216,28 +216,18 @@ function buildAttemptPrompt(task, basePrompt, attemptNumber, previousFailureCate
216
216
  "</retry_instruction>",
217
217
  ].join("\n");
218
218
  }
219
- if (previousFailureCategory !== "output-too-large") {
219
+ if (task.outputMode !== "structured-required" ||
220
+ previousFailureCategory !== "output-too-large") {
220
221
  return basePrompt;
221
222
  }
222
- if (task.outputMode === "structured-required") {
223
- return [
224
- basePrompt,
225
- "",
226
- "<retry_instruction>",
227
- "Previous attempt exceeded the structured output size limit.",
228
- "Return only the compact structured artifact required by this node's output contract.",
229
- "Do not include explanatory prose, duplicated upstream context, long evidence excerpts, or additional markdown sections.",
230
- "If a fenced JSON object is required, output exactly one fenced json block and nothing else.",
231
- "</retry_instruction>",
232
- ].join("\n");
233
- }
234
223
  return [
235
224
  basePrompt,
236
225
  "",
237
226
  "<retry_instruction>",
238
- "Previous attempt exceeded the output size limit.",
239
- "Return a minimal plan: ordered steps, narrow writeSet boundaries, and verification commands.",
240
- "Drop verbatim upstream quotes, long evidence excerpts, and repeated context.",
227
+ "Previous attempt exceeded the structured output size limit.",
228
+ "Return only the compact structured artifact required by this node's output contract.",
229
+ "Do not include explanatory prose, duplicated upstream context, long evidence excerpts, or additional markdown sections.",
230
+ "If a fenced JSON object is required, output exactly one fenced json block and nothing else.",
241
231
  "</retry_instruction>",
242
232
  ].join("\n");
243
233
  }
@@ -441,7 +441,6 @@ export async function buildDagRunReportEntry(input) {
441
441
  normalizedFailureCategory,
442
442
  nodeId,
443
443
  executor: node.executor,
444
- skippedReason: node.skippedReason,
445
444
  });
446
445
  const followUp = node.status === "ERROR" || node.status === "SKIPPED"
447
446
  ? recommendFollowUpForFailureCategory(node.failureCategory)
@@ -506,7 +505,6 @@ export async function buildDagRunReportEntry(input) {
506
505
  normalizedFailureCategory: normalizeDagFailureCategory(pausedNode.failureCategory, pausedNode.status),
507
506
  failureCategory: pausedNode.failureCategory,
508
507
  nodeId: input.state.pausedByNodeId,
509
- skippedReason: pausedNode.skippedReason,
510
508
  }
511
509
  : firstActionableNode
512
510
  ? {
@@ -515,7 +513,6 @@ export async function buildDagRunReportEntry(input) {
515
513
  normalizeDagFailureCategory(firstActionableNode.failureCategory, firstActionableNode.status),
516
514
  failureCategory: firstActionableNode.failureCategory,
517
515
  nodeId: firstActionableNode.nodeId,
518
- skippedReason: input.state.nodes[firstActionableNode.nodeId]?.skippedReason,
519
516
  }
520
517
  : {
521
518
  status: input.state.status,
@@ -532,9 +529,6 @@ export async function buildDagRunReportEntry(input) {
532
529
  rawFailureCategory: runRecoverySource.failureCategory,
533
530
  normalizedFailureCategory: runRecoverySource.normalizedFailureCategory,
534
531
  nodeId: "nodeId" in runRecoverySource ? runRecoverySource.nodeId : undefined,
535
- skippedReason: "skippedReason" in runRecoverySource
536
- ? runRecoverySource.skippedReason
537
- : undefined,
538
532
  });
539
533
  const runFollowUp = input.state.status === "failed" ||
540
534
  input.state.status === "partial_failed"
@@ -118,17 +118,6 @@ export const STRUCTURED_REQUIRED_PI_RETRY_POLICY = {
118
118
  ...DEFAULT_READ_ONLY_PI_RETRY_POLICY,
119
119
  retryCategories: [...STRUCTURED_REQUIRED_DAG_RETRY_CATEGORIES],
120
120
  };
121
- /**
122
- * Planner read-only nodes (standard DAG contract-pi / plan-pi) may retry with
123
- * a compact-output instruction when the model produced an oversized assistant
124
- * response. Only `output-too-large` is added on top of the default set;
125
- * report/review and other read-only roles keep the default set so they do not
126
- * silently learn new output semantics.
127
- */
128
- export const PLANNER_OUTPUT_LIMIT_RETRY_POLICY = {
129
- ...DEFAULT_READ_ONLY_PI_RETRY_POLICY,
130
- retryCategories: [...DEFAULT_DAG_RETRY_CATEGORIES, STRUCTURED_OUTPUT_RETRY_CATEGORY],
131
- };
132
121
  /**
133
122
  * Default retry for reviewer / recovery nodes that declare outputProtocol.
134
123
  * Includes protocol-invalid so missing VERDICT lines are corrected in-node.
@@ -8,6 +8,7 @@ import { readCandidateRecord } from "../../infrastructure/evaluation/candidate-s
8
8
  import { CANONICAL_TASK_ID_PATTERN, formatLocalCompactDate, } from "../../task/runtime.js";
9
9
  import { assertFrozenBudget, initRunBudgetLedger, preflightBudgetOrBreach, recordFinishedNodeBudget, writeBudgetLedgerArtifacts, } from "./budget-enforcement.js";
10
10
  import { getDagRunDir, isTerminalDagRunStatus, locateDagRun, readDagRunSpec, readDagRunState, readHumanApprovalArtifact, requireActiveDagRun, } from "./lifecycle.js";
11
+ import { markInterruptNeedsReconcile, mergeAbortSignals, readInterruptRequest, runnerIdentityFromState, settleDagInterrupt, startInterruptWatcher, } from "./interrupt-request.js";
11
12
  import { moveToCompletedRunDir, moveToPausedRunDir, prepareRunDir, writeRunSpec, writeRunState, } from "./run-store.js";
12
13
  import { evaluateNodeLiveness, resolveLivenessPolicy, } from "./liveness-policy.js";
13
14
  import { createDagNodeExecutor } from "./executor-registry.js";
@@ -450,6 +451,18 @@ async function executeDagCheckpoint(input) {
450
451
  void persistState().catch(() => { });
451
452
  }, runnerLivenessPolicy.heartbeatIntervalMs);
452
453
  heartbeatTimer.unref();
454
+ await persistState();
455
+ const interruptController = new AbortController();
456
+ const expectedRunnerIdentity = runnerIdentityFromState(state);
457
+ const interruptWatcher = expectedRunnerIdentity
458
+ ? startInterruptWatcher({
459
+ runDir,
460
+ runId: state.runId,
461
+ expectedRunnerIdentity,
462
+ controller: interruptController,
463
+ })
464
+ : undefined;
465
+ const abortSignal = mergeAbortSignals(input.abortSignal, interruptController.signal);
453
466
  // Graceful terminal persistence: an outer SIGTERM/SIGINT (operator hard
454
467
  // timeout, supervision layer, or shell wall-clock) must not leave the run
455
468
  // orphaned in RUNNING with a dead heartbeat. Persist a terminal failed
@@ -549,7 +562,7 @@ async function executeDagCheckpoint(input) {
549
562
  meta: { runDir, runId: state.runId, spec },
550
563
  }),
551
564
  executeScheduledNode: async (nodeId, executeNode, onPause) => {
552
- if (input.abortSignal?.aborted)
565
+ if (abortSignal?.aborted)
553
566
  return;
554
567
  if (isHardBudgetBreached(state.budgetLedger))
555
568
  return;
@@ -616,6 +629,7 @@ async function executeDagCheckpoint(input) {
616
629
  // still producing a truthful failure handoff for max-passes/non-retry.
617
630
  pausedByNodeId = await executeRanks(terminalCloseoutRanks);
618
631
  }
632
+ interruptWatcher?.stop();
619
633
  // Stop heartbeats before terminal archive. A late heartbeat writing
620
634
  // state.json under completed/ trips the completed-facts write guard and
621
635
  // makes run-dag exit non-zero after every node already finished.
@@ -626,12 +640,24 @@ async function executeDagCheckpoint(input) {
626
640
  await writeBudgetLedgerArtifacts(runDir, state.budgetLedger);
627
641
  if (pausedByNodeId) {
628
642
  state.status = "paused";
643
+ await finalizeRunOwnedInterrupt({
644
+ runDir,
645
+ state,
646
+ paused: true,
647
+ interruptAborted: interruptController.signal.aborted,
648
+ });
629
649
  await persistState();
630
650
  await notifyRunObserver(input.observer, "onRunFinish", state);
631
651
  runDir = await moveToPausedRunDir(runDir, pausedRunDir);
632
652
  }
633
653
  else {
634
654
  const { recoveryPending } = await finalizeTerminalRunStatus(state, spec.tasks.length, runDir, cwd);
655
+ await finalizeRunOwnedInterrupt({
656
+ runDir,
657
+ state,
658
+ paused: false,
659
+ interruptAborted: interruptController.signal.aborted,
660
+ });
635
661
  await persistState();
636
662
  if (recoveryPending) {
637
663
  // Recovery coordinator (phase 3c AC-2/AC-3): materialize the reserved
@@ -664,7 +690,7 @@ async function executeDagCheckpoint(input) {
664
690
  maxConcurrent,
665
691
  executeNode: input.executeNode,
666
692
  observer: input.observer,
667
- abortSignal: input.abortSignal,
693
+ abortSignal,
668
694
  });
669
695
  }
670
696
  }
@@ -701,10 +727,51 @@ async function executeDagCheckpoint(input) {
701
727
  }
702
728
  finally {
703
729
  clearInterval(heartbeatTimer);
730
+ interruptWatcher?.stop();
704
731
  process.removeListener("SIGTERM", onSigTerm);
705
732
  process.removeListener("SIGINT", onSigInt);
706
733
  }
707
734
  }
735
+ async function finalizeRunOwnedInterrupt(input) {
736
+ const request = await readInterruptRequest(input.runDir);
737
+ if (!request)
738
+ return;
739
+ if (request.status === "settled" || request.status === "needs-reconcile") {
740
+ return;
741
+ }
742
+ const hasUnconfirmed = Object.values(input.state.nodes).some((node) => node.failureCategory === "termination-unconfirmed");
743
+ if (hasUnconfirmed || input.paused) {
744
+ await markInterruptNeedsReconcile({
745
+ runDir: input.runDir,
746
+ outcome: hasUnconfirmed
747
+ ? "termination-unconfirmed"
748
+ : "paused-before-settle",
749
+ });
750
+ return;
751
+ }
752
+ if (input.interruptAborted &&
753
+ request.status === "acknowledged" &&
754
+ (input.state.status === "failed" ||
755
+ input.state.status === "partial_failed" ||
756
+ input.state.failureCategory === "controller-interrupted")) {
757
+ await settleDagInterrupt({
758
+ runDir: input.runDir,
759
+ outcome: "controller-interrupted",
760
+ });
761
+ return;
762
+ }
763
+ if (request.status === "requested") {
764
+ await markInterruptNeedsReconcile({
765
+ runDir: input.runDir,
766
+ outcome: "runner-did-not-acknowledge",
767
+ });
768
+ return;
769
+ }
770
+ await markInterruptNeedsReconcile({
771
+ runDir: input.runDir,
772
+ outcome: "run-exited-before-interrupt-settled",
773
+ });
774
+ }
708
775
  async function notifyRunObserver(observer, event, state) {
709
776
  try {
710
777
  await observer?.[event]?.(state);
@@ -859,8 +926,9 @@ export async function finalizeTerminalRunStatus(state, taskCount, runDir, cwd) {
859
926
  // mechanical partial_failed just because prewrite FINISHED while the writer
860
927
  // was SKIPPED. Fail closed on retryable-invalid/blocked before aggregation.
861
928
  let recoveryPending = false;
929
+ const skipFrontendRecovery = state.failureCategory === "controller-interrupted";
862
930
  const hasFrontendWriter = Object.keys(state.nodes).some((id) => FRONTEND_WRITER_NODE_IDS.includes(id));
863
- if (hasFrontendWriter) {
931
+ if (hasFrontendWriter && !skipFrontendRecovery) {
864
932
  const admission = await readFrontendPrewriteResult(runDir);
865
933
  if (admission.ok) {
866
934
  if (admission.result.classification === "blocked") {
@@ -888,7 +956,7 @@ export async function finalizeTerminalRunStatus(state, taskCount, runDir, cwd) {
888
956
  // for frontend-implementation writers with a remaining root continuation, and
889
957
  // only when the rollback journal (captured before the provider call) restores
890
958
  // cleanly; otherwise the run stays failed with auto-recovery-blocked.
891
- if (hasFrontendWriter && !recoveryPending) {
959
+ if (hasFrontendWriter && !recoveryPending && !skipFrontendRecovery) {
892
960
  const writerNodeId = FRONTEND_WRITER_NODE_IDS.find((id) => {
893
961
  const node = state.nodes[id];
894
962
  return (node &&
@@ -18,19 +18,13 @@ TODO
18
18
 
19
19
  ## Mock 约束(数据型任务)
20
20
 
21
- 本段决定生成期 `allowedMockStrategies` 与冻结的 Mock 验证命令。只有当生成期**能证明任务确实包含 Mock 且有确定性验证命令**时,DAG 才会放行 `native` / `browser-intercept` / `request-adapter`;否则自动收敛为 `not-needed`,design review 若按项目规范改选 `native`,会在 prewrite 被 `mock-strategy-outside-allowed` 拦截、`implement` 被跳过。
22
-
23
- - **何时必须填**:项目规范(`openspec/project-specs/**`、`ai_workspace/**`、mock 规则或类似 DEC-* 决策)要求/建议 Mock;或需求涉及远端接口且后端未就绪。规范要求 Mock 时,优先 `task.json.frontendMock.policy: "required"`。
24
- - **命令要探索、不要硬编码**:到项目 `package.json` scripts 里找实际存在的 Mock 相关脚本(如 `mock`、`mock:*`、`dev:mock`,或名字含 mock 的脚本),结合既有 service root 的 handler/fixture/bootstrap 启动方式,确定一条**真实存在、确定、可自终止**(启动→断言→退出 0)的验证命令。常驻 dev server 必须包装成自终止脚本(start→assert→stop),否则 verify shell 会超时 fail-closed。
25
- - **写入 task.json**:把探索到的命令原样写进 `frontendMock.verifyCommands`(`label` 唯一、`command` 与项目脚本一致)。`label` 会进入冻结命令集,implementation contract 的 `verificationTarget.commandLabel` 必须逐字引用它。
26
- - **命令来源白名单**:只允许项目 `package.json` 已有脚本、Mock capability seed、或本段声明的 `verifyCommands`;plan/design 阶段不能发明 shell 命令。
27
- - **Mock/API/schema 规范路径**:TODO
28
- - **既有 Mock service root、handler/fixture/bootstrap**:TODO
29
- - **既有 browser/e2e interception 或 request adapter/DI seam**:TODO
30
- - **production 禁用边界**:TODO
31
- - **真实请求默认路径与 Mock 显式启用方式**:TODO
32
- - **`task.json.frontendMock.policy`**:`auto | required | disabled`(规范强制 Mock 用 `required`)
33
- - **被拦截时怎么修**:`mock-strategy-outside-allowed` / `no authorized Mock verification commands` 是**生成期契约问题,不是 plan 问题**——补 `frontendMock.verifyCommands`(或 `policy: "required"` + 命令)后**重新生成 DAG** 再跑,plan-revision 无法修复它。
21
+ - Mock/API/schema 规范路径:TODO
22
+ - 既有 Mock service root、handler/fixture/bootstrap:TODO
23
+ - 既有 browser/e2e interception request adapter/DI seam:TODO
24
+ - 启动、健康检查和专项验证命令:TODO
25
+ - production 禁用边界:TODO
26
+ - 真实请求默认路径与 Mock 显式启用方式:TODO
27
+ - `task.json.frontendMock.policy`:`auto | required | disabled`
34
28
 
35
29
  ## allowedPaths
36
30
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@tea-agent/loop-agent",
3
- "version": "0.36.1-beta.0",
3
+ "version": "0.36.2",
4
4
  "type": "module",
5
5
  "bin": {
6
6
  "loop-agent": "bin/loop-agent.js",
@@ -321,6 +321,7 @@ loop-agent dag report [--run-id <run-id>] [--lifecycle active|paused|completed|a
321
321
  loop-agent dag reconcile-run --run-id <run-id> # 只读检查 effectiveStatus 与恢复/收口资格
322
322
  loop-agent dag reconcile-run --run-id <run-id> --action supersede --reason "..." # 显式保留证据并标记为任务已另行完成
323
323
  loop-agent dag reconcile-run --run-id <run-id> --action abandon --reason "..." # 显式保留证据并收口为已放弃
324
+ loop-agent dag request-interrupt --run-id <run-id> --request-id <id> --target-operation-id <id> --reason-code <code> --reason <detail> [--json] # 协作中止:写入 interrupt.json(identity CAS);需配 --expected-runner-pid/--expected-runner-hostname 校验 runner 身份
324
325
  loop-agent dag rerun --run-id <run-id> --from-node <node-id> --plan [--json] # 从节点重跑资格预检(不执行)
325
326
  loop-agent dag rerun --run-id <run-id> --from-node <node-id> --plan-hash <sha256> --request-id <key> --reason "..." [--json] # 安全子图 continuation
326
327
  loop-agent dag rerun-task --run-id <run-id> --reason "..." --request-id <key> [--profile auto] [--task-id <id>] [--json] # standalone 完整任务重跑