@tea-agent/loop-agent 0.35.0 → 0.35.1-beta.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (152) hide show
  1. package/AGENTS.md +108 -108
  2. package/CHANGELOG.md +30 -0
  3. package/README.md +165 -165
  4. package/bin/agent-worker.js +0 -0
  5. package/bin/loop-agent.js +21 -21
  6. package/dist/application/task-lifecycle/advance.js +1 -0
  7. package/dist/commands/cursor-prompt.js +6 -6
  8. package/dist/commands/init-upgrade.js +351 -19
  9. package/dist/commands/init.js +14 -67
  10. package/dist/commands/loop-benchmark.js +11 -11
  11. package/dist/commands/pi-reuse-benchmark.js +16 -16
  12. package/dist/commands/run-dag-progress.js +14 -0
  13. package/dist/commands/task-advance.js +33 -3
  14. package/dist/shared/operator/capabilities.js +38 -1
  15. package/dist/sidecars/cursor-prompt/executor.js +1 -1
  16. package/dist/worker/console/chat/pi-runtime.js +41 -25
  17. package/dist/worker/console/chat/routes.js +27 -4
  18. package/dist/worker/console/operation-runner.js +24 -0
  19. package/dist/worker/console/operation-wait.js +241 -0
  20. package/dist/worker/console/operator-actions.js +58 -0
  21. package/dist/worker/console/static/assets/{index-hJqCPs_g.css → index-Dups4sSM.css} +1 -1
  22. package/dist/worker/console/static/assets/index-SjjjZnV3.js +56 -0
  23. package/dist/worker/console/static/index.html +2 -2
  24. package/dist/worker/console/static-src/operator-chat/slash-palette-nav.js +141 -0
  25. package/dist/worker/console/static-src/operator-chat/useChatSessions.js +13 -2
  26. package/dist/worker/console/static-src/operator-chat/useComposer.js +30 -7
  27. package/dist/worker/observe/static/copy.js +67 -67
  28. package/dist/worker/observe/static/dag-layout.d.ts +36 -36
  29. package/dist/worker/observe/static/dom.js +220 -220
  30. package/dist/worker/observe/static/relations.js +133 -133
  31. package/dist/worker/observe/static/run-processing.js +148 -148
  32. package/dist/worker/observe/static/views/batch.js +227 -227
  33. package/dist/worker/observe/static/views/failures.js +143 -143
  34. package/dist/worker/observe/static/views/feature.js +492 -492
  35. package/dist/worker/observe/static/views/run.js +453 -453
  36. package/dist/worker/observe/static/views/shell.js +7 -7
  37. package/dist/worker/observe/static/views/timeline.js +163 -163
  38. package/dist/workflows/dag/canvas-observer.js +275 -275
  39. package/docs/architecture/evolution.md +73 -73
  40. package/docs/architecture/system-overview.md +100 -100
  41. package/docs/architecture/worker-and-feature.md +122 -122
  42. package/docs/skills/README.md +7 -7
  43. package/docs/templates/adr.md +60 -60
  44. package/docs/templates/agent-dag-authority-surface-audit.prompt.md +94 -94
  45. package/docs/templates/agent-dag-decision-envelope.schema.json +213 -213
  46. package/docs/templates/agent-dag-decision-gate.prompt.md +246 -246
  47. package/docs/templates/agent-dag-process-supervisor.prompt.md +98 -98
  48. package/docs/templates/agent-dag-report.schema.json +473 -473
  49. package/docs/templates/agent-dag-review-verdict.prompt.md +68 -68
  50. package/docs/templates/backend-test-result.schema.json +99 -99
  51. package/docs/templates/evaluation/agents-map-slim-v1.md +87 -87
  52. package/docs/templates/evaluation/agents-map-verbose-v0.md +153 -153
  53. package/docs/templates/feature-spec.md +53 -53
  54. package/docs/templates/frontend-design-contract.md +42 -42
  55. package/docs/templates/frontend-eval/fixtures/failures/01-type-build-error.md +17 -17
  56. package/docs/templates/frontend-eval/fixtures/failures/02-unit-component-test-fail.md +16 -16
  57. package/docs/templates/frontend-eval/fixtures/failures/03-fixture-schema-drift.md +16 -16
  58. package/docs/templates/frontend-eval/fixtures/failures/04-missing-loading-empty-error-state.md +16 -16
  59. package/docs/templates/frontend-eval/fixtures/failures/05-forbidden-write-writeset-expansion.md +16 -16
  60. package/docs/templates/frontend-eval/fixtures/failures/06-unapproved-dependency-add.md +16 -16
  61. package/docs/templates/frontend-eval/fixtures/failures/07-mock-production-on.md +21 -21
  62. package/docs/templates/frontend-eval/fixtures/functional/01-simple-component-style.md +29 -29
  63. package/docs/templates/frontend-eval/fixtures/functional/02-form-validation.md +28 -28
  64. package/docs/templates/frontend-eval/fixtures/functional/03-list-detail-page.md +28 -28
  65. package/docs/templates/frontend-eval/fixtures/functional/04-api-mock.md +29 -29
  66. package/docs/templates/frontend-eval/fixtures/functional/05-permission-auth-gated-ui.md +27 -27
  67. package/docs/templates/frontend-eval/fixtures/functional/06-ssr-server-client-boundary.md +28 -28
  68. package/docs/templates/frontend-eval/fixtures/functional/07-shared-public-component-api.md +28 -28
  69. package/docs/templates/frontend-eval/fixtures/functional/08-pure-local-no-remote.md +27 -27
  70. package/docs/templates/frontend-eval/metrics.md +138 -138
  71. package/docs/templates/frontend-eval/smoke-targets.md +53 -53
  72. package/docs/templates/frontend-task-constraints.md +35 -35
  73. package/docs/templates/frontend-task-requirement.md +70 -70
  74. package/docs/templates/init-evolution-review.md +35 -35
  75. package/docs/templates/init-managed-agents.md +156 -154
  76. package/docs/templates/interactive-ui-round2-experiment.md +66 -66
  77. package/docs/templates/knowledge-graph-bootstrap-dag.json +118 -118
  78. package/docs/templates/knowledge-sync-dag.json +178 -178
  79. package/docs/templates/knowledge-sync-draft.schema.json +71 -71
  80. package/docs/templates/product-line/closeout.yaml +9 -9
  81. package/docs/templates/product-line/design.md +13 -13
  82. package/docs/templates/product-line/links.md +10 -10
  83. package/docs/templates/product-line/requirement.md +17 -17
  84. package/docs/templates/product-line/test-plan.md +7 -7
  85. package/docs/templates/project-start-checklist.md +9 -9
  86. package/docs/templates/qa-report.md +48 -48
  87. package/docs/templates/sprint-contract.md +29 -29
  88. package/docs/templates/worker-dogfood-evidence.md +80 -80
  89. package/docs/templates/worker-dogfood-setup.md +68 -68
  90. package/harness.json +5 -2
  91. package/package.json +1 -1
  92. package/scripts/kb-bootstrap-init-skeleton.sh +0 -0
  93. package/scripts/kb-graph-incremental-prepare.mjs +0 -0
  94. package/scripts/kb-graph-materialize.mjs +105 -105
  95. package/scripts/kb-graph-promote.mjs +164 -164
  96. package/scripts/kb-query.mjs +554 -554
  97. package/skills/agent-worker/SKILL.md +48 -48
  98. package/skills/agent-worker/references/agent-worker-operator.md +159 -159
  99. package/skills/ai-engineering-context/SKILL.md +48 -48
  100. package/skills/analyze-product-dependencies/scripts/test-validators.mjs +0 -0
  101. package/skills/analyze-product-dependencies/scripts/validate-api-documentation.mjs +0 -0
  102. package/skills/analyze-product-dependencies/scripts/validate-dependency-analysis.mjs +0 -0
  103. package/skills/analyze-product-dependencies/scripts/validate-product-requirement-input.mjs +0 -0
  104. package/skills/analyze-product-requirements/scripts/compute-source-identity.mjs +0 -0
  105. package/skills/analyze-product-requirements/scripts/test-validators.mjs +0 -0
  106. package/skills/analyze-product-requirements/scripts/validate-product-analysis.mjs +0 -0
  107. package/skills/analyze-product-requirements/scripts/validate-product-requirement.mjs +0 -0
  108. package/skills/analyze-product-requirements/scripts/validate-requirement-clarification.mjs +0 -0
  109. package/skills/browser-tools/browser-content.js +103 -103
  110. package/skills/browser-tools/browser-cookies.js +35 -35
  111. package/skills/browser-tools/browser-eval.js +53 -53
  112. package/skills/browser-tools/browser-hn-scraper.js +108 -108
  113. package/skills/browser-tools/browser-nav.js +44 -44
  114. package/skills/browser-tools/browser-pick.js +162 -162
  115. package/skills/browser-tools/browser-screenshot.js +34 -34
  116. package/skills/browser-tools/browser-start.js +86 -86
  117. package/skills/browser-tools/package-lock.json +2556 -2556
  118. package/skills/browser-tools/package.json +19 -19
  119. package/skills/code-review-core/SKILL.md +20 -20
  120. package/skills/codebase-scout/SKILL.md +19 -19
  121. package/skills/grill-me/SKILL.md +10 -10
  122. package/skills/local-jacoco-coverage/scripts/run-coverage-analysis.sh +0 -0
  123. package/skills/local-jacoco-coverage/scripts/start-jacoco-agent.sh +0 -0
  124. package/skills/loop-agent/SKILL.md +1 -0
  125. package/skills/loop-agent/references/command-reference.md +641 -639
  126. package/skills/loop-agent/references/docs-converge.md +126 -126
  127. package/skills/loop-agent/references/learned/README.md +21 -21
  128. package/skills/loop-agent/references/pi-prompt.md +23 -23
  129. package/skills/loop-agent/references/pi-subagent-assisted-mode.md +84 -84
  130. package/skills/playwright-cli/references/element-attributes.md +23 -23
  131. package/skills/playwright-cli/references/playwright-tests.md +39 -39
  132. package/skills/playwright-cli/references/request-mocking.md +87 -87
  133. package/skills/playwright-cli/references/running-code.md +241 -241
  134. package/skills/playwright-cli/references/session-management.md +225 -225
  135. package/skills/playwright-cli/references/storage-state.md +275 -275
  136. package/skills/playwright-cli/references/test-generation.md +433 -433
  137. package/skills/requesting-code-review/SKILL.md +101 -101
  138. package/skills/requesting-code-review/code-reviewer.md +168 -168
  139. package/skills/systematic-debugging/CREATION-LOG.md +119 -119
  140. package/skills/systematic-debugging/condition-based-waiting-example.ts +158 -158
  141. package/skills/systematic-debugging/condition-based-waiting.md +115 -115
  142. package/skills/systematic-debugging/defense-in-depth.md +122 -122
  143. package/skills/systematic-debugging/find-polluter.sh +63 -63
  144. package/skills/systematic-debugging/root-cause-tracing.md +169 -169
  145. package/skills/systematic-debugging/test-academic.md +14 -14
  146. package/skills/systematic-debugging/test-pressure-1.md +58 -58
  147. package/skills/systematic-debugging/test-pressure-2.md +68 -68
  148. package/skills/systematic-debugging/test-pressure-3.md +69 -69
  149. package/skills/using-git-worktrees/SKILL.md +215 -215
  150. package/skills/verification-before-completion/SKILL.md +154 -154
  151. package/skills/webapp-testing/SKILL.md +19 -19
  152. package/dist/worker/console/static/assets/index-fsjzREob.js +0 -56
@@ -846,55 +846,19 @@ function mergeGitignoreManagedBlock(existing, block) {
846
846
  return `${block}\n`;
847
847
  return `${existing.trimEnd()}\n\n${block}\n`;
848
848
  }
849
- /** Shared ignore rules for loop-agent runtime facts; keep prompts/ and directory structure shareable. */
849
+ /**
850
+ * Shared ignore rules for loop-agent runtime facts. The four directories are
851
+ * local, rebuildable runtime surface: every developer runs loop-agent init
852
+ * after cloning, so nothing inside them is team-shareable by default.
853
+ * `scripts/` and `ai_workspace/loop-agent/` stay on the shared commit surface.
854
+ */
850
855
  export function buildManagedGitignoreBlock() {
851
856
  return [
852
857
  GITIGNORE_BLOCK_START,
853
- "# loop-agent runtime: ignore personal/session facts; keep prompts and directory placeholders shareable",
854
- "",
855
- "# tasks (source + runtime state per developer/session)",
856
- ".harness/tasks/*",
857
- "!.harness/tasks/.gitkeep",
858
- "",
859
- "# DAG runs",
860
- ".harness/dag-runs/active/*",
861
- "!.harness/dag-runs/active/.gitkeep",
862
- ".harness/dag-runs/completed/*",
863
- "!.harness/dag-runs/completed/.gitkeep",
864
- ".harness/dag-runs/paused/*",
865
- "!.harness/dag-runs/paused/.gitkeep",
866
- "",
867
- "# one-shot executor runs",
868
- ".harness/runs/*",
869
- "!.harness/runs/.gitkeep",
870
- "!.harness/runs/active/",
871
- "!.harness/runs/active/.gitkeep",
872
- "!.harness/runs/completed/",
873
- "!.harness/runs/completed/.gitkeep",
874
- "!.harness/runs/failed/",
875
- "!.harness/runs/failed/.gitkeep",
876
- "",
877
- "# session / cache / logs / recomputable surface state",
878
- ".harness/live/",
879
- ".harness/cache/",
880
- ".harness/*.log",
881
- ".harness/init-surface.json",
882
- ".harness/init-upgrades/",
883
- "",
884
- "# Eval Lab runtime (candidates, campaigns, scorecards, dogfood scratch)",
885
- ".harness/evaluation/",
886
- "",
887
- "# fullstack / worker dogfood evidence (local campaign scratch)",
888
- ".harness/dogfood-evidence/",
889
- "",
890
- "# worker task pool",
891
- ".harness/task-pool/*",
858
+ "# loop-agent runtime: local, rebuildable facts; do not ignore scripts/ or ai_workspace/loop-agent/",
859
+ ".harness/",
860
+ ".agents/",
892
861
  ".task-pool/",
893
- "",
894
- "# repo-local skill dependencies installed inside projected skills",
895
- ".agents/skills/*/node_modules/",
896
- "",
897
- "# multi-worktree parallel mode",
898
862
  ".worktrees/",
899
863
  GITIGNORE_BLOCK_END,
900
864
  ].join("\n");
@@ -1087,26 +1051,8 @@ async function ensureHarnessDirs(repoRoot, written) {
1087
1051
  await mkdir(path.join(repoRoot, dir), { recursive: true });
1088
1052
  written.push(dir);
1089
1053
  }
1090
- const placeholders = [
1091
- ".harness/tasks/.gitkeep",
1092
- ".harness/dag-runs/active/.gitkeep",
1093
- ".harness/dag-runs/completed/.gitkeep",
1094
- ".harness/dag-runs/paused/.gitkeep",
1095
- ".harness/runs/.gitkeep",
1096
- ".harness/runs/active/.gitkeep",
1097
- ".harness/runs/completed/.gitkeep",
1098
- ".harness/runs/failed/.gitkeep",
1099
- ];
1100
- for (const relativePath of placeholders) {
1101
- const target = path.join(repoRoot, relativePath);
1102
- try {
1103
- await access(target);
1104
- }
1105
- catch {
1106
- await writeFile(target, "", "utf8");
1107
- written.push(relativePath);
1108
- }
1109
- }
1054
+ // No .gitkeep placeholders: .harness/ is fully ignored by the managed
1055
+ // gitignore block, and runtime dirs are recreated on demand by init/doctor.
1110
1056
  }
1111
1057
  async function writeCompatPrompts(input) {
1112
1058
  for (const [name, content] of Object.entries(COMPAT_PROMPTS)) {
@@ -2490,6 +2436,7 @@ export function buildInitInstructions(input) {
2490
2436
  "When the user says `loop-agent初始化更新`, `loop-agent 初始化更新`, `更新 loop-agent 初始化内容`, or an equivalent write request, run `loop-agent init upgrade --repo-root . --json` as the single controller-owned entry. Complete returned single-file allowedPaths merge tasks and call `--continue` until a stable terminal result; do not stop at check-update, needs-safe-update, needs-model-merge, or verification-pending.",
2491
2437
  "Explicit `检查初始化更新`, `初始化更新校验`, or `只检查,不要修改` stays strictly read-only: run only `loop-agent init check-update --repo-root . --markdown` and do not create an upgrade run.",
2492
2438
  "Upgrade manages project `.opencode/plugins/`, `.pi/extensions/`, and `.pi/settings.json` by nested merge. Pi must trust the project before loading these settings; do not read or write `~/.pi/agent/settings.json` by default.",
2439
+ "The upgrade also writes a read-only gitignore migration assessment (`.harness/init-upgrades/<run-id>/gitignore-migration.json`): tracked `.harness/**` and `.agents/**` get index-only `git rm -r --cached --ignore-unmatch` guidance (working-tree files are kept). Pause for human review when `.agents` content is tracked or when staged changes / git query failures block safe guidance; never run git rm yourself.",
2493
2440
  "",
2494
2441
  "## Apply Defaults",
2495
2442
  "",
@@ -2505,7 +2452,7 @@ export function buildInitInstructions(input) {
2505
2452
  "- Enrich the root `README.md`: keep the deterministic project title and the loop-agent managed block intact, and fill the human-authored sections (项目概览, 技术栈与目录结构, 开发与验证) from the target project's actual files. The root README must serve both as a human-first project entry and as an agent work entry; replace the initialization-model supplement comments when the project files provide the information.",
2506
2453
  `- Populate \`${governanceRoot}/verification-matrix.md\` with the target project's actual quick, standard, and full verification commands derived from its real language and toolchain, keeping the governance rows intact.`,
2507
2454
  "- Project repo-local skills live in `.agents/skills/`. Do not create a root `skills/` directory in the target project; the package's bundled `skills/` remains the built-in fallback.",
2508
- "- Merge a loop-agent managed block into `.gitignore` that ignores harness runtime facts (tasks, dag-runs, runs, evaluation, dogfood-evidence, live, cache, init-surface.json, .harness/task-pool, legacy .task-pool residue) while keeping prompts and directory placeholders shareable.",
2455
+ "- Merge a loop-agent managed block into `.gitignore` that ignores exactly the four local, rebuildable runtime directories (`.harness/`, `.agents/`, `.task-pool/`, `.worktrees/`) while keeping `scripts/` and `ai_workspace/loop-agent/` on the shared commit surface; keep all user rules outside the block untouched.",
2509
2456
  "- Do not copy examples by default; examples stay bundled in the tool and are available through `loop-agent examples`.",
2510
2457
  "- Add or update a loop-agent managed block in AGENTS.md.",
2511
2458
  "- The generated AGENTS.md must include documentation convergence and structured DAG write-boundary rules so target projects keep the same working discipline as this repository.",
@@ -3382,7 +3329,7 @@ export async function runInitDoctor(input) {
3382
3329
  ? await readFile(gitignorePath, "utf-8")
3383
3330
  : "";
3384
3331
  add("gitignore loop-agent block", gitignoreContent.includes(GITIGNORE_BLOCK_START) &&
3385
- gitignoreContent.includes(".harness/tasks/*"), ".gitignore managed runtime ignores");
3332
+ gitignoreContent.includes(".harness/"), ".gitignore managed runtime ignores");
3386
3333
  const requiredScripts = Object.keys(INIT_SCRIPT_FILES);
3387
3334
  const missingScripts = [];
3388
3335
  for (const script of requiredScripts) {
@@ -37,17 +37,17 @@ export function parseLoopBenchmarkArgs(args) {
37
37
  return { json, markdown, outputPath };
38
38
  }
39
39
  export function printLoopBenchmarkUsage() {
40
- console.log(`usage: loop-benchmark [options]
41
-
42
- Deterministic loop-agent loop benchmark baseline (no live Pi/Cursor calls).
43
-
44
- Options:
45
- --json Emit JSON (default when no format flag is set)
46
- --markdown Emit Markdown report
47
- --output <path> Write Markdown report to a repo-relative or absolute path
48
- -h, --help Show this help
49
-
50
- Control groups: single-repair, 3-pass-convergence, 3-pass-convergence+quota.
40
+ console.log(`usage: loop-benchmark [options]
41
+
42
+ Deterministic loop-agent loop benchmark baseline (no live Pi/Cursor calls).
43
+
44
+ Options:
45
+ --json Emit JSON (default when no format flag is set)
46
+ --markdown Emit Markdown report
47
+ --output <path> Write Markdown report to a repo-relative or absolute path
48
+ -h, --help Show this help
49
+
50
+ Control groups: single-repair, 3-pass-convergence, 3-pass-convergence+quota.
51
51
  Recommendation never changes convergence.enabled default.`);
52
52
  }
53
53
  export async function runLoopBenchmark(repoRoot, rawArgs) {
@@ -106,22 +106,22 @@ export function parsePiReuseBenchmarkArgs(args) {
106
106
  };
107
107
  }
108
108
  export function printPiReuseBenchmarkUsage() {
109
- console.log(`usage: pi-reuse-benchmark [options]
110
-
111
- Deterministic Pi runtime reuse benchmark/decision summary (no live Pi calls).
112
-
113
- Options:
114
- --report <path> Benchmark report markdown (approval status)
115
- --approval <path> Explicit approval JSON artifact
116
- --off-executor <path> Baseline executor.jsonl (reuse off)
117
- --on-executor <path> Treatment executor.jsonl (reuse on)
118
- --off-task <task-id> Resolve baseline from .harness/tasks/<id>/logs/executor.jsonl
119
- --on-task <task-id> Resolve treatment from .harness/tasks/<id>/logs/executor.jsonl
120
- --json Emit JSON (default when no format flag is set)
121
- --markdown Emit Markdown summary
122
- -h, --help Show this help
123
-
124
- Recommendations: defer | maintain-opt-in | eligible-for-human-review
109
+ console.log(`usage: pi-reuse-benchmark [options]
110
+
111
+ Deterministic Pi runtime reuse benchmark/decision summary (no live Pi calls).
112
+
113
+ Options:
114
+ --report <path> Benchmark report markdown (approval status)
115
+ --approval <path> Explicit approval JSON artifact
116
+ --off-executor <path> Baseline executor.jsonl (reuse off)
117
+ --on-executor <path> Treatment executor.jsonl (reuse on)
118
+ --off-task <task-id> Resolve baseline from .harness/tasks/<id>/logs/executor.jsonl
119
+ --on-task <task-id> Resolve treatment from .harness/tasks/<id>/logs/executor.jsonl
120
+ --json Emit JSON (default when no format flag is set)
121
+ --markdown Emit Markdown summary
122
+ -h, --help Show this help
123
+
124
+ Recommendations: defer | maintain-opt-in | eligible-for-human-review
125
125
  Never changes CODE_AGENT_PI_REUSE_RUNTIME default (off).`);
126
126
  }
127
127
  function resolveRepoRelative(repoRoot, filePath) {
@@ -1,4 +1,18 @@
1
1
  export const DEFAULT_RUN_DAG_PROGRESS_INTERVAL_MS = 30_000;
2
+ /**
3
+ * Contract floor for periodic progress output. Values below this are rejected
4
+ * at CLI parse time (aligned with `dag execute`'s parseProgressIntervalMs).
5
+ * Lives here (not src/application/dag/args.ts) because task-advance shares it
6
+ * and src/application/dag/args.ts is outside the task-advance write boundary.
7
+ */
8
+ export const MIN_RUN_DAG_PROGRESS_INTERVAL_MS = 1_000;
9
+ /** Validate a progress interval; throws with the dag execute error contract. */
10
+ export function validateRunDagProgressIntervalMs(value) {
11
+ if (!Number.isInteger(value) || value < MIN_RUN_DAG_PROGRESS_INTERVAL_MS) {
12
+ throw new Error("progress-interval-ms must be an integer >= 1000");
13
+ }
14
+ return value;
15
+ }
2
16
  function formatDuration(durationMs) {
3
17
  const totalSeconds = Math.max(0, Math.floor(durationMs / 1_000));
4
18
  const hours = Math.floor(totalSeconds / 3_600);
@@ -1,5 +1,6 @@
1
1
  import { advanceTaskLifecycle, } from "../application/task-lifecycle/index.js";
2
2
  import { buildOperatorResult, operatorFailed, processExitCodeForOutcome, writeOperatorJson, } from "../shared/operator/index.js";
3
+ import { createRunDagProgressObserver, validateRunDagProgressIntervalMs, } from "./run-dag-progress.js";
3
4
  const COMMAND = "task advance";
4
5
  const USAGE = `usage:
5
6
  task advance <task-id> [title]
@@ -24,6 +25,8 @@ const USAGE = `usage:
24
25
  [--dag-output <path>]
25
26
  [--skip-finalize]
26
27
  [--no-strict-models]
28
+ [--quiet]
29
+ [--progress-interval-ms <ms>]
27
30
  [--dry-run]
28
31
  [--json]`;
29
32
  function pushList(target, value) {
@@ -170,6 +173,19 @@ export function parseTaskAdvanceArgs(args) {
170
173
  options.strictModels = false;
171
174
  continue;
172
175
  }
176
+ if (token === "--quiet") {
177
+ options.quiet = true;
178
+ continue;
179
+ }
180
+ if (token === "--progress-interval-ms") {
181
+ const raw = next();
182
+ const parsed = Number(raw);
183
+ if (!Number.isFinite(parsed)) {
184
+ throw new Error(`progress-interval-ms must be an integer >= 1000\n${USAGE}`);
185
+ }
186
+ options.progressIntervalMs = validateRunDagProgressIntervalMs(parsed);
187
+ continue;
188
+ }
173
189
  if (token === "--help" || token === "-h") {
174
190
  throw new Error(USAGE);
175
191
  }
@@ -208,7 +224,7 @@ function mapOutcome(result) {
208
224
  return "blocked";
209
225
  return "succeeded";
210
226
  }
211
- function toUseCaseInput(repoRoot, taskId, options) {
227
+ export function toUseCaseInput(repoRoot, taskId, options, observer) {
212
228
  const timeouts = new Map((options.verifyTimeout ?? []).map((entry) => {
213
229
  const parsed = parseVerifyTimeout(entry);
214
230
  return [parsed.label, parsed.timeoutMs];
@@ -270,7 +286,8 @@ function toUseCaseInput(repoRoot, taskId, options) {
270
286
  dagOutputPath: options.dagOutputPath,
271
287
  skipFinalize: options.skipFinalize,
272
288
  strictModels: options.strictModels,
273
- onProgress: options.json
289
+ ...(observer ? { observer } : {}),
290
+ onProgress: options.quiet
274
291
  ? undefined
275
292
  : (message) => {
276
293
  process.stderr.write(`[task advance] ${message}\n`);
@@ -295,8 +312,18 @@ export async function runTaskAdvance(repoRoot, args) {
295
312
  process.exitCode = processExitCodeForOutcome(envelope.outcome);
296
313
  return;
297
314
  }
315
+ let progress;
298
316
  try {
299
- const result = await advanceTaskLifecycle(toUseCaseInput(repoRoot, taskId, options));
317
+ // AC-006/AC-007: periodic DAG progress goes to stderr only (stdout stays
318
+ // the single final OperatorCommandResultV1 JSON). The observer is created
319
+ // ONLY for the approve-gate execution path and disposed on every exit;
320
+ // its timer starts only after onRunStart (double guard, no stray timer).
321
+ if (options.approveGate && !options.dryRun && !options.quiet) {
322
+ progress = createRunDagProgressObserver({
323
+ intervalMs: options.progressIntervalMs,
324
+ });
325
+ }
326
+ const result = await advanceTaskLifecycle(toUseCaseInput(repoRoot, taskId, options, progress?.observer));
300
327
  const outcome = mapOutcome(result);
301
328
  const gateStop = result.lifecycleState === "awaiting-write-set-approval" &&
302
329
  result.blockers.length === 0;
@@ -332,4 +359,7 @@ export async function runTaskAdvance(repoRoot, args) {
332
359
  writeOperatorJson(envelope);
333
360
  process.exitCode = processExitCodeForOutcome(envelope.outcome);
334
361
  }
362
+ finally {
363
+ progress?.dispose();
364
+ }
335
365
  }
@@ -445,6 +445,43 @@ export function buildOperatorCapabilitiesDocument() {
445
445
  modelCallable: "always",
446
446
  humanConfirmation: "none",
447
447
  },
448
+ {
449
+ action: "operationWait",
450
+ cli: "console canonical operation wait (read-only event-driven long poll)",
451
+ kind: "read",
452
+ inputSchemaVersion: 1,
453
+ resultSchemaVersion: 1,
454
+ resultPolicy: { readOnly: true, bounded: true, redacted: true },
455
+ envelopeSchemaVersion: 1,
456
+ requiredErrorCodes: [
457
+ "NOT_FOUND",
458
+ "INVALID_INPUT",
459
+ "EVENT_CURSOR_EXPIRED",
460
+ ],
461
+ description: "Read-only event-driven wait on the canonical operation event ring. Returns immediately on existing events, terminal state or needs-reconcile; otherwise resolves on the first new event/state change or after maxWaitMs with timedOut:true (a success summary, not a command failure).",
462
+ inputParams: [
463
+ {
464
+ name: "operationId",
465
+ type: "string",
466
+ required: true,
467
+ description: "canonical operation id",
468
+ },
469
+ {
470
+ name: "afterSeq",
471
+ type: "number",
472
+ required: false,
473
+ description: "event cursor; only events with seq > afterSeq count (default 0, >= 0)",
474
+ },
475
+ {
476
+ name: "maxWaitMs",
477
+ type: "number",
478
+ required: false,
479
+ description: "bounded wait budget; clamped server-side (default 180000, floor 60000)",
480
+ },
481
+ ],
482
+ modelCallable: "always",
483
+ humanConfirmation: "none",
484
+ },
448
485
  {
449
486
  action: "contractShow",
450
487
  cli: "loop-agent task status <taskId> --json",
@@ -965,7 +1002,7 @@ export function buildOperatorCapabilitiesDocument() {
965
1002
  "CONTROLLER_MISMATCH",
966
1003
  "INVALID_INPUT",
967
1004
  ],
968
- description: "Consume a single-use execution receipt and execute the reviewed DAG (prefer task advance --approve-gate when gate token present). accepted/queued/running/operationId are NOT completion — supervise via operationGet/status/dagReport/dagDoctor.",
1005
+ description: "Consume a single-use execution receipt and execute the reviewed DAG (prefer task advance --approve-gate when gate token present). accepted/queued/running/operationId are NOT completion — supervise via operationGet/status/dagReport/dagDoctor/operationWait.",
969
1006
  inputParams: [
970
1007
  {
971
1008
  name: "executionId",
@@ -29,7 +29,7 @@ export function resolveArtifactWriteDir(options) {
29
29
  }
30
30
  export function buildArtifactPathPrompt(writeDir) {
31
31
  if (!writeDir)
32
- return `After changes, write artifacts/修改记录.md and artifacts/验证结果.md with verification evidence.
32
+ return `After changes, write artifacts/修改记录.md and artifacts/验证结果.md with verification evidence.
33
33
  ${ARTIFACT_INSTRUCTIONS}`;
34
34
  return [
35
35
  `After changes, write the following files:`,
@@ -58,11 +58,11 @@ import { RUNTIME_CONTEXT_TEXT_MAX, redactRuntimeText, } from "./runtime-context.
58
58
  */
59
59
  export const OPERATOR_CHAT_SYSTEM_PROMPT_BASE = [
60
60
  "You are the General Operator Chat for loop-agent / agent-worker.",
61
- "Operate through operator_* tools first. Pi read/write/edit/bash/grep/find/ls are available, but obey the loaded repository AGENTS.md. apply_patch, full-tools, shell, and coding-chat are denied alternate runtimes. Prefer safe-read/safe-grep for sensitive probes.",
62
- "If a repository has no loop-agent harness and the user requests initialization, run `loop-agent init instructions --repo-root .` then `loop-agent init --repo-root . --profile full --merge`; finish the generated setup, init doctor, inspect, docs audit, and quick verification. For updates use `loop-agent init upgrade --repo-root . --json` until stable; use init check-update only for an explicitly read-only request.",
63
- "After a DAG is started or accepted, do not end on accepted/queued/running or an operationId. Supervise it to a terminal outcome. Poll operationGet, task status, dagReport, and dagDoctor after about 15 seconds on start/change, every 30 seconds during progress, and every 60 seconds after 3 minutes unchanged. Report only meaningful node/rank changes, review/verify/closeout, recovery, liveness concerns, and terminal outcomes.",
64
- "On failure read primaryFailure, primaryRecovery, and doctor evidence. If meaningful progress exists, wait. Otherwise use a fresh eligible dagRerunPlan and rerun its safe node; when ineligible follow AGENTS.md/runtime recovery for same-task rerun/advance, resume, or Worker retry. Auto-approve only a bounded writeSet inside allowedPaths, outside forbiddenPaths, without broad/destructive risk, and with structured verification.",
65
- "Continue until success, user stop, or no safe eligible recovery remains because limits, bindings, auth/quota recovery, or required external authorization are exhausted. Never create a new task for a transient failure or replace repair-pi with direct edits.",
61
+ "Operate through operator_* tools first. Pi read/write/edit/bash/grep/find/ls are available; obey the loaded repository AGENTS.md. apply_patch, full-tools, shell, coding-chat are denied. Prefer safe-read/safe-grep for sensitive probes.",
62
+ "If a repository has no loop-agent harness and the user requests initialization, run `loop-agent init instructions --repo-root .` then `loop-agent init --repo-root . --profile full --merge`; finish with init doctor, inspect, docs audit, quick verification. For updates use `loop-agent init upgrade --repo-root . --json` until stable; init check-update only for explicitly read-only requests.",
63
+ "Long-running DAGs must run via prepareDagExecution → runDag → operationId held by a Console operation; never foreground-Bash `task advance --approve-gate`, no tail/head pipes, no hand-rolled nohup/Start-Process/start; do not end on accepted/queued/running or an operationId NOT completion. Supervise via operationGet/status/dagReport/dagDoctor/operationWait: 60s 90s 120s 180s backoff; reset to 60s on state change; 30-60s re-checks when stall suspected. Report only meaningful node/rank changes, review/verify/closeout, recovery, liveness, terminal outcomes.",
64
+ "On failure read primaryFailure, primaryRecovery, doctor evidence. If meaningful progress exists, wait. Else use a fresh eligible dagRerunPlan and rerun its safe node; when ineligible follow AGENTS.md/runtime recovery for same-task rerun/advance, resume, or Worker retry. Auto-approve only a bounded writeSet inside allowedPaths, outside forbiddenPaths, without broad/destructive risk, with structured verification.",
65
+ "Continue until success, user stop, or no safe eligible recovery remains (limits, bindings, auth/quota, external authorization exhausted). Never create a new task for a transient failure or replace repair-pi with direct edits.",
66
66
  ].join("\n");
67
67
  /** Compose the inspectable system prompt actually injected into Operator Chat. */
68
68
  export function composeOperatorChatSystemPrompt(input) {
@@ -1280,6 +1280,7 @@ export class ConsolePiRuntime {
1280
1280
  * session is not active — callers fail closed instead of fabricating data.
1281
1281
  */
1282
1282
  /**
1283
+ * ConsolePiRuntime.listSlashCommands
1283
1284
  * Browser-safe slash command projection for UI-11.
1284
1285
  * Sources: current Session extension commands, prompt templates, skills.
1285
1286
  * Never returns prompt/skill file bodies.
@@ -1292,29 +1293,36 @@ export class ConsolePiRuntime {
1292
1293
  .replace(/[A-Za-z]:\\Users\\[^\s]+/gi, "~")
1293
1294
  .slice(0, 240);
1294
1295
  };
1295
- try {
1296
- await this.ensureSessionReady(sessionId);
1297
- }
1298
- catch {
1299
- return [];
1300
- }
1296
+ // Fail soft at the HTTP layer: bubble readiness errors so routes can attach
1297
+ // a compact warning while the UI keeps local Operator/Pi Web commands.
1298
+ await this.ensureSessionReady(sessionId);
1301
1299
  const session = this.sessions.get(sessionId);
1302
1300
  if (!session)
1303
- return [];
1304
- const out = [];
1305
- const seen = new Set();
1301
+ return { commands: [] };
1302
+ // Same-name conflict priority (AC-3): extension > skill > prompt.
1303
+ // Enum order must not decide the winner — prompt-before-skill still loses.
1304
+ const SOURCE_RANK = {
1305
+ extension: 0,
1306
+ skill: 1,
1307
+ prompt: 2,
1308
+ };
1309
+ const byKey = new Map();
1310
+ const sourceWarnings = [];
1306
1311
  const push = (command, label, description, source) => {
1307
1312
  const name = command.startsWith("/") ? command : `/${command}`;
1308
1313
  const key = name.toLowerCase();
1309
- if (!key || key === "/" || seen.has(key))
1314
+ if (!key || key === "/")
1310
1315
  return;
1311
- seen.add(key);
1312
- out.push({
1316
+ const next = {
1313
1317
  command: name,
1314
1318
  label: (label || name).slice(0, 80),
1315
1319
  description: scrub(description),
1316
1320
  source,
1317
- });
1321
+ };
1322
+ const existing = byKey.get(key);
1323
+ if (!existing || SOURCE_RANK[source] < SOURCE_RANK[existing.source]) {
1324
+ byKey.set(key, next);
1325
+ }
1318
1326
  };
1319
1327
  try {
1320
1328
  const cmds = session.extensionRunner?.getRegisteredCommands?.() ?? [];
@@ -1325,18 +1333,20 @@ export class ConsolePiRuntime {
1325
1333
  push(String(name), String(cmd.name || name), String(cmd.description || ""), "extension");
1326
1334
  }
1327
1335
  }
1328
- catch {
1329
- // non-blocking
1336
+ catch (error) {
1337
+ // Source isolation: keep other sources; surface compact warning (AC-3).
1338
+ sourceWarnings.push(`extension: ${scrub(error instanceof Error ? error.message : String(error))}`);
1330
1339
  }
1331
1340
  try {
1341
+ // Enumerate prompts before skills on purpose: priority must still let skill win.
1332
1342
  for (const tpl of session.promptTemplates ?? []) {
1333
1343
  if (!tpl?.name)
1334
1344
  continue;
1335
1345
  push(String(tpl.name), String(tpl.name), String(tpl.description || ""), "prompt");
1336
1346
  }
1337
1347
  }
1338
- catch {
1339
- // non-blocking
1348
+ catch (error) {
1349
+ sourceWarnings.push(`prompt: ${scrub(error instanceof Error ? error.message : String(error))}`);
1340
1350
  }
1341
1351
  try {
1342
1352
  const services = this.serviceScopes.get(sessionId);
@@ -1351,10 +1361,16 @@ export class ConsolePiRuntime {
1351
1361
  push(command, skill.name, String(skill.description || ""), "skill");
1352
1362
  }
1353
1363
  }
1354
- catch {
1355
- // non-blocking
1364
+ catch (error) {
1365
+ sourceWarnings.push(`skill: ${scrub(error instanceof Error ? error.message : String(error))}`);
1356
1366
  }
1357
- return out;
1367
+ const warning = sourceWarnings.length > 0
1368
+ ? sourceWarnings.join("; ").slice(0, 160)
1369
+ : undefined;
1370
+ return {
1371
+ commands: [...byKey.values()],
1372
+ ...(warning ? { warning } : {}),
1373
+ };
1358
1374
  }
1359
1375
  async getRuntimeSnapshot(sessionId, options) {
1360
1376
  try {
@@ -1645,14 +1645,24 @@ async function handleChatFiles(res, deps, sessionId, query) {
1645
1645
  }
1646
1646
  export async function handleChatCommands(res, deps, sessionId) {
1647
1647
  try {
1648
- const commands = typeof deps.runtime.listSlashCommands === "function"
1648
+ const listed = typeof deps.runtime.listSlashCommands === "function"
1649
1649
  ? await deps.runtime.listSlashCommands(sessionId)
1650
- : [];
1650
+ : { commands: [] };
1651
+ // Partial single-source failures keep successful commands + compact warning (AC-3).
1652
+ const commands = listed.commands ?? [];
1653
+ const warning = typeof listed.warning === "string" && listed.warning.trim()
1654
+ ? listed.warning.trim().slice(0, 160)
1655
+ : undefined;
1651
1656
  res.statusCode = 200;
1652
1657
  res.setHeader("content-type", "application/json; charset=utf-8");
1653
- res.end(JSON.stringify({ ok: true, commands }));
1658
+ res.end(JSON.stringify({
1659
+ ok: true,
1660
+ commands,
1661
+ ...(warning ? { warning } : {}),
1662
+ }));
1654
1663
  }
1655
1664
  catch (error) {
1665
+ // Whole-list / readiness failures stay fail-soft with empty commands + warning.
1656
1666
  res.statusCode = 200;
1657
1667
  res.setHeader("content-type", "application/json; charset=utf-8");
1658
1668
  res.end(JSON.stringify({
@@ -2799,8 +2809,21 @@ export async function handleChatCompact(req, res, deps, sessionId) {
2799
2809
  return;
2800
2810
  }
2801
2811
  }
2812
+ let body = {};
2813
+ try {
2814
+ body = await readJsonBody(req);
2815
+ }
2816
+ catch {
2817
+ sendJson(res, 400, {
2818
+ ok: false,
2819
+ error: { code: "INVALID_INPUT", message: "invalid json body" },
2820
+ });
2821
+ return;
2822
+ }
2823
+ const rawInstructions = typeof body.instructions === "string" ? body.instructions : undefined;
2824
+ const customInstructions = rawInstructions?.trim() || undefined;
2802
2825
  try {
2803
- const snapshot = await deps.runtime.compact(sessionId);
2826
+ const snapshot = await deps.runtime.compact(sessionId, customInstructions);
2804
2827
  const event = deps.events.append(sessionId, deps.events.latestTurnId(sessionId) ?? `${sessionId}:compact`, { kind: "compact", data: snapshot });
2805
2828
  sendJson(res, 200, { ok: true, snapshot, eventId: event.eventId });
2806
2829
  }
@@ -74,6 +74,30 @@ export async function runOperation(operationId, deps) {
74
74
  message: chunk,
75
75
  });
76
76
  },
77
+ onHeartbeat: (info) => {
78
+ // P1 (2026-08-13): project sibling CLI heartbeats into canonical
79
+ // operation events so operationWait/operationEventSummary can consume
80
+ // them. The projection is best-effort: a failed append must never
81
+ // terminate the sibling CLI execution (AC-002). The injected test
82
+ // runCommand path has no LoopAgentClient callback guard, so the
83
+ // runner protects itself here.
84
+ try {
85
+ deps.events.append(operationId, {
86
+ at: info.at,
87
+ kind: "heartbeat",
88
+ message: `heartbeat elapsedMs=${info.elapsedMs}`,
89
+ data: {
90
+ elapsedMs: info.elapsedMs,
91
+ action: op.action,
92
+ ...(op.taskId ? { taskId: op.taskId } : {}),
93
+ ...(op.dagRunId ? { dagRunId: op.dagRunId } : {}),
94
+ },
95
+ });
96
+ }
97
+ catch {
98
+ // Heartbeat is derived telemetry; ignore projection failures.
99
+ }
100
+ },
77
101
  });
78
102
  await spawnUpdate;
79
103
  finished = await finalizeFromWorkerResult(operationId, result, deps);