@tea-agent/loop-agent 0.35.0-beta.2 → 0.35.1-beta.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (154) hide show
  1. package/AGENTS.md +108 -108
  2. package/CHANGELOG.md +55 -4
  3. package/README.md +165 -165
  4. package/bin/agent-worker.js +0 -0
  5. package/bin/loop-agent.js +21 -21
  6. package/dist/application/task-lifecycle/advance.js +1 -0
  7. package/dist/commands/cursor-prompt.js +6 -6
  8. package/dist/commands/init-upgrade.js +351 -19
  9. package/dist/commands/init.js +14 -67
  10. package/dist/commands/loop-benchmark.js +11 -11
  11. package/dist/commands/pi-reuse-benchmark.js +16 -16
  12. package/dist/commands/run-dag-progress.js +14 -0
  13. package/dist/commands/task-advance.js +33 -3
  14. package/dist/shared/operator/capabilities.js +38 -1
  15. package/dist/sidecars/cursor-prompt/executor.js +1 -1
  16. package/dist/worker/console/chat/pi-runtime.js +41 -25
  17. package/dist/worker/console/chat/routes.js +27 -4
  18. package/dist/worker/console/operation-runner.js +24 -0
  19. package/dist/worker/console/operation-wait.js +241 -0
  20. package/dist/worker/console/operator-actions.js +58 -0
  21. package/dist/worker/console/static/assets/{index-hJqCPs_g.css → index-Dups4sSM.css} +1 -1
  22. package/dist/worker/console/static/assets/index-SjjjZnV3.js +56 -0
  23. package/dist/worker/console/static/index.html +2 -2
  24. package/dist/worker/console/static-src/operator-chat/slash-palette-nav.js +141 -0
  25. package/dist/worker/console/static-src/operator-chat/useChatSessions.js +13 -2
  26. package/dist/worker/console/static-src/operator-chat/useComposer.js +30 -7
  27. package/dist/worker/observe/static/copy.js +67 -67
  28. package/dist/worker/observe/static/dag-layout.d.ts +36 -36
  29. package/dist/worker/observe/static/dom.js +220 -220
  30. package/dist/worker/observe/static/relations.js +133 -133
  31. package/dist/worker/observe/static/run-processing.js +148 -148
  32. package/dist/worker/observe/static/views/batch.js +227 -227
  33. package/dist/worker/observe/static/views/failures.js +143 -143
  34. package/dist/worker/observe/static/views/feature.js +492 -492
  35. package/dist/worker/observe/static/views/run.js +453 -453
  36. package/dist/worker/observe/static/views/shell.js +7 -7
  37. package/dist/worker/observe/static/views/timeline.js +163 -163
  38. package/dist/workflows/dag/canvas-observer.js +275 -275
  39. package/dist/workflows/dag/frontend-prewrite-gate.js +9 -1
  40. package/dist/workflows/dag/init-hybrid.js +2 -0
  41. package/docs/architecture/evolution.md +73 -73
  42. package/docs/architecture/system-overview.md +100 -100
  43. package/docs/architecture/worker-and-feature.md +122 -122
  44. package/docs/skills/README.md +7 -7
  45. package/docs/templates/adr.md +60 -60
  46. package/docs/templates/agent-dag-authority-surface-audit.prompt.md +94 -94
  47. package/docs/templates/agent-dag-decision-envelope.schema.json +213 -213
  48. package/docs/templates/agent-dag-decision-gate.prompt.md +246 -246
  49. package/docs/templates/agent-dag-process-supervisor.prompt.md +98 -98
  50. package/docs/templates/agent-dag-report.schema.json +473 -473
  51. package/docs/templates/agent-dag-review-verdict.prompt.md +68 -68
  52. package/docs/templates/backend-test-result.schema.json +99 -99
  53. package/docs/templates/evaluation/agents-map-slim-v1.md +87 -87
  54. package/docs/templates/evaluation/agents-map-verbose-v0.md +153 -153
  55. package/docs/templates/feature-spec.md +53 -53
  56. package/docs/templates/frontend-design-contract.md +42 -42
  57. package/docs/templates/frontend-eval/fixtures/failures/01-type-build-error.md +17 -17
  58. package/docs/templates/frontend-eval/fixtures/failures/02-unit-component-test-fail.md +16 -16
  59. package/docs/templates/frontend-eval/fixtures/failures/03-fixture-schema-drift.md +16 -16
  60. package/docs/templates/frontend-eval/fixtures/failures/04-missing-loading-empty-error-state.md +16 -16
  61. package/docs/templates/frontend-eval/fixtures/failures/05-forbidden-write-writeset-expansion.md +16 -16
  62. package/docs/templates/frontend-eval/fixtures/failures/06-unapproved-dependency-add.md +16 -16
  63. package/docs/templates/frontend-eval/fixtures/failures/07-mock-production-on.md +21 -21
  64. package/docs/templates/frontend-eval/fixtures/functional/01-simple-component-style.md +29 -29
  65. package/docs/templates/frontend-eval/fixtures/functional/02-form-validation.md +28 -28
  66. package/docs/templates/frontend-eval/fixtures/functional/03-list-detail-page.md +28 -28
  67. package/docs/templates/frontend-eval/fixtures/functional/04-api-mock.md +29 -29
  68. package/docs/templates/frontend-eval/fixtures/functional/05-permission-auth-gated-ui.md +27 -27
  69. package/docs/templates/frontend-eval/fixtures/functional/06-ssr-server-client-boundary.md +28 -28
  70. package/docs/templates/frontend-eval/fixtures/functional/07-shared-public-component-api.md +28 -28
  71. package/docs/templates/frontend-eval/fixtures/functional/08-pure-local-no-remote.md +27 -27
  72. package/docs/templates/frontend-eval/metrics.md +138 -138
  73. package/docs/templates/frontend-eval/smoke-targets.md +53 -53
  74. package/docs/templates/frontend-task-constraints.md +35 -35
  75. package/docs/templates/frontend-task-requirement.md +70 -70
  76. package/docs/templates/init-evolution-review.md +35 -35
  77. package/docs/templates/init-managed-agents.md +156 -154
  78. package/docs/templates/interactive-ui-round2-experiment.md +66 -66
  79. package/docs/templates/knowledge-graph-bootstrap-dag.json +118 -118
  80. package/docs/templates/knowledge-sync-dag.json +178 -178
  81. package/docs/templates/knowledge-sync-draft.schema.json +71 -71
  82. package/docs/templates/product-line/closeout.yaml +9 -9
  83. package/docs/templates/product-line/design.md +13 -13
  84. package/docs/templates/product-line/links.md +10 -10
  85. package/docs/templates/product-line/requirement.md +17 -17
  86. package/docs/templates/product-line/test-plan.md +7 -7
  87. package/docs/templates/project-start-checklist.md +9 -9
  88. package/docs/templates/qa-report.md +48 -48
  89. package/docs/templates/sprint-contract.md +29 -29
  90. package/docs/templates/worker-dogfood-evidence.md +80 -80
  91. package/docs/templates/worker-dogfood-setup.md +68 -68
  92. package/harness.json +5 -2
  93. package/package.json +1 -1
  94. package/scripts/kb-bootstrap-init-skeleton.sh +0 -0
  95. package/scripts/kb-graph-incremental-prepare.mjs +0 -0
  96. package/scripts/kb-graph-materialize.mjs +105 -105
  97. package/scripts/kb-graph-promote.mjs +164 -164
  98. package/scripts/kb-query.mjs +554 -554
  99. package/skills/agent-worker/SKILL.md +48 -48
  100. package/skills/agent-worker/references/agent-worker-operator.md +159 -159
  101. package/skills/ai-engineering-context/SKILL.md +48 -48
  102. package/skills/analyze-product-dependencies/scripts/test-validators.mjs +0 -0
  103. package/skills/analyze-product-dependencies/scripts/validate-api-documentation.mjs +0 -0
  104. package/skills/analyze-product-dependencies/scripts/validate-dependency-analysis.mjs +0 -0
  105. package/skills/analyze-product-dependencies/scripts/validate-product-requirement-input.mjs +0 -0
  106. package/skills/analyze-product-requirements/scripts/compute-source-identity.mjs +0 -0
  107. package/skills/analyze-product-requirements/scripts/test-validators.mjs +0 -0
  108. package/skills/analyze-product-requirements/scripts/validate-product-analysis.mjs +0 -0
  109. package/skills/analyze-product-requirements/scripts/validate-product-requirement.mjs +0 -0
  110. package/skills/analyze-product-requirements/scripts/validate-requirement-clarification.mjs +0 -0
  111. package/skills/browser-tools/browser-content.js +103 -103
  112. package/skills/browser-tools/browser-cookies.js +35 -35
  113. package/skills/browser-tools/browser-eval.js +53 -53
  114. package/skills/browser-tools/browser-hn-scraper.js +108 -108
  115. package/skills/browser-tools/browser-nav.js +44 -44
  116. package/skills/browser-tools/browser-pick.js +162 -162
  117. package/skills/browser-tools/browser-screenshot.js +34 -34
  118. package/skills/browser-tools/browser-start.js +86 -86
  119. package/skills/browser-tools/package-lock.json +2556 -2556
  120. package/skills/browser-tools/package.json +19 -19
  121. package/skills/code-review-core/SKILL.md +20 -20
  122. package/skills/codebase-scout/SKILL.md +19 -19
  123. package/skills/grill-me/SKILL.md +10 -10
  124. package/skills/local-jacoco-coverage/scripts/run-coverage-analysis.sh +0 -0
  125. package/skills/local-jacoco-coverage/scripts/start-jacoco-agent.sh +0 -0
  126. package/skills/loop-agent/SKILL.md +1 -0
  127. package/skills/loop-agent/references/command-reference.md +641 -639
  128. package/skills/loop-agent/references/docs-converge.md +126 -126
  129. package/skills/loop-agent/references/learned/README.md +21 -21
  130. package/skills/loop-agent/references/pi-prompt.md +23 -23
  131. package/skills/loop-agent/references/pi-subagent-assisted-mode.md +84 -84
  132. package/skills/playwright-cli/references/element-attributes.md +23 -23
  133. package/skills/playwright-cli/references/playwright-tests.md +39 -39
  134. package/skills/playwright-cli/references/request-mocking.md +87 -87
  135. package/skills/playwright-cli/references/running-code.md +241 -241
  136. package/skills/playwright-cli/references/session-management.md +225 -225
  137. package/skills/playwright-cli/references/storage-state.md +275 -275
  138. package/skills/playwright-cli/references/test-generation.md +433 -433
  139. package/skills/requesting-code-review/SKILL.md +101 -101
  140. package/skills/requesting-code-review/code-reviewer.md +168 -168
  141. package/skills/systematic-debugging/CREATION-LOG.md +119 -119
  142. package/skills/systematic-debugging/condition-based-waiting-example.ts +158 -158
  143. package/skills/systematic-debugging/condition-based-waiting.md +115 -115
  144. package/skills/systematic-debugging/defense-in-depth.md +122 -122
  145. package/skills/systematic-debugging/find-polluter.sh +63 -63
  146. package/skills/systematic-debugging/root-cause-tracing.md +169 -169
  147. package/skills/systematic-debugging/test-academic.md +14 -14
  148. package/skills/systematic-debugging/test-pressure-1.md +58 -58
  149. package/skills/systematic-debugging/test-pressure-2.md +68 -68
  150. package/skills/systematic-debugging/test-pressure-3.md +69 -69
  151. package/skills/using-git-worktrees/SKILL.md +215 -215
  152. package/skills/verification-before-completion/SKILL.md +154 -154
  153. package/skills/webapp-testing/SKILL.md +19 -19
  154. package/dist/worker/console/static/assets/index-fsjzREob.js +0 -56
@@ -106,22 +106,22 @@ export function parsePiReuseBenchmarkArgs(args) {
106
106
  };
107
107
  }
108
108
  export function printPiReuseBenchmarkUsage() {
109
- console.log(`usage: pi-reuse-benchmark [options]
110
-
111
- Deterministic Pi runtime reuse benchmark/decision summary (no live Pi calls).
112
-
113
- Options:
114
- --report <path> Benchmark report markdown (approval status)
115
- --approval <path> Explicit approval JSON artifact
116
- --off-executor <path> Baseline executor.jsonl (reuse off)
117
- --on-executor <path> Treatment executor.jsonl (reuse on)
118
- --off-task <task-id> Resolve baseline from .harness/tasks/<id>/logs/executor.jsonl
119
- --on-task <task-id> Resolve treatment from .harness/tasks/<id>/logs/executor.jsonl
120
- --json Emit JSON (default when no format flag is set)
121
- --markdown Emit Markdown summary
122
- -h, --help Show this help
123
-
124
- Recommendations: defer | maintain-opt-in | eligible-for-human-review
109
+ console.log(`usage: pi-reuse-benchmark [options]
110
+
111
+ Deterministic Pi runtime reuse benchmark/decision summary (no live Pi calls).
112
+
113
+ Options:
114
+ --report <path> Benchmark report markdown (approval status)
115
+ --approval <path> Explicit approval JSON artifact
116
+ --off-executor <path> Baseline executor.jsonl (reuse off)
117
+ --on-executor <path> Treatment executor.jsonl (reuse on)
118
+ --off-task <task-id> Resolve baseline from .harness/tasks/<id>/logs/executor.jsonl
119
+ --on-task <task-id> Resolve treatment from .harness/tasks/<id>/logs/executor.jsonl
120
+ --json Emit JSON (default when no format flag is set)
121
+ --markdown Emit Markdown summary
122
+ -h, --help Show this help
123
+
124
+ Recommendations: defer | maintain-opt-in | eligible-for-human-review
125
125
  Never changes CODE_AGENT_PI_REUSE_RUNTIME default (off).`);
126
126
  }
127
127
  function resolveRepoRelative(repoRoot, filePath) {
@@ -1,4 +1,18 @@
1
1
  export const DEFAULT_RUN_DAG_PROGRESS_INTERVAL_MS = 30_000;
2
+ /**
3
+ * Contract floor for periodic progress output. Values below this are rejected
4
+ * at CLI parse time (aligned with `dag execute`'s parseProgressIntervalMs).
5
+ * Lives here (not src/application/dag/args.ts) because task-advance shares it
6
+ * and src/application/dag/args.ts is outside the task-advance write boundary.
7
+ */
8
+ export const MIN_RUN_DAG_PROGRESS_INTERVAL_MS = 1_000;
9
+ /** Validate a progress interval; throws with the dag execute error contract. */
10
+ export function validateRunDagProgressIntervalMs(value) {
11
+ if (!Number.isInteger(value) || value < MIN_RUN_DAG_PROGRESS_INTERVAL_MS) {
12
+ throw new Error("progress-interval-ms must be an integer >= 1000");
13
+ }
14
+ return value;
15
+ }
2
16
  function formatDuration(durationMs) {
3
17
  const totalSeconds = Math.max(0, Math.floor(durationMs / 1_000));
4
18
  const hours = Math.floor(totalSeconds / 3_600);
@@ -1,5 +1,6 @@
1
1
  import { advanceTaskLifecycle, } from "../application/task-lifecycle/index.js";
2
2
  import { buildOperatorResult, operatorFailed, processExitCodeForOutcome, writeOperatorJson, } from "../shared/operator/index.js";
3
+ import { createRunDagProgressObserver, validateRunDagProgressIntervalMs, } from "./run-dag-progress.js";
3
4
  const COMMAND = "task advance";
4
5
  const USAGE = `usage:
5
6
  task advance <task-id> [title]
@@ -24,6 +25,8 @@ const USAGE = `usage:
24
25
  [--dag-output <path>]
25
26
  [--skip-finalize]
26
27
  [--no-strict-models]
28
+ [--quiet]
29
+ [--progress-interval-ms <ms>]
27
30
  [--dry-run]
28
31
  [--json]`;
29
32
  function pushList(target, value) {
@@ -170,6 +173,19 @@ export function parseTaskAdvanceArgs(args) {
170
173
  options.strictModels = false;
171
174
  continue;
172
175
  }
176
+ if (token === "--quiet") {
177
+ options.quiet = true;
178
+ continue;
179
+ }
180
+ if (token === "--progress-interval-ms") {
181
+ const raw = next();
182
+ const parsed = Number(raw);
183
+ if (!Number.isFinite(parsed)) {
184
+ throw new Error(`progress-interval-ms must be an integer >= 1000\n${USAGE}`);
185
+ }
186
+ options.progressIntervalMs = validateRunDagProgressIntervalMs(parsed);
187
+ continue;
188
+ }
173
189
  if (token === "--help" || token === "-h") {
174
190
  throw new Error(USAGE);
175
191
  }
@@ -208,7 +224,7 @@ function mapOutcome(result) {
208
224
  return "blocked";
209
225
  return "succeeded";
210
226
  }
211
- function toUseCaseInput(repoRoot, taskId, options) {
227
+ export function toUseCaseInput(repoRoot, taskId, options, observer) {
212
228
  const timeouts = new Map((options.verifyTimeout ?? []).map((entry) => {
213
229
  const parsed = parseVerifyTimeout(entry);
214
230
  return [parsed.label, parsed.timeoutMs];
@@ -270,7 +286,8 @@ function toUseCaseInput(repoRoot, taskId, options) {
270
286
  dagOutputPath: options.dagOutputPath,
271
287
  skipFinalize: options.skipFinalize,
272
288
  strictModels: options.strictModels,
273
- onProgress: options.json
289
+ ...(observer ? { observer } : {}),
290
+ onProgress: options.quiet
274
291
  ? undefined
275
292
  : (message) => {
276
293
  process.stderr.write(`[task advance] ${message}\n`);
@@ -295,8 +312,18 @@ export async function runTaskAdvance(repoRoot, args) {
295
312
  process.exitCode = processExitCodeForOutcome(envelope.outcome);
296
313
  return;
297
314
  }
315
+ let progress;
298
316
  try {
299
- const result = await advanceTaskLifecycle(toUseCaseInput(repoRoot, taskId, options));
317
+ // AC-006/AC-007: periodic DAG progress goes to stderr only (stdout stays
318
+ // the single final OperatorCommandResultV1 JSON). The observer is created
319
+ // ONLY for the approve-gate execution path and disposed on every exit;
320
+ // its timer starts only after onRunStart (double guard, no stray timer).
321
+ if (options.approveGate && !options.dryRun && !options.quiet) {
322
+ progress = createRunDagProgressObserver({
323
+ intervalMs: options.progressIntervalMs,
324
+ });
325
+ }
326
+ const result = await advanceTaskLifecycle(toUseCaseInput(repoRoot, taskId, options, progress?.observer));
300
327
  const outcome = mapOutcome(result);
301
328
  const gateStop = result.lifecycleState === "awaiting-write-set-approval" &&
302
329
  result.blockers.length === 0;
@@ -332,4 +359,7 @@ export async function runTaskAdvance(repoRoot, args) {
332
359
  writeOperatorJson(envelope);
333
360
  process.exitCode = processExitCodeForOutcome(envelope.outcome);
334
361
  }
362
+ finally {
363
+ progress?.dispose();
364
+ }
335
365
  }
@@ -445,6 +445,43 @@ export function buildOperatorCapabilitiesDocument() {
445
445
  modelCallable: "always",
446
446
  humanConfirmation: "none",
447
447
  },
448
+ {
449
+ action: "operationWait",
450
+ cli: "console canonical operation wait (read-only event-driven long poll)",
451
+ kind: "read",
452
+ inputSchemaVersion: 1,
453
+ resultSchemaVersion: 1,
454
+ resultPolicy: { readOnly: true, bounded: true, redacted: true },
455
+ envelopeSchemaVersion: 1,
456
+ requiredErrorCodes: [
457
+ "NOT_FOUND",
458
+ "INVALID_INPUT",
459
+ "EVENT_CURSOR_EXPIRED",
460
+ ],
461
+ description: "Read-only event-driven wait on the canonical operation event ring. Returns immediately on existing events, terminal state or needs-reconcile; otherwise resolves on the first new event/state change or after maxWaitMs with timedOut:true (a success summary, not a command failure).",
462
+ inputParams: [
463
+ {
464
+ name: "operationId",
465
+ type: "string",
466
+ required: true,
467
+ description: "canonical operation id",
468
+ },
469
+ {
470
+ name: "afterSeq",
471
+ type: "number",
472
+ required: false,
473
+ description: "event cursor; only events with seq > afterSeq count (default 0, >= 0)",
474
+ },
475
+ {
476
+ name: "maxWaitMs",
477
+ type: "number",
478
+ required: false,
479
+ description: "bounded wait budget; clamped server-side (default 180000, floor 60000)",
480
+ },
481
+ ],
482
+ modelCallable: "always",
483
+ humanConfirmation: "none",
484
+ },
448
485
  {
449
486
  action: "contractShow",
450
487
  cli: "loop-agent task status <taskId> --json",
@@ -965,7 +1002,7 @@ export function buildOperatorCapabilitiesDocument() {
965
1002
  "CONTROLLER_MISMATCH",
966
1003
  "INVALID_INPUT",
967
1004
  ],
968
- description: "Consume a single-use execution receipt and execute the reviewed DAG (prefer task advance --approve-gate when gate token present). accepted/queued/running/operationId are NOT completion — supervise via operationGet/status/dagReport/dagDoctor.",
1005
+ description: "Consume a single-use execution receipt and execute the reviewed DAG (prefer task advance --approve-gate when gate token present). accepted/queued/running/operationId are NOT completion — supervise via operationGet/status/dagReport/dagDoctor/operationWait.",
969
1006
  inputParams: [
970
1007
  {
971
1008
  name: "executionId",
@@ -29,7 +29,7 @@ export function resolveArtifactWriteDir(options) {
29
29
  }
30
30
  export function buildArtifactPathPrompt(writeDir) {
31
31
  if (!writeDir)
32
- return `After changes, write artifacts/修改记录.md and artifacts/验证结果.md with verification evidence.
32
+ return `After changes, write artifacts/修改记录.md and artifacts/验证结果.md with verification evidence.
33
33
  ${ARTIFACT_INSTRUCTIONS}`;
34
34
  return [
35
35
  `After changes, write the following files:`,
@@ -58,11 +58,11 @@ import { RUNTIME_CONTEXT_TEXT_MAX, redactRuntimeText, } from "./runtime-context.
58
58
  */
59
59
  export const OPERATOR_CHAT_SYSTEM_PROMPT_BASE = [
60
60
  "You are the General Operator Chat for loop-agent / agent-worker.",
61
- "Operate through operator_* tools first. Pi read/write/edit/bash/grep/find/ls are available, but obey the loaded repository AGENTS.md. apply_patch, full-tools, shell, and coding-chat are denied alternate runtimes. Prefer safe-read/safe-grep for sensitive probes.",
62
- "If a repository has no loop-agent harness and the user requests initialization, run `loop-agent init instructions --repo-root .` then `loop-agent init --repo-root . --profile full --merge`; finish the generated setup, init doctor, inspect, docs audit, and quick verification. For updates use `loop-agent init upgrade --repo-root . --json` until stable; use init check-update only for an explicitly read-only request.",
63
- "After a DAG is started or accepted, do not end on accepted/queued/running or an operationId. Supervise it to a terminal outcome. Poll operationGet, task status, dagReport, and dagDoctor after about 15 seconds on start/change, every 30 seconds during progress, and every 60 seconds after 3 minutes unchanged. Report only meaningful node/rank changes, review/verify/closeout, recovery, liveness concerns, and terminal outcomes.",
64
- "On failure read primaryFailure, primaryRecovery, and doctor evidence. If meaningful progress exists, wait. Otherwise use a fresh eligible dagRerunPlan and rerun its safe node; when ineligible follow AGENTS.md/runtime recovery for same-task rerun/advance, resume, or Worker retry. Auto-approve only a bounded writeSet inside allowedPaths, outside forbiddenPaths, without broad/destructive risk, and with structured verification.",
65
- "Continue until success, user stop, or no safe eligible recovery remains because limits, bindings, auth/quota recovery, or required external authorization are exhausted. Never create a new task for a transient failure or replace repair-pi with direct edits.",
61
+ "Operate through operator_* tools first. Pi read/write/edit/bash/grep/find/ls are available; obey the loaded repository AGENTS.md. apply_patch, full-tools, shell, coding-chat are denied. Prefer safe-read/safe-grep for sensitive probes.",
62
+ "If a repository has no loop-agent harness and the user requests initialization, run `loop-agent init instructions --repo-root .` then `loop-agent init --repo-root . --profile full --merge`; finish with init doctor, inspect, docs audit, quick verification. For updates use `loop-agent init upgrade --repo-root . --json` until stable; init check-update only for explicitly read-only requests.",
63
+ "Long-running DAGs must run via prepareDagExecution → runDag → operationId held by a Console operation; never foreground-Bash `task advance --approve-gate`, no tail/head pipes, no hand-rolled nohup/Start-Process/start; do not end on accepted/queued/running or an operationId NOT completion. Supervise via operationGet/status/dagReport/dagDoctor/operationWait: 60s 90s 120s 180s backoff; reset to 60s on state change; 30-60s re-checks when stall suspected. Report only meaningful node/rank changes, review/verify/closeout, recovery, liveness, terminal outcomes.",
64
+ "On failure read primaryFailure, primaryRecovery, doctor evidence. If meaningful progress exists, wait. Else use a fresh eligible dagRerunPlan and rerun its safe node; when ineligible follow AGENTS.md/runtime recovery for same-task rerun/advance, resume, or Worker retry. Auto-approve only a bounded writeSet inside allowedPaths, outside forbiddenPaths, without broad/destructive risk, with structured verification.",
65
+ "Continue until success, user stop, or no safe eligible recovery remains (limits, bindings, auth/quota, external authorization exhausted). Never create a new task for a transient failure or replace repair-pi with direct edits.",
66
66
  ].join("\n");
67
67
  /** Compose the inspectable system prompt actually injected into Operator Chat. */
68
68
  export function composeOperatorChatSystemPrompt(input) {
@@ -1280,6 +1280,7 @@ export class ConsolePiRuntime {
1280
1280
  * session is not active — callers fail closed instead of fabricating data.
1281
1281
  */
1282
1282
  /**
1283
+ * ConsolePiRuntime.listSlashCommands
1283
1284
  * Browser-safe slash command projection for UI-11.
1284
1285
  * Sources: current Session extension commands, prompt templates, skills.
1285
1286
  * Never returns prompt/skill file bodies.
@@ -1292,29 +1293,36 @@ export class ConsolePiRuntime {
1292
1293
  .replace(/[A-Za-z]:\\Users\\[^\s]+/gi, "~")
1293
1294
  .slice(0, 240);
1294
1295
  };
1295
- try {
1296
- await this.ensureSessionReady(sessionId);
1297
- }
1298
- catch {
1299
- return [];
1300
- }
1296
+ // Fail soft at the HTTP layer: bubble readiness errors so routes can attach
1297
+ // a compact warning while the UI keeps local Operator/Pi Web commands.
1298
+ await this.ensureSessionReady(sessionId);
1301
1299
  const session = this.sessions.get(sessionId);
1302
1300
  if (!session)
1303
- return [];
1304
- const out = [];
1305
- const seen = new Set();
1301
+ return { commands: [] };
1302
+ // Same-name conflict priority (AC-3): extension > skill > prompt.
1303
+ // Enum order must not decide the winner — prompt-before-skill still loses.
1304
+ const SOURCE_RANK = {
1305
+ extension: 0,
1306
+ skill: 1,
1307
+ prompt: 2,
1308
+ };
1309
+ const byKey = new Map();
1310
+ const sourceWarnings = [];
1306
1311
  const push = (command, label, description, source) => {
1307
1312
  const name = command.startsWith("/") ? command : `/${command}`;
1308
1313
  const key = name.toLowerCase();
1309
- if (!key || key === "/" || seen.has(key))
1314
+ if (!key || key === "/")
1310
1315
  return;
1311
- seen.add(key);
1312
- out.push({
1316
+ const next = {
1313
1317
  command: name,
1314
1318
  label: (label || name).slice(0, 80),
1315
1319
  description: scrub(description),
1316
1320
  source,
1317
- });
1321
+ };
1322
+ const existing = byKey.get(key);
1323
+ if (!existing || SOURCE_RANK[source] < SOURCE_RANK[existing.source]) {
1324
+ byKey.set(key, next);
1325
+ }
1318
1326
  };
1319
1327
  try {
1320
1328
  const cmds = session.extensionRunner?.getRegisteredCommands?.() ?? [];
@@ -1325,18 +1333,20 @@ export class ConsolePiRuntime {
1325
1333
  push(String(name), String(cmd.name || name), String(cmd.description || ""), "extension");
1326
1334
  }
1327
1335
  }
1328
- catch {
1329
- // non-blocking
1336
+ catch (error) {
1337
+ // Source isolation: keep other sources; surface compact warning (AC-3).
1338
+ sourceWarnings.push(`extension: ${scrub(error instanceof Error ? error.message : String(error))}`);
1330
1339
  }
1331
1340
  try {
1341
+ // Enumerate prompts before skills on purpose: priority must still let skill win.
1332
1342
  for (const tpl of session.promptTemplates ?? []) {
1333
1343
  if (!tpl?.name)
1334
1344
  continue;
1335
1345
  push(String(tpl.name), String(tpl.name), String(tpl.description || ""), "prompt");
1336
1346
  }
1337
1347
  }
1338
- catch {
1339
- // non-blocking
1348
+ catch (error) {
1349
+ sourceWarnings.push(`prompt: ${scrub(error instanceof Error ? error.message : String(error))}`);
1340
1350
  }
1341
1351
  try {
1342
1352
  const services = this.serviceScopes.get(sessionId);
@@ -1351,10 +1361,16 @@ export class ConsolePiRuntime {
1351
1361
  push(command, skill.name, String(skill.description || ""), "skill");
1352
1362
  }
1353
1363
  }
1354
- catch {
1355
- // non-blocking
1364
+ catch (error) {
1365
+ sourceWarnings.push(`skill: ${scrub(error instanceof Error ? error.message : String(error))}`);
1356
1366
  }
1357
- return out;
1367
+ const warning = sourceWarnings.length > 0
1368
+ ? sourceWarnings.join("; ").slice(0, 160)
1369
+ : undefined;
1370
+ return {
1371
+ commands: [...byKey.values()],
1372
+ ...(warning ? { warning } : {}),
1373
+ };
1358
1374
  }
1359
1375
  async getRuntimeSnapshot(sessionId, options) {
1360
1376
  try {
@@ -1645,14 +1645,24 @@ async function handleChatFiles(res, deps, sessionId, query) {
1645
1645
  }
1646
1646
  export async function handleChatCommands(res, deps, sessionId) {
1647
1647
  try {
1648
- const commands = typeof deps.runtime.listSlashCommands === "function"
1648
+ const listed = typeof deps.runtime.listSlashCommands === "function"
1649
1649
  ? await deps.runtime.listSlashCommands(sessionId)
1650
- : [];
1650
+ : { commands: [] };
1651
+ // Partial single-source failures keep successful commands + compact warning (AC-3).
1652
+ const commands = listed.commands ?? [];
1653
+ const warning = typeof listed.warning === "string" && listed.warning.trim()
1654
+ ? listed.warning.trim().slice(0, 160)
1655
+ : undefined;
1651
1656
  res.statusCode = 200;
1652
1657
  res.setHeader("content-type", "application/json; charset=utf-8");
1653
- res.end(JSON.stringify({ ok: true, commands }));
1658
+ res.end(JSON.stringify({
1659
+ ok: true,
1660
+ commands,
1661
+ ...(warning ? { warning } : {}),
1662
+ }));
1654
1663
  }
1655
1664
  catch (error) {
1665
+ // Whole-list / readiness failures stay fail-soft with empty commands + warning.
1656
1666
  res.statusCode = 200;
1657
1667
  res.setHeader("content-type", "application/json; charset=utf-8");
1658
1668
  res.end(JSON.stringify({
@@ -2799,8 +2809,21 @@ export async function handleChatCompact(req, res, deps, sessionId) {
2799
2809
  return;
2800
2810
  }
2801
2811
  }
2812
+ let body = {};
2813
+ try {
2814
+ body = await readJsonBody(req);
2815
+ }
2816
+ catch {
2817
+ sendJson(res, 400, {
2818
+ ok: false,
2819
+ error: { code: "INVALID_INPUT", message: "invalid json body" },
2820
+ });
2821
+ return;
2822
+ }
2823
+ const rawInstructions = typeof body.instructions === "string" ? body.instructions : undefined;
2824
+ const customInstructions = rawInstructions?.trim() || undefined;
2802
2825
  try {
2803
- const snapshot = await deps.runtime.compact(sessionId);
2826
+ const snapshot = await deps.runtime.compact(sessionId, customInstructions);
2804
2827
  const event = deps.events.append(sessionId, deps.events.latestTurnId(sessionId) ?? `${sessionId}:compact`, { kind: "compact", data: snapshot });
2805
2828
  sendJson(res, 200, { ok: true, snapshot, eventId: event.eventId });
2806
2829
  }
@@ -74,6 +74,30 @@ export async function runOperation(operationId, deps) {
74
74
  message: chunk,
75
75
  });
76
76
  },
77
+ onHeartbeat: (info) => {
78
+ // P1 (2026-08-13): project sibling CLI heartbeats into canonical
79
+ // operation events so operationWait/operationEventSummary can consume
80
+ // them. The projection is best-effort: a failed append must never
81
+ // terminate the sibling CLI execution (AC-002). The injected test
82
+ // runCommand path has no LoopAgentClient callback guard, so the
83
+ // runner protects itself here.
84
+ try {
85
+ deps.events.append(operationId, {
86
+ at: info.at,
87
+ kind: "heartbeat",
88
+ message: `heartbeat elapsedMs=${info.elapsedMs}`,
89
+ data: {
90
+ elapsedMs: info.elapsedMs,
91
+ action: op.action,
92
+ ...(op.taskId ? { taskId: op.taskId } : {}),
93
+ ...(op.dagRunId ? { dagRunId: op.dagRunId } : {}),
94
+ },
95
+ });
96
+ }
97
+ catch {
98
+ // Heartbeat is derived telemetry; ignore projection failures.
99
+ }
100
+ },
77
101
  });
78
102
  await spawnUpdate;
79
103
  finished = await finalizeFromWorkerResult(operationId, result, deps);
@@ -0,0 +1,241 @@
1
+ import { projectOperationEventSummary, projectOperationForChat, } from "./chat/chat-event-store.js";
2
+ import { isTerminalOperationState, } from "./operation-store.js";
3
+ /**
4
+ * P2: read-only event-driven long poll on the canonical operation event ring
5
+ * (2026-08-13 Operator Chat long-run supervision). Server contract bounds:
6
+ * the model-facing `maxWaitMs` is clamped to [minWaitMs, maxWaitMsBound];
7
+ * defaults allow the 60–180s model supervision cadence and tests may inject
8
+ * short bounds.
9
+ */
10
+ export const DEFAULT_OPERATION_WAIT_MIN_MS = 60_000;
11
+ export const DEFAULT_OPERATION_WAIT_MAX_MS = 180_000;
12
+ export class OperationWaitError extends Error {
13
+ code;
14
+ constructor(code, message) {
15
+ super(message);
16
+ this.code = code;
17
+ this.name = "OperationWaitError";
18
+ }
19
+ }
20
+ function isFocusedOperation(operation) {
21
+ return (isTerminalOperationState(operation.state) ||
22
+ operation.state === "needs-reconcile");
23
+ }
24
+ function settledSummary(input) {
25
+ const projectedEvents = input.newEvents.length > 0
26
+ ? input.newEvents.map(projectOperationEventSummary)
27
+ : [];
28
+ const nextSeq = input.newEvents.length > 0
29
+ ? input.newEvents[input.newEvents.length - 1].seq
30
+ : input.afterSeq;
31
+ return {
32
+ operationId: input.operationId,
33
+ state: input.operation.state,
34
+ changed: input.changed,
35
+ timedOut: input.timedOut,
36
+ events: projectedEvents,
37
+ nextSeq,
38
+ operation: projectOperationForChat(input.operation),
39
+ };
40
+ }
41
+ /**
42
+ * Deterministic read-only wait over the canonical operation event stream
43
+ * (AC-003 / AC-004). Completion paths:
44
+ * - existing events (seq > afterSeq) → immediate (changed: true);
45
+ * - operation terminal/needs-reconcile with unconsumed events → immediate summary (changed: true);
46
+ * - operation terminal/needs-reconcile with no new events → immediate summary (changed: false);
47
+ * - EVENT_CURSOR_EXPIRED when the cursor fell out of the retained ring;
48
+ * - first subscribed event/state change → immediate settle;
49
+ * - maxWaitMs elapsed with no change → timedOut: true summary (not a failure).
50
+ *
51
+ * Race safety: the listener is registered BEFORE listFrom, closing the
52
+ * listFrom/subscribe gap; every path settles exactly once through a guarded
53
+ * `finish`, and listener + timer are always cleaned up on settle.
54
+ */
55
+ export async function waitForOperationChange(input) {
56
+ const operationId = input.operationId?.trim();
57
+ if (!operationId) {
58
+ throw new OperationWaitError("INVALID_INPUT", "operationId is required");
59
+ }
60
+ if (!Number.isInteger(input.afterSeq) || input.afterSeq < 0) {
61
+ throw new OperationWaitError("INVALID_INPUT", "afterSeq must be a non-negative integer");
62
+ }
63
+ const minWaitMs = input.minWaitMs ?? DEFAULT_OPERATION_WAIT_MIN_MS;
64
+ const maxWaitMsBound = input.maxWaitMsBound ?? DEFAULT_OPERATION_WAIT_MAX_MS;
65
+ if (!Number.isFinite(minWaitMs) ||
66
+ !Number.isFinite(maxWaitMsBound) ||
67
+ minWaitMs < 0 ||
68
+ maxWaitMsBound < minWaitMs) {
69
+ throw new OperationWaitError("INVALID_INPUT", "invalid wait bounds");
70
+ }
71
+ const rawMaxWaitMs = input.maxWaitMs ?? maxWaitMsBound;
72
+ if (!Number.isFinite(rawMaxWaitMs) || rawMaxWaitMs < 0) {
73
+ throw new OperationWaitError("INVALID_INPUT", "maxWaitMs must be a non-negative number");
74
+ }
75
+ const maxWaitMs = Math.max(minWaitMs, Math.min(maxWaitMsBound, Math.floor(rawMaxWaitMs)));
76
+ const afterSeq = input.afterSeq;
77
+ const operation = await input.operations.get(operationId);
78
+ if (!operation) {
79
+ throw new OperationWaitError("NOT_FOUND", `operation not found: ${operationId}`);
80
+ }
81
+ const events = input.events;
82
+ if (isFocusedOperation(operation)) {
83
+ // Flush unconsumed events before the terminal summary (AC-002): the
84
+ // caller's cursor must advance past every retained canonical event.
85
+ const listed = events.listFrom(operationId, afterSeq);
86
+ if ("error" in listed) {
87
+ throw new OperationWaitError("EVENT_CURSOR_EXPIRED", "event cursor expired; re-read operation snapshot");
88
+ }
89
+ if (listed.events.length > 0) {
90
+ return settledSummary({
91
+ operationId,
92
+ operation,
93
+ afterSeq,
94
+ changed: true,
95
+ timedOut: false,
96
+ newEvents: listed.events,
97
+ });
98
+ }
99
+ return settledSummary({
100
+ operationId,
101
+ operation,
102
+ afterSeq,
103
+ changed: false,
104
+ timedOut: false,
105
+ newEvents: [],
106
+ });
107
+ }
108
+ return new Promise((resolve, reject) => {
109
+ let settled = false;
110
+ let timer;
111
+ let unsubscribe;
112
+ const cleanup = () => {
113
+ if (timer !== undefined) {
114
+ clearTimeout(timer);
115
+ timer = undefined;
116
+ }
117
+ if (unsubscribe) {
118
+ unsubscribe();
119
+ unsubscribe = undefined;
120
+ }
121
+ };
122
+ const finish = (result) => {
123
+ if (settled)
124
+ return;
125
+ settled = true;
126
+ cleanup();
127
+ if (result instanceof OperationWaitError)
128
+ reject(result);
129
+ else
130
+ resolve(result);
131
+ };
132
+ const settleWithEvent = (event) => {
133
+ // Future-cursor guard (AC-003): events at or below the caller's
134
+ // afterSeq are already consumed and must not settle this wait.
135
+ if (event.seq <= afterSeq)
136
+ return;
137
+ // Listener path: async re-read of the operation for a fresh summary.
138
+ void (async () => {
139
+ try {
140
+ const current = await input.operations.get(operationId);
141
+ finish(settledSummary({
142
+ operationId,
143
+ operation: current ?? operation,
144
+ afterSeq,
145
+ changed: true,
146
+ timedOut: false,
147
+ newEvents: [event],
148
+ }));
149
+ }
150
+ catch (error) {
151
+ finish(error instanceof OperationWaitError
152
+ ? error
153
+ : new OperationWaitError("INVALID_INPUT", error instanceof Error ? error.message : String(error)));
154
+ }
155
+ })();
156
+ };
157
+ // Terminal recheck before subscribing: events landing during this await
158
+ // are still in the ring and are caught by listFrom below.
159
+ void (async () => {
160
+ try {
161
+ const current = await input.operations.get(operationId);
162
+ if (current && isFocusedOperation(current)) {
163
+ // Same terminal flush as the initial path (AC-002): events
164
+ // landing during the await are still in the retained ring.
165
+ const listed = events.listFrom(operationId, afterSeq);
166
+ if ("error" in listed) {
167
+ finish(new OperationWaitError("EVENT_CURSOR_EXPIRED", "event cursor expired; re-read operation snapshot"));
168
+ return;
169
+ }
170
+ if (listed.events.length > 0) {
171
+ finish(settledSummary({
172
+ operationId,
173
+ operation: current,
174
+ afterSeq,
175
+ changed: true,
176
+ timedOut: false,
177
+ newEvents: listed.events,
178
+ }));
179
+ return;
180
+ }
181
+ finish(settledSummary({
182
+ operationId,
183
+ operation: current,
184
+ afterSeq,
185
+ changed: false,
186
+ timedOut: false,
187
+ newEvents: [],
188
+ }));
189
+ return;
190
+ }
191
+ // Subscribe BEFORE listFrom: any event appended after this point
192
+ // reaches the listener, closing the listFrom/subscribe race.
193
+ unsubscribe = events.subscribe(operationId, settleWithEvent);
194
+ const listed = events.listFrom(operationId, afterSeq);
195
+ if ("error" in listed) {
196
+ finish(new OperationWaitError("EVENT_CURSOR_EXPIRED", "event cursor expired; re-read operation snapshot"));
197
+ return;
198
+ }
199
+ if (listed.events.length > 0) {
200
+ finish(settledSummary({
201
+ operationId,
202
+ operation: current ?? operation,
203
+ afterSeq,
204
+ changed: true,
205
+ timedOut: false,
206
+ newEvents: listed.events,
207
+ }));
208
+ return;
209
+ }
210
+ // No events: arm the bounded wait; the listener settles on the
211
+ // first new event/state change, the timer settles on timeout.
212
+ timer = setTimeout(() => {
213
+ void (async () => {
214
+ try {
215
+ const latest = await input.operations.get(operationId);
216
+ finish(settledSummary({
217
+ operationId,
218
+ operation: latest ?? operation,
219
+ afterSeq,
220
+ changed: false,
221
+ timedOut: true,
222
+ newEvents: [],
223
+ }));
224
+ }
225
+ catch (error) {
226
+ finish(error instanceof OperationWaitError
227
+ ? error
228
+ : new OperationWaitError("INVALID_INPUT", error instanceof Error ? error.message : String(error)));
229
+ }
230
+ })();
231
+ }, maxWaitMs);
232
+ timer.unref?.();
233
+ }
234
+ catch (error) {
235
+ finish(error instanceof OperationWaitError
236
+ ? error
237
+ : new OperationWaitError("INVALID_INPUT", error instanceof Error ? error.message : String(error)));
238
+ }
239
+ })();
240
+ });
241
+ }