pi-better-harness 0.8.0 → 0.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (28) hide show
  1. package/node_modules/pi-better-background-tasks/README.md +51 -0
  2. package/node_modules/pi-better-background-tasks/package.json +1 -1
  3. package/node_modules/pi-better-background-tasks/src/output.ts +96 -7
  4. package/node_modules/pi-better-background-tasks/src/runtime.ts +164 -12
  5. package/node_modules/pi-better-background-tasks/src/sandbox.ts +3 -0
  6. package/node_modules/pi-better-background-tasks/src/shared-failure-observations.ts +22 -9
  7. package/node_modules/pi-better-background-tasks/src/shared-navigator.ts +57 -2
  8. package/node_modules/pi-better-background-tasks/src/shared-sandbox-core.ts +129 -21
  9. package/node_modules/pi-better-background-tasks/src/tools.ts +52 -10
  10. package/node_modules/pi-better-background-tasks/src/types.ts +20 -0
  11. package/node_modules/pi-better-sandbox/package.json +1 -1
  12. package/node_modules/pi-better-sandbox/shared-sandbox-core.ts +129 -21
  13. package/node_modules/pi-better-sandbox/shared-task-files.ts +1 -1
  14. package/node_modules/pi-better-sandbox/shared-task-sandbox.ts +1 -0
  15. package/node_modules/pi-better-sandbox/shell.ts +4 -1
  16. package/node_modules/pi-better-subagents/child-incidents.ts +6 -4
  17. package/node_modules/pi-better-subagents/docs/failure-observations.md +2 -0
  18. package/node_modules/pi-better-subagents/failures.ts +49 -3
  19. package/node_modules/pi-better-subagents/incident-model.ts +7 -7
  20. package/node_modules/pi-better-subagents/output-payload.ts +2 -10
  21. package/node_modules/pi-better-subagents/package.json +1 -1
  22. package/node_modules/pi-better-subagents/shared-failure-observations.ts +22 -9
  23. package/node_modules/pi-better-subagents/shared-navigator.ts +57 -2
  24. package/node_modules/pi-better-subagents/shared-sandbox-core.ts +129 -21
  25. package/node_modules/pi-better-subagents/shared-task-files.ts +1 -1
  26. package/node_modules/pi-better-subagents/shared-task-sandbox.ts +1 -0
  27. package/node_modules/pi-better-subagents/task-policy.ts +1 -1
  28. package/package.json +4 -4
@@ -124,6 +124,57 @@ Missing JSON fields or invalid JSON output remain retryable; task status shows
124
124
  the condition evaluation error until a subsequent poll recovers. Keep a finite
125
125
  timeout to bound watches whose output never becomes evaluable.
126
126
 
127
+ ### Writing a watch check
128
+
129
+ A check that swallows its own errors reports "still pending" forever. Keep
130
+ failures visible:
131
+
132
+ - Do not end the check with `exit 0` or `|| true`. A check that exits non-zero
133
+ is recorded as a failure and escalates to the parent session.
134
+ - Map an unknown or unparseable state to failure (a non-zero exit), not to
135
+ pending.
136
+ - Prefer structured output parsed with `jq -e` over hand-written format
137
+ strings. `jq -e` exits non-zero when the field is missing, so a broken query
138
+ shows up at once.
139
+
140
+ For example, a Cloud Run job execution:
141
+
142
+ ```sh
143
+ status=$(gcloud run jobs executions describe "$EXECUTION" --region="$REGION" --format=json \
144
+ | jq -er '.status.conditions[] | select(.type == "Completed") | .status') || exit 2
145
+ case "$status" in
146
+ True) echo TERMINAL_SUCCESS ;;
147
+ False) echo TERMINAL_FAILURE ;;
148
+ Unknown) echo STILL_RUNNING ;;
149
+ *) echo "unexpected Completed status: $status" >&2; exit 2 ;;
150
+ esac
151
+ ```
152
+
153
+ with `success_when: {type: "stdout_contains", value: "TERMINAL_SUCCESS"}` and
154
+ `failure_when: {type: "stdout_contains", value: "TERMINAL_FAILURE"}`.
155
+
156
+ `bg_task_watch` (and `bg_task` with `action: "watch"`) waits up to 15 seconds
157
+ for the first check and puts its exit code, the newest few lines of stderr and
158
+ of stdout in the tool result, so a broken check is visible at launch. When the
159
+ result is short on room, stdout is cut first. If the first check is still
160
+ running after 15 seconds, or you press Esc during the wait, the result says so
161
+ at once and the watch continues.
162
+
163
+ A running watch also guards against a blind check. When 3 checks in a row exit
164
+ 0, write to stderr, and match neither `success_when` nor `failure_when`, the
165
+ watch records one incident that needs action, with the latest stderr line, and
166
+ wakes the parent session once. The watch keeps running. A later check with
167
+ empty stderr recovers the incident whatever its exit code (a non-zero or
168
+ failed check is then recorded as its own incident), and so does a check that
169
+ matches a condition. A non-zero check that writes stderr restarts the count but
170
+ leaves the incident open. A clean pending check (exit 0, no stderr) never
171
+ counts.
172
+
173
+ Some tools write to stderr on success (`gcloud … list` prints "Listed 0
174
+ items.", and kubectl and npm print warnings), which can raise a false alarm.
175
+ If the stderr is expected, redirect it (`2>/dev/null`) or set
176
+ `blind_checks: 0`. Set `blind_checks` to another number to change the count.
177
+
127
178
  ## Install
128
179
 
129
180
  ```sh
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pi-better-background-tasks",
3
- "version": "0.5.0",
3
+ "version": "0.6.0",
4
4
  "description": "Pi extension for durable background shell tasks, watchers, logs, and status inspection.",
5
5
  "license": "MIT",
6
6
  "type": "module",
@@ -40,7 +40,10 @@ import {
40
40
  import { failurePath } from "./failures.js";
41
41
  import { captureGapsFor, pageTaskLog, readLog, type LogRead } from "./logs.js";
42
42
  import { belongsToOrigin, inspectMeta, listTaskRecords, originOf, type MetaInspection } from "./registry.js";
43
- import type { BackgroundTaskCallbackOrigin, BackgroundTaskMeta, Condition } from "./types.js";
43
+ import type { BackgroundTaskCallbackOrigin, BackgroundTaskMeta, Condition, FirstWatchCheck } from "./types.js";
44
+
45
+ /** The launch tool stopped waiting while a watch's first check was still running (#359). */
46
+ export type FirstCheckPending = { pending: "timeout" | "aborted" | "suspended"; waitedMs: number };
44
47
 
45
48
  /**
46
49
  * Issue #312 consumer budgets. Defaults follow OUTPUT-POLICY / shared
@@ -240,9 +243,10 @@ export function formatCallbackFacts(meta: BackgroundTaskMeta): {
240
243
  meta.captureDiscardedBytes ? `capture overflow discarded ${meta.captureDiscardedBytes} bytes; not full history` : undefined,
241
244
  ].filter((line): line is string => Boolean(line));
242
245
  // History (expected and closed incidents) is not a row, but the callback still says it exists, so a
243
- // declared expected exit is not read as a plain failure.
246
+ // declared expected exit is not read as a plain failure. It leads the decision: a tight callback
247
+ // budget keeps a prefix of the decision, and this short line must survive it (#332).
244
248
  const history = rows.length ? historyTailLine(state) : quietFailureLine(state);
245
- const decision = [formatDecision(meta), ...gapLines, history].filter(Boolean).join("\n") || undefined;
249
+ const decision = [history, formatDecision(meta), ...gapLines].filter(Boolean).join("\n") || undefined;
246
250
  return {
247
251
  outcome: meta.status,
248
252
  ...(rows.length ? { failureRows: rows, incidentCount: rows.length } : {}),
@@ -434,24 +438,109 @@ function asInspection(inspection: MetaInspection | BackgroundTaskMeta | undefine
434
438
  return { id: meta.id, meta, found: true, readable: true };
435
439
  }
436
440
 
437
- export function formatLaunch(meta: BackgroundTaskMeta): string {
441
+ /**
442
+ * Launch result. For a watch, `firstCheck` reports its first check (#359): the check's result,
443
+ * or why the launch stopped waiting while it was still running.
444
+ *
445
+ * The first check gets whatever the status budget leaves after the rest of the launch text, so
446
+ * nothing else is clipped for it. The log path is kept whole or dropped whole: a clipped path
447
+ * looks valid but points nowhere.
448
+ */
449
+ export function formatLaunch(meta: BackgroundTaskMeta, firstCheck?: FirstWatchCheck | FirstCheckPending): string {
438
450
  const label = meta.name ? `${meta.name} (${meta.id})` : meta.id;
439
451
  const remoteLines = [
440
452
  ...(meta.ssh ? [`Remote: ${meta.ssh.target}${meta.remote?.session ? ` mode=${meta.remote.session}` : ""}${meta.remote?.sessionName ? ` session=${meta.remote.sessionName}` : ""}.`] : []),
441
453
  ...(meta.remote?.bootstrapMessage ? [`Remote setup: ${meta.remote.bootstrapMessage}`] : []),
442
454
  ...(meta.remote?.warning ? [`Warning: ${meta.remote.warning}`] : []),
443
455
  ];
444
- return assembleBackgroundContent({
456
+ const logLine = `Log: ${meta.logPath}`;
457
+ const build = (checkText: string | undefined, withLog: boolean) => assembleBackgroundContent({
445
458
  surface: "status",
446
459
  sections: {
447
460
  identity: `Started background ${meta.kind} ${label}. Status: ${meta.status}.`,
448
461
  failure: incidentSection(meta.id, {}),
449
462
  decision: formatDecision(meta),
450
- diagnostics: remoteLines.join("\n") || undefined,
451
- progress: `Log: ${meta.logPath}`,
463
+ diagnostics: [...remoteLines, ...(checkText ? [checkText] : [])].join("\n") || undefined,
464
+ progress: withLog ? logLine : undefined,
452
465
  },
453
466
  gaps: taskGaps(meta),
454
467
  });
468
+ let checkText: string | undefined;
469
+ if (firstCheck) {
470
+ const room = backgroundBudget("status") - utf8ByteLength(build(undefined, true)) - 1;
471
+ checkText = formatFirstWatchCheck(meta, firstCheck, room);
472
+ }
473
+ const text = build(checkText, true);
474
+ return text.includes(logLine) ? text : build(checkText, false);
475
+ }
476
+
477
+ const FIRST_CHECK_TAIL_LINES = 3;
478
+ const FIRST_CHECK_LINE_CHARS = 200;
479
+
480
+ function tailLines(text: string): string[] {
481
+ return text.split(/\r?\n/).map((line) => line.replace(/\s+/g, " ").trim()).filter(Boolean)
482
+ .slice(-FIRST_CHECK_TAIL_LINES)
483
+ .map((line) => line.length > FIRST_CHECK_LINE_CHARS ? `${line.slice(0, FIRST_CHECK_LINE_CHARS - 1)}…` : line);
484
+ }
485
+
486
+ function pendingFirstCheckText(check: FirstCheckPending): string {
487
+ switch (check.pending) {
488
+ case "aborted":
489
+ return "First check still running; stopped waiting because the tool call was cancelled. The watch continues in the background; check it later with bg_task_status.";
490
+ case "suspended":
491
+ return "First check still running when the session shut down. The watch continues when its session resumes; check it later with bg_task_status.";
492
+ default:
493
+ return `First check still running after ${formatDuration(check.waitedMs)}; the watch continues in the background. Check it later with bg_task_status.`;
494
+ }
495
+ }
496
+
497
+ /**
498
+ * The first check of a watch in at most `maxBytes` (#359). What matters most is kept first: the
499
+ * outcome, the exit-0-with-stderr warning, the newest stderr lines, then the newest stdout lines,
500
+ * so stdout is cut first. Lines are shown oldest to newest.
501
+ */
502
+ export function formatFirstWatchCheck(meta: BackgroundTaskMeta, check: FirstWatchCheck | FirstCheckPending, maxBytes = Number.MAX_SAFE_INTEGER): string {
503
+ if ("pending" in check) return pendingFirstCheckText(check);
504
+ if (check.error) return `First check could not run: ${oneLine(check.error, 300)}`;
505
+ const outcome = check.timedOut ? "timed out" : check.signal ? `signal ${check.signal}` : `exit ${check.exitCode ?? "unknown"}`;
506
+ const took = check.durationMs < 1000 ? `${check.durationMs}ms` : formatDuration(check.durationMs);
507
+ const header = `First check: ${outcome} in ${took}.`;
508
+ const stdout = tailLines(check.stdout);
509
+ const stderr = tailLines(check.stderr);
510
+ const warning = check.exitCode === 0 && stderr.length && meta.status === "running"
511
+ ? "The check exited 0 but wrote stderr: if it is broken, the watch cannot tell. Let errors exit non-zero."
512
+ : undefined;
513
+ const keptErr: string[] = [];
514
+ const keptOut: string[] = [];
515
+ let keptEmptyStdout = false;
516
+ const render = () => [
517
+ header,
518
+ ...(warning ? [warning] : []),
519
+ ...(keptErr.length ? [`stderr tail:\n${keptErr.map((line) => ` ${line}`).join("\n")}`] : []),
520
+ ...(keptOut.length ? [`stdout tail:\n${keptOut.map((line) => ` ${line}`).join("\n")}`] : []),
521
+ ...(keptEmptyStdout ? ["stdout: (empty)"] : []),
522
+ ].join("\n");
523
+ const fits = () => utf8ByteLength(render()) <= maxBytes;
524
+ // Newest first within each stream; each kept line goes in front to stay in order.
525
+ for (const line of [...stderr].reverse()) {
526
+ keptErr.unshift(line);
527
+ if (!fits()) { keptErr.shift(); break; }
528
+ }
529
+ for (const line of [...stdout].reverse()) {
530
+ keptOut.unshift(line);
531
+ if (!fits()) { keptOut.shift(); break; }
532
+ }
533
+ if (!stdout.length) {
534
+ keptEmptyStdout = true;
535
+ if (!fits()) keptEmptyStdout = false;
536
+ }
537
+ const omitted = (stderr.length - keptErr.length) + (stdout.length - keptOut.length);
538
+ if (omitted > 0) {
539
+ const note = `(${omitted} output line${omitted === 1 ? "" : "s"} omitted; see bg_task_log)`;
540
+ const text = `${render()}\n${note}`;
541
+ if (utf8ByteLength(text) <= maxBytes) return text;
542
+ }
543
+ return render();
455
544
  }
456
545
 
457
546
  function redactedVerbose(meta: BackgroundTaskMeta): unknown {
@@ -19,6 +19,7 @@ import type {
19
19
  CommandResult,
20
20
  CommandSpec,
21
21
  Condition,
22
+ FirstWatchCheck,
22
23
  RemoteTaskParams,
23
24
  SshConnectionParams,
24
25
  TerminalResult,
@@ -31,6 +32,8 @@ const processTimeoutTimers = new Map<string, ReturnType<typeof setTimeout>>();
31
32
  const activeProcessTimeouts = new Set<string>();
32
33
  const activeRemoteTasks = new Map<string, ResolvedSshRemoteTask>();
33
34
  const remoteSessionStarts = new Map<string, Promise<CommandResult>>();
35
+ /** Tasks whose stop in this instance is waiting on an in-flight tmux start, and kills the session itself. */
36
+ const remoteStopsAwaitingStart = new Set<string>();
34
37
  const activePolls = new Set<string>();
35
38
  const logRetentionTimers = new Map<string, ReturnType<typeof setInterval>>();
36
39
  const LOG_RETENTION_CHECK_MS = 1000;
@@ -44,6 +47,63 @@ let scheduledWorkSuspended = false;
44
47
  const REMOTE_SESSION_POLL_MS = 100;
45
48
 
46
49
  export const DEFAULT_WATCH_TIMEOUT_SECONDS = 15 * 60;
50
+ /** Consecutive blind checks (exit 0, stderr, no condition matched) before a watch is flagged (#359). */
51
+ export const DEFAULT_BLIND_CHECKS = 3;
52
+ /** How long bg_task_watch waits for the first check before returning (#359). */
53
+ export const FIRST_WATCH_CHECK_WAIT_MS = 15_000;
54
+ const BLIND_OPERATION = "watch-blind";
55
+ /** How to fix a false alarm: some tools write progress or warnings to stderr on success. */
56
+ export const BLIND_CHECK_HINT = "If the stderr is expected (progress or warnings), redirect it (2>/dev/null) or set blind_checks:0.";
57
+
58
+ /**
59
+ * Why the launch stopped waiting before the first check finished (#359): the bounded wait
60
+ * ran out, the tool call was aborted (Esc), or the session shut down.
61
+ */
62
+ export type FirstCheckWaitEnd = "timeout" | "aborted" | "suspended";
63
+ /** A first check's outcome, or why it is not known yet; undefined when the watch ended without one. */
64
+ export type FirstCheckOutcome = FirstWatchCheck | { pending: FirstCheckWaitEnd; waitedMs: number } | undefined;
65
+
66
+ /** In-flight first checks of watches launched by this instance, keyed by task id. */
67
+ const firstCheckWaiters = new Map<string, { promise: Promise<FirstWatchCheck | "suspended" | undefined>; resolve: (check: FirstWatchCheck | "suspended" | undefined) => void }>();
68
+
69
+ /**
70
+ * Wait for a watch's first check, bounded by `timeoutMs` and `signal` (#359). Resolves a pending
71
+ * outcome when the wait ends first, and undefined when this instance did not launch the watch
72
+ * or the watch ended without a check.
73
+ */
74
+ export async function awaitFirstWatchCheck(id: string, timeoutMs = FIRST_WATCH_CHECK_WAIT_MS, signal?: AbortSignal): Promise<FirstCheckOutcome> {
75
+ const waiter = firstCheckWaiters.get(id);
76
+ if (!waiter) return undefined;
77
+ const started = Date.now();
78
+ const pending = (reason: FirstCheckWaitEnd) => ({ pending: reason, waitedMs: Date.now() - started });
79
+ if (signal?.aborted) return pending("aborted");
80
+ let timer: ReturnType<typeof setTimeout> | undefined;
81
+ let onAbort: (() => void) | undefined;
82
+ const timeout = new Promise<"timeout">((resolve) => {
83
+ // Kept referenced: the launching tool call is waiting on it, and the watch timers are
84
+ // unref'd, so an unref'd wait could let the event loop drain before the first check.
85
+ timer = setTimeout(() => resolve("timeout"), Math.max(0, timeoutMs));
86
+ });
87
+ const aborted = new Promise<"aborted">((resolve) => {
88
+ onAbort = () => resolve("aborted");
89
+ signal?.addEventListener("abort", onAbort, { once: true });
90
+ });
91
+ try {
92
+ const outcome = await Promise.race([waiter.promise, timeout, aborted]);
93
+ if (outcome === "timeout" || outcome === "aborted" || outcome === "suspended") return pending(outcome);
94
+ return outcome;
95
+ } finally {
96
+ if (timer) clearTimeout(timer);
97
+ if (onAbort) signal?.removeEventListener("abort", onAbort);
98
+ }
99
+ }
100
+
101
+ function settleFirstCheck(id: string, check: FirstWatchCheck | "suspended" | undefined): void {
102
+ const waiter = firstCheckWaiters.get(id);
103
+ if (!waiter) return;
104
+ firstCheckWaiters.delete(id);
105
+ waiter.resolve(check);
106
+ }
47
107
 
48
108
  /**
49
109
  * Stop this extension instance's scheduled work when its session shuts down (#324).
@@ -81,6 +141,8 @@ export function suspendScheduledWork(): void {
81
141
  remoteSessionTimers.clear();
82
142
  processTimeoutTimers.clear();
83
143
  logRetentionTimers.clear();
144
+ // A launch still waiting on a first check that will not run here reports it as still running.
145
+ for (const id of [...firstCheckWaiters.keys()]) settleFirstCheck(id, "suspended");
84
146
  suspendFailureAttention();
85
147
  }
86
148
 
@@ -94,12 +156,16 @@ export function resumeScheduledWork(): void {
94
156
 
95
157
  export type ActiveSessionProvider = () => BackgroundTaskCallbackOrigin | undefined;
96
158
 
97
- export interface SpawnTaskParams extends CommandSpec {
98
- name?: string;
99
- /** Structured intent (#325): stable id shared by modified retries of one operation. */
159
+ /** Structured intent (#325), declared by the launching agent and shared by spawns and watchers. */
160
+ export interface TaskIntentParams {
161
+ /** Stable id shared by modified retries of one operation. */
100
162
  operation_id?: string | null;
101
- /** Structured intent (#325): non-zero exit codes declared intentional before launch. */
163
+ /** Non-zero exit codes declared intentional before launch. */
102
164
  expected_exit_codes?: number[] | null;
165
+ }
166
+
167
+ export interface SpawnTaskParams extends CommandSpec, TaskIntentParams {
168
+ name?: string;
103
169
  callback?: boolean;
104
170
  timeout_seconds?: number;
105
171
  max_log_bytes?: number;
@@ -107,18 +173,16 @@ export interface SpawnTaskParams extends CommandSpec {
107
173
  remote?: RemoteTaskParams;
108
174
  }
109
175
 
110
- export interface WatchTaskParams extends CommandSpec {
176
+ export interface WatchTaskParams extends CommandSpec, TaskIntentParams {
111
177
  name?: string;
112
- /** Structured intent (#325): stable id shared by modified retries of one operation. */
113
- operation_id?: string | null;
114
- /** Structured intent (#325): non-zero exit codes declared intentional before launch. */
115
- expected_exit_codes?: number[] | null;
116
178
  callback?: boolean;
117
179
  interval_seconds?: number;
118
180
  timeout_seconds?: number;
119
181
  max_log_bytes?: number;
120
182
  success_when: Condition;
121
183
  failure_when?: Condition;
184
+ /** Consecutive blind checks before the watch is flagged (#359). Default 3; 0 turns it off. */
185
+ blind_checks?: number;
122
186
  ssh?: SshConnectionParams;
123
187
  remote?: RemoteTaskParams;
124
188
  }
@@ -157,9 +221,10 @@ export function spawnTask(
157
221
  }, dependencies.remoteRunner)
158
222
  : undefined;
159
223
  const commandSpec: CommandSpec = remoteTask?.commandSpec ?? { ...params, cwd, shell: params.shell ?? true };
224
+ const sandboxNotices: string[] = [];
160
225
  const launchSpec = remoteTask
161
226
  ? commandSpec
162
- : confineCommandSpec(commandSpec, sandboxPlan, sandboxProfilePathFor(id));
227
+ : confineCommandSpec(commandSpec, sandboxPlan, sandboxProfilePathFor(id), {}, (line) => sandboxNotices.push(line));
163
228
  const tmuxBacked = remoteTask?.metadata.remote.session === "tmux";
164
229
  // Outside project = Write or Write & delete can change files outside the
165
230
  // project: start an APFS local snapshot (macOS, background, never blocks).
@@ -169,6 +234,9 @@ export function spawnTask(
169
234
  : remoteTask
170
235
  ? remoteTask.spawn(logPath, true)
171
236
  : spawnCommand(launchSpec, logPath, true);
237
+ for (const line of sandboxNotices) {
238
+ try { appendLine(logPath, `--- ${line} ---`); } catch { /* log gone */ }
239
+ }
172
240
  if (snapshot?.started) {
173
241
  void snapshot.done.then((outcome) => {
174
242
  if (!outcome.ok) {
@@ -276,7 +344,15 @@ async function launchRemoteTmux(
276
344
  if (remoteSessionStarts.get(id) === startAttempt) remoteSessionStarts.delete(id);
277
345
  }
278
346
  const afterStart = readMeta(id);
279
- if (!afterStart || afterStart.status !== "running" || afterStart.stopRequestedAt) return;
347
+ if (!afterStart) return;
348
+ if (afterStart.status !== "running" || afterStart.stopRequestedAt) {
349
+ // Stopped (or timed out) while the start was in flight. A stop in this instance awaited the
350
+ // attempt and kills the session itself. Anything else could not see the attempt: a stop from
351
+ // an instance loaded by /reload, or a deadline that found the session not yet started. Only
352
+ // this launch can still reach the session it may just have created, so it kills it (#332).
353
+ if (!remoteStopsAwaitingStart.has(id)) await killSessionStartedAfterStop(afterStart, remoteTask);
354
+ return;
355
+ }
280
356
  if (started.exitCode !== 0) {
281
357
  const detail = started.stderr.trim() || started.stdout.trim() || "remote tmux returned no diagnostic";
282
358
  const reason = `Could not create remote tmux session ${afterStart.remote?.sessionName} on ${afterStart.ssh?.target} (exit ${started.exitCode ?? "unknown"}): ${detail}`;
@@ -296,6 +372,19 @@ async function launchRemoteTmux(
296
372
  }
297
373
  }
298
374
 
375
+ /** Kill a tmux session whose start completed after its task stopped; a failed start may still have left it running. */
376
+ async function killSessionStartedAfterStop(meta: BackgroundTaskMeta, remoteTask: ResolvedSshRemoteTask): Promise<void> {
377
+ const session = `remote tmux session ${meta.remote?.sessionName} on ${meta.ssh?.target}`;
378
+ try {
379
+ const stopped = await remoteTask.killTmuxSession();
380
+ appendLine(meta.logPath, stopped.exitCode === 0
381
+ ? `--- Killed ${session}: it started after the task was ${meta.status === "running" ? "stopped" : meta.status} ---`
382
+ : `--- Could not kill ${session} that started after the task stopped (exit ${stopped.exitCode ?? "unknown"}) ---`);
383
+ } catch (error) {
384
+ appendLine(meta.logPath, `--- Could not kill ${session} that started after the task stopped: ${error instanceof Error ? error.message : String(error)} ---`);
385
+ }
386
+ }
387
+
299
388
  function scheduleRemoteSessionPoll(
300
389
  pi: ExtensionAPI,
301
390
  id: string,
@@ -404,6 +493,7 @@ export function startWatchTask(
404
493
  const error = condition && validateCondition(condition);
405
494
  if (error) throw new Error(`${name}: ${error}`);
406
495
  }
496
+ const blindChecks = readBlindChecks(params.blind_checks);
407
497
  const intent = readTaskIntent(params);
408
498
  const sandboxPlan = resolveForegroundSandboxPlan(pi, !!params.ssh);
409
499
  const id = nextTaskId();
@@ -421,9 +511,10 @@ export function startWatchTask(
421
511
  }, dependencies.remoteRunner)
422
512
  : undefined;
423
513
  const commandSpec: CommandSpec = remoteTask?.commandSpec ?? { ...params, cwd, shell: params.shell ?? true };
514
+ const sandboxNotices: string[] = [];
424
515
  const launchSpec = remoteTask
425
516
  ? commandSpec
426
- : confineCommandSpec(commandSpec, sandboxPlan, sandboxProfilePathFor(id));
517
+ : confineCommandSpec(commandSpec, sandboxPlan, sandboxProfilePathFor(id), {}, (line) => sandboxNotices.push(line));
427
518
  const meta: BackgroundTaskMeta = {
428
519
  id,
429
520
  name: params.name,
@@ -447,6 +538,7 @@ export function startWatchTask(
447
538
  spawnPidStartTime: currentProcessStartToken(),
448
539
  successWhen: params.success_when,
449
540
  failureWhen: params.failure_when,
541
+ ...(blindChecks !== undefined ? { blindChecks } : {}),
450
542
  notifyOn: "terminal",
451
543
  ssh: remoteTask?.metadata.ssh,
452
544
  remote: remoteTask?.metadata.remote,
@@ -454,13 +546,25 @@ export function startWatchTask(
454
546
  };
455
547
  ensureTaskDir(id);
456
548
  appendLine(meta.logPath, `--- watch ${new Date(now).toISOString()} interval_ms=${meta.intervalMs} ---`);
549
+ for (const line of sandboxNotices) appendLine(meta.logPath, `--- ${line} ---`);
457
550
  writeMeta(meta);
551
+ let resolveFirst!: (check: FirstWatchCheck | "suspended" | undefined) => void;
552
+ const firstCheck = new Promise<FirstWatchCheck | "suspended" | undefined>((resolve) => { resolveFirst = resolve; });
553
+ firstCheckWaiters.set(id, { promise: firstCheck, resolve: resolveFirst });
458
554
  scheduleWatch(pi, id, 0, getActiveSession, remoteTask
459
555
  ? (timeoutMs) => remoteTask.runOnce(undefined, timeoutMs)
460
556
  : undefined);
461
557
  return meta;
462
558
  }
463
559
 
560
+ function readBlindChecks(value: unknown): number | undefined {
561
+ if (value === undefined || value === null) return undefined;
562
+ if (typeof value !== "number" || !Number.isInteger(value) || value < 0) {
563
+ throw new Error("blind_checks must be a non-negative integer (0 turns the blind-check rule off).");
564
+ }
565
+ return value;
566
+ }
567
+
464
568
  function resolveWatchTimeoutSeconds(timeoutSeconds: number | undefined): number | undefined {
465
569
  if (timeoutSeconds === undefined) return DEFAULT_WATCH_TIMEOUT_SECONDS;
466
570
  if (timeoutSeconds <= 0) return undefined;
@@ -641,7 +745,9 @@ export async function stopTask(
641
745
  const remoteSessionMayExist = remote?.session === "tmux"
642
746
  && (remote.sessionStarted !== false || remoteStartAttempt !== undefined);
643
747
  if (remoteStartAttempt) {
748
+ remoteStopsAwaitingStart.add(id);
644
749
  try { await remoteStartAttempt; } catch { /* A failed SSH result can still leave the detached session running. */ }
750
+ finally { remoteStopsAwaitingStart.delete(id); }
645
751
  }
646
752
 
647
753
  if (remoteSessionMayExist) {
@@ -729,6 +835,7 @@ async function pollWatch(
729
835
  ): Promise<void> {
730
836
  if (activePolls.has(id)) return;
731
837
  activePolls.add(id);
838
+ let checked: FirstWatchCheck | undefined;
732
839
  try {
733
840
  const meta = readMeta(id);
734
841
  if (!meta || meta.status !== "running" || meta.kind !== "command_watch") return;
@@ -741,6 +848,10 @@ async function pollWatch(
741
848
  const result = runOnce
742
849
  ? await runOnce(timeoutMs)
743
850
  : await runCommandOnce(commandSpecFromMeta(meta), undefined, timeoutMs);
851
+ checked = {
852
+ exitCode: result.exitCode, signal: result.signal, durationMs: Math.max(0, result.endedAt - result.startedAt),
853
+ ...(result.timedOut ? { timedOut: true } : {}), stdout: result.stdout, stderr: result.stderr,
854
+ };
744
855
  // A poll that was in flight when the session shut down belongs to a stale instance.
745
856
  if (scheduledWorkSuspended) return;
746
857
  appendWatchResult(meta.logPath, result);
@@ -793,6 +904,10 @@ async function pollWatch(
793
904
  else if (result.exitCode !== 0) recordExitFailure(latest, "watch-poll", `Watch poll exited with code ${result.exitCode ?? "unknown"}`, result.exitCode, pollKey,
794
905
  { expected: expectedPollExit, at: result.endedAt });
795
906
  else recoverFailure(latest, "watch-poll", pollKey, result.endedAt);
907
+ observeBlindCheck(latest, result, pollKey, {
908
+ clean: !transportFailure && !conditionErrors.length,
909
+ matched: failure?.matched === true || success?.matched === true,
910
+ });
796
911
  if (conditionErrors.length) latest.error = conditionErrors.join("; ");
797
912
  if (failure?.matched) {
798
913
  recordFailure(latest, "failure_when", "failure condition matched", pollKey, { category: "condition", at: result.endedAt });
@@ -840,10 +955,47 @@ async function pollWatch(
840
955
  }
841
956
  recordFailure(meta, "watch-poll", reason, `throw:${meta.lastCheckedAt ?? meta.startedAt}`, { category: meta.ssh ? "ssh" : "execution" });
842
957
  finalize(meta, { status: "failed", reason }, pi, getActiveSession);
958
+ checked ??= { exitCode: null, signal: null, durationMs: 0, stdout: "", stderr: "", error: reason };
843
959
  }
844
960
  } finally {
845
961
  activePolls.delete(id);
962
+ if (firstCheckWaiters.has(id)) {
963
+ if (checked) settleFirstCheck(id, checked);
964
+ else if (readMeta(id)?.status !== "running") settleFirstCheck(id, undefined);
965
+ }
966
+ }
967
+ }
968
+
969
+ /**
970
+ * Blind-check rule (#359). A check that exits 0, writes stderr, and matches neither condition
971
+ * is probably broken: it prints an error and then reports "still pending" forever (the real
972
+ * incident was a gcloud --format error followed by `exit 0`). After `blindChecks` such checks
973
+ * in a row, record one actionable incident; the existing attention path wakes the parent once.
974
+ * The watch keeps running. A later check with empty stderr, whatever its exit code (including a
975
+ * non-zero, SSH-transport-failed or condition-error check), or one matching a condition,
976
+ * recovers it; those failures are recorded as their own incidents. Any other check (non-zero
977
+ * with stderr) resets the count and leaves the incident open. A clean pending check (exit 0,
978
+ * no stderr) never counts.
979
+ */
980
+ function observeBlindCheck(meta: BackgroundTaskMeta, result: CommandResult, pollKey: unknown,
981
+ check: { clean: boolean; matched: boolean }): void {
982
+ const threshold = meta.blindChecks ?? DEFAULT_BLIND_CHECKS;
983
+ const stderr = result.stderr.trim();
984
+ const blind = threshold > 0 && check.clean && !check.matched && result.exitCode === 0 && stderr.length > 0;
985
+ if (!blind) {
986
+ meta.blindCheckStreak = 0;
987
+ if (!stderr || check.matched) recoverFailure(meta, BLIND_OPERATION, pollKey, result.endedAt);
988
+ return;
846
989
  }
990
+ const streak = (meta.blindCheckStreak ?? 0) + 1;
991
+ meta.blindCheckStreak = streak;
992
+ if (streak !== threshold) return;
993
+ const lastLine = stderr.split(/\r?\n/).map((line) => line.trim()).filter(Boolean).at(-1) ?? "";
994
+ const clipped = lastLine.length > 240 ? `${lastLine.slice(0, 239)}…` : lastLine;
995
+ // The summary is capped in compact rows; the stderr line rides as evidence, which rows show whole.
996
+ recordFailure(meta, BLIND_OPERATION,
997
+ `Blind watch check: ${streak} checks in a row exited 0 with stderr and matched no condition; the check may be broken. The watch keeps running.`,
998
+ pollKey, { category: "blind-check", at: result.endedAt, evidence: `latest stderr: ${clipped} · ${BLIND_CHECK_HINT}` });
847
999
  }
848
1000
 
849
1001
  function finalize(
@@ -279,6 +279,8 @@ export function confineCommandSpec(
279
279
  plan: ForegroundSandboxPlan,
280
280
  profilePath: string,
281
281
  seams: SandboxSeams = {},
282
+ /** Receives lines to show with the launch (e.g. a placeholder left in the user's files). */
283
+ onNotice: (line: string) => void = () => {},
282
284
  ): CommandSpec {
283
285
  if (!plan.confined) return spec;
284
286
 
@@ -331,6 +333,7 @@ export function confineCommandSpec(
331
333
  throw blocked(plan, error instanceof Error ? error.message : String(error));
332
334
  }
333
335
  if (!command) throw blocked(plan, "no sandbox backend was applied");
336
+ for (const line of command.notices ?? []) onNotice(line);
334
337
 
335
338
  return { ...spec, argv: [command.file, ...command.fileArgs], shell: false };
336
339
  }
@@ -75,7 +75,8 @@ export function failureIdentity(...parts: unknown[]): string {
75
75
  return createHash("sha256").update(JSON.stringify(parts)).digest("hex").slice(0, 32);
76
76
  }
77
77
  function text(value: string | undefined, fallback: string): string {
78
- return (value || fallback).replace(/[\x00-\x1f\x7f]/g, " ").slice(0, 400);
78
+ // The 400-unit cut can split a surrogate pair; a lone half would render as U+FFFD (#332).
79
+ return dropLoneSurrogates((value || fallback).replace(/[\x00-\x1f\x7f]/g, " ").slice(0, 400));
79
80
  }
80
81
 
81
82
  /** Agent tool attempts: the agent handles its own tool errors, so one failure is not yet actionable. */
@@ -517,6 +518,8 @@ export interface TerminalFailureParts {
517
518
  rows: string[];
518
519
  /** Counts and separate facts: earlier-reported, unclassified, expected, closed history, correctness. */
519
520
  notes: string[];
521
+ /** The unclassified-count note, when it is among `notes`: a byte budget keeps it longer than the others. */
522
+ unclassifiedNote?: string;
520
523
  }
521
524
  /**
522
525
  * Terminal/completion facts: actionable and incomplete incidents from `incidents` as rows,
@@ -530,12 +533,14 @@ export function terminalFailureParts(state: FailureState, incidents: readonly st
530
533
  const earlier = activeFailures(state).filter((x) => !wanted.has(x.id) && reportable(x)).length;
531
534
  if (earlier) notes.push(`${earlier} actionable incident${earlier === 1 ? " was" : "s were"} reported earlier; not repeated here.`);
532
535
  const counts = failureCounts(state);
533
- if (counts.unclassified) notes.push(`${counts.unclassified} earlier tool failure${counts.unclassified === 1 ? "" : "s"} remain${counts.unclassified === 1 ? "s" : ""} unclassified.`);
536
+ const unclassifiedNote = counts.unclassified
537
+ ? `${counts.unclassified} earlier tool failure${counts.unclassified === 1 ? "" : "s"} remain${counts.unclassified === 1 ? "s" : ""} unclassified.` : undefined;
538
+ if (unclassifiedNote) notes.push(unclassifiedNote);
534
539
  if (counts.expected) notes.push(`${counts.expected} expected failure${counts.expected === 1 ? "" : "s"} recorded.`);
535
540
  const history = closedHistoryLine(state);
536
541
  if (history) notes.push(history);
537
542
  if (counts.unclassified || counts.actionRequired) notes.push(CORRECTNESS_NOTE);
538
- return { rows, notes };
543
+ return { rows, notes, ...(unclassifiedNote ? { unclassifiedNote } : {}) };
539
544
  }
540
545
  export function formatTerminalFailureFacts(state: FailureState, incidents: readonly string[] = []): string {
541
546
  const { rows, notes } = terminalFailureParts(state, incidents);
@@ -643,7 +648,10 @@ export interface IncidentPageRequest extends IncidentRowOptions {
643
648
  * Caller-owned incident pages. Whole rows are preferred; a row larger than the
644
649
  * page is split at a code-point boundary and resumes at that byte, so pages
645
650
  * reconstruct formatFailureLines() exactly (join with "\n" except after a page
646
- * that `endsPartial`). A page never exceeds `maxBytes`.
651
+ * that `endsPartial`). A page never exceeds `maxBytes`, so one smaller than the
652
+ * next code point (up to 4 bytes) returns no text, `hasMore`, and a `nextCursor`
653
+ * equal to `cursor`: the caller retries with a larger page, as with
654
+ * pageVerbatimText. Consumers never ask for fewer than 4 bytes.
647
655
  */
648
656
  export function pageFailureIncidents(state: FailureState, request: IncidentPageRequest = {}): FailureIncidentPage {
649
657
  // A cursor carries its view: following a history cursor stays in history without the flag.
@@ -761,14 +769,19 @@ export function incidentVerbatimPage(state: FailureState, request: IncidentPageR
761
769
  * it, so a cache keyed by it never serves stale incident counts.
762
770
  */
763
771
  export function failureJournalFingerprint(path: string): string {
764
- let file: string;
772
+ return `${fileChangeIdentity(path)}|${existsSync(`${path}.observed`) ? 1 : 0}|${pendingWrites.get(path)?.length ?? 0}`;
773
+ }
774
+ /**
775
+ * A file's identity and change stamp (device, inode, size, and nanosecond mtime/ctime), or its read
776
+ * error. Any append, rewrite, replacement, or deletion changes it; for cache keys, never for trust.
777
+ */
778
+ export function fileChangeIdentity(path: string): string {
765
779
  try {
766
780
  const stats = statSync(path, { bigint: true });
767
- file = `${stats.dev}:${stats.ino}:${stats.size}:${stats.mtimeNs}:${stats.ctimeNs}`;
781
+ return `${stats.dev}:${stats.ino}:${stats.size}:${stats.mtimeNs}:${stats.ctimeNs}`;
768
782
  } catch (error) {
769
- file = `unreadable:${(error as NodeJS.ErrnoException).code ?? "error"}`;
783
+ return `unreadable:${(error as NodeJS.ErrnoException).code ?? "error"}`;
770
784
  }
771
- return `${file}|${existsSync(`${path}.observed`) ? 1 : 0}|${pendingWrites.get(path)?.length ?? 0}`;
772
785
  }
773
786
 
774
787
  export function incidentCursorAt(state: FailureState, offset: number, resource?: string, options: IncidentRowOptions = {}): string {
@@ -903,7 +916,7 @@ export function formatTerminalIncidentSummary(state: FailureState, options: Inci
903
916
  // cursor's retrieval hint, the unclassified count, the cursor, and the count line. The
904
917
  // correctness note goes last, and only when it alone does not fit.
905
918
  const correctness: string[] = notes.filter((x) => x === CORRECTNESS_NOTE);
906
- const unclassified = notes.filter((x) => x !== CORRECTNESS_NOTE && /remains? unclassified\.$/.test(x));
919
+ const unclassified = notes.filter((x) => x === parts.unclassifiedNote);
907
920
  const low = notes.filter((x) => !correctness.includes(x) && !unclassified.includes(x));
908
921
  const ladder: Array<{ mode: Header | undefined; notes: string[] }> = [];
909
922
  for (let keep = low.length - 1; keep >= 0; keep -= 1) ladder.push({ mode: "full", notes: [...low.slice(0, keep), ...unclassified, ...correctness] });