pi-better-harness 0.8.0 → 0.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/node_modules/pi-better-background-tasks/README.md +51 -0
- package/node_modules/pi-better-background-tasks/package.json +1 -1
- package/node_modules/pi-better-background-tasks/src/output.ts +96 -7
- package/node_modules/pi-better-background-tasks/src/runtime.ts +164 -12
- package/node_modules/pi-better-background-tasks/src/sandbox.ts +3 -0
- package/node_modules/pi-better-background-tasks/src/shared-failure-observations.ts +22 -9
- package/node_modules/pi-better-background-tasks/src/shared-navigator.ts +57 -2
- package/node_modules/pi-better-background-tasks/src/shared-sandbox-core.ts +129 -21
- package/node_modules/pi-better-background-tasks/src/tools.ts +52 -10
- package/node_modules/pi-better-background-tasks/src/types.ts +20 -0
- package/node_modules/pi-better-goal/README.md +1 -1
- package/node_modules/pi-better-goal/package.json +1 -1
- package/node_modules/pi-better-goal/src/index.ts +165 -57
- package/node_modules/pi-better-goal/src/types.ts +6 -4
- package/node_modules/pi-better-sandbox/package.json +1 -1
- package/node_modules/pi-better-sandbox/shared-sandbox-core.ts +129 -21
- package/node_modules/pi-better-sandbox/shared-task-files.ts +1 -1
- package/node_modules/pi-better-sandbox/shared-task-sandbox.ts +1 -0
- package/node_modules/pi-better-sandbox/shell.ts +4 -1
- package/node_modules/pi-better-subagents/child-incidents.ts +6 -4
- package/node_modules/pi-better-subagents/docs/failure-observations.md +2 -0
- package/node_modules/pi-better-subagents/failures.ts +49 -3
- package/node_modules/pi-better-subagents/incident-model.ts +7 -7
- package/node_modules/pi-better-subagents/output-payload.ts +2 -10
- package/node_modules/pi-better-subagents/package.json +1 -1
- package/node_modules/pi-better-subagents/shared-failure-observations.ts +22 -9
- package/node_modules/pi-better-subagents/shared-navigator.ts +57 -2
- package/node_modules/pi-better-subagents/shared-sandbox-core.ts +129 -21
- package/node_modules/pi-better-subagents/shared-task-files.ts +1 -1
- package/node_modules/pi-better-subagents/shared-task-sandbox.ts +1 -0
- package/node_modules/pi-better-subagents/task-policy.ts +1 -1
- package/package.json +5 -5
|
@@ -124,6 +124,57 @@ Missing JSON fields or invalid JSON output remain retryable; task status shows
|
|
|
124
124
|
the condition evaluation error until a subsequent poll recovers. Keep a finite
|
|
125
125
|
timeout to bound watches whose output never becomes evaluable.
|
|
126
126
|
|
|
127
|
+
### Writing a watch check
|
|
128
|
+
|
|
129
|
+
A check that swallows its own errors reports "still pending" forever. Keep
|
|
130
|
+
failures visible:
|
|
131
|
+
|
|
132
|
+
- Do not end the check with `exit 0` or `|| true`. A check that exits non-zero
|
|
133
|
+
is recorded as a failure and escalates to the parent session.
|
|
134
|
+
- Map an unknown or unparseable state to failure (a non-zero exit), not to
|
|
135
|
+
pending.
|
|
136
|
+
- Prefer structured output parsed with `jq -e` over hand-written format
|
|
137
|
+
strings. `jq -e` exits non-zero when the field is missing, so a broken query
|
|
138
|
+
shows up at once.
|
|
139
|
+
|
|
140
|
+
For example, a Cloud Run job execution:
|
|
141
|
+
|
|
142
|
+
```sh
|
|
143
|
+
status=$(gcloud run jobs executions describe "$EXECUTION" --region="$REGION" --format=json \
|
|
144
|
+
| jq -er '.status.conditions[] | select(.type == "Completed") | .status') || exit 2
|
|
145
|
+
case "$status" in
|
|
146
|
+
True) echo TERMINAL_SUCCESS ;;
|
|
147
|
+
False) echo TERMINAL_FAILURE ;;
|
|
148
|
+
Unknown) echo STILL_RUNNING ;;
|
|
149
|
+
*) echo "unexpected Completed status: $status" >&2; exit 2 ;;
|
|
150
|
+
esac
|
|
151
|
+
```
|
|
152
|
+
|
|
153
|
+
with `success_when: {type: "stdout_contains", value: "TERMINAL_SUCCESS"}` and
|
|
154
|
+
`failure_when: {type: "stdout_contains", value: "TERMINAL_FAILURE"}`.
|
|
155
|
+
|
|
156
|
+
`bg_task_watch` (and `bg_task` with `action: "watch"`) waits up to 15 seconds
|
|
157
|
+
for the first check and puts its exit code, the newest few lines of stderr and
|
|
158
|
+
of stdout in the tool result, so a broken check is visible at launch. When the
|
|
159
|
+
result is short on room, stdout is cut first. If the first check is still
|
|
160
|
+
running after 15 seconds, or you press Esc during the wait, the result says so
|
|
161
|
+
at once and the watch continues.
|
|
162
|
+
|
|
163
|
+
A running watch also guards against a blind check. When 3 checks in a row exit
|
|
164
|
+
0, write to stderr, and match neither `success_when` nor `failure_when`, the
|
|
165
|
+
watch records one incident that needs action, with the latest stderr line, and
|
|
166
|
+
wakes the parent session once. The watch keeps running. A later check with
|
|
167
|
+
empty stderr recovers the incident whatever its exit code (a non-zero or
|
|
168
|
+
failed check is then recorded as its own incident), and so does a check that
|
|
169
|
+
matches a condition. A non-zero check that writes stderr restarts the count but
|
|
170
|
+
leaves the incident open. A clean pending check (exit 0, no stderr) never
|
|
171
|
+
counts.
|
|
172
|
+
|
|
173
|
+
Some tools write to stderr on success (`gcloud … list` prints "Listed 0
|
|
174
|
+
items.", and kubectl and npm print warnings), which can raise a false alarm.
|
|
175
|
+
If the stderr is expected, redirect it (`2>/dev/null`) or set
|
|
176
|
+
`blind_checks: 0`. Set `blind_checks` to another number to change the count.
|
|
177
|
+
|
|
127
178
|
## Install
|
|
128
179
|
|
|
129
180
|
```sh
|
|
@@ -40,7 +40,10 @@ import {
|
|
|
40
40
|
import { failurePath } from "./failures.js";
|
|
41
41
|
import { captureGapsFor, pageTaskLog, readLog, type LogRead } from "./logs.js";
|
|
42
42
|
import { belongsToOrigin, inspectMeta, listTaskRecords, originOf, type MetaInspection } from "./registry.js";
|
|
43
|
-
import type { BackgroundTaskCallbackOrigin, BackgroundTaskMeta, Condition } from "./types.js";
|
|
43
|
+
import type { BackgroundTaskCallbackOrigin, BackgroundTaskMeta, Condition, FirstWatchCheck } from "./types.js";
|
|
44
|
+
|
|
45
|
+
/** The launch tool stopped waiting while a watch's first check was still running (#359). */
|
|
46
|
+
export type FirstCheckPending = { pending: "timeout" | "aborted" | "suspended"; waitedMs: number };
|
|
44
47
|
|
|
45
48
|
/**
|
|
46
49
|
* Issue #312 consumer budgets. Defaults follow OUTPUT-POLICY / shared
|
|
@@ -240,9 +243,10 @@ export function formatCallbackFacts(meta: BackgroundTaskMeta): {
|
|
|
240
243
|
meta.captureDiscardedBytes ? `capture overflow discarded ${meta.captureDiscardedBytes} bytes; not full history` : undefined,
|
|
241
244
|
].filter((line): line is string => Boolean(line));
|
|
242
245
|
// History (expected and closed incidents) is not a row, but the callback still says it exists, so a
|
|
243
|
-
// declared expected exit is not read as a plain failure.
|
|
246
|
+
// declared expected exit is not read as a plain failure. It leads the decision: a tight callback
|
|
247
|
+
// budget keeps a prefix of the decision, and this short line must survive it (#332).
|
|
244
248
|
const history = rows.length ? historyTailLine(state) : quietFailureLine(state);
|
|
245
|
-
const decision = [formatDecision(meta), ...gapLines
|
|
249
|
+
const decision = [history, formatDecision(meta), ...gapLines].filter(Boolean).join("\n") || undefined;
|
|
246
250
|
return {
|
|
247
251
|
outcome: meta.status,
|
|
248
252
|
...(rows.length ? { failureRows: rows, incidentCount: rows.length } : {}),
|
|
@@ -434,24 +438,109 @@ function asInspection(inspection: MetaInspection | BackgroundTaskMeta | undefine
|
|
|
434
438
|
return { id: meta.id, meta, found: true, readable: true };
|
|
435
439
|
}
|
|
436
440
|
|
|
437
|
-
|
|
441
|
+
/**
|
|
442
|
+
* Launch result. For a watch, `firstCheck` reports its first check (#359): the check's result,
|
|
443
|
+
* or why the launch stopped waiting while it was still running.
|
|
444
|
+
*
|
|
445
|
+
* The first check gets whatever the status budget leaves after the rest of the launch text, so
|
|
446
|
+
* nothing else is clipped for it. The log path is kept whole or dropped whole: a clipped path
|
|
447
|
+
* looks valid but points nowhere.
|
|
448
|
+
*/
|
|
449
|
+
export function formatLaunch(meta: BackgroundTaskMeta, firstCheck?: FirstWatchCheck | FirstCheckPending): string {
|
|
438
450
|
const label = meta.name ? `${meta.name} (${meta.id})` : meta.id;
|
|
439
451
|
const remoteLines = [
|
|
440
452
|
...(meta.ssh ? [`Remote: ${meta.ssh.target}${meta.remote?.session ? ` mode=${meta.remote.session}` : ""}${meta.remote?.sessionName ? ` session=${meta.remote.sessionName}` : ""}.`] : []),
|
|
441
453
|
...(meta.remote?.bootstrapMessage ? [`Remote setup: ${meta.remote.bootstrapMessage}`] : []),
|
|
442
454
|
...(meta.remote?.warning ? [`Warning: ${meta.remote.warning}`] : []),
|
|
443
455
|
];
|
|
444
|
-
|
|
456
|
+
const logLine = `Log: ${meta.logPath}`;
|
|
457
|
+
const build = (checkText: string | undefined, withLog: boolean) => assembleBackgroundContent({
|
|
445
458
|
surface: "status",
|
|
446
459
|
sections: {
|
|
447
460
|
identity: `Started background ${meta.kind} ${label}. Status: ${meta.status}.`,
|
|
448
461
|
failure: incidentSection(meta.id, {}),
|
|
449
462
|
decision: formatDecision(meta),
|
|
450
|
-
diagnostics: remoteLines.join("\n") || undefined,
|
|
451
|
-
progress:
|
|
463
|
+
diagnostics: [...remoteLines, ...(checkText ? [checkText] : [])].join("\n") || undefined,
|
|
464
|
+
progress: withLog ? logLine : undefined,
|
|
452
465
|
},
|
|
453
466
|
gaps: taskGaps(meta),
|
|
454
467
|
});
|
|
468
|
+
let checkText: string | undefined;
|
|
469
|
+
if (firstCheck) {
|
|
470
|
+
const room = backgroundBudget("status") - utf8ByteLength(build(undefined, true)) - 1;
|
|
471
|
+
checkText = formatFirstWatchCheck(meta, firstCheck, room);
|
|
472
|
+
}
|
|
473
|
+
const text = build(checkText, true);
|
|
474
|
+
return text.includes(logLine) ? text : build(checkText, false);
|
|
475
|
+
}
|
|
476
|
+
|
|
477
|
+
const FIRST_CHECK_TAIL_LINES = 3;
|
|
478
|
+
const FIRST_CHECK_LINE_CHARS = 200;
|
|
479
|
+
|
|
480
|
+
function tailLines(text: string): string[] {
|
|
481
|
+
return text.split(/\r?\n/).map((line) => line.replace(/\s+/g, " ").trim()).filter(Boolean)
|
|
482
|
+
.slice(-FIRST_CHECK_TAIL_LINES)
|
|
483
|
+
.map((line) => line.length > FIRST_CHECK_LINE_CHARS ? `${line.slice(0, FIRST_CHECK_LINE_CHARS - 1)}…` : line);
|
|
484
|
+
}
|
|
485
|
+
|
|
486
|
+
function pendingFirstCheckText(check: FirstCheckPending): string {
|
|
487
|
+
switch (check.pending) {
|
|
488
|
+
case "aborted":
|
|
489
|
+
return "First check still running; stopped waiting because the tool call was cancelled. The watch continues in the background; check it later with bg_task_status.";
|
|
490
|
+
case "suspended":
|
|
491
|
+
return "First check still running when the session shut down. The watch continues when its session resumes; check it later with bg_task_status.";
|
|
492
|
+
default:
|
|
493
|
+
return `First check still running after ${formatDuration(check.waitedMs)}; the watch continues in the background. Check it later with bg_task_status.`;
|
|
494
|
+
}
|
|
495
|
+
}
|
|
496
|
+
|
|
497
|
+
/**
|
|
498
|
+
* The first check of a watch in at most `maxBytes` (#359). What matters most is kept first: the
|
|
499
|
+
* outcome, the exit-0-with-stderr warning, the newest stderr lines, then the newest stdout lines,
|
|
500
|
+
* so stdout is cut first. Lines are shown oldest to newest.
|
|
501
|
+
*/
|
|
502
|
+
export function formatFirstWatchCheck(meta: BackgroundTaskMeta, check: FirstWatchCheck | FirstCheckPending, maxBytes = Number.MAX_SAFE_INTEGER): string {
|
|
503
|
+
if ("pending" in check) return pendingFirstCheckText(check);
|
|
504
|
+
if (check.error) return `First check could not run: ${oneLine(check.error, 300)}`;
|
|
505
|
+
const outcome = check.timedOut ? "timed out" : check.signal ? `signal ${check.signal}` : `exit ${check.exitCode ?? "unknown"}`;
|
|
506
|
+
const took = check.durationMs < 1000 ? `${check.durationMs}ms` : formatDuration(check.durationMs);
|
|
507
|
+
const header = `First check: ${outcome} in ${took}.`;
|
|
508
|
+
const stdout = tailLines(check.stdout);
|
|
509
|
+
const stderr = tailLines(check.stderr);
|
|
510
|
+
const warning = check.exitCode === 0 && stderr.length && meta.status === "running"
|
|
511
|
+
? "The check exited 0 but wrote stderr: if it is broken, the watch cannot tell. Let errors exit non-zero."
|
|
512
|
+
: undefined;
|
|
513
|
+
const keptErr: string[] = [];
|
|
514
|
+
const keptOut: string[] = [];
|
|
515
|
+
let keptEmptyStdout = false;
|
|
516
|
+
const render = () => [
|
|
517
|
+
header,
|
|
518
|
+
...(warning ? [warning] : []),
|
|
519
|
+
...(keptErr.length ? [`stderr tail:\n${keptErr.map((line) => ` ${line}`).join("\n")}`] : []),
|
|
520
|
+
...(keptOut.length ? [`stdout tail:\n${keptOut.map((line) => ` ${line}`).join("\n")}`] : []),
|
|
521
|
+
...(keptEmptyStdout ? ["stdout: (empty)"] : []),
|
|
522
|
+
].join("\n");
|
|
523
|
+
const fits = () => utf8ByteLength(render()) <= maxBytes;
|
|
524
|
+
// Newest first within each stream; each kept line goes in front to stay in order.
|
|
525
|
+
for (const line of [...stderr].reverse()) {
|
|
526
|
+
keptErr.unshift(line);
|
|
527
|
+
if (!fits()) { keptErr.shift(); break; }
|
|
528
|
+
}
|
|
529
|
+
for (const line of [...stdout].reverse()) {
|
|
530
|
+
keptOut.unshift(line);
|
|
531
|
+
if (!fits()) { keptOut.shift(); break; }
|
|
532
|
+
}
|
|
533
|
+
if (!stdout.length) {
|
|
534
|
+
keptEmptyStdout = true;
|
|
535
|
+
if (!fits()) keptEmptyStdout = false;
|
|
536
|
+
}
|
|
537
|
+
const omitted = (stderr.length - keptErr.length) + (stdout.length - keptOut.length);
|
|
538
|
+
if (omitted > 0) {
|
|
539
|
+
const note = `(${omitted} output line${omitted === 1 ? "" : "s"} omitted; see bg_task_log)`;
|
|
540
|
+
const text = `${render()}\n${note}`;
|
|
541
|
+
if (utf8ByteLength(text) <= maxBytes) return text;
|
|
542
|
+
}
|
|
543
|
+
return render();
|
|
455
544
|
}
|
|
456
545
|
|
|
457
546
|
function redactedVerbose(meta: BackgroundTaskMeta): unknown {
|
|
@@ -19,6 +19,7 @@ import type {
|
|
|
19
19
|
CommandResult,
|
|
20
20
|
CommandSpec,
|
|
21
21
|
Condition,
|
|
22
|
+
FirstWatchCheck,
|
|
22
23
|
RemoteTaskParams,
|
|
23
24
|
SshConnectionParams,
|
|
24
25
|
TerminalResult,
|
|
@@ -31,6 +32,8 @@ const processTimeoutTimers = new Map<string, ReturnType<typeof setTimeout>>();
|
|
|
31
32
|
const activeProcessTimeouts = new Set<string>();
|
|
32
33
|
const activeRemoteTasks = new Map<string, ResolvedSshRemoteTask>();
|
|
33
34
|
const remoteSessionStarts = new Map<string, Promise<CommandResult>>();
|
|
35
|
+
/** Tasks whose stop in this instance is waiting on an in-flight tmux start, and kills the session itself. */
|
|
36
|
+
const remoteStopsAwaitingStart = new Set<string>();
|
|
34
37
|
const activePolls = new Set<string>();
|
|
35
38
|
const logRetentionTimers = new Map<string, ReturnType<typeof setInterval>>();
|
|
36
39
|
const LOG_RETENTION_CHECK_MS = 1000;
|
|
@@ -44,6 +47,63 @@ let scheduledWorkSuspended = false;
|
|
|
44
47
|
const REMOTE_SESSION_POLL_MS = 100;
|
|
45
48
|
|
|
46
49
|
export const DEFAULT_WATCH_TIMEOUT_SECONDS = 15 * 60;
|
|
50
|
+
/** Consecutive blind checks (exit 0, stderr, no condition matched) before a watch is flagged (#359). */
|
|
51
|
+
export const DEFAULT_BLIND_CHECKS = 3;
|
|
52
|
+
/** How long bg_task_watch waits for the first check before returning (#359). */
|
|
53
|
+
export const FIRST_WATCH_CHECK_WAIT_MS = 15_000;
|
|
54
|
+
const BLIND_OPERATION = "watch-blind";
|
|
55
|
+
/** How to fix a false alarm: some tools write progress or warnings to stderr on success. */
|
|
56
|
+
export const BLIND_CHECK_HINT = "If the stderr is expected (progress or warnings), redirect it (2>/dev/null) or set blind_checks:0.";
|
|
57
|
+
|
|
58
|
+
/**
|
|
59
|
+
* Why the launch stopped waiting before the first check finished (#359): the bounded wait
|
|
60
|
+
* ran out, the tool call was aborted (Esc), or the session shut down.
|
|
61
|
+
*/
|
|
62
|
+
export type FirstCheckWaitEnd = "timeout" | "aborted" | "suspended";
|
|
63
|
+
/** A first check's outcome, or why it is not known yet; undefined when the watch ended without one. */
|
|
64
|
+
export type FirstCheckOutcome = FirstWatchCheck | { pending: FirstCheckWaitEnd; waitedMs: number } | undefined;
|
|
65
|
+
|
|
66
|
+
/** In-flight first checks of watches launched by this instance, keyed by task id. */
|
|
67
|
+
const firstCheckWaiters = new Map<string, { promise: Promise<FirstWatchCheck | "suspended" | undefined>; resolve: (check: FirstWatchCheck | "suspended" | undefined) => void }>();
|
|
68
|
+
|
|
69
|
+
/**
|
|
70
|
+
* Wait for a watch's first check, bounded by `timeoutMs` and `signal` (#359). Resolves a pending
|
|
71
|
+
* outcome when the wait ends first, and undefined when this instance did not launch the watch
|
|
72
|
+
* or the watch ended without a check.
|
|
73
|
+
*/
|
|
74
|
+
export async function awaitFirstWatchCheck(id: string, timeoutMs = FIRST_WATCH_CHECK_WAIT_MS, signal?: AbortSignal): Promise<FirstCheckOutcome> {
|
|
75
|
+
const waiter = firstCheckWaiters.get(id);
|
|
76
|
+
if (!waiter) return undefined;
|
|
77
|
+
const started = Date.now();
|
|
78
|
+
const pending = (reason: FirstCheckWaitEnd) => ({ pending: reason, waitedMs: Date.now() - started });
|
|
79
|
+
if (signal?.aborted) return pending("aborted");
|
|
80
|
+
let timer: ReturnType<typeof setTimeout> | undefined;
|
|
81
|
+
let onAbort: (() => void) | undefined;
|
|
82
|
+
const timeout = new Promise<"timeout">((resolve) => {
|
|
83
|
+
// Kept referenced: the launching tool call is waiting on it, and the watch timers are
|
|
84
|
+
// unref'd, so an unref'd wait could let the event loop drain before the first check.
|
|
85
|
+
timer = setTimeout(() => resolve("timeout"), Math.max(0, timeoutMs));
|
|
86
|
+
});
|
|
87
|
+
const aborted = new Promise<"aborted">((resolve) => {
|
|
88
|
+
onAbort = () => resolve("aborted");
|
|
89
|
+
signal?.addEventListener("abort", onAbort, { once: true });
|
|
90
|
+
});
|
|
91
|
+
try {
|
|
92
|
+
const outcome = await Promise.race([waiter.promise, timeout, aborted]);
|
|
93
|
+
if (outcome === "timeout" || outcome === "aborted" || outcome === "suspended") return pending(outcome);
|
|
94
|
+
return outcome;
|
|
95
|
+
} finally {
|
|
96
|
+
if (timer) clearTimeout(timer);
|
|
97
|
+
if (onAbort) signal?.removeEventListener("abort", onAbort);
|
|
98
|
+
}
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
function settleFirstCheck(id: string, check: FirstWatchCheck | "suspended" | undefined): void {
|
|
102
|
+
const waiter = firstCheckWaiters.get(id);
|
|
103
|
+
if (!waiter) return;
|
|
104
|
+
firstCheckWaiters.delete(id);
|
|
105
|
+
waiter.resolve(check);
|
|
106
|
+
}
|
|
47
107
|
|
|
48
108
|
/**
|
|
49
109
|
* Stop this extension instance's scheduled work when its session shuts down (#324).
|
|
@@ -81,6 +141,8 @@ export function suspendScheduledWork(): void {
|
|
|
81
141
|
remoteSessionTimers.clear();
|
|
82
142
|
processTimeoutTimers.clear();
|
|
83
143
|
logRetentionTimers.clear();
|
|
144
|
+
// A launch still waiting on a first check that will not run here reports it as still running.
|
|
145
|
+
for (const id of [...firstCheckWaiters.keys()]) settleFirstCheck(id, "suspended");
|
|
84
146
|
suspendFailureAttention();
|
|
85
147
|
}
|
|
86
148
|
|
|
@@ -94,12 +156,16 @@ export function resumeScheduledWork(): void {
|
|
|
94
156
|
|
|
95
157
|
export type ActiveSessionProvider = () => BackgroundTaskCallbackOrigin | undefined;
|
|
96
158
|
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
/**
|
|
159
|
+
/** Structured intent (#325), declared by the launching agent and shared by spawns and watchers. */
|
|
160
|
+
export interface TaskIntentParams {
|
|
161
|
+
/** Stable id shared by modified retries of one operation. */
|
|
100
162
|
operation_id?: string | null;
|
|
101
|
-
/**
|
|
163
|
+
/** Non-zero exit codes declared intentional before launch. */
|
|
102
164
|
expected_exit_codes?: number[] | null;
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
export interface SpawnTaskParams extends CommandSpec, TaskIntentParams {
|
|
168
|
+
name?: string;
|
|
103
169
|
callback?: boolean;
|
|
104
170
|
timeout_seconds?: number;
|
|
105
171
|
max_log_bytes?: number;
|
|
@@ -107,18 +173,16 @@ export interface SpawnTaskParams extends CommandSpec {
|
|
|
107
173
|
remote?: RemoteTaskParams;
|
|
108
174
|
}
|
|
109
175
|
|
|
110
|
-
export interface WatchTaskParams extends CommandSpec {
|
|
176
|
+
export interface WatchTaskParams extends CommandSpec, TaskIntentParams {
|
|
111
177
|
name?: string;
|
|
112
|
-
/** Structured intent (#325): stable id shared by modified retries of one operation. */
|
|
113
|
-
operation_id?: string | null;
|
|
114
|
-
/** Structured intent (#325): non-zero exit codes declared intentional before launch. */
|
|
115
|
-
expected_exit_codes?: number[] | null;
|
|
116
178
|
callback?: boolean;
|
|
117
179
|
interval_seconds?: number;
|
|
118
180
|
timeout_seconds?: number;
|
|
119
181
|
max_log_bytes?: number;
|
|
120
182
|
success_when: Condition;
|
|
121
183
|
failure_when?: Condition;
|
|
184
|
+
/** Consecutive blind checks before the watch is flagged (#359). Default 3; 0 turns it off. */
|
|
185
|
+
blind_checks?: number;
|
|
122
186
|
ssh?: SshConnectionParams;
|
|
123
187
|
remote?: RemoteTaskParams;
|
|
124
188
|
}
|
|
@@ -157,9 +221,10 @@ export function spawnTask(
|
|
|
157
221
|
}, dependencies.remoteRunner)
|
|
158
222
|
: undefined;
|
|
159
223
|
const commandSpec: CommandSpec = remoteTask?.commandSpec ?? { ...params, cwd, shell: params.shell ?? true };
|
|
224
|
+
const sandboxNotices: string[] = [];
|
|
160
225
|
const launchSpec = remoteTask
|
|
161
226
|
? commandSpec
|
|
162
|
-
: confineCommandSpec(commandSpec, sandboxPlan, sandboxProfilePathFor(id));
|
|
227
|
+
: confineCommandSpec(commandSpec, sandboxPlan, sandboxProfilePathFor(id), {}, (line) => sandboxNotices.push(line));
|
|
163
228
|
const tmuxBacked = remoteTask?.metadata.remote.session === "tmux";
|
|
164
229
|
// Outside project = Write or Write & delete can change files outside the
|
|
165
230
|
// project: start an APFS local snapshot (macOS, background, never blocks).
|
|
@@ -169,6 +234,9 @@ export function spawnTask(
|
|
|
169
234
|
: remoteTask
|
|
170
235
|
? remoteTask.spawn(logPath, true)
|
|
171
236
|
: spawnCommand(launchSpec, logPath, true);
|
|
237
|
+
for (const line of sandboxNotices) {
|
|
238
|
+
try { appendLine(logPath, `--- ${line} ---`); } catch { /* log gone */ }
|
|
239
|
+
}
|
|
172
240
|
if (snapshot?.started) {
|
|
173
241
|
void snapshot.done.then((outcome) => {
|
|
174
242
|
if (!outcome.ok) {
|
|
@@ -276,7 +344,15 @@ async function launchRemoteTmux(
|
|
|
276
344
|
if (remoteSessionStarts.get(id) === startAttempt) remoteSessionStarts.delete(id);
|
|
277
345
|
}
|
|
278
346
|
const afterStart = readMeta(id);
|
|
279
|
-
if (!afterStart
|
|
347
|
+
if (!afterStart) return;
|
|
348
|
+
if (afterStart.status !== "running" || afterStart.stopRequestedAt) {
|
|
349
|
+
// Stopped (or timed out) while the start was in flight. A stop in this instance awaited the
|
|
350
|
+
// attempt and kills the session itself. Anything else could not see the attempt: a stop from
|
|
351
|
+
// an instance loaded by /reload, or a deadline that found the session not yet started. Only
|
|
352
|
+
// this launch can still reach the session it may just have created, so it kills it (#332).
|
|
353
|
+
if (!remoteStopsAwaitingStart.has(id)) await killSessionStartedAfterStop(afterStart, remoteTask);
|
|
354
|
+
return;
|
|
355
|
+
}
|
|
280
356
|
if (started.exitCode !== 0) {
|
|
281
357
|
const detail = started.stderr.trim() || started.stdout.trim() || "remote tmux returned no diagnostic";
|
|
282
358
|
const reason = `Could not create remote tmux session ${afterStart.remote?.sessionName} on ${afterStart.ssh?.target} (exit ${started.exitCode ?? "unknown"}): ${detail}`;
|
|
@@ -296,6 +372,19 @@ async function launchRemoteTmux(
|
|
|
296
372
|
}
|
|
297
373
|
}
|
|
298
374
|
|
|
375
|
+
/** Kill a tmux session whose start completed after its task stopped; a failed start may still have left it running. */
|
|
376
|
+
async function killSessionStartedAfterStop(meta: BackgroundTaskMeta, remoteTask: ResolvedSshRemoteTask): Promise<void> {
|
|
377
|
+
const session = `remote tmux session ${meta.remote?.sessionName} on ${meta.ssh?.target}`;
|
|
378
|
+
try {
|
|
379
|
+
const stopped = await remoteTask.killTmuxSession();
|
|
380
|
+
appendLine(meta.logPath, stopped.exitCode === 0
|
|
381
|
+
? `--- Killed ${session}: it started after the task was ${meta.status === "running" ? "stopped" : meta.status} ---`
|
|
382
|
+
: `--- Could not kill ${session} that started after the task stopped (exit ${stopped.exitCode ?? "unknown"}) ---`);
|
|
383
|
+
} catch (error) {
|
|
384
|
+
appendLine(meta.logPath, `--- Could not kill ${session} that started after the task stopped: ${error instanceof Error ? error.message : String(error)} ---`);
|
|
385
|
+
}
|
|
386
|
+
}
|
|
387
|
+
|
|
299
388
|
function scheduleRemoteSessionPoll(
|
|
300
389
|
pi: ExtensionAPI,
|
|
301
390
|
id: string,
|
|
@@ -404,6 +493,7 @@ export function startWatchTask(
|
|
|
404
493
|
const error = condition && validateCondition(condition);
|
|
405
494
|
if (error) throw new Error(`${name}: ${error}`);
|
|
406
495
|
}
|
|
496
|
+
const blindChecks = readBlindChecks(params.blind_checks);
|
|
407
497
|
const intent = readTaskIntent(params);
|
|
408
498
|
const sandboxPlan = resolveForegroundSandboxPlan(pi, !!params.ssh);
|
|
409
499
|
const id = nextTaskId();
|
|
@@ -421,9 +511,10 @@ export function startWatchTask(
|
|
|
421
511
|
}, dependencies.remoteRunner)
|
|
422
512
|
: undefined;
|
|
423
513
|
const commandSpec: CommandSpec = remoteTask?.commandSpec ?? { ...params, cwd, shell: params.shell ?? true };
|
|
514
|
+
const sandboxNotices: string[] = [];
|
|
424
515
|
const launchSpec = remoteTask
|
|
425
516
|
? commandSpec
|
|
426
|
-
: confineCommandSpec(commandSpec, sandboxPlan, sandboxProfilePathFor(id));
|
|
517
|
+
: confineCommandSpec(commandSpec, sandboxPlan, sandboxProfilePathFor(id), {}, (line) => sandboxNotices.push(line));
|
|
427
518
|
const meta: BackgroundTaskMeta = {
|
|
428
519
|
id,
|
|
429
520
|
name: params.name,
|
|
@@ -447,6 +538,7 @@ export function startWatchTask(
|
|
|
447
538
|
spawnPidStartTime: currentProcessStartToken(),
|
|
448
539
|
successWhen: params.success_when,
|
|
449
540
|
failureWhen: params.failure_when,
|
|
541
|
+
...(blindChecks !== undefined ? { blindChecks } : {}),
|
|
450
542
|
notifyOn: "terminal",
|
|
451
543
|
ssh: remoteTask?.metadata.ssh,
|
|
452
544
|
remote: remoteTask?.metadata.remote,
|
|
@@ -454,13 +546,25 @@ export function startWatchTask(
|
|
|
454
546
|
};
|
|
455
547
|
ensureTaskDir(id);
|
|
456
548
|
appendLine(meta.logPath, `--- watch ${new Date(now).toISOString()} interval_ms=${meta.intervalMs} ---`);
|
|
549
|
+
for (const line of sandboxNotices) appendLine(meta.logPath, `--- ${line} ---`);
|
|
457
550
|
writeMeta(meta);
|
|
551
|
+
let resolveFirst!: (check: FirstWatchCheck | "suspended" | undefined) => void;
|
|
552
|
+
const firstCheck = new Promise<FirstWatchCheck | "suspended" | undefined>((resolve) => { resolveFirst = resolve; });
|
|
553
|
+
firstCheckWaiters.set(id, { promise: firstCheck, resolve: resolveFirst });
|
|
458
554
|
scheduleWatch(pi, id, 0, getActiveSession, remoteTask
|
|
459
555
|
? (timeoutMs) => remoteTask.runOnce(undefined, timeoutMs)
|
|
460
556
|
: undefined);
|
|
461
557
|
return meta;
|
|
462
558
|
}
|
|
463
559
|
|
|
560
|
+
function readBlindChecks(value: unknown): number | undefined {
|
|
561
|
+
if (value === undefined || value === null) return undefined;
|
|
562
|
+
if (typeof value !== "number" || !Number.isInteger(value) || value < 0) {
|
|
563
|
+
throw new Error("blind_checks must be a non-negative integer (0 turns the blind-check rule off).");
|
|
564
|
+
}
|
|
565
|
+
return value;
|
|
566
|
+
}
|
|
567
|
+
|
|
464
568
|
function resolveWatchTimeoutSeconds(timeoutSeconds: number | undefined): number | undefined {
|
|
465
569
|
if (timeoutSeconds === undefined) return DEFAULT_WATCH_TIMEOUT_SECONDS;
|
|
466
570
|
if (timeoutSeconds <= 0) return undefined;
|
|
@@ -641,7 +745,9 @@ export async function stopTask(
|
|
|
641
745
|
const remoteSessionMayExist = remote?.session === "tmux"
|
|
642
746
|
&& (remote.sessionStarted !== false || remoteStartAttempt !== undefined);
|
|
643
747
|
if (remoteStartAttempt) {
|
|
748
|
+
remoteStopsAwaitingStart.add(id);
|
|
644
749
|
try { await remoteStartAttempt; } catch { /* A failed SSH result can still leave the detached session running. */ }
|
|
750
|
+
finally { remoteStopsAwaitingStart.delete(id); }
|
|
645
751
|
}
|
|
646
752
|
|
|
647
753
|
if (remoteSessionMayExist) {
|
|
@@ -729,6 +835,7 @@ async function pollWatch(
|
|
|
729
835
|
): Promise<void> {
|
|
730
836
|
if (activePolls.has(id)) return;
|
|
731
837
|
activePolls.add(id);
|
|
838
|
+
let checked: FirstWatchCheck | undefined;
|
|
732
839
|
try {
|
|
733
840
|
const meta = readMeta(id);
|
|
734
841
|
if (!meta || meta.status !== "running" || meta.kind !== "command_watch") return;
|
|
@@ -741,6 +848,10 @@ async function pollWatch(
|
|
|
741
848
|
const result = runOnce
|
|
742
849
|
? await runOnce(timeoutMs)
|
|
743
850
|
: await runCommandOnce(commandSpecFromMeta(meta), undefined, timeoutMs);
|
|
851
|
+
checked = {
|
|
852
|
+
exitCode: result.exitCode, signal: result.signal, durationMs: Math.max(0, result.endedAt - result.startedAt),
|
|
853
|
+
...(result.timedOut ? { timedOut: true } : {}), stdout: result.stdout, stderr: result.stderr,
|
|
854
|
+
};
|
|
744
855
|
// A poll that was in flight when the session shut down belongs to a stale instance.
|
|
745
856
|
if (scheduledWorkSuspended) return;
|
|
746
857
|
appendWatchResult(meta.logPath, result);
|
|
@@ -793,6 +904,10 @@ async function pollWatch(
|
|
|
793
904
|
else if (result.exitCode !== 0) recordExitFailure(latest, "watch-poll", `Watch poll exited with code ${result.exitCode ?? "unknown"}`, result.exitCode, pollKey,
|
|
794
905
|
{ expected: expectedPollExit, at: result.endedAt });
|
|
795
906
|
else recoverFailure(latest, "watch-poll", pollKey, result.endedAt);
|
|
907
|
+
observeBlindCheck(latest, result, pollKey, {
|
|
908
|
+
clean: !transportFailure && !conditionErrors.length,
|
|
909
|
+
matched: failure?.matched === true || success?.matched === true,
|
|
910
|
+
});
|
|
796
911
|
if (conditionErrors.length) latest.error = conditionErrors.join("; ");
|
|
797
912
|
if (failure?.matched) {
|
|
798
913
|
recordFailure(latest, "failure_when", "failure condition matched", pollKey, { category: "condition", at: result.endedAt });
|
|
@@ -840,10 +955,47 @@ async function pollWatch(
|
|
|
840
955
|
}
|
|
841
956
|
recordFailure(meta, "watch-poll", reason, `throw:${meta.lastCheckedAt ?? meta.startedAt}`, { category: meta.ssh ? "ssh" : "execution" });
|
|
842
957
|
finalize(meta, { status: "failed", reason }, pi, getActiveSession);
|
|
958
|
+
checked ??= { exitCode: null, signal: null, durationMs: 0, stdout: "", stderr: "", error: reason };
|
|
843
959
|
}
|
|
844
960
|
} finally {
|
|
845
961
|
activePolls.delete(id);
|
|
962
|
+
if (firstCheckWaiters.has(id)) {
|
|
963
|
+
if (checked) settleFirstCheck(id, checked);
|
|
964
|
+
else if (readMeta(id)?.status !== "running") settleFirstCheck(id, undefined);
|
|
965
|
+
}
|
|
966
|
+
}
|
|
967
|
+
}
|
|
968
|
+
|
|
969
|
+
/**
|
|
970
|
+
* Blind-check rule (#359). A check that exits 0, writes stderr, and matches neither condition
|
|
971
|
+
* is probably broken: it prints an error and then reports "still pending" forever (the real
|
|
972
|
+
* incident was a gcloud --format error followed by `exit 0`). After `blindChecks` such checks
|
|
973
|
+
* in a row, record one actionable incident; the existing attention path wakes the parent once.
|
|
974
|
+
* The watch keeps running. A later check with empty stderr, whatever its exit code (including a
|
|
975
|
+
* non-zero, SSH-transport-failed or condition-error check), or one matching a condition,
|
|
976
|
+
* recovers it; those failures are recorded as their own incidents. Any other check (non-zero
|
|
977
|
+
* with stderr) resets the count and leaves the incident open. A clean pending check (exit 0,
|
|
978
|
+
* no stderr) never counts.
|
|
979
|
+
*/
|
|
980
|
+
function observeBlindCheck(meta: BackgroundTaskMeta, result: CommandResult, pollKey: unknown,
|
|
981
|
+
check: { clean: boolean; matched: boolean }): void {
|
|
982
|
+
const threshold = meta.blindChecks ?? DEFAULT_BLIND_CHECKS;
|
|
983
|
+
const stderr = result.stderr.trim();
|
|
984
|
+
const blind = threshold > 0 && check.clean && !check.matched && result.exitCode === 0 && stderr.length > 0;
|
|
985
|
+
if (!blind) {
|
|
986
|
+
meta.blindCheckStreak = 0;
|
|
987
|
+
if (!stderr || check.matched) recoverFailure(meta, BLIND_OPERATION, pollKey, result.endedAt);
|
|
988
|
+
return;
|
|
846
989
|
}
|
|
990
|
+
const streak = (meta.blindCheckStreak ?? 0) + 1;
|
|
991
|
+
meta.blindCheckStreak = streak;
|
|
992
|
+
if (streak !== threshold) return;
|
|
993
|
+
const lastLine = stderr.split(/\r?\n/).map((line) => line.trim()).filter(Boolean).at(-1) ?? "";
|
|
994
|
+
const clipped = lastLine.length > 240 ? `${lastLine.slice(0, 239)}…` : lastLine;
|
|
995
|
+
// The summary is capped in compact rows; the stderr line rides as evidence, which rows show whole.
|
|
996
|
+
recordFailure(meta, BLIND_OPERATION,
|
|
997
|
+
`Blind watch check: ${streak} checks in a row exited 0 with stderr and matched no condition; the check may be broken. The watch keeps running.`,
|
|
998
|
+
pollKey, { category: "blind-check", at: result.endedAt, evidence: `latest stderr: ${clipped} · ${BLIND_CHECK_HINT}` });
|
|
847
999
|
}
|
|
848
1000
|
|
|
849
1001
|
function finalize(
|
|
@@ -279,6 +279,8 @@ export function confineCommandSpec(
|
|
|
279
279
|
plan: ForegroundSandboxPlan,
|
|
280
280
|
profilePath: string,
|
|
281
281
|
seams: SandboxSeams = {},
|
|
282
|
+
/** Receives lines to show with the launch (e.g. a placeholder left in the user's files). */
|
|
283
|
+
onNotice: (line: string) => void = () => {},
|
|
282
284
|
): CommandSpec {
|
|
283
285
|
if (!plan.confined) return spec;
|
|
284
286
|
|
|
@@ -331,6 +333,7 @@ export function confineCommandSpec(
|
|
|
331
333
|
throw blocked(plan, error instanceof Error ? error.message : String(error));
|
|
332
334
|
}
|
|
333
335
|
if (!command) throw blocked(plan, "no sandbox backend was applied");
|
|
336
|
+
for (const line of command.notices ?? []) onNotice(line);
|
|
334
337
|
|
|
335
338
|
return { ...spec, argv: [command.file, ...command.fileArgs], shell: false };
|
|
336
339
|
}
|
|
@@ -75,7 +75,8 @@ export function failureIdentity(...parts: unknown[]): string {
|
|
|
75
75
|
return createHash("sha256").update(JSON.stringify(parts)).digest("hex").slice(0, 32);
|
|
76
76
|
}
|
|
77
77
|
function text(value: string | undefined, fallback: string): string {
|
|
78
|
-
|
|
78
|
+
// The 400-unit cut can split a surrogate pair; a lone half would render as U+FFFD (#332).
|
|
79
|
+
return dropLoneSurrogates((value || fallback).replace(/[\x00-\x1f\x7f]/g, " ").slice(0, 400));
|
|
79
80
|
}
|
|
80
81
|
|
|
81
82
|
/** Agent tool attempts: the agent handles its own tool errors, so one failure is not yet actionable. */
|
|
@@ -517,6 +518,8 @@ export interface TerminalFailureParts {
|
|
|
517
518
|
rows: string[];
|
|
518
519
|
/** Counts and separate facts: earlier-reported, unclassified, expected, closed history, correctness. */
|
|
519
520
|
notes: string[];
|
|
521
|
+
/** The unclassified-count note, when it is among `notes`: a byte budget keeps it longer than the others. */
|
|
522
|
+
unclassifiedNote?: string;
|
|
520
523
|
}
|
|
521
524
|
/**
|
|
522
525
|
* Terminal/completion facts: actionable and incomplete incidents from `incidents` as rows,
|
|
@@ -530,12 +533,14 @@ export function terminalFailureParts(state: FailureState, incidents: readonly st
|
|
|
530
533
|
const earlier = activeFailures(state).filter((x) => !wanted.has(x.id) && reportable(x)).length;
|
|
531
534
|
if (earlier) notes.push(`${earlier} actionable incident${earlier === 1 ? " was" : "s were"} reported earlier; not repeated here.`);
|
|
532
535
|
const counts = failureCounts(state);
|
|
533
|
-
|
|
536
|
+
const unclassifiedNote = counts.unclassified
|
|
537
|
+
? `${counts.unclassified} earlier tool failure${counts.unclassified === 1 ? "" : "s"} remain${counts.unclassified === 1 ? "s" : ""} unclassified.` : undefined;
|
|
538
|
+
if (unclassifiedNote) notes.push(unclassifiedNote);
|
|
534
539
|
if (counts.expected) notes.push(`${counts.expected} expected failure${counts.expected === 1 ? "" : "s"} recorded.`);
|
|
535
540
|
const history = closedHistoryLine(state);
|
|
536
541
|
if (history) notes.push(history);
|
|
537
542
|
if (counts.unclassified || counts.actionRequired) notes.push(CORRECTNESS_NOTE);
|
|
538
|
-
return { rows, notes };
|
|
543
|
+
return { rows, notes, ...(unclassifiedNote ? { unclassifiedNote } : {}) };
|
|
539
544
|
}
|
|
540
545
|
export function formatTerminalFailureFacts(state: FailureState, incidents: readonly string[] = []): string {
|
|
541
546
|
const { rows, notes } = terminalFailureParts(state, incidents);
|
|
@@ -643,7 +648,10 @@ export interface IncidentPageRequest extends IncidentRowOptions {
|
|
|
643
648
|
* Caller-owned incident pages. Whole rows are preferred; a row larger than the
|
|
644
649
|
* page is split at a code-point boundary and resumes at that byte, so pages
|
|
645
650
|
* reconstruct formatFailureLines() exactly (join with "\n" except after a page
|
|
646
|
-
* that `endsPartial`). A page never exceeds `maxBytes
|
|
651
|
+
* that `endsPartial`). A page never exceeds `maxBytes`, so one smaller than the
|
|
652
|
+
* next code point (up to 4 bytes) returns no text, `hasMore`, and a `nextCursor`
|
|
653
|
+
* equal to `cursor`: the caller retries with a larger page, as with
|
|
654
|
+
* pageVerbatimText. Consumers never ask for fewer than 4 bytes.
|
|
647
655
|
*/
|
|
648
656
|
export function pageFailureIncidents(state: FailureState, request: IncidentPageRequest = {}): FailureIncidentPage {
|
|
649
657
|
// A cursor carries its view: following a history cursor stays in history without the flag.
|
|
@@ -761,14 +769,19 @@ export function incidentVerbatimPage(state: FailureState, request: IncidentPageR
|
|
|
761
769
|
* it, so a cache keyed by it never serves stale incident counts.
|
|
762
770
|
*/
|
|
763
771
|
export function failureJournalFingerprint(path: string): string {
|
|
764
|
-
|
|
772
|
+
return `${fileChangeIdentity(path)}|${existsSync(`${path}.observed`) ? 1 : 0}|${pendingWrites.get(path)?.length ?? 0}`;
|
|
773
|
+
}
|
|
774
|
+
/**
|
|
775
|
+
* A file's identity and change stamp (device, inode, size, and nanosecond mtime/ctime), or its read
|
|
776
|
+
* error. Any append, rewrite, replacement, or deletion changes it; for cache keys, never for trust.
|
|
777
|
+
*/
|
|
778
|
+
export function fileChangeIdentity(path: string): string {
|
|
765
779
|
try {
|
|
766
780
|
const stats = statSync(path, { bigint: true });
|
|
767
|
-
|
|
781
|
+
return `${stats.dev}:${stats.ino}:${stats.size}:${stats.mtimeNs}:${stats.ctimeNs}`;
|
|
768
782
|
} catch (error) {
|
|
769
|
-
|
|
783
|
+
return `unreadable:${(error as NodeJS.ErrnoException).code ?? "error"}`;
|
|
770
784
|
}
|
|
771
|
-
return `${file}|${existsSync(`${path}.observed`) ? 1 : 0}|${pendingWrites.get(path)?.length ?? 0}`;
|
|
772
785
|
}
|
|
773
786
|
|
|
774
787
|
export function incidentCursorAt(state: FailureState, offset: number, resource?: string, options: IncidentRowOptions = {}): string {
|
|
@@ -903,7 +916,7 @@ export function formatTerminalIncidentSummary(state: FailureState, options: Inci
|
|
|
903
916
|
// cursor's retrieval hint, the unclassified count, the cursor, and the count line. The
|
|
904
917
|
// correctness note goes last, and only when it alone does not fit.
|
|
905
918
|
const correctness: string[] = notes.filter((x) => x === CORRECTNESS_NOTE);
|
|
906
|
-
const unclassified = notes.filter((x) => x
|
|
919
|
+
const unclassified = notes.filter((x) => x === parts.unclassifiedNote);
|
|
907
920
|
const low = notes.filter((x) => !correctness.includes(x) && !unclassified.includes(x));
|
|
908
921
|
const ladder: Array<{ mode: Header | undefined; notes: string[] }> = [];
|
|
909
922
|
for (let keep = low.length - 1; keep >= 0; keep -= 1) ladder.push({ mode: "full", notes: [...low.slice(0, keep), ...unclassified, ...correctness] });
|