pi-better-harness 0.7.0 → 0.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (48) hide show
  1. package/node_modules/pi-better-background-tasks/README.md +57 -4
  2. package/node_modules/pi-better-background-tasks/package.json +1 -1
  3. package/node_modules/pi-better-background-tasks/src/failures.ts +9 -1
  4. package/node_modules/pi-better-background-tasks/src/navigator-provider.ts +9 -6
  5. package/node_modules/pi-better-background-tasks/src/output.ts +101 -8
  6. package/node_modules/pi-better-background-tasks/src/runtime.ts +189 -21
  7. package/node_modules/pi-better-background-tasks/src/sandbox.ts +3 -0
  8. package/node_modules/pi-better-background-tasks/src/shared-failure-observations.ts +70 -21
  9. package/node_modules/pi-better-background-tasks/src/shared-log-utils.ts +6 -12
  10. package/node_modules/pi-better-background-tasks/src/shared-navigator.ts +224 -125
  11. package/node_modules/pi-better-background-tasks/src/shared-sandbox-core.ts +194 -22
  12. package/node_modules/pi-better-background-tasks/src/tools.ts +53 -11
  13. package/node_modules/pi-better-background-tasks/src/types.ts +20 -0
  14. package/node_modules/pi-better-goal/package.json +1 -1
  15. package/node_modules/pi-better-sandbox/README.md +32 -6
  16. package/node_modules/pi-better-sandbox/index.ts +4 -0
  17. package/node_modules/pi-better-sandbox/package.json +1 -1
  18. package/node_modules/pi-better-sandbox/permissions-page.ts +74 -7
  19. package/node_modules/pi-better-sandbox/permissions.ts +13 -2
  20. package/node_modules/pi-better-sandbox/shared-sandbox-core.ts +194 -22
  21. package/node_modules/pi-better-sandbox/shared-task-apply-patch.ts +409 -0
  22. package/node_modules/pi-better-sandbox/shared-task-files.ts +69 -7
  23. package/node_modules/pi-better-sandbox/shared-task-sandbox.ts +12 -0
  24. package/node_modules/pi-better-sandbox/shared-task-tools.ts +111 -0
  25. package/node_modules/pi-better-sandbox/shell.ts +4 -1
  26. package/node_modules/pi-better-sandbox/state.ts +5 -1
  27. package/node_modules/pi-better-subagents/README.md +1 -1
  28. package/node_modules/pi-better-subagents/child-incidents.ts +8 -6
  29. package/node_modules/pi-better-subagents/docs/failure-observations.md +3 -1
  30. package/node_modules/pi-better-subagents/failures.ts +94 -6
  31. package/node_modules/pi-better-subagents/incident-model.ts +18 -11
  32. package/node_modules/pi-better-subagents/index.ts +47 -18
  33. package/node_modules/pi-better-subagents/output-payload.ts +4 -13
  34. package/node_modules/pi-better-subagents/package.json +1 -1
  35. package/node_modules/pi-better-subagents/permission-policy.ts +9 -3
  36. package/node_modules/pi-better-subagents/registry.ts +19 -2
  37. package/node_modules/pi-better-subagents/shared-failure-observations.ts +70 -21
  38. package/node_modules/pi-better-subagents/shared-log-utils.ts +6 -12
  39. package/node_modules/pi-better-subagents/shared-navigator.ts +224 -125
  40. package/node_modules/pi-better-subagents/shared-sandbox-core.ts +194 -22
  41. package/node_modules/pi-better-subagents/shared-task-apply-patch.ts +409 -0
  42. package/node_modules/pi-better-subagents/shared-task-files.ts +69 -7
  43. package/node_modules/pi-better-subagents/shared-task-sandbox.ts +12 -0
  44. package/node_modules/pi-better-subagents/shared-task-tools.ts +111 -0
  45. package/node_modules/pi-better-subagents/subagent-tools.ts +95 -0
  46. package/node_modules/pi-better-subagents/task-guard.ts +42 -7
  47. package/node_modules/pi-better-subagents/task-policy.ts +29 -1
  48. package/package.json +4 -4
@@ -99,9 +99,11 @@ After `/new`, `/resume`, fork, or switching to another session, the previous
99
99
  session's tasks keep running but are paused from Pi's side: watches do not poll,
100
100
  remote tmux output is not collected, and `timeout_seconds` deadlines are not
101
101
  enforced until that session is active again. An overdue deadline is enforced as
102
- soon as the session resumes, so a timeout can land late but is never skipped. A
103
- local process that exits in the meantime is recorded as finished, and its
104
- callback is delivered when its session resumes.
102
+ soon as the session resumes, so a timeout can land late but is never skipped.
103
+ While the same Pi process stays open, a local process that exits in the meantime
104
+ is recorded as finished, and its callback is delivered when its session resumes.
105
+ If you quit Pi first, nothing records that exit: when the session is resumed in a
106
+ new Pi process, a task whose process is gone is marked lost.
105
107
 
106
108
  ## Watch conditions
107
109
 
@@ -122,6 +124,57 @@ Missing JSON fields or invalid JSON output remain retryable; task status shows
122
124
  the condition evaluation error until a subsequent poll recovers. Keep a finite
123
125
  timeout to bound watches whose output never becomes evaluable.
124
126
 
127
+ ### Writing a watch check
128
+
129
+ A check that swallows its own errors reports "still pending" forever. Keep
130
+ failures visible:
131
+
132
+ - Do not end the check with `exit 0` or `|| true`. A check that exits non-zero
133
+ is recorded as a failure and escalates to the parent session.
134
+ - Map an unknown or unparseable state to failure (a non-zero exit), not to
135
+ pending.
136
+ - Prefer structured output parsed with `jq -e` over hand-written format
137
+ strings. `jq -e` exits non-zero when the field is missing, so a broken query
138
+ shows up at once.
139
+
140
+ For example, a Cloud Run job execution:
141
+
142
+ ```sh
143
+ status=$(gcloud run jobs executions describe "$EXECUTION" --region="$REGION" --format=json \
144
+ | jq -er '.status.conditions[] | select(.type == "Completed") | .status') || exit 2
145
+ case "$status" in
146
+ True) echo TERMINAL_SUCCESS ;;
147
+ False) echo TERMINAL_FAILURE ;;
148
+ Unknown) echo STILL_RUNNING ;;
149
+ *) echo "unexpected Completed status: $status" >&2; exit 2 ;;
150
+ esac
151
+ ```
152
+
153
+ with `success_when: {type: "stdout_contains", value: "TERMINAL_SUCCESS"}` and
154
+ `failure_when: {type: "stdout_contains", value: "TERMINAL_FAILURE"}`.
155
+
156
+ `bg_task_watch` (and `bg_task` with `action: "watch"`) waits up to 15 seconds
157
+ for the first check and puts its exit code, the newest few lines of stderr and
158
+ of stdout in the tool result, so a broken check is visible at launch. When the
159
+ result is short on room, stdout is cut first. If the first check is still
160
+ running after 15 seconds, or you press Esc during the wait, the result says so
161
+ at once and the watch continues.
162
+
163
+ A running watch also guards against a blind check. When 3 checks in a row exit
164
+ 0, write to stderr, and match neither `success_when` nor `failure_when`, the
165
+ watch records one incident that needs action, with the latest stderr line, and
166
+ wakes the parent session once. The watch keeps running. A later check with
167
+ empty stderr recovers the incident whatever its exit code (a non-zero or
168
+ failed check is then recorded as its own incident), and so does a check that
169
+ matches a condition. A non-zero check that writes stderr restarts the count but
170
+ leaves the incident open. A clean pending check (exit 0, no stderr) never
171
+ counts.
172
+
173
+ Some tools write to stderr on success (`gcloud … list` prints "Listed 0
174
+ items.", and kubectl and npm print warnings), which can raise a false alarm.
175
+ If the stderr is expected, redirect it (`2>/dev/null`) or set
176
+ `blind_checks: 0`. Set `blind_checks` to another number to change the count.
177
+
125
178
  ## Install
126
179
 
127
180
  ```sh
@@ -151,7 +204,7 @@ evidence is reported as **observation incomplete**.
151
204
  Pass `operation_id` and `expected_exit_codes` on `bg_task_spawn`, `bg_task_watch`,
152
205
  or `bg_task` to declare intent before launch; a malformed declaration starts
153
206
  nothing. An exit code in `expected_exit_codes` (distinct integers 1-255, such as
154
- `[1]` for a no-match probe) is recorded as an **Expected failure** that needs no
207
+ `[1]` for a no-match probe; a `0` is ignored) is recorded as an **Expected failure** that needs no
155
208
  action; signals and timeouts never are. When a task with an `operation_id`
156
209
  succeeds, earlier failed tasks with the same `operation_id`, kind, cwd, SSH
157
210
  target, and owner (the same session id, or for sessionless tasks the same Pi
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pi-better-background-tasks",
3
- "version": "0.4.0",
3
+ "version": "0.6.0",
4
4
  "description": "Pi extension for durable background shell tasks, watchers, logs, and status inspection.",
5
5
  "license": "MIT",
6
6
  "type": "module",
@@ -1,7 +1,7 @@
1
1
  import { join } from "node:path";
2
2
  import type { ExtensionAPI } from "@earendil-works/pi-coding-agent";
3
3
  import {
4
- activeFailures, failureAttentionHandled, failureIdentity, formatFailureSummary, markFailureAttentionDelivered,
4
+ actionableFailures, activeFailures, failureAttentionHandled, failureIdentity, formatFailureSummary, markFailureAttentionDelivered,
5
5
  observeFailures, pendingAttentionNote, pendingAttentionRows, pendingFailureAttention, readCommandIntent, readFailureState,
6
6
  type FailureEvent, type FailureState,
7
7
  } from "./shared-failure-observations.js";
@@ -12,6 +12,14 @@ import type { BackgroundTaskMeta } from "./types.js";
12
12
 
13
13
  export const failurePath = (id: string): string => join(taskDir(id), "failures.jsonl");
14
14
  export const failureSummary = (id: string): string => formatFailureSummary(readFailureState(failurePath(id)));
15
+ /**
16
+ * The failure summary plus whether anything needs action. A navigator row leads with failure text
17
+ * only when something does; the quiet history line never displaces the command (#332).
18
+ */
19
+ export function failureView(id: string): { text: string; actionable: boolean } {
20
+ const state = readFailureState(failurePath(id));
21
+ return { text: formatFailureSummary(state), actionable: actionableFailures(state).length > 0 };
22
+ }
15
23
 
16
24
  export function recordFailure(meta: BackgroundTaskMeta, operation: string, summary: string, eventKey: unknown,
17
25
  options: { category?: string; expected?: boolean; incomplete?: boolean; evidence?: string; at?: number } = {}): void {
@@ -10,8 +10,8 @@ import {
10
10
  } from "./shared-navigator.ts";
11
11
  import { CustomEditor } from "@earendil-works/pi-coding-agent";
12
12
  import { Key, matchesKey, truncateToWidth } from "@earendil-works/pi-tui";
13
- import { activeFailures, failureLabel, readFailureState } from "./shared-failure-observations.js";
14
- import { failurePath, failureSummary } from "./failures.js";
13
+ import { actionableFailures, failureLabel, readFailureState } from "./shared-failure-observations.js";
14
+ import { failurePath, failureView } from "./failures.js";
15
15
  import { readLog } from "./logs.js";
16
16
  import { listMetasForOrigin, onMetaChanged, readMeta, writeMeta } from "./registry.js";
17
17
  import { stopTask } from "./runtime.js";
@@ -105,7 +105,9 @@ function isExpiredTerminalNavigatorRow(meta: BackgroundTaskMeta, now: number): b
105
105
  }
106
106
 
107
107
  function rowFromMeta(meta: BackgroundTaskMeta, now: number): BackgroundWorkRow {
108
- const failure = failureSummary(meta.id);
108
+ // Only a failure that needs action replaces the row's command; history stays in the detail view (#332).
109
+ const view = failureView(meta.id);
110
+ const failure = view.actionable ? view.text : "";
109
111
  const elapsed = formatDuration((meta.endedAt ?? now) - meta.startedAt);
110
112
  return {
111
113
  providerId: "background-tasks",
@@ -129,7 +131,8 @@ function rowFromMeta(meta: BackgroundTaskMeta, now: number): BackgroundWorkRow {
129
131
 
130
132
  function detailFromMeta(meta: BackgroundTaskMeta | undefined, now: number, options?: { logTailLines?: number }): BackgroundWorkDetail | null {
131
133
  if (!meta) return null;
132
- const failure = failureSummary(meta.id);
134
+ const view = failureView(meta.id);
135
+ const failure = view.text;
133
136
  const log = readLog(meta.logPath, options?.logTailLines ?? NAVIGATOR_DETAIL_ROWS);
134
137
  const command = commandLabel(meta);
135
138
  const metadata = [
@@ -164,7 +167,7 @@ function detailFromMeta(meta: BackgroundTaskMeta | undefined, now: number, optio
164
167
  title: meta.name || meta.id,
165
168
  status: meta.status,
166
169
  statusTone: toneForStatus(meta.status),
167
- subtitle: failure ? failure.split("\n")[0]! : compactCommandLabel(meta),
170
+ subtitle: view.actionable ? failure.split("\n")[0]! : compactCommandLabel(meta),
168
171
  metadata,
169
172
  foldedSections: [{
170
173
  id: "command",
@@ -217,7 +220,7 @@ function secondaryLabel(meta: BackgroundTaskMeta): string | undefined {
217
220
 
218
221
  function factsForMeta(meta: BackgroundTaskMeta, now: number): string[] {
219
222
  const facts: string[] = [];
220
- const incident = activeFailures(readFailureState(failurePath(meta.id)))[0];
223
+ const incident = actionableFailures(readFailureState(failurePath(meta.id)))[0];
221
224
  if (incident) facts.push(`${failureLabel(incident)}: ${incident.summary}`);
222
225
  if (meta.status === "running") {
223
226
  const stall = observeBackgroundTaskStall(meta, now);
@@ -40,7 +40,10 @@ import {
40
40
  import { failurePath } from "./failures.js";
41
41
  import { captureGapsFor, pageTaskLog, readLog, type LogRead } from "./logs.js";
42
42
  import { belongsToOrigin, inspectMeta, listTaskRecords, originOf, type MetaInspection } from "./registry.js";
43
- import type { BackgroundTaskCallbackOrigin, BackgroundTaskMeta, Condition } from "./types.js";
43
+ import type { BackgroundTaskCallbackOrigin, BackgroundTaskMeta, Condition, FirstWatchCheck } from "./types.js";
44
+
45
+ /** The launch tool stopped waiting while a watch's first check was still running (#359). */
46
+ export type FirstCheckPending = { pending: "timeout" | "aborted" | "suspended"; waitedMs: number };
44
47
 
45
48
  /**
46
49
  * Issue #312 consumer budgets. Defaults follow OUTPUT-POLICY / shared
@@ -240,9 +243,10 @@ export function formatCallbackFacts(meta: BackgroundTaskMeta): {
240
243
  meta.captureDiscardedBytes ? `capture overflow discarded ${meta.captureDiscardedBytes} bytes; not full history` : undefined,
241
244
  ].filter((line): line is string => Boolean(line));
242
245
  // History (expected and closed incidents) is not a row, but the callback still says it exists, so a
243
- // declared expected exit is not read as a plain failure.
246
+ // declared expected exit is not read as a plain failure. It leads the decision: a tight callback
247
+ // budget keeps a prefix of the decision, and this short line must survive it (#332).
244
248
  const history = rows.length ? historyTailLine(state) : quietFailureLine(state);
245
- const decision = [formatDecision(meta), ...gapLines, history].filter(Boolean).join("\n") || undefined;
249
+ const decision = [history, formatDecision(meta), ...gapLines].filter(Boolean).join("\n") || undefined;
246
250
  return {
247
251
  outcome: meta.status,
248
252
  ...(rows.length ? { failureRows: rows, incidentCount: rows.length } : {}),
@@ -434,24 +438,109 @@ function asInspection(inspection: MetaInspection | BackgroundTaskMeta | undefine
434
438
  return { id: meta.id, meta, found: true, readable: true };
435
439
  }
436
440
 
437
- export function formatLaunch(meta: BackgroundTaskMeta): string {
441
+ /**
442
+ * Launch result. For a watch, `firstCheck` reports its first check (#359): the check's result,
443
+ * or why the launch stopped waiting while it was still running.
444
+ *
445
+ * The first check gets whatever the status budget leaves after the rest of the launch text, so
446
+ * nothing else is clipped for it. The log path is kept whole or dropped whole: a clipped path
447
+ * looks valid but points nowhere.
448
+ */
449
+ export function formatLaunch(meta: BackgroundTaskMeta, firstCheck?: FirstWatchCheck | FirstCheckPending): string {
438
450
  const label = meta.name ? `${meta.name} (${meta.id})` : meta.id;
439
451
  const remoteLines = [
440
452
  ...(meta.ssh ? [`Remote: ${meta.ssh.target}${meta.remote?.session ? ` mode=${meta.remote.session}` : ""}${meta.remote?.sessionName ? ` session=${meta.remote.sessionName}` : ""}.`] : []),
441
453
  ...(meta.remote?.bootstrapMessage ? [`Remote setup: ${meta.remote.bootstrapMessage}`] : []),
442
454
  ...(meta.remote?.warning ? [`Warning: ${meta.remote.warning}`] : []),
443
455
  ];
444
- return assembleBackgroundContent({
456
+ const logLine = `Log: ${meta.logPath}`;
457
+ const build = (checkText: string | undefined, withLog: boolean) => assembleBackgroundContent({
445
458
  surface: "status",
446
459
  sections: {
447
460
  identity: `Started background ${meta.kind} ${label}. Status: ${meta.status}.`,
448
461
  failure: incidentSection(meta.id, {}),
449
462
  decision: formatDecision(meta),
450
- diagnostics: remoteLines.join("\n") || undefined,
451
- progress: `Log: ${meta.logPath}`,
463
+ diagnostics: [...remoteLines, ...(checkText ? [checkText] : [])].join("\n") || undefined,
464
+ progress: withLog ? logLine : undefined,
452
465
  },
453
466
  gaps: taskGaps(meta),
454
467
  });
468
+ let checkText: string | undefined;
469
+ if (firstCheck) {
470
+ const room = backgroundBudget("status") - utf8ByteLength(build(undefined, true)) - 1;
471
+ checkText = formatFirstWatchCheck(meta, firstCheck, room);
472
+ }
473
+ const text = build(checkText, true);
474
+ return text.includes(logLine) ? text : build(checkText, false);
475
+ }
476
+
477
+ const FIRST_CHECK_TAIL_LINES = 3;
478
+ const FIRST_CHECK_LINE_CHARS = 200;
479
+
480
+ function tailLines(text: string): string[] {
481
+ return text.split(/\r?\n/).map((line) => line.replace(/\s+/g, " ").trim()).filter(Boolean)
482
+ .slice(-FIRST_CHECK_TAIL_LINES)
483
+ .map((line) => line.length > FIRST_CHECK_LINE_CHARS ? `${line.slice(0, FIRST_CHECK_LINE_CHARS - 1)}…` : line);
484
+ }
485
+
486
+ function pendingFirstCheckText(check: FirstCheckPending): string {
487
+ switch (check.pending) {
488
+ case "aborted":
489
+ return "First check still running; stopped waiting because the tool call was cancelled. The watch continues in the background; check it later with bg_task_status.";
490
+ case "suspended":
491
+ return "First check still running when the session shut down. The watch continues when its session resumes; check it later with bg_task_status.";
492
+ default:
493
+ return `First check still running after ${formatDuration(check.waitedMs)}; the watch continues in the background. Check it later with bg_task_status.`;
494
+ }
495
+ }
496
+
497
+ /**
498
+ * The first check of a watch in at most `maxBytes` (#359). What matters most is kept first: the
499
+ * outcome, the exit-0-with-stderr warning, the newest stderr lines, then the newest stdout lines,
500
+ * so stdout is cut first. Lines are shown oldest to newest.
501
+ */
502
+ export function formatFirstWatchCheck(meta: BackgroundTaskMeta, check: FirstWatchCheck | FirstCheckPending, maxBytes = Number.MAX_SAFE_INTEGER): string {
503
+ if ("pending" in check) return pendingFirstCheckText(check);
504
+ if (check.error) return `First check could not run: ${oneLine(check.error, 300)}`;
505
+ const outcome = check.timedOut ? "timed out" : check.signal ? `signal ${check.signal}` : `exit ${check.exitCode ?? "unknown"}`;
506
+ const took = check.durationMs < 1000 ? `${check.durationMs}ms` : formatDuration(check.durationMs);
507
+ const header = `First check: ${outcome} in ${took}.`;
508
+ const stdout = tailLines(check.stdout);
509
+ const stderr = tailLines(check.stderr);
510
+ const warning = check.exitCode === 0 && stderr.length && meta.status === "running"
511
+ ? "The check exited 0 but wrote stderr: if it is broken, the watch cannot tell. Let errors exit non-zero."
512
+ : undefined;
513
+ const keptErr: string[] = [];
514
+ const keptOut: string[] = [];
515
+ let keptEmptyStdout = false;
516
+ const render = () => [
517
+ header,
518
+ ...(warning ? [warning] : []),
519
+ ...(keptErr.length ? [`stderr tail:\n${keptErr.map((line) => ` ${line}`).join("\n")}`] : []),
520
+ ...(keptOut.length ? [`stdout tail:\n${keptOut.map((line) => ` ${line}`).join("\n")}`] : []),
521
+ ...(keptEmptyStdout ? ["stdout: (empty)"] : []),
522
+ ].join("\n");
523
+ const fits = () => utf8ByteLength(render()) <= maxBytes;
524
+ // Newest first within each stream; each kept line goes in front to stay in order.
525
+ for (const line of [...stderr].reverse()) {
526
+ keptErr.unshift(line);
527
+ if (!fits()) { keptErr.shift(); break; }
528
+ }
529
+ for (const line of [...stdout].reverse()) {
530
+ keptOut.unshift(line);
531
+ if (!fits()) { keptOut.shift(); break; }
532
+ }
533
+ if (!stdout.length) {
534
+ keptEmptyStdout = true;
535
+ if (!fits()) keptEmptyStdout = false;
536
+ }
537
+ const omitted = (stderr.length - keptErr.length) + (stdout.length - keptOut.length);
538
+ if (omitted > 0) {
539
+ const note = `(${omitted} output line${omitted === 1 ? "" : "s"} omitted; see bg_task_log)`;
540
+ const text = `${render()}\n${note}`;
541
+ if (utf8ByteLength(text) <= maxBytes) return text;
542
+ }
543
+ return render();
455
544
  }
456
545
 
457
546
  function redactedVerbose(meta: BackgroundTaskMeta): unknown {
@@ -548,13 +637,16 @@ export function formatStatus(
548
637
  if (pageKind === "f") return formatLog(meta.id, { ...options, raw: true });
549
638
  if (pageKind === "t") return formatVerbose(meta, options);
550
639
  if (options.verbose) return formatVerbose(meta, options);
640
+ // A bg_task_list page cursor pages the list, not this task: say so rather than report it as a
641
+ // stale status cursor (#332).
642
+ const listCursor = pageKind === "l";
551
643
  const state = failureStateFor(meta.id);
552
644
  const resource = `status:${scopeKey(options)}:${meta.id}`;
553
645
  const revision = inspectStatusRevision({
554
646
  resource,
555
647
  contentRevision: contentRevision(meta),
556
648
  failureRevision: failureRevision(state),
557
- cursor: options.cursor,
649
+ cursor: listCursor ? undefined : options.cursor,
558
650
  });
559
651
  if (options.cursor && revision.change === "none") {
560
652
  return assembleBackgroundContent({
@@ -574,6 +666,7 @@ export function formatStatus(
574
666
  // Change/reset and read-gap facts are short and decision-relevant: they are
575
667
  // budgeted with the decision section, ahead of long incident rows.
576
668
  const changeFacts = [
669
+ ...(listCursor ? ["cursor ignored: it is a bg_task_list page cursor; pass it to bg_task_list, or pass this status's statusCursor here."] : []),
577
670
  ...(revision.change === "failure" ? ["change=failure"] : []),
578
671
  ...(revision.reset ? [`reset=${revision.reset}`] : []),
579
672
  ...(log.error ? [`log unreadable: ${oneLine(log.error, 200)}; cannot treat this as an empty healthy log.`] : []),