pi-goal-list-loop-audit 0.38.3 → 0.38.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -1,5 +1,18 @@
1
1
  # Changelog
2
2
 
3
+ ## 0.38.5 — delta-only goal continuation: marker-only steady-state (2026-09-03)
4
+
5
+ ### Fixed
6
+ Ongoing-conversation resend: steady-state goal turns now send the 45-char `[GOAL CHECKPOINT goalId=…]` marker only instead of the full ~23k continuation prompt every turn (history already holds T0 objective/contract/tasks + `complete_task` deltas). Post-compact sends `resync + marker` (~250 chars); full sends only on first-send per process or dynamic deltas (`repairTarget` / `autoResumedAt` / auditor TODOs / audit report / stale-approval mismatch / designer / full-audit). Marker still carries the dispatch marker so `before_agent_start` start-proof keeps matching; `agent_start`/`turn_start` fallback needs no prompt. Ledger `goal_continuation_sent` gains `kind` (`full`/`full+resync`/`resync`/`marker`) + `payloadChars`. Loop turns unchanged (per-iteration prompts vary). `extensions/goal-continuation.ts` (`buildMarkerContent` / `needsFullContinuation` / `buildContinuationContent`); `tests/delta-only-continuation.test.ts` pins the matrix.
7
+
8
+ ## 0.38.4 — human-input zombie stand-down (PR #36 slice) + AVO close (2026-09-02)
9
+
10
+ ### Fixed
11
+ Zombie watchdog no longer aborts while waiting on human input: `pause_goal`, `propose_goal_draft`, `propose_loop_draft`, `propose_loop_refine`, `propose_task_list`, `list_add`, `list_activate`, and `ask_user_question` now stand down the `BUSY + zero stream` abort exactly like `subagent` waits (field: drafting confirm / decision popup / `ask_user_question` looked like a hung provider, the bounded abort parked the dialog and the retry re-opened the same dialog). Own ledger `zombie_run_stood_down_user_input` vs `zombie_run_stood_down_subagent_wait`; genuinely hung streams still abort after the grace window. Selective port of PR #36 `USER_INPUT_WAIT_TOOL_NAMES` + `isUserInputWaitCall` with `heartbeatTick` carve-out; `tests/zombie-user-input-standdown.test.ts` pins the set and the branch.
12
+
13
+ ### Changed
14
+ AVO consideration closed: `audit/AVO-DEEP-DIVE-2026-09-02.md` remains the full read (`Vary(Pt)=Agent(Pt,K,f)`, §3.3 3-sentence supervisor). PR #22 (stagnation nudge, true AVO pattern but P1s: raw HEAD, no fencing, cycling-cap bypass) and PR #36 (commissar not AVO) dispositioned as `no merge as-is`; only the zombie stand-down ships in this release. PRs to be closed.
15
+
3
16
  ## 0.38.3 — truncate picker/settings titles to terminal width (PR #41) (2026-09-02)
4
17
 
5
18
  ### Fixed
package/docs/INDEX.md CHANGED
@@ -18,7 +18,7 @@ For shipped docs, the relevant entry points are:
18
18
  failback; v0.35.9 hardened cross-version npm tarball checks; v0.35.10
19
19
  handles multi-entry npm dry-run reports; v0.35.11 accepts both npm report
20
20
  shapes; v0.35.12 supports npm 12's keyed pack reports; v0.35.13 fixes stale-API recovery loops.
21
- v0.35.14–v0.38.3 continue through the supervisor freeze (`/glla pause`),
21
+ v0.35.14–v0.38.5 continue through the supervisor freeze (`/glla pause`),
22
22
  load hold, auditor picker parity, Windows launch fix, zombie-watchdog
23
23
  subagent carve-out, due-wait backstop, the `/glla agents` visibility panel,
24
24
  durable state-root selection, blank-until-resume auditor context, frozen
@@ -1054,6 +1054,52 @@ export function scheduleContinuation(ctx: ExtensionContext, force = false, delay
1054
1054
  continuationTimer = scheduleSessionTimeout(() => sendContinuation(goalId), delay);
1055
1055
  }
1056
1056
 
1057
+ export function buildMarkerContent(goalId: string): string {
1058
+ return `[GOAL CHECKPOINT goalId=${goalId}]`;
1059
+ }
1060
+
1061
+ /** v0.38.5 (delta-only): steady-state live turns need only the wake-up marker —
1062
+ * history already holds T0 objective/contract/tasks + complete_task deltas.
1063
+ * Full is required only for dynamic deltas the model has not seen: repair,
1064
+ * recovery, auditor TODOs/report, stale-approval mismatch, plus the rare
1065
+ * designer / full-audit paths (preserved verbatim in v1). Stable guidance
1066
+ * (vision, discipline) stays clean — it was in the first full send.
1067
+ */
1068
+ export function needsFullContinuation(goal: Goal): boolean {
1069
+ if (goal.repairTarget) return true;
1070
+ if (goal.autoResumedAt) return true;
1071
+ if (!auditorSurfaceSuppressed() && goal.pendingTasks && goal.pendingTasks.length > 0) return true;
1072
+ const lastAudit = goal.auditHistory?.[goal.auditHistory.length - 1];
1073
+ if (lastAudit && lastAudit.report && !auditorSurfaceSuppressed()) return true;
1074
+ if (
1075
+ !auditorSurfaceSuppressed() &&
1076
+ goal.status === "active" &&
1077
+ !goal.pendingCompletion &&
1078
+ lastAudit?.approved &&
1079
+ !lastAudit.impossible &&
1080
+ lastAudit.regressionShieldPassed !== false &&
1081
+ typeof lastAudit.revision === "number" &&
1082
+ lastAudit.revision !== (goal.revision ?? 0)
1083
+ ) return true;
1084
+ const next = findNextPendingTask(goal.taskList?.tasks ?? []);
1085
+ if (goal.agentRole === "designer" || next?.agentRole === "designer") return true;
1086
+ try {
1087
+ const settingsCwd = freshCtx?.()?.cwd ?? process.cwd();
1088
+ const eff = resolveEffectiveAggressiveSettings(loadSettings(settingsCwd));
1089
+ if (eff.aggressiveMode && isFullAuditObjective(goal.objective)) return true;
1090
+ } catch { /* settings read failure must not force full */ }
1091
+ return false;
1092
+ }
1093
+
1094
+ export function buildContinuationContent(goal: Goal, opts: { resync?: string; firstSend?: boolean } = {}): { content: string; kind: string } {
1095
+ const resync = opts.resync ?? "";
1096
+ const marker = buildMarkerContent(goal.id);
1097
+ const needsFull = (opts.firstSend ?? false) || needsFullContinuation(goal);
1098
+ if (needsFull) return { content: resync + continuationPrompt(goal), kind: resync ? "full+resync" : "full" };
1099
+ if (resync) return { content: resync + marker, kind: "resync" };
1100
+ return { content: marker, kind: "marker" };
1101
+ }
1102
+
1057
1103
  export function sendContinuation(goalId: string): void {
1058
1104
  // v0.35.15: `/glla pause` — a continuation timer armed BEFORE the pause
1059
1105
  // must not fire into the frozen window. scheduleContinuation already
@@ -1118,15 +1164,24 @@ export function sendContinuation(goalId: string): void {
1118
1164
  resync: Boolean(resync),
1119
1165
  });
1120
1166
  if (!attempt) return;
1167
+ // v0.38.5 (delta-only, ongoing conversation): clean live turns send the
1168
+ // 45-char marker only — history already holds objective/contract/tasks.
1169
+ // Resync (post-compact) sends resync+marker (~250 chars). Full 23k sends
1170
+ // only on first-send per process or dynamic deltas (repair/recovery/
1171
+ // audit). Marker still carries the dispatch marker so before_agent_start
1172
+ // start-proof matching keeps working; fallback agent_start/turn_start
1173
+ // needs no prompt at all.
1174
+ const goalForSend = state.goal!;
1175
+ const { content, kind } = buildContinuationContent(goalForSend, { resync, firstSend: lastContinuationSentAt === 0 });
1121
1176
  flags.extensionApi.sendMessage({
1122
1177
  customType: GOAL_EVENT_ENTRY,
1123
- content: resync + continuationPrompt(state.goal!),
1178
+ content,
1124
1179
  display: false,
1125
1180
  }, { triggerTurn: true, deliverAs: "followUp" });
1126
- lastContinuationSentPayload = { content: resync + continuationPrompt(state.goal!), display: false }; // v0.34.88: verbatim retry payload
1181
+ lastContinuationSentPayload = { content, display: false }; // v0.34.88: verbatim retry payload
1127
1182
  if (!dispatchAccepted(ctx, attempt)) return;
1128
1183
  continuationRearmStreak = 0; continuationRearmSince = 0; // v0.28.5 (E3): an accepted dispatch clears the storm
1129
- appendLedger(ctx.cwd, "goal_continuation_sent", { goalId, attemptId: attempt.id, generation: attempt.generation });
1184
+ appendLedger(ctx.cwd, "goal_continuation_sent", { goalId, attemptId: attempt.id, generation: attempt.generation, kind, payloadChars: content.length });
1130
1185
  // v0.35.37 (audit finding): the welcome-back recovery notice must fire
1131
1186
  // EXACTLY ONCE per auto-resume. The payload above was built with the
1132
1187
  // notice in it; once the dispatch is ACCEPTED the message has landed in
@@ -202,6 +202,32 @@ const SUBAGENT_WAIT_TOOL_NAMES: ReadonlySet<string> = new Set([
202
202
  const isSubagentWaitCall = (t: { name?: string }): boolean =>
203
203
  typeof t.name === "string" && SUBAGENT_WAIT_TOOL_NAMES.has(t.name);
204
204
 
205
+ // v0.36.1 (field: resume-after-long-decision-wait): a turn blocked INSIDE a
206
+ // tool handler that is waiting on HUMAN input is stream-silent by design —
207
+ // glla's own goal-toolkit verbs await Confirm/select pickers (drafting
208
+ // confirms, decision popups, task-list review), and the structured-question
209
+ // provider blocks in `ask_user_question` for as long as the user is away.
210
+ // The zombie watchdog read that silence as a hung provider stream, aborted
211
+ // the dialog mid-answer ("Operation aborted"), parked the goal, and its
212
+ // one bounded auto-retry re-opened the SAME dialog — so the second silence
213
+ // parked permanently. A human-input wait must stand the zombie branch down
214
+ // exactly like a subagent wait: detection+notify territory (wedge alert),
215
+ // never an abort.
216
+ const USER_INPUT_WAIT_TOOL_NAMES: ReadonlySet<string> = new Set([
217
+ // glla goal-toolkit verbs whose handlers can await user UI:
218
+ "pause_goal", // decision popup (maybeDecisionPopup)
219
+ "propose_goal_draft", // drafting confirm + interview
220
+ "propose_loop_draft",
221
+ "propose_loop_refine",
222
+ "propose_task_list",
223
+ "list_add", // batch drafting interview
224
+ "list_activate", // item activation → drafting
225
+ // structured-question providers (the companion we test against):
226
+ "ask_user_question",
227
+ ]);
228
+ export const isUserInputWaitCall = (t: { name?: string }): boolean =>
229
+ typeof t.name === "string" && USER_INPUT_WAIT_TOOL_NAMES.has(t.name);
230
+
205
231
  // ============================================================================
206
232
  // v0.35.28 (issue #16): due-wait backstop — the durable invariant that a
207
233
  // pauseKind "wait" actually resumes when its pauseResumeAt lapses.
@@ -1350,9 +1376,25 @@ function heartbeatTick(): void {
1350
1376
  // event has not reached the probe registry yet. Once a probe is already
1351
1377
  // stale, however, the same ambiguous tool name must not shield cleanup.
1352
1378
  const subagentWaitInFlight = healthySubagentWait || (inFlightSubagentTool && !staleSubagent);
1353
- if (subagentWaitInFlight) {
1379
+ // v0.37.0: a tool blocking on HUMAN input (decision popups, drafting
1380
+ // confirms, ask_user_question) is legitimately stream-silent for as long
1381
+ // as the user is away — never abort it (see USER_INPUT_WAIT_TOOL_NAMES).
1382
+ // This shield is independent of child-probe health: the parent turn is
1383
+ // waiting on its operator, not on a subagent.
1384
+ const userInputWaitInFlight = [...flags.inFlightToolCalls.values()].some(
1385
+ isUserInputWaitCall,
1386
+ );
1387
+ if (subagentWaitInFlight || userInputWaitInFlight) {
1354
1388
  if (streamSilentMs >= zombieAbortMs && !flags.abortedStandDown) {
1355
- appendLedger(ctx.cwd, "zombie_run_stood_down_subagent_wait", { streamSilentMs });
1389
+ appendLedger(
1390
+ ctx.cwd,
1391
+ subagentWaitInFlight
1392
+ ? "zombie_run_stood_down_subagent_wait"
1393
+ : "zombie_run_stood_down_user_input",
1394
+ {
1395
+ streamSilentMs,
1396
+ },
1397
+ );
1356
1398
  }
1357
1399
  return;
1358
1400
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pi-goal-list-loop-audit",
3
- "version": "0.38.3",
3
+ "version": "0.38.5",
4
4
  "description": "Mission control for autonomous pi: interview-drafted goals, an audited task queue, and forever-loops (metric, spec, project-audit) that run for hours. A detached extension-less auditor process re-verifies every completion with raw evidence without holding the main pi turn; confirmed drafts, decision pauses and consent gates keep you in charge.",
5
5
  "license": "AGPL-3.0-only",
6
6
  "author": "dracon",