pi-goal-list-loop-audit 0.38.3 → 0.38.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +13 -0
- package/docs/INDEX.md +1 -1
- package/extensions/goal-continuation.ts +58 -3
- package/extensions/goal-heartbeat.ts +44 -2
- package/package.json +1 -1
package/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,18 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## 0.38.5 — delta-only goal continuation: marker-only steady-state (2026-09-03)
|
|
4
|
+
|
|
5
|
+
### Fixed
|
|
6
|
+
Ongoing-conversation resend: steady-state goal turns now send the 45-char `[GOAL CHECKPOINT goalId=…]` marker only instead of the full ~23k continuation prompt every turn (history already holds T0 objective/contract/tasks + `complete_task` deltas). Post-compact sends `resync + marker` (~250 chars); full sends only on first-send per process or dynamic deltas (`repairTarget` / `autoResumedAt` / auditor TODOs / audit report / stale-approval mismatch / designer / full-audit). Marker still carries the dispatch marker so `before_agent_start` start-proof keeps matching; `agent_start`/`turn_start` fallback needs no prompt. Ledger `goal_continuation_sent` gains `kind` (`full`/`full+resync`/`resync`/`marker`) + `payloadChars`. Loop turns unchanged (per-iteration prompts vary). `extensions/goal-continuation.ts` (`buildMarkerContent` / `needsFullContinuation` / `buildContinuationContent`); `tests/delta-only-continuation.test.ts` pins the matrix.
|
|
7
|
+
|
|
8
|
+
## 0.38.4 — human-input zombie stand-down (PR #36 slice) + AVO close (2026-09-02)
|
|
9
|
+
|
|
10
|
+
### Fixed
|
|
11
|
+
Zombie watchdog no longer aborts while waiting on human input: `pause_goal`, `propose_goal_draft`, `propose_loop_draft`, `propose_loop_refine`, `propose_task_list`, `list_add`, `list_activate`, and `ask_user_question` now stand down the `BUSY + zero stream` abort exactly like `subagent` waits (field: drafting confirm / decision popup / `ask_user_question` looked like a hung provider, the bounded abort parked the dialog and the retry re-opened the same dialog). Own ledger `zombie_run_stood_down_user_input` vs `zombie_run_stood_down_subagent_wait`; genuinely hung streams still abort after the grace window. Selective port of PR #36 `USER_INPUT_WAIT_TOOL_NAMES` + `isUserInputWaitCall` with `heartbeatTick` carve-out; `tests/zombie-user-input-standdown.test.ts` pins the set and the branch.
|
|
12
|
+
|
|
13
|
+
### Changed
|
|
14
|
+
AVO consideration closed: `audit/AVO-DEEP-DIVE-2026-09-02.md` remains the full read (`Vary(Pt)=Agent(Pt,K,f)`, §3.3 3-sentence supervisor). PR #22 (stagnation nudge, true AVO pattern but P1s: raw HEAD, no fencing, cycling-cap bypass) and PR #36 (commissar not AVO) dispositioned as `no merge as-is`; only the zombie stand-down ships in this release. PRs to be closed.
|
|
15
|
+
|
|
3
16
|
## 0.38.3 — truncate picker/settings titles to terminal width (PR #41) (2026-09-02)
|
|
4
17
|
|
|
5
18
|
### Fixed
|
package/docs/INDEX.md
CHANGED
|
@@ -18,7 +18,7 @@ For shipped docs, the relevant entry points are:
|
|
|
18
18
|
failback; v0.35.9 hardened cross-version npm tarball checks; v0.35.10
|
|
19
19
|
handles multi-entry npm dry-run reports; v0.35.11 accepts both npm report
|
|
20
20
|
shapes; v0.35.12 supports npm 12's keyed pack reports; v0.35.13 fixes stale-API recovery loops.
|
|
21
|
-
v0.35.14–v0.38.
|
|
21
|
+
v0.35.14–v0.38.5 continue through the supervisor freeze (`/glla pause`),
|
|
22
22
|
load hold, auditor picker parity, Windows launch fix, zombie-watchdog
|
|
23
23
|
subagent carve-out, due-wait backstop, the `/glla agents` visibility panel,
|
|
24
24
|
durable state-root selection, blank-until-resume auditor context, frozen
|
|
@@ -1054,6 +1054,52 @@ export function scheduleContinuation(ctx: ExtensionContext, force = false, delay
|
|
|
1054
1054
|
continuationTimer = scheduleSessionTimeout(() => sendContinuation(goalId), delay);
|
|
1055
1055
|
}
|
|
1056
1056
|
|
|
1057
|
+
export function buildMarkerContent(goalId: string): string {
|
|
1058
|
+
return `[GOAL CHECKPOINT goalId=${goalId}]`;
|
|
1059
|
+
}
|
|
1060
|
+
|
|
1061
|
+
/** v0.38.5 (delta-only): steady-state live turns need only the wake-up marker —
|
|
1062
|
+
* history already holds T0 objective/contract/tasks + complete_task deltas.
|
|
1063
|
+
* Full is required only for dynamic deltas the model has not seen: repair,
|
|
1064
|
+
* recovery, auditor TODOs/report, stale-approval mismatch, plus the rare
|
|
1065
|
+
* designer / full-audit paths (preserved verbatim in v1). Stable guidance
|
|
1066
|
+
* (vision, discipline) stays clean — it was in the first full send.
|
|
1067
|
+
*/
|
|
1068
|
+
export function needsFullContinuation(goal: Goal): boolean {
|
|
1069
|
+
if (goal.repairTarget) return true;
|
|
1070
|
+
if (goal.autoResumedAt) return true;
|
|
1071
|
+
if (!auditorSurfaceSuppressed() && goal.pendingTasks && goal.pendingTasks.length > 0) return true;
|
|
1072
|
+
const lastAudit = goal.auditHistory?.[goal.auditHistory.length - 1];
|
|
1073
|
+
if (lastAudit && lastAudit.report && !auditorSurfaceSuppressed()) return true;
|
|
1074
|
+
if (
|
|
1075
|
+
!auditorSurfaceSuppressed() &&
|
|
1076
|
+
goal.status === "active" &&
|
|
1077
|
+
!goal.pendingCompletion &&
|
|
1078
|
+
lastAudit?.approved &&
|
|
1079
|
+
!lastAudit.impossible &&
|
|
1080
|
+
lastAudit.regressionShieldPassed !== false &&
|
|
1081
|
+
typeof lastAudit.revision === "number" &&
|
|
1082
|
+
lastAudit.revision !== (goal.revision ?? 0)
|
|
1083
|
+
) return true;
|
|
1084
|
+
const next = findNextPendingTask(goal.taskList?.tasks ?? []);
|
|
1085
|
+
if (goal.agentRole === "designer" || next?.agentRole === "designer") return true;
|
|
1086
|
+
try {
|
|
1087
|
+
const settingsCwd = freshCtx?.()?.cwd ?? process.cwd();
|
|
1088
|
+
const eff = resolveEffectiveAggressiveSettings(loadSettings(settingsCwd));
|
|
1089
|
+
if (eff.aggressiveMode && isFullAuditObjective(goal.objective)) return true;
|
|
1090
|
+
} catch { /* settings read failure must not force full */ }
|
|
1091
|
+
return false;
|
|
1092
|
+
}
|
|
1093
|
+
|
|
1094
|
+
export function buildContinuationContent(goal: Goal, opts: { resync?: string; firstSend?: boolean } = {}): { content: string; kind: string } {
|
|
1095
|
+
const resync = opts.resync ?? "";
|
|
1096
|
+
const marker = buildMarkerContent(goal.id);
|
|
1097
|
+
const needsFull = (opts.firstSend ?? false) || needsFullContinuation(goal);
|
|
1098
|
+
if (needsFull) return { content: resync + continuationPrompt(goal), kind: resync ? "full+resync" : "full" };
|
|
1099
|
+
if (resync) return { content: resync + marker, kind: "resync" };
|
|
1100
|
+
return { content: marker, kind: "marker" };
|
|
1101
|
+
}
|
|
1102
|
+
|
|
1057
1103
|
export function sendContinuation(goalId: string): void {
|
|
1058
1104
|
// v0.35.15: `/glla pause` — a continuation timer armed BEFORE the pause
|
|
1059
1105
|
// must not fire into the frozen window. scheduleContinuation already
|
|
@@ -1118,15 +1164,24 @@ export function sendContinuation(goalId: string): void {
|
|
|
1118
1164
|
resync: Boolean(resync),
|
|
1119
1165
|
});
|
|
1120
1166
|
if (!attempt) return;
|
|
1167
|
+
// v0.38.5 (delta-only, ongoing conversation): clean live turns send the
|
|
1168
|
+
// 45-char marker only — history already holds objective/contract/tasks.
|
|
1169
|
+
// Resync (post-compact) sends resync+marker (~250 chars). Full 23k sends
|
|
1170
|
+
// only on first-send per process or dynamic deltas (repair/recovery/
|
|
1171
|
+
// audit). Marker still carries the dispatch marker so before_agent_start
|
|
1172
|
+
// start-proof matching keeps working; fallback agent_start/turn_start
|
|
1173
|
+
// needs no prompt at all.
|
|
1174
|
+
const goalForSend = state.goal!;
|
|
1175
|
+
const { content, kind } = buildContinuationContent(goalForSend, { resync, firstSend: lastContinuationSentAt === 0 });
|
|
1121
1176
|
flags.extensionApi.sendMessage({
|
|
1122
1177
|
customType: GOAL_EVENT_ENTRY,
|
|
1123
|
-
content
|
|
1178
|
+
content,
|
|
1124
1179
|
display: false,
|
|
1125
1180
|
}, { triggerTurn: true, deliverAs: "followUp" });
|
|
1126
|
-
lastContinuationSentPayload = { content
|
|
1181
|
+
lastContinuationSentPayload = { content, display: false }; // v0.34.88: verbatim retry payload
|
|
1127
1182
|
if (!dispatchAccepted(ctx, attempt)) return;
|
|
1128
1183
|
continuationRearmStreak = 0; continuationRearmSince = 0; // v0.28.5 (E3): an accepted dispatch clears the storm
|
|
1129
|
-
appendLedger(ctx.cwd, "goal_continuation_sent", { goalId, attemptId: attempt.id, generation: attempt.generation });
|
|
1184
|
+
appendLedger(ctx.cwd, "goal_continuation_sent", { goalId, attemptId: attempt.id, generation: attempt.generation, kind, payloadChars: content.length });
|
|
1130
1185
|
// v0.35.37 (audit finding): the welcome-back recovery notice must fire
|
|
1131
1186
|
// EXACTLY ONCE per auto-resume. The payload above was built with the
|
|
1132
1187
|
// notice in it; once the dispatch is ACCEPTED the message has landed in
|
|
@@ -202,6 +202,32 @@ const SUBAGENT_WAIT_TOOL_NAMES: ReadonlySet<string> = new Set([
|
|
|
202
202
|
const isSubagentWaitCall = (t: { name?: string }): boolean =>
|
|
203
203
|
typeof t.name === "string" && SUBAGENT_WAIT_TOOL_NAMES.has(t.name);
|
|
204
204
|
|
|
205
|
+
// v0.36.1 (field: resume-after-long-decision-wait): a turn blocked INSIDE a
|
|
206
|
+
// tool handler that is waiting on HUMAN input is stream-silent by design —
|
|
207
|
+
// glla's own goal-toolkit verbs await Confirm/select pickers (drafting
|
|
208
|
+
// confirms, decision popups, task-list review), and the structured-question
|
|
209
|
+
// provider blocks in `ask_user_question` for as long as the user is away.
|
|
210
|
+
// The zombie watchdog read that silence as a hung provider stream, aborted
|
|
211
|
+
// the dialog mid-answer ("Operation aborted"), parked the goal, and its
|
|
212
|
+
// one bounded auto-retry re-opened the SAME dialog — so the second silence
|
|
213
|
+
// parked permanently. A human-input wait must stand the zombie branch down
|
|
214
|
+
// exactly like a subagent wait: detection+notify territory (wedge alert),
|
|
215
|
+
// never an abort.
|
|
216
|
+
const USER_INPUT_WAIT_TOOL_NAMES: ReadonlySet<string> = new Set([
|
|
217
|
+
// glla goal-toolkit verbs whose handlers can await user UI:
|
|
218
|
+
"pause_goal", // decision popup (maybeDecisionPopup)
|
|
219
|
+
"propose_goal_draft", // drafting confirm + interview
|
|
220
|
+
"propose_loop_draft",
|
|
221
|
+
"propose_loop_refine",
|
|
222
|
+
"propose_task_list",
|
|
223
|
+
"list_add", // batch drafting interview
|
|
224
|
+
"list_activate", // item activation → drafting
|
|
225
|
+
// structured-question providers (the companion we test against):
|
|
226
|
+
"ask_user_question",
|
|
227
|
+
]);
|
|
228
|
+
export const isUserInputWaitCall = (t: { name?: string }): boolean =>
|
|
229
|
+
typeof t.name === "string" && USER_INPUT_WAIT_TOOL_NAMES.has(t.name);
|
|
230
|
+
|
|
205
231
|
// ============================================================================
|
|
206
232
|
// v0.35.28 (issue #16): due-wait backstop — the durable invariant that a
|
|
207
233
|
// pauseKind "wait" actually resumes when its pauseResumeAt lapses.
|
|
@@ -1350,9 +1376,25 @@ function heartbeatTick(): void {
|
|
|
1350
1376
|
// event has not reached the probe registry yet. Once a probe is already
|
|
1351
1377
|
// stale, however, the same ambiguous tool name must not shield cleanup.
|
|
1352
1378
|
const subagentWaitInFlight = healthySubagentWait || (inFlightSubagentTool && !staleSubagent);
|
|
1353
|
-
|
|
1379
|
+
// v0.37.0: a tool blocking on HUMAN input (decision popups, drafting
|
|
1380
|
+
// confirms, ask_user_question) is legitimately stream-silent for as long
|
|
1381
|
+
// as the user is away — never abort it (see USER_INPUT_WAIT_TOOL_NAMES).
|
|
1382
|
+
// This shield is independent of child-probe health: the parent turn is
|
|
1383
|
+
// waiting on its operator, not on a subagent.
|
|
1384
|
+
const userInputWaitInFlight = [...flags.inFlightToolCalls.values()].some(
|
|
1385
|
+
isUserInputWaitCall,
|
|
1386
|
+
);
|
|
1387
|
+
if (subagentWaitInFlight || userInputWaitInFlight) {
|
|
1354
1388
|
if (streamSilentMs >= zombieAbortMs && !flags.abortedStandDown) {
|
|
1355
|
-
appendLedger(
|
|
1389
|
+
appendLedger(
|
|
1390
|
+
ctx.cwd,
|
|
1391
|
+
subagentWaitInFlight
|
|
1392
|
+
? "zombie_run_stood_down_subagent_wait"
|
|
1393
|
+
: "zombie_run_stood_down_user_input",
|
|
1394
|
+
{
|
|
1395
|
+
streamSilentMs,
|
|
1396
|
+
},
|
|
1397
|
+
);
|
|
1356
1398
|
}
|
|
1357
1399
|
return;
|
|
1358
1400
|
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pi-goal-list-loop-audit",
|
|
3
|
-
"version": "0.38.
|
|
3
|
+
"version": "0.38.5",
|
|
4
4
|
"description": "Mission control for autonomous pi: interview-drafted goals, an audited task queue, and forever-loops (metric, spec, project-audit) that run for hours. A detached extension-less auditor process re-verifies every completion with raw evidence without holding the main pi turn; confirmed drafts, decision pauses and consent gates keep you in charge.",
|
|
5
5
|
"license": "AGPL-3.0-only",
|
|
6
6
|
"author": "dracon",
|