pi-goal-list-loop-audit 0.38.53 → 0.38.55
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +75 -0
- package/INSTALL.md +9 -3
- package/README.md +9 -5
- package/docs/DESIGN-long-running-supervision.md +8 -4
- package/docs/DESIGN-subagent-visibility.md +2 -1
- package/docs/DESIGN.md +1 -1
- package/docs/INDEX.md +9 -7
- package/docs/SETTINGS.md +2 -0
- package/extensions/approval-render-store.ts +5 -2
- package/extensions/completion-summary.ts +165 -73
- package/extensions/goal-commands.ts +4 -1
- package/extensions/goal-loop-auditor-process.ts +36 -0
- package/extensions/goal-loop-core.ts +11 -3
- package/extensions/goal-loop-display.ts +13 -37
- package/extensions/goal-recovery.ts +42 -4
- package/extensions/goal-settings.ts +15 -4
- package/extensions/loops/goal-activation.ts +7 -1
- package/extensions/loops/goal-auditor-hooks.ts +69 -22
- package/extensions/loops/goal-list-queue.ts +3 -0
- package/extensions/loops/goal-runtime-globals.ts +2 -3
- package/extensions/loops/goal-settings-ui.ts +11 -0
- package/extensions/loops/goal-tools.ts +234 -11
- package/extensions/loops/goal-ui.ts +3 -1
- package/extensions/settings-menu.ts +9 -0
- package/package.json +1 -1
- package/prompts/goal-loop-continuation.md +4 -4
- package/prompts/goal-loop-draft.md +1 -1
- package/scripts/release-pack-smoke.mjs +31 -0
- package/skills/glla-delegate/SKILL.md +2 -0
package/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,80 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## 0.38.55 — Full-parity terminal card, lenient auditor, agent resume (2026-09-16)
|
|
4
|
+
|
|
5
|
+
### Full-parity terminal card
|
|
6
|
+
|
|
7
|
+
The post-objective summary now shares everything the archive knows (owner:
|
|
8
|
+
full parity with the long example closes): a `## Done — auditor …`
|
|
9
|
+
verdict banner opens the card, findings/values render uncapped (Next
|
|
10
|
+
keeps the one-concrete-action rule — DECIDED 2026-09-15), the
|
|
11
|
+
verification table always renders in full (the green PASS-line collapse is
|
|
12
|
+
retired), gate rows carry an optional repro `command` with its own Command
|
|
13
|
+
column, commit hashes stay visible, and a `### Final Repository State`
|
|
14
|
+
section (branch, HEAD, tree cleanliness, best-effort) closes the card
|
|
15
|
+
ahead of the record trailer. One surface per fact is preserved: the banner
|
|
16
|
+
owns the verdict, the footer stays liveness-only.
|
|
17
|
+
|
|
18
|
+
### Agent-side resume
|
|
19
|
+
|
|
20
|
+
Agent-side resume (field incident Screenshot_20260916_090307): a paused goal
|
|
21
|
+
whose blocker the user waives in conversation no longer bounces back to the
|
|
22
|
+
user as a pointless `/goal resume` round-trip. The new `resume_goal` tool is
|
|
23
|
+
the agent-side equivalent of `/goal resume` — same ownership/staleness
|
|
24
|
+
admission, but it schedules nothing (the live turn owns what happens next)
|
|
25
|
+
and refuses everything it must not touch: live loops, `/glla pause` freezes,
|
|
26
|
+
pending main-model recovery, stale sessions, and non-paused states name the
|
|
27
|
+
real user command. A cold-load hold releases like any explicit work command
|
|
28
|
+
(same as `/goal resume`); a stored completion claim re-fires through a new
|
|
29
|
+
`agent` audit origin with the manual-equivalent fresh cycle. The paused
|
|
30
|
+
`complete_goal` refusal now points at `resume_goal` first.
|
|
31
|
+
|
|
32
|
+
### Auditor retries like main (unified recovery envelope)
|
|
33
|
+
|
|
34
|
+
Field evidence (hellhunter/junk-runner 2026-09-16): a burned auditor
|
|
35
|
+
candidate chain parked with "automatic recovery is stopped" while the
|
|
36
|
+
main model would have kept probing for hours. The auditor is just
|
|
37
|
+
provider requests, so it now reuses the same retry/accounting machinery:
|
|
38
|
+
a burned chain clears its cursor and enters the shared durable-retry
|
|
39
|
+
ladder (eager 5s, then hourly :00:30 probes, 24h horizon; no
|
|
40
|
+
auditor-only attempt cap), a recognizably backend-side failure (5xx, network,
|
|
41
|
+
auth, or quota-wall wording) skips its backend's other rungs instead of
|
|
42
|
+
burning a launch each — while an ambiguous worker death still walks the
|
|
43
|
+
chain so the session fallback is tried — and parked auditor claims ride the
|
|
44
|
+
shared hourly ticker as a backstop. The card now reads "next:
|
|
45
|
+
auto-retry in …" while the ladder owns the wait; "automatic recovery
|
|
46
|
+
is stopped" remains only for a genuinely unrecoverable cursor-
|
|
47
|
+
persistence failure.
|
|
48
|
+
|
|
49
|
+
## 0.38.54 — Opt-in context-checkpoint projection (2026-09-14)
|
|
50
|
+
|
|
51
|
+
The per-turn `context`-hook projection rewrote history on every turn, busting
|
|
52
|
+
the provider prefix-cache (upstream issue #53). Projection is now opt-in:
|
|
53
|
+
the hook leaves the transcript append-only by default so the cache holds
|
|
54
|
+
across turns, while the fresh continuation prompt still carries live durable
|
|
55
|
+
state every turn.
|
|
56
|
+
|
|
57
|
+
### Added
|
|
58
|
+
|
|
59
|
+
- **`contextCheckpointProjection` setting (default off):** explicit opt-in
|
|
60
|
+
restores the legacy per-turn splice of a bounded continuation checkpoint.
|
|
61
|
+
Carried in `SETTINGS_KEYS` with junk-normalization to unset, a row in the
|
|
62
|
+
`/glla` settings Other tab with an on/off editor, and a `docs/SETTINGS.md`
|
|
63
|
+
entry (release-gate docs coverage holds).
|
|
64
|
+
|
|
65
|
+
### Fixed
|
|
66
|
+
|
|
67
|
+
- **Prompt-cache continuity:** with projection off (the default) the hook
|
|
68
|
+
returns no message rewrite and records no projection event, preserving
|
|
69
|
+
cache-prefix reuse turn over turn.
|
|
70
|
+
|
|
71
|
+
### Tests
|
|
72
|
+
|
|
73
|
+
- Both hook paths pinned (default-off passthrough + opt-in splice), junk
|
|
74
|
+
normalization, menu row off/default, editor round-trip, and docs coverage.
|
|
75
|
+
Full suite 2163 pass / 2 skip / 0 fail; `tsc` clean; `git diff --check`
|
|
76
|
+
clean.
|
|
77
|
+
|
|
3
78
|
## 0.38.53 — Consent-safe GLLA delegation skill and list-draft staging (2026-09-13)
|
|
4
79
|
|
|
5
80
|
Added the packaged `glla-delegate` skill for normal-chat goal/list delegation.
|
package/INSTALL.md
CHANGED
|
@@ -31,11 +31,11 @@ GLLA loads into new pi sessions. If pi is already open, reload that session:
|
|
|
31
31
|
|
|
32
32
|
## Updating
|
|
33
33
|
|
|
34
|
-
The status line always shows the running version (`glla: … ·
|
|
34
|
+
The status line always shows the running version (`glla: … · vX.Y.Z`).
|
|
35
35
|
When the npm registry is ahead, it also nudges:
|
|
36
36
|
|
|
37
37
|
```text
|
|
38
|
-
glla: … ·
|
|
38
|
+
glla: … · vX.Y.Z · update vA.B.C available
|
|
39
39
|
```
|
|
40
40
|
|
|
41
41
|
The nudge comes from a daily sidecar check (`.pi-glla/update-check.json`),
|
|
@@ -117,10 +117,16 @@ waiting for a decision.
|
|
|
117
117
|
```text
|
|
118
118
|
/list "refactor the cache. Done when: tests pass"
|
|
119
119
|
/list plan.md
|
|
120
|
-
/list
|
|
120
|
+
/list # interview + Confirm for a context draft
|
|
121
|
+
/list show # show active and waiting items
|
|
122
|
+
/list add <text...> # queue one item directly (no interview)
|
|
123
|
+
/list import <file> # import a file: one Confirm for the whole batch
|
|
121
124
|
/list start
|
|
122
125
|
/list next
|
|
123
126
|
/list resume
|
|
127
|
+
/list remove <n>
|
|
128
|
+
/list clear
|
|
129
|
+
/list cancel
|
|
124
130
|
|
|
125
131
|
/loop
|
|
126
132
|
/loop start # one clear recent target, metricless
|
package/README.md
CHANGED
|
@@ -166,18 +166,22 @@ quietly inventing an unbounded backlog.
|
|
|
166
166
|
/list remove <n>
|
|
167
167
|
/list clear
|
|
168
168
|
/list cancel # stop the active item and drop waiting items
|
|
169
|
+
/list add <text...> # queue one item directly (no interview)
|
|
170
|
+
/list import <file> # import a file: one Confirm for the whole batch
|
|
169
171
|
```
|
|
170
172
|
|
|
171
173
|
A pasted multi-line, bulleted, numbered, or checklist-style list keeps its
|
|
172
174
|
item wording and boundaries; GLLA does not ask for an “exact or refined” choice.
|
|
173
|
-
Only a genuinely ambiguous individual item needs clarification. The
|
|
175
|
+
Only a genuinely ambiguous individual item needs clarification. The direct
|
|
176
|
+
`/list add` path skips the interview but not consent: pasted text and files
|
|
177
|
+
still get one Confirm for the whole batch before anything is queued. The packaged
|
|
174
178
|
`skills/glla-delegate/SKILL.md` records the safe normal-chat delegation path:
|
|
175
179
|
explicit queue requests may use `list_add`, while discovered follow-ups are
|
|
176
180
|
offered first and durable goals remain Confirm-gated.
|
|
177
181
|
|
|
178
182
|
Order is the default, not the law. Automatic advance normally uses the head of
|
|
179
183
|
the queue, while `/list next <n>` or the agent's `list_activate` tool can choose
|
|
180
|
-
another item. Numbering always matches `/list` output. After a list item is
|
|
184
|
+
another item. Numbering always matches `/list show` output. After a list item is
|
|
181
185
|
approved and archived, the next queued item starts automatically; no manual
|
|
182
186
|
`/list next` is needed between items.
|
|
183
187
|
|
|
@@ -313,9 +317,9 @@ GLLA is the supervisor. These companions add capabilities around it:
|
|
|
313
317
|
count + `/glla agents`. If you see the same run stacked 3× or panels
|
|
314
318
|
swapping order, set pi-subagents `inlineToolDisplay: "summary"` (one
|
|
315
319
|
stable row per run) and pin `fleetViewPlacement`; GLLA's own richness is
|
|
316
|
-
the `subagentDisplayRichness` setting (`/glla` → Subagents): `
|
|
317
|
-
(default:
|
|
318
|
-
(
|
|
320
|
+
the `subagentDisplayRichness` setting (`/glla` → Subagents): `quiet`
|
|
321
|
+
(default: troubled workers + the count line; HUNG is never silent),
|
|
322
|
+
`compact` (count line), `rich` (all worker rows). Upstream triple-render report:
|
|
319
323
|
nicobailon/pi-subagents#1931.
|
|
320
324
|
|
|
321
325
|
Install (or keep pinned):
|
|
@@ -115,10 +115,14 @@ and the status/history surfaces. The human layer in chat, transcript, and the
|
|
|
115
115
|
archive's `## Terminal summary` renders the same facts as rich sections
|
|
116
116
|
(`## Done: <objective> — <outcome>`, duration line, Key Findings grouped by area
|
|
117
117
|
or tabulated at 4+ groups with per-finding `Test Results:` proof lines,
|
|
118
|
-
Verification Summary table — widened to Quality Gate |
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
118
|
+
Verification Summary table — widened to Quality Gate | Command | Scope |
|
|
119
|
+
Status | Notes when any gate row carries a repro command, else Quality
|
|
120
|
+
Gate | Scope | Status | Notes — with derived statuses when the agent
|
|
121
|
+
supplies a gate inventory — uncapped findings/values (Next keeps the
|
|
122
|
+
one-concrete-action rule), and a Final
|
|
123
|
+
Repository State section (branch, HEAD, tree) closing the card behind a
|
|
124
|
+
verdict banner that opens it (`## Done — auditor approved (N verdicts)`
|
|
125
|
+
and siblings). Width-bound and external surfaces (status line, widget card,
|
|
122
126
|
external notifies) use the compact single-line projection of all six labels.
|
|
123
127
|
Every terminal goal notification, including version-bearing already-shipped claims,
|
|
124
128
|
explicit goal/list cancellation, and `/glla wipe`, carries either the rich sections
|
|
@@ -96,7 +96,8 @@ silence ages re-laid out the editor every tick). v0.38.22 keeps the doctrine's
|
|
|
96
96
|
safety property by other means — `renderAgentsWidgetLines` buckets silence
|
|
97
97
|
ages exactly like the compact line, so the widget key only moves on genuine
|
|
98
98
|
state transitions — and restores rich ambient rows behind the
|
|
99
|
-
`subagentDisplayRichness` ladder (`
|
|
99
|
+
`subagentDisplayRichness` ladder (`quiet` canonical default since the
|
|
100
|
+
2026-09-07 flip — `rich` was the default before / `compact` / `quiet`, HUNG
|
|
100
101
|
never silent), plus — until v0.38.23 — the task-linkage header
|
|
101
102
|
(`→ <objective>`) only GLLA can show. v0.38.23 removes the header (the
|
|
102
103
|
card head already names the objective), collapses worker rows to one
|
package/docs/DESIGN.md
CHANGED
|
@@ -546,7 +546,7 @@ active → auditing (complete_goal called)
|
|
|
546
546
|
auditing → complete (auditor <approved/>)
|
|
547
547
|
auditing → active (auditor <disapproved/>; reset iteration counter)
|
|
548
548
|
active → paused (pause_goal called, or stuck > 5 min, or empty turn)
|
|
549
|
-
paused → active (user /goal resume)
|
|
549
|
+
paused → active (user /goal resume, or resume_goal when the user authorized continuation in conversation)
|
|
550
550
|
active → aborted (user /goal cancel)
|
|
551
551
|
```
|
|
552
552
|
|
package/docs/INDEX.md
CHANGED
|
@@ -16,7 +16,7 @@ Policy contracts and recent changes live in the `audit/` directory of the
|
|
|
16
16
|
failback; v0.35.9 hardened cross-version npm tarball checks; v0.35.10
|
|
17
17
|
handles multi-entry npm dry-run reports; v0.35.11 accepts both npm report
|
|
18
18
|
shapes; v0.35.12 supports npm 12's keyed pack reports; v0.35.13 fixes stale-API recovery loops.
|
|
19
|
-
v0.35.14–v0.38.
|
|
19
|
+
v0.35.14–v0.38.55 continue through the supervisor freeze (`/glla pause`),
|
|
20
20
|
load hold, auditor picker parity, Windows launch fix, zombie-watchdog
|
|
21
21
|
subagent carve-out, due-wait backstop, the `/glla agents` visibility panel,
|
|
22
22
|
durable state-root selection, blank-until-resume auditor context, frozen
|
|
@@ -25,14 +25,16 @@ Policy contracts and recent changes live in the `audit/` directory of the
|
|
|
25
25
|
extensions, bounded zero-stream retry containment, crash-safe persistence,
|
|
26
26
|
packed-artifact release verification, and the 2026-09-07 display/lifecycle/
|
|
27
27
|
settings audit pass (abort-latch send guards, ownership compare-and-swap,
|
|
28
|
-
auditor inherit/clear parity), the v0.38.
|
|
29
|
-
|
|
30
|
-
|
|
28
|
+
auditor inherit/clear parity), the v0.38.25 post-objective summary
|
|
29
|
+
(canonical approval render, persist + replay, audit-goal counts line), the v0.38.27 selective port of PRs #45/#46 (`/loop pause` soft-hold, widget subtask count), the v0.38.28 provenance-row closure fix, the v0.38.29 compact active-card recovery/judgment projection, and the v0.38.53 consent-safe `glla-delegate`
|
|
30
|
+
skill/list-drafting release, and the v0.38.54 opt-in context-checkpoint
|
|
31
|
+
projection (prompt-cache continuity, issue #53), and the v0.38.55 full-parity
|
|
32
|
+
terminal card, unified auditor-retry envelope, and agent-side resume_goal;
|
|
33
|
+
see CHANGELOG.md for the full trail.
|
|
31
34
|
- `../README.md`: what the plugin is, install, quickstart, and the
|
|
32
35
|
architectural guarantee (drafting + confirm + detached auditor).
|
|
33
|
-
- `../INSTALL.md`: source install / local development setup
|
|
34
|
-
companion plugins
|
|
35
|
-
note.
|
|
36
|
+
- `../INSTALL.md`: source install / local development setup and the
|
|
37
|
+
recommended companion plugins.
|
|
36
38
|
|
|
37
39
|
## Entry points
|
|
38
40
|
- `../README.md`: what the plugin is, install, quickstart
|
package/docs/SETTINGS.md
CHANGED
|
@@ -73,6 +73,7 @@ copies are ignored (the recovery runtime reads the global file):
|
|
|
73
73
|
| `subagentModelStrategy` | `"inherit-parent"` | Default subagent model policy for new sessions. |
|
|
74
74
|
| `subagentModelOverrides` | unset | Per-agent-type model pin; always wins over strategy. |
|
|
75
75
|
| `subagentFallbacks` | unset | Per-role fallback chains (first eligible ref wins). |
|
|
76
|
+
| `subagentDisplayRichness` | `"quiet"` | Ambient worker UI: `"quiet"` (default, troubled workers + count line) / `"compact"` / `"rich"`. |
|
|
76
77
|
| `aggressiveMode` | `true` | Keep-going defaults (`false` = pause-first policy). |
|
|
77
78
|
| `stuckMaxInterventions` | `5` (10 aggressive) | Consecutive stuck interventions before a loop stops. |
|
|
78
79
|
| `subagentHangEscalationMinutes` | `30` | Confirmed no-progress minutes before child-specific abort (`0` = warn only). |
|
|
@@ -82,4 +83,5 @@ copies are ignored (the recovery runtime reads the global file):
|
|
|
82
83
|
| `stallSimilarityThreshold` | `0.6` | Trigram similarity above this (tool-less) is a nudge. |
|
|
83
84
|
| `postaudit` | unset | Post-completion audit config (same shape as legacy `reviewer`). |
|
|
84
85
|
| `toolOverrides` | unset | Per-tool allow/hide/per-tool-config overrides. |
|
|
86
|
+
| `contextCheckpointProjection` | `false` (off) | Per-turn `context`-hook splice of a bounded continuation checkpoint. Off (default) leaves the transcript append-only so the provider prefix-cache holds; the fresh continuation prompt still carries live state. On restores the legacy projection (busts the cache). |
|
|
85
87
|
| `reviewer` | legacy | Deprecated alias for `postaudit`; migrated on load, `postaudit` wins. |
|
|
@@ -20,8 +20,11 @@ const MAX_STORED_RENDERS = 20;
|
|
|
20
20
|
const MAX_REPLAY_PER_CONTACT = 5;
|
|
21
21
|
// v0.38.30 audit: bound the sidecar behind the "each is ~1KB" comment — a
|
|
22
22
|
// long approval trailer used to grow the 20-entry file without bound.
|
|
23
|
-
|
|
24
|
-
|
|
23
|
+
// v0.38.55 audit: raised for the uncapped full-parity card (findings +
|
|
24
|
+
// full table + repo state routinely exceed 60 lines) — a stored render
|
|
25
|
+
// must replay verbatim, so the store bound stays above realistic cards.
|
|
26
|
+
const MAX_RENDER_CHAT_LINES = 150;
|
|
27
|
+
const MAX_RENDER_LINE_CHARS = 2000;
|
|
25
28
|
|
|
26
29
|
export function approvalRenderStorePath(cwd: string): string {
|
|
27
30
|
return path.join(piGlaDir(cwd), "pending-approval-renders.json");
|