pi-goal-list-loop-audit 0.38.68 → 0.38.69
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +23 -0
- package/README.md +7 -7
- package/docs/INDEX.md +1 -1
- package/docs/SETTINGS.md +1 -0
- package/extensions/completion-summary.ts +91 -44
- package/extensions/goal-continuation.ts +44 -3
- package/extensions/goal-loop-core.ts +30 -0
- package/extensions/goal-recovery.ts +87 -16
- package/extensions/goal-settings.ts +10 -0
- package/extensions/loops/goal-auditor-hooks.ts +7 -9
- package/extensions/loops/goal-tools.ts +71 -12
- package/extensions/main-model-recovery.ts +44 -5
- package/package.json +2 -2
- package/schemas/goal.schema.json +14 -0
- package/scripts/release-pack-smoke.mjs +56 -19
package/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,28 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## 0.38.69 — About and README refresh (2026-09-20)
|
|
4
|
+
|
|
5
|
+
- Refined npm about line and README intro: tighter mission control wording, no em dashes.
|
|
6
|
+
- README range wording uses plain "to" (0 to 10, 1 to 10).
|
|
7
|
+
|
|
8
|
+
### Antigravity auto-continue port (survey 2026-09-19)
|
|
9
|
+
|
|
10
|
+
- Mid-run interruption budget (`decisionPauseBudget`): decision pauses past
|
|
11
|
+
budget auto-default to the recommended option and ledger
|
|
12
|
+
`decision_budget_auto_default` instead of stalling mid-run.
|
|
13
|
+
- Quota sleep-until-reset: quota-class failures with an explicit upstream
|
|
14
|
+
hint sleep until reset (5h cap; 5s eager first retry preserved); quota
|
|
15
|
+
waits never park at the 24h horizon; background primary probe re-arms at
|
|
16
|
+
reset while fallback serves.
|
|
17
|
+
- Completion-claim evidence-density lint: zero path:line tokens AND zero
|
|
18
|
+
gate rows rides a ledgered NOTE annotation (claim still audits); every
|
|
19
|
+
terminal path already ends with the archive record pointer.
|
|
20
|
+
- Per-repo pitfall registry: `.pi-glla/pitfalls.md` (absent stays absent)
|
|
21
|
+
rides the continuation prompt under REPO PITFALLS, ledgered once per goal.
|
|
22
|
+
- Objection-attached retries: every disapproval extracts durable TODOs via
|
|
23
|
+
`durableObjectionsForDisapproval` (aggressive keeps only cap/no-progress
|
|
24
|
+
behaviors); two pre-existing source-shape pins revised to the new shape.
|
|
25
|
+
|
|
3
26
|
## 0.38.68 — Relentless auto-continue: heat-routed exhaustion, quota retries, truthful recovering (2026-09-19)
|
|
4
27
|
|
|
5
28
|
### Length-exhaustion wedge fixed at the root: heat, not budget (field 162348)
|
package/README.md
CHANGED
|
@@ -4,20 +4,20 @@
|
|
|
4
4
|
<img src="media/glla2.png" alt="GLLA mission control" width="960">
|
|
5
5
|
</p>
|
|
6
6
|
|
|
7
|
-
> **Long
|
|
7
|
+
> **Long running, high leverage autonomy for pi.**
|
|
8
8
|
>
|
|
9
9
|
> Give pi a meaningful outcome. GLLA helps it research, plan, execute,
|
|
10
10
|
> recover, and prove the result over hours or days instead of treating one
|
|
11
11
|
> chat turn as the whole job.
|
|
12
12
|
|
|
13
13
|
`pi-goal-list-loop-audit` (GLLA) is mission control for autonomous work in
|
|
14
|
-
[pi](https://github.com/badlogic/pi-mono). It
|
|
14
|
+
[pi](https://github.com/badlogic/pi-mono). It fits work that is too broad,
|
|
15
15
|
too long, or too important to leave to a single uninterrupted prompt:
|
|
16
16
|
repo-wide changes, migrations, audits, research, documentation overhauls,
|
|
17
17
|
large refactors, and continuous improvement.
|
|
18
18
|
|
|
19
|
-
GLLA does not promise that an agent
|
|
20
|
-
agent's work **more effective, durable, recoverable, and
|
|
19
|
+
GLLA does not promise that an agent never makes a mistake. It makes the
|
|
20
|
+
agent's work **more effective, durable, recoverable, and hard to declare
|
|
21
21
|
finished without evidence**:
|
|
22
22
|
|
|
23
23
|
- You state the outcome and what “done” means.
|
|
@@ -30,7 +30,7 @@ finished without evidence**:
|
|
|
30
30
|
- A separate detached auditor checks the saved completion claim before GLLA
|
|
31
31
|
accepts it.
|
|
32
32
|
|
|
33
|
-
The aim is not
|
|
33
|
+
The aim is not "run forever." The aim is **more useful work per unit of
|
|
34
34
|
attention, with event-driven progress instead of guessed-duration waiting, and
|
|
35
35
|
better evidence at the end**. See `docs/DESIGN-long-running-supervision.md` for
|
|
36
36
|
the long-running policy.
|
|
@@ -398,7 +398,7 @@ proof of a quota or billing state.
|
|
|
398
398
|
|
|
399
399
|
- automatic retries are bounded and visible; a BUSY/no-stream turn is parked
|
|
400
400
|
and re-dispatched within the configurable **Zero-stream retries** budget
|
|
401
|
-
(default 3, range 0
|
|
401
|
+
(default 3, range 0 to 10), then requires explicit resume;
|
|
402
402
|
- `/goal resume`, `/list resume`, and `/loop resume` are explicit recovery
|
|
403
403
|
paths;
|
|
404
404
|
- a user abort means stop, not “try again behind my back”;
|
|
@@ -431,7 +431,7 @@ Open `/glla` for the settings table. The most important choices are:
|
|
|
431
431
|
- **Subagent hang escalation:** warning-only at `0`, or one child-specific
|
|
432
432
|
action after a confirmed frozen interval;
|
|
433
433
|
- **Zero-stream retries:** automatic GLLA recovery attempts after a busy,
|
|
434
|
-
stream-silent Pi turn; `0` keeps recovery manual and `1
|
|
434
|
+
stream-silent Pi turn; `0` keeps recovery manual and `1 to 10` bounds repeats;
|
|
435
435
|
- **Audit cap and retry cadence:** bounds for repeated objections and
|
|
436
436
|
infrastructure recovery.
|
|
437
437
|
|
package/docs/INDEX.md
CHANGED
|
@@ -16,7 +16,7 @@ Policy contracts and recent changes live in the `audit/` directory of the
|
|
|
16
16
|
failback; v0.35.9 hardened cross-version npm tarball checks; v0.35.10
|
|
17
17
|
handles multi-entry npm dry-run reports; v0.35.11 accepts both npm report
|
|
18
18
|
shapes; v0.35.12 supports npm 12's keyed pack reports; v0.35.13 fixes stale-API recovery loops.
|
|
19
|
-
v0.35.14–v0.38.
|
|
19
|
+
v0.35.14–v0.38.69 continue through the supervisor freeze (`/glla pause`),
|
|
20
20
|
load hold, auditor picker parity, Windows launch fix, zombie-watchdog
|
|
21
21
|
subagent carve-out, due-wait backstop, the `/glla agents` visibility panel,
|
|
22
22
|
durable state-root selection, blank-until-resume auditor context, frozen
|
package/docs/SETTINGS.md
CHANGED
|
@@ -64,6 +64,7 @@ copies are ignored (the recovery runtime reads the global file):
|
|
|
64
64
|
| `wedgeAlertMinutes` | unset (30, or off while aggressive mode is on — the default) | Busy-but-silent minutes before the wedge alert; `0` = off. The menu shows the effective value. |
|
|
65
65
|
| `autoResume` | `false` | Restored goals/loops/lists auto-resume in fresh sessions. Global-only. |
|
|
66
66
|
| `decisionPopup` | `true` | Decision pauses pop the picker (`false` = widget card only). |
|
|
67
|
+
| `decisionPauseBudget` | unset (unlimited) | Max agent-authored decision pauses per goal before auto-default: the (N+1)-th adopts the recommended option as a logged assumption without pausing. `0` = relentless from the first decision. |
|
|
67
68
|
| `carryover` | `"pause"` | Stale carryover on new activation: `"pause"` / `"clear"` / `"resume"`. |
|
|
68
69
|
| `autoAcceptDrafts` | `false` | Drafts activate without the Confirm dialog (unattended rigs). |
|
|
69
70
|
| `auditCap` | `5` (10 aggressive) | Pause after N consecutive auditor disapprovals (`0` = unlimited). |
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { execFileSync } from "node:child_process";
|
|
2
|
-
import type
|
|
2
|
+
import { sanitizeDisplayText, type FindingGroup, type GateRow, type Goal, type Status } from "./goal-loop-core.js";
|
|
3
3
|
import { fmtElapsed, truncateCells } from "./goal-loop-display.js";
|
|
4
4
|
|
|
5
5
|
/**
|
|
@@ -180,9 +180,10 @@ export interface HumanCompletionBrief {
|
|
|
180
180
|
* the durable archive keeps the full text. Falls back to the original
|
|
181
181
|
* when stripping would empty the value. */
|
|
182
182
|
export function chatSafeDetailValue(value: string): string {
|
|
183
|
+
const safeValue = sanitizeDisplayText(value);
|
|
183
184
|
// v0.38.45 audit: loop the innermost-group strip to a fixpoint
|
|
184
185
|
// (bounded) — a single pass left nested husks like "(log )" behind.
|
|
185
|
-
let withoutGroups = stripMachineGroups(
|
|
186
|
+
let withoutGroups = stripMachineGroups(safeValue);
|
|
186
187
|
const stripped = withoutGroups
|
|
187
188
|
.replace(/(?:\/var)?\/tmp\/\S+/g, "")
|
|
188
189
|
.replace(/\btarballs?\s+\S+\.tgz\b/gi, "")
|
|
@@ -190,7 +191,7 @@ export function chatSafeDetailValue(value: string): string {
|
|
|
190
191
|
.replace(/\(\s*\)/g, "")
|
|
191
192
|
.replace(/\s{2,}/g, " ")
|
|
192
193
|
.trim();
|
|
193
|
-
return stripped ||
|
|
194
|
+
return stripped || safeValue;
|
|
194
195
|
}
|
|
195
196
|
|
|
196
197
|
/** The human briefing: outcome first in its own words, then only the
|
|
@@ -384,7 +385,7 @@ function stripMachineTokens(line: string): string {
|
|
|
384
385
|
export function structuredSummaryLines(summary: string | undefined, label = "Outcome:"): string[] | null {
|
|
385
386
|
const raw = rawLabelValue(summary ?? "", label);
|
|
386
387
|
if (!raw || !isSectionStructured(raw)) return null;
|
|
387
|
-
const lines = raw.split("\n").map(stripMachineTokens);
|
|
388
|
+
const lines = raw.split("\n").map((line) => stripMachineTokens(sanitizeDisplayText(line)));
|
|
388
389
|
while (lines.length > 0 && !(lines[0] ?? "").trim()) lines.shift();
|
|
389
390
|
while (lines.length > 0 && !(lines[lines.length - 1] ?? "").trim()) lines.pop();
|
|
390
391
|
let text = lines.join("\n");
|
|
@@ -416,7 +417,7 @@ export interface RichTerminalParts {
|
|
|
416
417
|
}
|
|
417
418
|
|
|
418
419
|
function escapeTableCell(value: string): string {
|
|
419
|
-
return value.replace(/\|/g, "\\|").replace(/\s+/g, " ").trim();
|
|
420
|
+
return sanitizeDisplayText(value).replace(/\|/g, "\\|").replace(/\s+/g, " ").trim();
|
|
420
421
|
}
|
|
421
422
|
|
|
422
423
|
function testsRowStatus(value: string): string {
|
|
@@ -470,9 +471,10 @@ function auditRowStatus(history: Goal["auditHistory"]): string {
|
|
|
470
471
|
|
|
471
472
|
/** Split a stale-filtered `Label: value` detail into a bold lead + body. */
|
|
472
473
|
function leadBody(detail: string): { lead: string; body: string } {
|
|
473
|
-
const
|
|
474
|
-
|
|
475
|
-
return { lead:
|
|
474
|
+
const safeDetail = sanitizeDisplayText(detail);
|
|
475
|
+
const separator = safeDetail.indexOf(":");
|
|
476
|
+
if (separator < 0) return { lead: "Note", body: safeDetail };
|
|
477
|
+
return { lead: safeDetail.slice(0, separator).trim() || "Note", body: safeDetail.slice(separator + 1).trim() };
|
|
476
478
|
}
|
|
477
479
|
|
|
478
480
|
/**
|
|
@@ -487,7 +489,7 @@ const EVIDENCE_TOKEN_PATTERN = /(?<![/~+\w])[\w.+][\w.+/-]*\.[A-Za-z0-9]{1,8}:\d
|
|
|
487
489
|
|
|
488
490
|
export function extractEvidenceTokens(text: string): { text: string; evidence: string[] } {
|
|
489
491
|
const evidence: string[] = [];
|
|
490
|
-
const stripped = text
|
|
492
|
+
const stripped = sanitizeDisplayText(text)
|
|
491
493
|
.replace(EVIDENCE_TOKEN_PATTERN, (match) => {
|
|
492
494
|
if (evidence.length < RICH_EVIDENCE_TOKENS_PER_FINDING && !evidence.includes(match)) evidence.push(match);
|
|
493
495
|
// The token often rides in parentheses — move the wrapper too, so
|
|
@@ -501,6 +503,29 @@ export function extractEvidenceTokens(text: string): { text: string; evidence: s
|
|
|
501
503
|
return { text: stripped, evidence };
|
|
502
504
|
}
|
|
503
505
|
|
|
506
|
+
/**
|
|
507
|
+
* v0.38.69 (Antigravity port): completion-claim evidence-density lint.
|
|
508
|
+
* A walkthrough always ships areas + counts; GLLA richness rode optional
|
|
509
|
+
* agent params, so a flat six-label claim with zero path:line tokens and
|
|
510
|
+
* zero gate rows still passed. Returns a NOTE annotation when the claim
|
|
511
|
+
* carries no verifiable pointer at all (no evidence tokens across the
|
|
512
|
+
* summary and every group finding, and no gate rows) — the claim still
|
|
513
|
+
* audits, but the terminal render falls back to recorded facts. Tokens in
|
|
514
|
+
* finding groups count; gate rows count as density without tokens.
|
|
515
|
+
*/
|
|
516
|
+
export function completionSummaryDensityNote(
|
|
517
|
+
summary: string | undefined,
|
|
518
|
+
groups?: FindingGroup[],
|
|
519
|
+
gates?: GateRow[],
|
|
520
|
+
): string | undefined {
|
|
521
|
+
if (!summary?.trim()) return undefined;
|
|
522
|
+
const pool = [summary, ...(groups ?? []).flatMap((group) => group.findings ?? [])].join("\n");
|
|
523
|
+
const { evidence } = extractEvidenceTokens(pool);
|
|
524
|
+
if (evidence.length > 0) return undefined;
|
|
525
|
+
if ((gates ?? []).length > 0) return undefined;
|
|
526
|
+
return "low evidence density: no path:line evidence tokens and no verification gate rows — add file:line pointers (e.g. extensions/goal-recovery.ts:847) or a gateRows inventory so the terminal render is verifiable";
|
|
527
|
+
}
|
|
528
|
+
|
|
504
529
|
/**
|
|
505
530
|
* v0.38.50: one compact duration line from durable goal state — turns,
|
|
506
531
|
* wall-clock elapsed since creation, and audit count. Only known facts
|
|
@@ -509,7 +534,9 @@ export function extractEvidenceTokens(text: string): { text: string; evidence: s
|
|
|
509
534
|
export function buildDurationLine(goal: Goal, now = Date.now()): string | null {
|
|
510
535
|
const segs: string[] = [];
|
|
511
536
|
const turns = goal.telemetry?.turns;
|
|
512
|
-
|
|
537
|
+
// Field 20260918_172705: a zero turns count means untracked telemetry,
|
|
538
|
+
// not a known fact — omit it while elapsed/audits still render.
|
|
539
|
+
if (typeof turns === "number" && Number.isFinite(turns) && turns > 0) {
|
|
513
540
|
segs.push(`${turns} turn${turns === 1 ? "" : "s"}`);
|
|
514
541
|
}
|
|
515
542
|
const started = Date.parse(goal.createdAt ?? "");
|
|
@@ -519,15 +546,25 @@ export function buildDurationLine(goal: Goal, now = Date.now()): string | null {
|
|
|
519
546
|
return segs.length > 0 ? `\u2014 ${segs.join(" \u00b7 ")}` : null;
|
|
520
547
|
}
|
|
521
548
|
|
|
522
|
-
/** Partition informing details into findings / Tests / next buckets.
|
|
549
|
+
/** Partition informing details into findings / Tests / next buckets.
|
|
550
|
+
* Field 20260918_172705: exact-duplicate lines collapse to one per bucket
|
|
551
|
+
* (repeated claim details rendered as doubled rows). Near-duplicates with
|
|
552
|
+
* different wording still render — that prose belongs to the claim. */
|
|
523
553
|
export function partitionRichDetails(details: string[]): { findings: string[]; tests: string[]; next: string[] } {
|
|
554
|
+
const seen = new Set<string>();
|
|
555
|
+
const push = (bucket: string[], detail: string) => {
|
|
556
|
+
const key = detail.trim();
|
|
557
|
+
if (seen.has(key)) return;
|
|
558
|
+
seen.add(key);
|
|
559
|
+
bucket.push(detail);
|
|
560
|
+
};
|
|
524
561
|
const findings: string[] = [];
|
|
525
562
|
const tests: string[] = [];
|
|
526
563
|
const next: string[] = [];
|
|
527
564
|
for (const detail of details) {
|
|
528
|
-
if (/^\s*Tests\s*:/i.test(detail))
|
|
529
|
-
else if (/^\s*(Next|Unresolved|Left out)\s*:/i.test(detail))
|
|
530
|
-
else
|
|
565
|
+
if (/^\s*Tests\s*:/i.test(detail)) push(tests, detail);
|
|
566
|
+
else if (/^\s*(Next|Unresolved|Left out)\s*:/i.test(detail)) push(next, detail);
|
|
567
|
+
else push(findings, detail);
|
|
531
568
|
}
|
|
532
569
|
return { findings, tests, next };
|
|
533
570
|
}
|
|
@@ -538,15 +575,17 @@ export function partitionRichDetails(details: string[]): { findings: string[]; t
|
|
|
538
575
|
* `## Done — outcome` shape stays, so direct unit callers are unaffected.
|
|
539
576
|
*/
|
|
540
577
|
export function requestEchoHeadline(kind: "Done" | "Aborted", objective: string | undefined, outcome: string): string {
|
|
541
|
-
const
|
|
542
|
-
|
|
578
|
+
const safeObjective = objective ? sanitizeDisplayText(objective) : "";
|
|
579
|
+
const safeOutcome = sanitizeDisplayText(outcome);
|
|
580
|
+
const echo = safeObjective.trim() ? clipSummaryValue(safeObjective.trim(), RICH_OBJECTIVE_ECHO_CHARS) : null;
|
|
581
|
+
return echo ? `## ${kind}: ${echo} \u2014 ${safeOutcome}` : `## ${kind} \u2014 ${safeOutcome}`;
|
|
543
582
|
}
|
|
544
583
|
|
|
545
584
|
/** Chat-only projection: keep explanations and counts, not command/hash receipts.
|
|
546
585
|
* Repository-only findings are filtered separately; useful implementation
|
|
547
586
|
* references in substantive explanations are not themselves bookkeeping. */
|
|
548
587
|
function chatNarrative(value: string): string {
|
|
549
|
-
return stripMachineGroups(chatSafeDetailValue(value)
|
|
588
|
+
return stripMachineGroups(chatSafeDetailValue(sanitizeDisplayText(value))
|
|
550
589
|
.replace(/\b(?:fixed in|commit|HEAD(?: at)?|built from)\s+`?[a-f0-9]{7,64}`?/gi, "")
|
|
551
590
|
.replace(/\b(?=[a-f0-9]*[a-f])(?=[a-f0-9]*\d)[a-f0-9]{7,64}\b/gi, "")
|
|
552
591
|
.replace(/`(?:bun|npm|npx|node|git|tsc)\s+[^`]+`/g, "")
|
|
@@ -620,9 +659,12 @@ export function buildFinalRepoStateLines(cwd: string): string[] | undefined {
|
|
|
620
659
|
return undefined;
|
|
621
660
|
}
|
|
622
661
|
};
|
|
623
|
-
const
|
|
624
|
-
const
|
|
625
|
-
const
|
|
662
|
+
const branchRaw = run(["branch", "--show-current"]);
|
|
663
|
+
const headRaw = run(["log", "-1", "--format=%h %s"]);
|
|
664
|
+
const shortRaw = run(["status", "--short"]);
|
|
665
|
+
const branch = branchRaw === undefined ? undefined : sanitizeDisplayText(branchRaw);
|
|
666
|
+
const head = headRaw === undefined ? undefined : sanitizeDisplayText(headRaw);
|
|
667
|
+
const short = shortRaw === undefined ? undefined : shortRaw.split("\n").map((line) => sanitizeDisplayText(line)).join("\n");
|
|
626
668
|
if (!branch && !head && short === undefined) return undefined;
|
|
627
669
|
const lines = [`Branch ${branch ?? "detached"}${head ? ` @ ${head}` : ""}`];
|
|
628
670
|
if (short === undefined) lines.push("Tree state unreadable");
|
|
@@ -660,7 +702,7 @@ export function buildRichTerminalParts(args: {
|
|
|
660
702
|
chat?: boolean;
|
|
661
703
|
}): RichTerminalParts {
|
|
662
704
|
const { findings, tests, next } = partitionRichDetails(args.details);
|
|
663
|
-
const outcome = args.outcome;
|
|
705
|
+
const outcome = sanitizeDisplayText(args.outcome);
|
|
664
706
|
const kind = args.kind ?? "Done";
|
|
665
707
|
const headline = args.chat ? `## ${kind} — ${outcome}` : requestEchoHeadline(kind, args.objective, outcome);
|
|
666
708
|
const auditStatus = auditRowStatus(args.auditHistory);
|
|
@@ -676,11 +718,11 @@ export function buildRichTerminalParts(args: {
|
|
|
676
718
|
findingLines.push("| Area | Finding | Evidence |", "| --- | --- | --- |");
|
|
677
719
|
for (const group of groups) {
|
|
678
720
|
group.findings.forEach((finding, fi) => {
|
|
679
|
-
const { text, evidence } = extractEvidenceTokens(finding);
|
|
721
|
+
const { text, evidence } = extractEvidenceTokens(sanitizeDisplayText(finding));
|
|
680
722
|
const { lead, body } = leadBody(text);
|
|
681
723
|
// v0.38.52: test proof rides the Evidence cell (tables have no
|
|
682
724
|
// sub-bullets); cells stay pipe-escaped. v0.38.55: unclipped.
|
|
683
|
-
const proof = group.tests?.[fi]
|
|
725
|
+
const proof = group.tests?.[fi] ? sanitizeDisplayText(group.tests[fi]).trim() : "";
|
|
684
726
|
const evidenceCell = [evidence.join(", ") || "\u2014", ...(proof ? [`Tests: ${proof}`] : [])].join(" \u00b7 ");
|
|
685
727
|
findingLines.push(
|
|
686
728
|
`| ${escapeTableCell(group.title)} | ${escapeTableCell(`**${lead}** \u2014 ${body}`)} | ${escapeTableCell(evidenceCell)} |`,
|
|
@@ -689,12 +731,12 @@ export function buildRichTerminalParts(args: {
|
|
|
689
731
|
}
|
|
690
732
|
} else if (groups.length > 0) {
|
|
691
733
|
groups.forEach((group, i) => {
|
|
692
|
-
findingLines.push(`#### ${i + 1}. ${group.title}`);
|
|
734
|
+
findingLines.push(`#### ${i + 1}. ${sanitizeDisplayText(group.title)}`);
|
|
693
735
|
group.findings.forEach((finding, fi) => {
|
|
694
|
-
const { lead, body } = leadBody(args.chat ? chatNarrative(extractEvidenceTokens(finding).text) : finding);
|
|
736
|
+
const { lead, body } = leadBody(args.chat ? chatNarrative(extractEvidenceTokens(finding).text) : sanitizeDisplayText(finding));
|
|
695
737
|
findingLines.push(`- **${lead}** \u2014 ${body}`);
|
|
696
738
|
// v0.38.52: per-finding test proof (shot C) — absent stays absent.
|
|
697
|
-
const proof = group.tests?.[fi]
|
|
739
|
+
const proof = group.tests?.[fi] ? sanitizeDisplayText(group.tests[fi]).trim() : "";
|
|
698
740
|
if (proof) findingLines.push(` - Test Results: ${args.chat ? chatNarrative(proof) : proof}`);
|
|
699
741
|
});
|
|
700
742
|
});
|
|
@@ -716,22 +758,25 @@ export function buildRichTerminalParts(args: {
|
|
|
716
758
|
const showCommand = !args.chat && gates.some((row) => row.command?.trim());
|
|
717
759
|
if (gates.length > 0) {
|
|
718
760
|
for (const row of gates) {
|
|
719
|
-
const
|
|
720
|
-
const
|
|
721
|
-
const
|
|
761
|
+
const safeNotes = row.notes ? sanitizeDisplayText(row.notes).trim() : "";
|
|
762
|
+
const derived = testsRowStatus(safeNotes);
|
|
763
|
+
const notes = args.chat ? chatNarrative(safeNotes) : safeNotes;
|
|
764
|
+
const command = row.command ? sanitizeDisplayText(row.command).trim() : "\u2014";
|
|
765
|
+
const gate = sanitizeDisplayText(row.gate);
|
|
766
|
+
const scope = row.scope ? sanitizeDisplayText(row.scope).trim() : "\u2014";
|
|
722
767
|
tableRows.push(showCommand
|
|
723
|
-
? `| ${escapeTableCell(
|
|
724
|
-
: `| ${escapeTableCell(
|
|
768
|
+
? `| ${escapeTableCell(gate)} | ${escapeTableCell(command)} | ${escapeTableCell(scope)} | ${derived} | ${escapeTableCell(notes || "\u2014")} |`
|
|
769
|
+
: `| ${escapeTableCell(gate)} | ${escapeTableCell(scope)} | ${derived} | ${escapeTableCell(notes || "\u2014")} |`);
|
|
725
770
|
}
|
|
726
771
|
} else {
|
|
727
772
|
for (const detail of tests) {
|
|
728
773
|
const { body } = leadBody(detail);
|
|
729
774
|
const status = testsRowStatus(body);
|
|
730
|
-
tableRows.push(`| Tests | ${status} | ${escapeTableCell(args.chat ? chatNarrative(body) : body)} |`);
|
|
775
|
+
tableRows.push(`| Tests | ${status} | ${escapeTableCell(args.chat ? chatNarrative(body) : sanitizeDisplayText(body))} |`);
|
|
731
776
|
}
|
|
732
777
|
}
|
|
733
778
|
if (!args.chat && auditStatus !== "NO VERDICT") {
|
|
734
|
-
const auditBody = args.countsLine.replace(/^\u2014\s*/, "").replace(/\.\s*$/, "");
|
|
779
|
+
const auditBody = sanitizeDisplayText(args.countsLine).replace(/^\u2014\s*/, "").replace(/\.\s*$/, "");
|
|
735
780
|
// v0.38.55 audit: the Audit row's Scope names the row kind — the old
|
|
736
781
|
// shape duplicated the counts text in Scope and Notes.
|
|
737
782
|
if (gates.length > 0) {
|
|
@@ -763,8 +808,8 @@ export function buildRichTerminalParts(args: {
|
|
|
763
808
|
findingLines,
|
|
764
809
|
tableLines,
|
|
765
810
|
nextLines,
|
|
766
|
-
repoLines: args.chat ? [] : args.repoState ?? [],
|
|
767
|
-
summaryLines: args.summaryLines ?? [],
|
|
811
|
+
repoLines: args.chat ? [] : (args.repoState ?? []).map((line) => sanitizeDisplayText(line)),
|
|
812
|
+
summaryLines: (args.summaryLines ?? []).map((line) => sanitizeDisplayText(line)),
|
|
768
813
|
};
|
|
769
814
|
}
|
|
770
815
|
|
|
@@ -776,28 +821,30 @@ export function buildRichTerminalParts(args: {
|
|
|
776
821
|
* final repository state closes it — findings, verification, Next,
|
|
777
822
|
* repo state, in that order. */
|
|
778
823
|
export function composeRichTerminalLines(parts: RichTerminalParts): string[] {
|
|
779
|
-
const
|
|
780
|
-
|
|
781
|
-
|
|
824
|
+
const banner = sanitizeDisplayText(parts.banner ?? parts.headline);
|
|
825
|
+
const headline = sanitizeDisplayText(parts.headline);
|
|
826
|
+
const lines = [banner, ""];
|
|
827
|
+
if (headline !== lines[0]) lines.push(headline, "");
|
|
828
|
+
if (parts.durationLine) lines.push(sanitizeDisplayText(parts.durationLine), "");
|
|
782
829
|
// Structured-long (field 2026-09-16): the full Outcome body rides its
|
|
783
830
|
// own section between the headline and the findings — findings-first
|
|
784
831
|
// order is preserved (findings, verification, Next keep their relative
|
|
785
832
|
// order), the headline echo stays short, and the one-action Next rule
|
|
786
833
|
// is untouched.
|
|
787
834
|
if (parts.summaryLines.length > 0) {
|
|
788
|
-
lines.push("### Summary", ...parts.summaryLines, "");
|
|
835
|
+
lines.push("### Summary", ...parts.summaryLines.map((line) => sanitizeDisplayText(line)), "");
|
|
789
836
|
}
|
|
790
837
|
if (parts.findingLines.length > 0) {
|
|
791
|
-
lines.push("### Key Findings & Remediation", ...parts.findingLines, "");
|
|
838
|
+
lines.push("### Key Findings & Remediation", ...parts.findingLines.map((line) => sanitizeDisplayText(line)), "");
|
|
792
839
|
}
|
|
793
840
|
if (parts.tableLines.length > 0) {
|
|
794
|
-
lines.push("### Verification Summary", ...parts.tableLines, "");
|
|
841
|
+
lines.push("### Verification Summary", ...parts.tableLines.map((line) => sanitizeDisplayText(line)), "");
|
|
795
842
|
}
|
|
796
843
|
if (parts.nextLines.length > 0) {
|
|
797
|
-
lines.push("### Next", ...parts.nextLines, "");
|
|
844
|
+
lines.push("### Next", ...parts.nextLines.map((line) => sanitizeDisplayText(line)), "");
|
|
798
845
|
}
|
|
799
846
|
if (parts.repoLines.length > 0) {
|
|
800
|
-
lines.push("### Final Repository State", ...parts.repoLines.map((line) => `- ${line}`), "");
|
|
847
|
+
lines.push("### Final Repository State", ...parts.repoLines.map((line) => `- ${sanitizeDisplayText(line)}`), "");
|
|
801
848
|
}
|
|
802
849
|
return lines;
|
|
803
850
|
}
|
|
@@ -889,7 +936,7 @@ function stripApprovalModel(line: string): string {
|
|
|
889
936
|
return line.replace(/auditor\s+\S+\s+approved/, "auditor approved");
|
|
890
937
|
}
|
|
891
938
|
function trailerBullet(line: string): string {
|
|
892
|
-
return `• ${line.replace(/^—\s*/, "")}`;
|
|
939
|
+
return `• ${sanitizeDisplayText(line).replace(/^—\s*/, "")}`;
|
|
893
940
|
}
|
|
894
941
|
|
|
895
942
|
/** v0.38.25: the audit-goal counts line — verdict proof ONLY, built from
|
|
@@ -1420,7 +1420,7 @@ export function sendStallEscalation(ctx: ExtensionContext, nudges: number): void
|
|
|
1420
1420
|
if (supervisorPaused(state)) return;
|
|
1421
1421
|
// Audit 2026-09-07 (HIGH): a stall nudge must not resurrect a stood-down
|
|
1422
1422
|
// chain — same abort-latch reasoning as sendContinuation.
|
|
1423
|
-
if (flags.sessionHandoffPending || flags.initialSessionLoadPending || !flags.extensionApi || flags.extensionApiStale || continuationDispatchStoodDown || pendingContinuationDispatch || flags.abortedStandDown) return;
|
|
1423
|
+
if (flags.sessionHandoffPending || flags.initialSessionLoadPending || !flags.extensionApi || flags.extensionApiStale || flags.staleTerminalDone || flags.zombieStoodDown || continuationDispatchStoodDown || pendingContinuationDispatch || flags.abortedStandDown) return;
|
|
1424
1424
|
if (!state.goal || !guardGoalBeforeContinuation(ctx, "stall-escalation")) return;
|
|
1425
1425
|
const remaining = HEARTBEAT_MAX_NUDGES - nudges;
|
|
1426
1426
|
const text = [
|
|
@@ -1464,7 +1464,7 @@ export function sendLengthContinue(ctx: ExtensionContext, consecutive: number):
|
|
|
1464
1464
|
if (supervisorPaused(state)) return;
|
|
1465
1465
|
// Audit 2026-09-07 (HIGH): a length nudge must not resurrect a stood-down
|
|
1466
1466
|
// chain — same abort-latch reasoning as sendContinuation.
|
|
1467
|
-
if (flags.sessionHandoffPending || flags.initialSessionLoadPending || !flags.extensionApi || flags.extensionApiStale || continuationDispatchStoodDown || pendingContinuationDispatch || flags.abortedStandDown) return;
|
|
1467
|
+
if (flags.sessionHandoffPending || flags.initialSessionLoadPending || !flags.extensionApi || flags.extensionApiStale || flags.staleTerminalDone || flags.zombieStoodDown || continuationDispatchStoodDown || pendingContinuationDispatch || flags.abortedStandDown) return;
|
|
1468
1468
|
if (state.goal && !guardGoalBeforeContinuation(ctx, "length-continuation")) return;
|
|
1469
1469
|
// v0.38.66 (PR #55, FOF11): active goals get completion-aware recovery
|
|
1470
1470
|
// text — a truncated turn must offer closure, not another blind work
|
|
@@ -1525,7 +1525,34 @@ export function buildPostCompactResync(briefExcerpt?: string): string {
|
|
|
1525
1525
|
return lines.join("\n") + "\n\n";
|
|
1526
1526
|
}
|
|
1527
1527
|
|
|
1528
|
-
|
|
1528
|
+
/**
|
|
1529
|
+
* v0.38.69 (Antigravity port, 09-04 borrow candidate #1): per-repo pitfall
|
|
1530
|
+
* registry. `.pi-glla/pitfalls.md` holds distilled, answer-agnostic rakes
|
|
1531
|
+
* (never ledger prose — the ledger is forensics, never distilled). Read at
|
|
1532
|
+
* goal start (and re-read when edited mid-goal); absent/blank/unreadable
|
|
1533
|
+
* resolves absent so repos without one render byte-identical prompts.
|
|
1534
|
+
*/
|
|
1535
|
+
export const PITFALLS_BRIEF_MAX_CHARS = 1500;
|
|
1536
|
+
|
|
1537
|
+
export function readPitfallsBrief(cwd: string): string | undefined {
|
|
1538
|
+
let body: string;
|
|
1539
|
+
try {
|
|
1540
|
+
body = fs.readFileSync(path.join(cwd, ".pi-glla", "pitfalls.md"), "utf8");
|
|
1541
|
+
} catch {
|
|
1542
|
+
return undefined;
|
|
1543
|
+
}
|
|
1544
|
+
const trimmed = body.trim();
|
|
1545
|
+
if (!trimmed) return undefined;
|
|
1546
|
+
return trimmed.length > PITFALLS_BRIEF_MAX_CHARS
|
|
1547
|
+
? trimmed.slice(0, PITFALLS_BRIEF_MAX_CHARS)
|
|
1548
|
+
: trimmed;
|
|
1549
|
+
}
|
|
1550
|
+
|
|
1551
|
+
/** Goals already ledgered for their pitfalls consult this process — the
|
|
1552
|
+
* ledger records the consult once per goal, never once per turn. */
|
|
1553
|
+
const pitfallsLedgeredGoals = new Set<string>();
|
|
1554
|
+
|
|
1555
|
+
export function continuationPrompt(goal: Goal, opts: { includeRestartDetail?: boolean; pitfallsBrief?: string } = {}): string {
|
|
1529
1556
|
// Read the .md file as the template, then substitute {{tokens}}.
|
|
1530
1557
|
// For v0.1.0 we inline-substitute so we don't need fs at runtime.
|
|
1531
1558
|
const next = findNextPendingTask(goal.taskList?.tasks ?? []);
|
|
@@ -1583,6 +1610,20 @@ export function continuationPrompt(goal: Goal, opts: { includeRestartDetail?: bo
|
|
|
1583
1610
|
if (loadSettings(settingsCwd).visionAssist !== false) {
|
|
1584
1611
|
directives.push(VISION_ASSIST_GUIDANCE);
|
|
1585
1612
|
}
|
|
1613
|
+
// v0.38.69 (Antigravity port): the repo pitfall registry rides the
|
|
1614
|
+
// continuation prompt — consulted at goal start, re-read when edited.
|
|
1615
|
+
// Explicit opts win (tests); otherwise the repo file resolves, absent
|
|
1616
|
+
// keeping the prompt byte-identical for repos without one.
|
|
1617
|
+
const pitfallsBrief = opts.pitfallsBrief ?? readPitfallsBrief(settingsCwd);
|
|
1618
|
+
if (pitfallsBrief?.trim()) {
|
|
1619
|
+
directives.push(
|
|
1620
|
+
`## REPO PITFALLS (distilled — consult before acting)\n\nThese rakes already caught this repo. Check your plan against them before the first tool call and before every risky step:\n\n${pitfallsBrief.trim()}`,
|
|
1621
|
+
);
|
|
1622
|
+
if (!pitfallsLedgeredGoals.has(goal.id)) {
|
|
1623
|
+
pitfallsLedgeredGoals.add(goal.id);
|
|
1624
|
+
appendLedger(settingsCwd, "pitfalls_consulted", { goalId: goal.id });
|
|
1625
|
+
}
|
|
1626
|
+
}
|
|
1586
1627
|
if (effSettings.aggressiveMode && isFullAuditObjective(goal.objective)) {
|
|
1587
1628
|
directives.push(
|
|
1588
1629
|
"## FULL-AUDIT MODE (aggressiveMode + survey objective)\n\nThis objective is a survey, not a single fix. Spawn 3+ `scout` subagents NOW — one per subsystem, in a single message so they run in parallel — synthesize their findings, and call `propose_task_list` with the result. Do not start fixing before the task list exists.",
|
|
@@ -656,6 +656,15 @@ export interface Goal {
|
|
|
656
656
|
pauseOptions?: string[];
|
|
657
657
|
/** v0.28.22: 1-based index into pauseOptions the agent recommends. */
|
|
658
658
|
pauseRecommended?: number;
|
|
659
|
+
/** v0.38.69 (Antigravity port): mid-run decision-pause accounting — every
|
|
660
|
+
* agent-authored kind="decision" pause increments this, in-budget or
|
|
661
|
+
* not, so the interruption budget has a durable counter across reloads. */
|
|
662
|
+
midRunDecisionCount?: number;
|
|
663
|
+
/** v0.38.69 (Antigravity port): decisions auto-resolved to the
|
|
664
|
+
* recommended default when the budget was exhausted — the assumption
|
|
665
|
+
* the agent must carry into its completion recap's Left out. Bounded
|
|
666
|
+
* to the trailing 20. */
|
|
667
|
+
autoDefaultLog?: Array<{ at: string; reason: string; chosen: string; options: string[] }>;
|
|
659
668
|
/** v0.28.22: ISO time a wait-pause becomes resumable (countdown shown). */
|
|
660
669
|
pauseResumeAt?: string;
|
|
661
670
|
/** v0.35.28 (issue #16): set when glla AUTO-resumed a lapsed wait — the
|
|
@@ -1120,6 +1129,11 @@ export interface MainModelRecovery {
|
|
|
1120
1129
|
retryAt?: string;
|
|
1121
1130
|
/** Next preferred-primary health probe while a fallback is serving. */
|
|
1122
1131
|
primaryProbeAt?: string;
|
|
1132
|
+
/** v0.38.69 (Antigravity port): absolute reset of the quota-walled
|
|
1133
|
+
* primary, stamped when the episode failed over with an explicit
|
|
1134
|
+
* upstream reset hint. While a fallback serves, the background primary
|
|
1135
|
+
* probe fires at this reset instead of the generic probe cadence. */
|
|
1136
|
+
primaryResetAt?: string;
|
|
1123
1137
|
/** A preferred-primary switch was accepted and awaits one supervised turn. */
|
|
1124
1138
|
primaryProbeInFlight?: boolean;
|
|
1125
1139
|
/** Number of completed recovery waits; drives the bounded per-attempt exponential cadence. */
|
|
@@ -1228,6 +1242,7 @@ export function sanitizeMainModelRecovery(value: unknown): MainModelRecovery | u
|
|
|
1228
1242
|
...(Array.isArray(raw.recoveryNoticeKeys) ? { recoveryNoticeKeys: raw.recoveryNoticeKeys.filter((key): key is string => typeof key === "string").slice(-16).map((key) => key.slice(0, 300)) } : {}),
|
|
1229
1243
|
...(date(raw.retryAt) ? { retryAt: date(raw.retryAt) } : {}),
|
|
1230
1244
|
...(date(raw.primaryProbeAt) ? { primaryProbeAt: date(raw.primaryProbeAt) } : {}),
|
|
1245
|
+
...(date(raw.primaryResetAt) ? { primaryResetAt: date(raw.primaryResetAt) } : {}),
|
|
1231
1246
|
...(raw.primaryProbeInFlight === true ? { primaryProbeInFlight: true } : {}),
|
|
1232
1247
|
...(date(raw.firstFailureAt) ? { firstFailureAt: date(raw.firstFailureAt) } : {}),
|
|
1233
1248
|
...(date(raw.autoRetryUntil) ? { autoRetryUntil: date(raw.autoRetryUntil) } : {}),
|
|
@@ -3626,6 +3641,21 @@ export function extractPendingTasks(report: string, cap = 5): string[] {
|
|
|
3626
3641
|
return out;
|
|
3627
3642
|
}
|
|
3628
3643
|
|
|
3644
|
+
/**
|
|
3645
|
+
* v0.38.69 (Antigravity port): objection-attached retries. Every
|
|
3646
|
+
* disapproval — aggressive or not — becomes a durable TODO projection:
|
|
3647
|
+
* extracted objection bullets, or a single review-the-report TODO when
|
|
3648
|
+
* nothing extractable survives. Never an empty list: the retry argues the
|
|
3649
|
+
* objection, not generic effort. The aggressive gate keeps only the
|
|
3650
|
+
* cap-keep-going / no-progress-stop behaviors, never the TODOs.
|
|
3651
|
+
*/
|
|
3652
|
+
export function durableObjectionsForDisapproval(report: string, statusCommand = "/goal status"): string[] {
|
|
3653
|
+
const extracted = extractPendingTasks(report, 5);
|
|
3654
|
+
return extracted.length > 0
|
|
3655
|
+
? extracted
|
|
3656
|
+
: [`Review the latest auditor disapproval in ${statusCommand}.`];
|
|
3657
|
+
}
|
|
3658
|
+
|
|
3629
3659
|
/** Contract item 23: is the auditor's IMPOSSIBLE reason about the WHOLE
|
|
3630
3660
|
* goal or only part of it? Default "full" (safe — keeps the pause);
|
|
3631
3661
|
* partial only on explicit subset language. */
|
|
@@ -31,6 +31,8 @@ import {
|
|
|
31
31
|
mainModelFailureDelayMs,
|
|
32
32
|
mainModelPrimaryProbeDelayMs,
|
|
33
33
|
mainModelRetryDelayMs,
|
|
34
|
+
isQuotaHorizonExempt,
|
|
35
|
+
quotaResetSleepMs,
|
|
34
36
|
isCompactionInFlightSince,
|
|
35
37
|
MAIN_MODEL_AUTO_RETRY_HORIZON_MS,
|
|
36
38
|
modelRef,
|
|
@@ -352,7 +354,10 @@ export function observeCompactFailure(ctx: ExtensionContext, error: string | und
|
|
|
352
354
|
* current public event API returns false and the caller uses the terminal
|
|
353
355
|
* park with an explicit `/new` instruction instead of claiming recovery.
|
|
354
356
|
*
|
|
355
|
-
* The signature is `where: string` so every attempt/skip is observable.
|
|
357
|
+
* The signature is `where: string` so every attempt/skip is observable. A
|
|
358
|
+
* recovery is successful only when the host synchronously supplies the
|
|
359
|
+
* replacement context. Async host calls are fail-closed: callers must park
|
|
360
|
+
* the stale handle instead of treating invocation as recovery. */
|
|
356
361
|
export function attemptFreshSessionRecovery(ctx: ExtensionContext, where: string): boolean {
|
|
357
362
|
type FreshSessionContext = ExtensionContext & {
|
|
358
363
|
newSession?: (options?: {
|
|
@@ -384,8 +389,16 @@ export function attemptFreshSessionRecovery(ctx: ExtensionContext, where: string
|
|
|
384
389
|
return false;
|
|
385
390
|
}
|
|
386
391
|
try {
|
|
392
|
+
let rebound = false;
|
|
393
|
+
let failureRecorded = false;
|
|
394
|
+
const recordFailure = (error: string): void => {
|
|
395
|
+
if (failureRecorded) return;
|
|
396
|
+
failureRecorded = true;
|
|
397
|
+
appendLedger(recoveryCwd, "fresh_session_recovery_failed", { from: where, error });
|
|
398
|
+
};
|
|
387
399
|
const result = startNewSession.call(freshCtx, {
|
|
388
|
-
withSession:
|
|
400
|
+
withSession: (replacementCtx) => {
|
|
401
|
+
rebound = true;
|
|
389
402
|
appendLedger(replacementCtx.cwd, "fresh_session_recovery_rebound", { where });
|
|
390
403
|
replacementCtx.ui.notify(
|
|
391
404
|
"glla: stale ctx detected — host supplied a fresh-session capability; rehydrating the goal now.",
|
|
@@ -394,12 +407,23 @@ export function attemptFreshSessionRecovery(ctx: ExtensionContext, where: string
|
|
|
394
407
|
},
|
|
395
408
|
});
|
|
396
409
|
if (result && typeof (result as Promise<unknown>).then === "function") {
|
|
397
|
-
//
|
|
398
|
-
//
|
|
399
|
-
//
|
|
400
|
-
(result as Promise<unknown>).
|
|
401
|
-
|
|
402
|
-
|
|
410
|
+
// Do not let a rejected or callback-free async host call masquerade as
|
|
411
|
+
// recovery. The caller fails closed immediately when `rebound` is
|
|
412
|
+
// still false; this continuation only records the eventual outcome.
|
|
413
|
+
(result as Promise<unknown>).then(
|
|
414
|
+
() => {
|
|
415
|
+
if (!rebound) recordFailure("host completed without supplying a replacement context");
|
|
416
|
+
},
|
|
417
|
+
(err) => {
|
|
418
|
+
recordFailure(err instanceof Error ? err.message : String(err));
|
|
419
|
+
},
|
|
420
|
+
);
|
|
421
|
+
}
|
|
422
|
+
if (!rebound) {
|
|
423
|
+
recordFailure(result && typeof (result as Promise<unknown>).then === "function"
|
|
424
|
+
? "host recovery is asynchronous; no replacement context was supplied synchronously"
|
|
425
|
+
: "host recovery returned without supplying a replacement context");
|
|
426
|
+
return false;
|
|
403
427
|
}
|
|
404
428
|
appendLedger(recoveryCwd, "fresh_session_recovery_triggered", { from: where });
|
|
405
429
|
return true;
|
|
@@ -635,6 +659,18 @@ export async function tryMainModelFallback(ctx: ExtensionContext, failure: MainM
|
|
|
635
659
|
recoveryEpisodeKey: baseRecovery.recoveryEpisodeKey ?? `${baseRecovery.firstFailureAt ?? nowIso()}:${failureCopy.fingerprint}`,
|
|
636
660
|
recoveryNoticeKeys: baseRecovery.recoveryNoticeKeys ?? [],
|
|
637
661
|
};
|
|
662
|
+
// v0.38.69 (Antigravity port): a quota-walled primary carries its reset
|
|
663
|
+
// forward into the episode, so the background primary probe fires at
|
|
664
|
+
// reset instead of the generic cadence while fallback-chain work
|
|
665
|
+
// proceeds. Stamped only for failures ON the primary — a fallback's own
|
|
666
|
+
// wall says nothing about when the primary recovers — and never cleared
|
|
667
|
+
// here, so later fallback failures cannot erase the primary's reset.
|
|
668
|
+
if (sameModelRef(current, recovery.primary)) {
|
|
669
|
+
const resetSleep = quotaResetSleepMs(failure);
|
|
670
|
+
if (resetSleep !== undefined) {
|
|
671
|
+
recovery.primaryResetAt = new Date(Date.now() + resetSleep).toISOString();
|
|
672
|
+
}
|
|
673
|
+
}
|
|
638
674
|
if (!recovery.attempted.includes(current)) recovery.attempted.push(current);
|
|
639
675
|
const selector = sessionModelSelector(ctx);
|
|
640
676
|
const scope: ModelScope = { kind: "session" };
|
|
@@ -701,8 +737,9 @@ export async function tryMainModelFallback(ctx: ExtensionContext, failure: MainM
|
|
|
701
737
|
// never invoke setModel on that stale generation.
|
|
702
738
|
if (generation !== flags.sessionGeneration || !freshCtxForGeneration(generation)) return false;
|
|
703
739
|
const api = flags.extensionApi;
|
|
740
|
+
if (supervisorPaused(state)) return false;
|
|
704
741
|
const accepted = await api?.setModel(candidate);
|
|
705
|
-
if (generation !== flags.sessionGeneration || !freshCtxForGeneration(generation)) return false;
|
|
742
|
+
if (generation !== flags.sessionGeneration || !freshCtxForGeneration(generation) || supervisorPaused(state)) return false;
|
|
706
743
|
if (state.mainModelRecovery?.pendingModelSwitch?.toLowerCase() !== candidateRef.toLowerCase()) return false;
|
|
707
744
|
if (!accepted) {
|
|
708
745
|
state.mainModelRecovery = { ...state.mainModelRecovery, pendingModelSwitch: undefined, retryAt: undefined };
|
|
@@ -764,7 +801,13 @@ export function setMainModelRecoveryPause(ctx: ExtensionContext, recovery: MainM
|
|
|
764
801
|
const now = Date.now();
|
|
765
802
|
const deadlineMs = normalized.autoRetryUntil ? Date.parse(normalized.autoRetryUntil) : Number.NaN;
|
|
766
803
|
const requestedDelayMs = Math.max(1_000, delayMs);
|
|
767
|
-
|
|
804
|
+
// v0.38.69 (Antigravity port): quota waits never park at the horizon —
|
|
805
|
+
// a rate-limit/plan-quota wall is transient, so the hold below is
|
|
806
|
+
// skipped and the wait re-arms. Billing and non-quota failures keep
|
|
807
|
+
// their horizon park.
|
|
808
|
+
const horizonApplies = !aggressive
|
|
809
|
+
&& !isQuotaHorizonExempt(normalized.providerErrorDiagnostic ?? normalized.reason);
|
|
810
|
+
if (normalized.manualResumeRequired || (horizonApplies && Number.isFinite(deadlineMs) && (now >= deadlineMs || now + requestedDelayMs > deadlineMs))) {
|
|
768
811
|
holdMainModelRecovery(ctx, normalized, Number.isFinite(deadlineMs) && now >= deadlineMs
|
|
769
812
|
? "the 24h automatic recovery horizon was reached"
|
|
770
813
|
: "the automatic recovery horizon would be exceeded");
|
|
@@ -942,6 +985,12 @@ export function scheduleHourlyProbe(ctx: ExtensionContext): void {
|
|
|
942
985
|
* recovery path's timer cleanup, and a generation/host check prevents a
|
|
943
986
|
* stale session from creating a new timer. */
|
|
944
987
|
export async function fireHourlyProbe(ctx: ExtensionContext): Promise<void> {
|
|
988
|
+
// /glla pause may race an already-dequeued timer. Re-arm the parked slot
|
|
989
|
+
// without probing so explicit resume still has a durable backstop.
|
|
990
|
+
if (supervisorPaused(state)) {
|
|
991
|
+
scheduleHourlyProbe(ctx);
|
|
992
|
+
return;
|
|
993
|
+
}
|
|
945
994
|
if (!state.mainModelRecovery || state.mainModelRecovery.manualResumeRequired === true) {
|
|
946
995
|
await fireHourlyProbeForParkedAuditor(ctx);
|
|
947
996
|
// Keep the backstop alive: the consumed :00:30 slot cleared its own
|
|
@@ -972,6 +1021,7 @@ export async function fireHourlyProbe(ctx: ExtensionContext): Promise<void> {
|
|
|
972
1021
|
at: new Date().toISOString(),
|
|
973
1022
|
});
|
|
974
1023
|
try {
|
|
1024
|
+
if (supervisorPaused(state)) return;
|
|
975
1025
|
await probeMainModelRecovery(ctx);
|
|
976
1026
|
} catch (err) {
|
|
977
1027
|
if (isStaleApiError(err)) flags.extensionApiStale = true;
|
|
@@ -992,6 +1042,7 @@ export async function fireHourlyProbe(ctx: ExtensionContext): Promise<void> {
|
|
|
992
1042
|
* no-ops on the phase guard inside retryStoredCompletionAudit. Never
|
|
993
1043
|
* touches models: a detached audit launch cannot disturb a supervised turn. */
|
|
994
1044
|
async function fireHourlyProbeForParkedAuditor(ctx: ExtensionContext): Promise<void> {
|
|
1045
|
+
if (supervisorPaused(state)) return;
|
|
995
1046
|
const goal = state.goal;
|
|
996
1047
|
if (!goal || goal.status !== "paused") return;
|
|
997
1048
|
const pending = goal.pendingCompletion;
|
|
@@ -1014,6 +1065,7 @@ async function fireHourlyProbeForParkedAuditor(ctx: ExtensionContext): Promise<v
|
|
|
1014
1065
|
at: new Date().toISOString(),
|
|
1015
1066
|
pauseResumeAt: goal.pauseResumeAt ?? null,
|
|
1016
1067
|
});
|
|
1068
|
+
if (supervisorPaused(state)) return;
|
|
1017
1069
|
await retryStoredCompletionAudit("provider-retry");
|
|
1018
1070
|
}
|
|
1019
1071
|
|
|
@@ -1081,6 +1133,7 @@ let hourlyProbeGeneration: number | null = null;
|
|
|
1081
1133
|
* durable timer state is not enough to fence the two callbacks once both
|
|
1082
1134
|
* have fired, so serialize the actual async probe as well. */
|
|
1083
1135
|
export async function probeMainModelRecovery(ctx: ExtensionContext): Promise<void> {
|
|
1136
|
+
if (supervisorPaused(state)) return;
|
|
1084
1137
|
const generation = flags.sessionGeneration;
|
|
1085
1138
|
if (mainModelRecoveryProbeInFlight && mainModelRecoveryProbeGeneration !== generation) {
|
|
1086
1139
|
mainModelRecoveryProbeInFlight = false;
|
|
@@ -1205,8 +1258,9 @@ async function probePreferredPrimary(ctx: ExtensionContext, recovery: MainModelR
|
|
|
1205
1258
|
flags.mainModelSwitchInFlight = true;
|
|
1206
1259
|
try {
|
|
1207
1260
|
if (generation !== flags.sessionGeneration || !freshCtxForGeneration(generation)) return;
|
|
1261
|
+
if (supervisorPaused(state)) return;
|
|
1208
1262
|
const accepted = await flags.extensionApi?.setModel(candidate);
|
|
1209
|
-
if (generation !== flags.sessionGeneration || !freshCtxForGeneration(generation)) return;
|
|
1263
|
+
if (generation !== flags.sessionGeneration || !freshCtxForGeneration(generation) || supervisorPaused(state)) return;
|
|
1210
1264
|
if (state.mainModelRecovery?.pendingModelSwitch?.toLowerCase() !== primary.toLowerCase()) return;
|
|
1211
1265
|
if (!accepted) throw new Error("no configured auth for preferred primary");
|
|
1212
1266
|
const switched = {
|
|
@@ -1252,6 +1306,7 @@ async function probePreferredPrimary(ctx: ExtensionContext, recovery: MainModelR
|
|
|
1252
1306
|
}
|
|
1253
1307
|
|
|
1254
1308
|
async function probeMainModelRecoveryImpl(ctx: ExtensionContext): Promise<void> {
|
|
1309
|
+
if (supervisorPaused(state)) return;
|
|
1255
1310
|
const generation = flags.sessionGeneration;
|
|
1256
1311
|
const recovery = state.mainModelRecovery;
|
|
1257
1312
|
if (!recovery) return;
|
|
@@ -1399,7 +1454,10 @@ async function probeMainModelRecoveryImpl(ctx: ExtensionContext): Promise<void>
|
|
|
1399
1454
|
try { return resolveEffectiveAggressiveSettings(loadSettings(ctx.cwd)).aggressiveMode; } catch { return false; }
|
|
1400
1455
|
})();
|
|
1401
1456
|
const horizonMs = next.autoRetryUntil ? Date.parse(next.autoRetryUntil) : Number.NaN;
|
|
1402
|
-
|
|
1457
|
+
// v0.38.69: quota waits skip the horizon hold here too (see
|
|
1458
|
+
// setMainModelRecoveryPause) — a wall outliving 24h still resumes.
|
|
1459
|
+
const quotaExempt = isQuotaHorizonExempt(next.providerErrorDiagnostic ?? next.reason);
|
|
1460
|
+
if (next.manualResumeRequired || (!aggressive && !quotaExempt && Number.isFinite(horizonMs) && Date.now() >= horizonMs)) {
|
|
1403
1461
|
holdMainModelRecovery(ctx, next, "the 24h automatic recovery horizon was reached");
|
|
1404
1462
|
return;
|
|
1405
1463
|
}
|
|
@@ -1500,8 +1558,9 @@ async function probeMainModelRecoveryImpl(ctx: ExtensionContext): Promise<void>
|
|
|
1500
1558
|
resumeCurrent: (state.mainModelRecovery ?? recovery).resumeCurrent,
|
|
1501
1559
|
pendingModelSwitch: undefined,
|
|
1502
1560
|
});
|
|
1503
|
-
// All provider failures use the same bounded envelope.
|
|
1504
|
-
// upstream
|
|
1561
|
+
// All provider failures use the same bounded envelope. Only an explicit
|
|
1562
|
+
// upstream quota-reset hint alters the delay (sleep until reset, 5h
|
|
1563
|
+
// cap); all other error text and retry hints keep the ladder.
|
|
1505
1564
|
const delay = mainModelFailureDelayMs(failure, next.attempts, loadGlobalSettings().mainModelRetryMinutes);
|
|
1506
1565
|
if (setMainModelRecoveryPause(ctx, next, delay)) scheduleMainModelRecoveryTimer(ctx, delay);
|
|
1507
1566
|
} finally {
|
|
@@ -1539,7 +1598,9 @@ export function parkMainModelAfterFailure(ctx: ExtensionContext, failure: MainMo
|
|
|
1539
1598
|
resumeCurrent: existing.resumeCurrent,
|
|
1540
1599
|
});
|
|
1541
1600
|
// The generic envelope owns the wait; the 24h horizon ends automatic
|
|
1542
|
-
// probes regardless of the provider's wording.
|
|
1601
|
+
// probes regardless of the provider's wording. (A quota-class failure
|
|
1602
|
+
// with an explicit upstream reset hint sleeps until reset inside the
|
|
1603
|
+
// same 5h-capped envelope via mainModelFailureDelayMs.)
|
|
1543
1604
|
const delay = mainModelFailureDelayMs(failure, nextRecovery.attempts, loadGlobalSettings().mainModelRetryMinutes);
|
|
1544
1605
|
if (!setMainModelRecoveryPause(ctx, nextRecovery, delay)) return;
|
|
1545
1606
|
flags.mainModelAbortForRecovery = true;
|
|
@@ -1599,7 +1660,17 @@ export function mainModelRecoverySucceeded(ctx: ExtensionContext): void {
|
|
|
1599
1660
|
|
|
1600
1661
|
const probeAt = alreadyScheduled
|
|
1601
1662
|
? existingProbeAt!
|
|
1602
|
-
: new Date(
|
|
1663
|
+
: new Date(Math.max(
|
|
1664
|
+
Date.now() + mainModelPrimaryProbeDelay(),
|
|
1665
|
+
// v0.38.69 (Antigravity port): the primary failed with an explicit
|
|
1666
|
+
// quota reset, and the fallback is serving — probe AT the reset,
|
|
1667
|
+
// not on the generic cadence, so a known-walled primary is not
|
|
1668
|
+
// hammered every 15m while work proceeds on the chain.
|
|
1669
|
+
(() => {
|
|
1670
|
+
const resetMs = recovery.primaryResetAt ? Date.parse(recovery.primaryResetAt) : Number.NaN;
|
|
1671
|
+
return Number.isFinite(resetMs) ? resetMs : 0;
|
|
1672
|
+
})(),
|
|
1673
|
+
)).toISOString();
|
|
1603
1674
|
const nextRecovery: MainModelRecovery = {
|
|
1604
1675
|
...recovery,
|
|
1605
1676
|
active: current,
|
|
@@ -191,6 +191,15 @@ export interface Settings {
|
|
|
191
191
|
* widget card still shows the options; /goal decide opens it on demand).
|
|
192
192
|
* Default on; unattended rigs have no UI so this never fires there. */
|
|
193
193
|
decisionPopup?: boolean;
|
|
194
|
+
/** v0.38.69 (Antigravity port): maximum agent-authored decision pauses
|
|
195
|
+
* per goal before the run stops blocking on a human. The (N+1)-th
|
|
196
|
+
* decision pause does NOT pause: the goal stays active and the
|
|
197
|
+
* recommended option is adopted as a logged assumption (goal
|
|
198
|
+
* autoDefaultLog + `decision_budget_auto_default` ledger line), which
|
|
199
|
+
* the agent must carry into its completion recap's Left out. Unset =
|
|
200
|
+
* unlimited (legacy pause-every-time). `0` = relentless: the first
|
|
201
|
+
* decision already auto-defaults. */
|
|
202
|
+
decisionPauseBudget?: number;
|
|
194
203
|
/** v0.28.14: what happens to stale carryover (paused goal, waiting list,
|
|
195
204
|
* held loop from before this session) when NEW work activates.
|
|
196
205
|
* pause (default) = leave it + ONE summary; clear = drop it all honestly;
|
|
@@ -554,6 +563,7 @@ export function normalizeLoadedSettings(settings: Settings): Settings {
|
|
|
554
563
|
delete settings.carryover;
|
|
555
564
|
}
|
|
556
565
|
if (typeof settings.decisionPopup !== "boolean") delete settings.decisionPopup;
|
|
566
|
+
if (settings.decisionPauseBudget !== undefined && (!Number.isInteger(settings.decisionPauseBudget) || settings.decisionPauseBudget < 0)) delete settings.decisionPauseBudget;
|
|
557
567
|
if (typeof settings.aggressiveMode !== "boolean") delete settings.aggressiveMode;
|
|
558
568
|
if (typeof settings.autoResume !== "boolean") delete settings.autoResume;
|
|
559
569
|
if (typeof settings.autoAcceptDrafts !== "boolean") delete settings.autoAcceptDrafts;
|
|
@@ -51,7 +51,8 @@ import {
|
|
|
51
51
|
DEFAULT_STALL_ESCALATION_REFIRES,
|
|
52
52
|
DEFAULT_TOKEN_LIMIT,
|
|
53
53
|
classifyImpossibleReason,
|
|
54
|
-
|
|
54
|
+
durableObjectionsForDisapproval,
|
|
55
|
+
extractPendingTasks,
|
|
55
56
|
isFullAuditObjective,
|
|
56
57
|
resolveEffectiveAggressiveSettings,
|
|
57
58
|
appendAuditLog,
|
|
@@ -1958,15 +1959,12 @@ async function retryStoredCompletionAudit(origin: CompletionAuditOrigin = "provi
|
|
|
1958
1959
|
const effectiveCap = resolveEffectiveAggressiveSettings(loadSettings(liveCtx.cwd));
|
|
1959
1960
|
const auditCap = effectiveCap.auditCap;
|
|
1960
1961
|
const trailingDisapprovals = countTrailingDisapprovals(history);
|
|
1961
|
-
const durableObjections = result.disapproved
|
|
1962
|
-
? (()
|
|
1963
|
-
const extracted = extractPendingTasks(sanitizeProviderAuditReport(result.output), 5);
|
|
1964
|
-
return extracted.length > 0
|
|
1965
|
-
? extracted
|
|
1966
|
-
: [`Review the latest auditor disapproval in ${activeGoalStatusCommand()}.`];
|
|
1967
|
-
})()
|
|
1962
|
+
const durableObjections = result.disapproved
|
|
1963
|
+
? durableObjectionsForDisapproval(sanitizeProviderAuditReport(result.output), activeGoalStatusCommand())
|
|
1968
1964
|
: [];
|
|
1969
|
-
|
|
1965
|
+
// v0.38.69 (Antigravity port): objections attach on EVERY disapproval —
|
|
1966
|
+
// the retry argues the objection, not generic effort.
|
|
1967
|
+
if (result.disapproved) {
|
|
1970
1968
|
appendLedger(liveCtx.cwd, "audit_objections_todo", {
|
|
1971
1969
|
goalId,
|
|
1972
1970
|
attemptId: claim.attemptId,
|
|
@@ -56,7 +56,8 @@ import {
|
|
|
56
56
|
DEFAULT_STALL_ESCALATION_REFIRES,
|
|
57
57
|
DEFAULT_TOKEN_LIMIT,
|
|
58
58
|
classifyImpossibleReason,
|
|
59
|
-
|
|
59
|
+
durableObjectionsForDisapproval,
|
|
60
|
+
extractPendingTasks,
|
|
60
61
|
isFullAuditObjective,
|
|
61
62
|
resolveEffectiveAggressiveSettings,
|
|
62
63
|
appendAuditLog,
|
|
@@ -274,7 +275,7 @@ import {
|
|
|
274
275
|
pushCapped as pushRepetitionCapped,
|
|
275
276
|
} from "../goal-loop-repetition.js";
|
|
276
277
|
import { buildStatusText, buildWidgetLines, type AuditDisplayProgress } from "../goal-loop-display.js";
|
|
277
|
-
import { buildFinalRepoStateLines, buildTerminalApprovalRender, compactCompletionSummary, compactTerminalCompletionSummary } from "../completion-summary.js";
|
|
278
|
+
import { buildFinalRepoStateLines, buildTerminalApprovalRender, compactCompletionSummary, compactTerminalCompletionSummary, completionSummaryDensityNote } from "../completion-summary.js";
|
|
278
279
|
import { persistApprovalRender, replayUndeliveredApprovalRenders } from "../approval-render-store.js";
|
|
279
280
|
import { resolveAuditorThinkingLevel } from "../auditor-thinking.js";
|
|
280
281
|
import {
|
|
@@ -838,6 +839,17 @@ function registerAgentTools(pi: any): void {
|
|
|
838
839
|
// v0.38.52: same for the agent-supplied gate inventory.
|
|
839
840
|
const sanitizedGroups = sanitizeFindingGroups(p.findingGroups);
|
|
840
841
|
const sanitizedGates = sanitizeGateRows(p.gateRows);
|
|
842
|
+
// v0.38.69 (Antigravity port): evidence-density lint — a claim with
|
|
843
|
+
// no path:line tokens and no gate rows still audits, but rides with
|
|
844
|
+
// a NOTE annotation (ledgered) so the terminal render falls back to
|
|
845
|
+
// recorded facts instead of rendering unverifiable prose as proof.
|
|
846
|
+
const densityNote = completionSummaryDensityNote(finalSummary, sanitizedGroups, sanitizedGates);
|
|
847
|
+
const claimedSummary = densityNote && finalSummary
|
|
848
|
+
? `${finalSummary.trimEnd()} — NOTE: ${densityNote}`
|
|
849
|
+
: finalSummary;
|
|
850
|
+
if (densityNote) {
|
|
851
|
+
appendLedger(ctx.cwd, "completion_summary_low_density", { excerpt: (finalSummary ?? "").slice(0, 240) });
|
|
852
|
+
}
|
|
841
853
|
// 2026-09-16 whole-work recap: a re-claim after an auditor disapproval
|
|
842
854
|
// is usually a delta-only repair note. Keep the FIRST claim's text so
|
|
843
855
|
// the approved terminal render still opens with the whole work; the
|
|
@@ -849,7 +861,7 @@ function registerAgentTools(pi: any): void {
|
|
|
849
861
|
? state.goal.completionSummary
|
|
850
862
|
: undefined;
|
|
851
863
|
const completionClaim = beginCompletionAudit(ctx, {
|
|
852
|
-
completionSummary:
|
|
864
|
+
completionSummary: claimedSummary,
|
|
853
865
|
verificationSummary: p.verificationSummary,
|
|
854
866
|
// v0.38.37: the deliberate non-do rides the pending claim into
|
|
855
867
|
// the terminal render — the only source the summary may cite.
|
|
@@ -870,7 +882,7 @@ function registerAgentTools(pi: any): void {
|
|
|
870
882
|
isError: true,
|
|
871
883
|
};
|
|
872
884
|
}
|
|
873
|
-
updateGoal({ pendingTasks: undefined, ...(
|
|
885
|
+
updateGoal({ pendingTasks: undefined, ...(claimedSummary ? { completionSummary: claimedSummary } : {}) }, ctx);
|
|
874
886
|
const auditGoal = state.goal;
|
|
875
887
|
if (!auditGoal) return staleToolResult();
|
|
876
888
|
const auditGoalId = auditGoal.id;
|
|
@@ -1976,15 +1988,11 @@ function registerAgentTools(pi: any): void {
|
|
|
1976
1988
|
const auditFeedbackTruncationHint = auditFeedbackIsFull
|
|
1977
1989
|
? ""
|
|
1978
1990
|
: `\n\nReport truncated at the configured limit. ${activeGoalStatusCommand()} shows the full report; change Audit feedback chars in /glla settings (0 = full report).`;
|
|
1979
|
-
const durableObjections = result.disapproved
|
|
1980
|
-
? (()
|
|
1981
|
-
const extracted = extractPendingTasks(safeAuditOutput, 5);
|
|
1982
|
-
return extracted.length > 0
|
|
1983
|
-
? extracted
|
|
1984
|
-
: [`Review the latest auditor disapproval in ${activeGoalStatusCommand()}.`];
|
|
1985
|
-
})()
|
|
1991
|
+
const durableObjections = result.disapproved
|
|
1992
|
+
? durableObjectionsForDisapproval(safeAuditOutput, activeGoalStatusCommand())
|
|
1986
1993
|
: [];
|
|
1987
|
-
|
|
1994
|
+
// v0.38.69 (Antigravity port): objections attach on EVERY disapproval.
|
|
1995
|
+
if (result.disapproved) {
|
|
1988
1996
|
// v0.36.0: every ordinary disapproval becomes the current durable
|
|
1989
1997
|
// TODO projection, not only the post-cap case. Replacing (rather than
|
|
1990
1998
|
// appending) the bounded list makes repeated identical reports
|
|
@@ -2325,6 +2333,51 @@ function registerAgentTools(pi: any): void {
|
|
|
2325
2333
|
storedResumeAt = p.resumeAt;
|
|
2326
2334
|
}
|
|
2327
2335
|
}
|
|
2336
|
+
// v0.38.69 (Antigravity port): mid-run interruption budget with
|
|
2337
|
+
// auto-default-and-log fallback. Antigravity's no-mid-run-questions
|
|
2338
|
+
// trick is structural — the run cannot block on a human — while
|
|
2339
|
+
// GLLA's was rhetorical (the continuation prompt says don't ask).
|
|
2340
|
+
// With decisionPauseBudget set, the (N+1)-th agent-authored decision
|
|
2341
|
+
// pause does NOT pause and does NOT abort the turn: the goal stays
|
|
2342
|
+
// active, the recommended option (or option 1 when none is
|
|
2343
|
+
// recommended) is adopted as a logged assumption, and the agent is
|
|
2344
|
+
// told to carry it into its completion recap's Left out. Unset
|
|
2345
|
+
// budget = legacy pause-every-time; 0 = relentless from the start.
|
|
2346
|
+
if (p.kind === "decision" && p.options && p.options.length > 0) {
|
|
2347
|
+
const budgetRaw = loadSettings(ctx.cwd).decisionPauseBudget;
|
|
2348
|
+
const budget = typeof budgetRaw === "number" && Number.isInteger(budgetRaw) && budgetRaw >= 0 ? budgetRaw : undefined;
|
|
2349
|
+
const decisionCount = (state.goal.midRunDecisionCount ?? 0) + 1;
|
|
2350
|
+
if (budget !== undefined && decisionCount > budget) {
|
|
2351
|
+
const recIdx = p.recommended && p.recommended >= 1 && p.recommended <= p.options.length
|
|
2352
|
+
? Math.floor(p.recommended) - 1
|
|
2353
|
+
: 0;
|
|
2354
|
+
const chosen = p.options[recIdx]!;
|
|
2355
|
+
const entry = {
|
|
2356
|
+
at: new Date().toISOString(),
|
|
2357
|
+
reason: (p.reason ?? "").slice(0, 300),
|
|
2358
|
+
chosen: chosen.slice(0, 300),
|
|
2359
|
+
options: p.options.map((o) => o.slice(0, 200)).slice(0, 10),
|
|
2360
|
+
};
|
|
2361
|
+
updateGoal({
|
|
2362
|
+
midRunDecisionCount: decisionCount,
|
|
2363
|
+
autoDefaultLog: [...(state.goal.autoDefaultLog ?? []), entry].slice(-20),
|
|
2364
|
+
}, ctx);
|
|
2365
|
+
appendLedger(ctx.cwd, "decision_budget_auto_default", {
|
|
2366
|
+
goalId: state.goal.id,
|
|
2367
|
+
budget,
|
|
2368
|
+
decisionCount,
|
|
2369
|
+
chosen,
|
|
2370
|
+
reason: p.reason,
|
|
2371
|
+
});
|
|
2372
|
+
return {
|
|
2373
|
+
content: [{
|
|
2374
|
+
type: "text",
|
|
2375
|
+
text: `Decision budget exhausted (budget ${budget}, this is mid-run decision #${decisionCount}): NOT paused — proceeding with the recommended default, option ${recIdx + 1}: "${chosen}". This assumption is logged on the goal (autoDefaultLog) and ledgered as decision_budget_auto_default. Keep working — record it in your completion summary's Left out: "assumed ${chosen} for: ${(p.reason ?? "").slice(0, 160)}".`,
|
|
2376
|
+
}],
|
|
2377
|
+
details: {},
|
|
2378
|
+
};
|
|
2379
|
+
}
|
|
2380
|
+
}
|
|
2328
2381
|
updateGoal({
|
|
2329
2382
|
status: "paused",
|
|
2330
2383
|
pauseReason: safePauseReason,
|
|
@@ -2333,6 +2386,12 @@ function registerAgentTools(pi: any): void {
|
|
|
2333
2386
|
pauseOptions: p.kind === "decision" && p.options && p.options.length > 0 ? p.options : undefined,
|
|
2334
2387
|
pauseRecommended: p.kind === "decision" && p.recommended && p.recommended >= 1 ? Math.floor(p.recommended) : undefined,
|
|
2335
2388
|
pauseResumeAt: storedResumeAt,
|
|
2389
|
+
// v0.38.69 (Antigravity port): every agent-authored decision
|
|
2390
|
+
// pause counts against the mid-run interruption budget, in-budget
|
|
2391
|
+
// or not, so the counter survives reloads on the goal itself.
|
|
2392
|
+
...(p.kind === "decision" && p.options && p.options.length > 0
|
|
2393
|
+
? { midRunDecisionCount: (state.goal.midRunDecisionCount ?? 0) + 1 }
|
|
2394
|
+
: {}),
|
|
2336
2395
|
}, ctx);
|
|
2337
2396
|
if (p.kind === "decision" && p.options && p.options.length > 0) maybeDecisionPopup(ctx);
|
|
2338
2397
|
// v0.27.1: surface the FULL pause contract — reason AND suggested
|
|
@@ -5,6 +5,8 @@
|
|
|
5
5
|
// configured candidates, recognizes positively non-recoverable failures, and
|
|
6
6
|
// computes a bounded-but-persistent retry cadence.
|
|
7
7
|
|
|
8
|
+
import { DEFAULT_QUOTA_RETRY_SEC, parseQuotaError, quotaSignal } from "./quota-retry.js";
|
|
9
|
+
|
|
8
10
|
export const MAIN_MODEL_MAX_RETRY_DELAY_MS = 5 * 60 * 60_000;
|
|
9
11
|
export const MAIN_MODEL_AUTO_RETRY_HORIZON_MS = 24 * 60 * 60_000;
|
|
10
12
|
export const DEFAULT_MAIN_MODEL_PRIMARY_PROBE_MINUTES = 15;
|
|
@@ -291,16 +293,53 @@ export function hourAlignedRetryDelayMs(nowMs = Date.now()): number {
|
|
|
291
293
|
}
|
|
292
294
|
|
|
293
295
|
/** One uniform envelope for EVERY provider failure. Error text and upstream
|
|
294
|
-
* Retry-After prose are not trusted to choose a cadence
|
|
295
|
-
*
|
|
296
|
-
*
|
|
296
|
+
* Retry-After prose are not trusted to choose a cadence — EXCEPT one
|
|
297
|
+
* Antigravity-ported carve-out (v0.38.69): a quota-class failure
|
|
298
|
+
* (rate-limit/plan-quota signal) carrying an EXPLICIT upstream reset hint
|
|
299
|
+
* sleeps exactly until reset instead of laddering blindly into a known
|
|
300
|
+
* wall. The hint never widens the envelope (5h per-attempt cap), the eager
|
|
301
|
+
* first retry stays eager, and non-quota signals (transient, billing,
|
|
302
|
+
* unknown) keep the blind ladder — billing still heads for its park.
|
|
303
|
+
* Every other recoverable failure gets the same eager first retry, then
|
|
304
|
+
* the bounded configured ladder; the separate hourly retry adds the
|
|
305
|
+
* :00:30 slot. */
|
|
297
306
|
export function mainModelFailureDelayMs(failure: MainModelFailure, attempt: number, baseMinutes = 15, nowMs = Date.now()): number {
|
|
298
|
-
void failure;
|
|
299
|
-
void nowMs;
|
|
300
307
|
if (attempt <= 1) return 5_000;
|
|
308
|
+
const resetSleep = quotaResetSleepMs(failure, nowMs);
|
|
309
|
+
if (resetSleep !== undefined) return resetSleep;
|
|
301
310
|
return mainModelRetryDelayMs(attempt, baseMinutes);
|
|
302
311
|
}
|
|
303
312
|
|
|
313
|
+
/** v0.38.69 (Antigravity port): quota waits never park at the horizon.
|
|
314
|
+
* A rate-limit/plan-quota wall is transient by definition — agy
|
|
315
|
+
* busy-retries it until reset — so the 24h automatic-recovery hold must
|
|
316
|
+
* not end a quota wait. Billing (account wall, not a transient) and every
|
|
317
|
+
* non-quota failure keep their horizon park. Derived from the episode
|
|
318
|
+
* diagnostic text at the hold site, so no new durable state is needed and
|
|
319
|
+
* reloads classify identically. */
|
|
320
|
+
export function isQuotaHorizonExempt(raw: string | undefined): boolean {
|
|
321
|
+
if (typeof raw !== "string" || !raw.trim()) return false;
|
|
322
|
+
const signal = quotaSignal(raw);
|
|
323
|
+
return signal === "rate-limit" || signal === "plan-quota";
|
|
324
|
+
}
|
|
325
|
+
|
|
326
|
+
/** v0.38.69 (Antigravity port): sleep-until-reset for quota-class failures
|
|
327
|
+
* with an explicit upstream reset hint. Returns undefined (keep the blind
|
|
328
|
+
* ladder) unless ALL hold: a rate-limit/plan-quota signal, an upstream
|
|
329
|
+
* hint (header, JSON field, or explicit retry prose — never the silent
|
|
330
|
+
* fallback), and a finite non-negative window. The result is floored at
|
|
331
|
+
* the eager-retry quantum and capped at the per-attempt envelope, so a
|
|
332
|
+
* "retry in 1 week" wall sleeps 5h and re-evaluates instead of parking
|
|
333
|
+
* blind or sleeping unbounded. */
|
|
334
|
+
export function quotaResetSleepMs(failure: MainModelFailure, nowMs = Date.now()): number | undefined {
|
|
335
|
+
if (!failure || typeof failure.raw !== "string" || !failure.raw.trim()) return undefined;
|
|
336
|
+
const parsed = parseQuotaError(failure.raw, DEFAULT_QUOTA_RETRY_SEC, nowMs);
|
|
337
|
+
if (!parsed.fromUpstream) return undefined;
|
|
338
|
+
if (parsed.signal !== "rate-limit" && parsed.signal !== "plan-quota") return undefined;
|
|
339
|
+
if (!Number.isFinite(parsed.retryAfterSec) || parsed.retryAfterSec < 0) return undefined;
|
|
340
|
+
return Math.min(Math.max(Math.round(parsed.retryAfterSec * 1000), 5_000), MAIN_MODEL_MAX_RETRY_DELAY_MS);
|
|
341
|
+
}
|
|
342
|
+
|
|
304
343
|
/** Bound for positively-identified in-flight compaction. session_before_compact
|
|
305
344
|
* arms the marker and session_compact (or the next live host event) clears
|
|
306
345
|
* it; the cap only bounds a lost clear (cancelled compaction with no event,
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pi-goal-list-loop-audit",
|
|
3
|
-
"version": "0.38.
|
|
4
|
-
"description": "Mission control for
|
|
3
|
+
"version": "0.38.69",
|
|
4
|
+
"description": "Mission control for long running pi work: interview drafted goals, an audited task queue, and metric or audit loops that run for hours. Every completion is rechecked by a detached auditor with raw evidence, while Confirm drafts, decision pauses, and consent gates keep you in charge.",
|
|
5
5
|
"license": "AGPL-3.0-only",
|
|
6
6
|
"author": "dracon",
|
|
7
7
|
"type": "module",
|
package/schemas/goal.schema.json
CHANGED
|
@@ -59,6 +59,20 @@
|
|
|
59
59
|
"pauseKind": { "type": "string", "enum": ["decision", "error", "wait", "blocked", "standby"] },
|
|
60
60
|
"pauseOptions": { "type": "array", "items": { "type": "string" } },
|
|
61
61
|
"pauseRecommended": { "type": "number" },
|
|
62
|
+
"midRunDecisionCount": { "type": "number" },
|
|
63
|
+
"autoDefaultLog": {
|
|
64
|
+
"type": "array",
|
|
65
|
+
"maxItems": 20,
|
|
66
|
+
"items": {
|
|
67
|
+
"type": "object",
|
|
68
|
+
"properties": {
|
|
69
|
+
"at": { "type": "string" },
|
|
70
|
+
"reason": { "type": "string" },
|
|
71
|
+
"chosen": { "type": "string" },
|
|
72
|
+
"options": { "type": "array", "items": { "type": "string" } }
|
|
73
|
+
}
|
|
74
|
+
}
|
|
75
|
+
},
|
|
62
76
|
"pauseResumeAt": { "type": "string" },
|
|
63
77
|
"autoResumedAt": { "type": "string", "description": "v0.35.28 issue #16: ISO time glla auto-resumed a lapsed wait; renders the RECOVERY NOTICE in the continuation prompt; cleared by manual /goal resume" },
|
|
64
78
|
"autoResumedEvent": { "type": "string", "description": "v0.35.28 issue #16: what auto-resumed the goal (overdue wait route, provider recovery event)" },
|
|
@@ -8,9 +8,11 @@
|
|
|
8
8
|
*/
|
|
9
9
|
|
|
10
10
|
import { execFileSync } from "node:child_process";
|
|
11
|
+
import { createHash } from "node:crypto";
|
|
11
12
|
import * as fs from "node:fs";
|
|
12
13
|
import * as os from "node:os";
|
|
13
14
|
import * as path from "node:path";
|
|
15
|
+
import { pathToFileURL } from "node:url";
|
|
14
16
|
import { createJiti } from "jiti";
|
|
15
17
|
|
|
16
18
|
const repoRoot = path.resolve(import.meta.dirname, "..");
|
|
@@ -80,7 +82,6 @@ try {
|
|
|
80
82
|
tarball,
|
|
81
83
|
"--ignore-scripts",
|
|
82
84
|
"--omit=dev",
|
|
83
|
-
"--legacy-peer-deps",
|
|
84
85
|
"--no-save",
|
|
85
86
|
"--prefix",
|
|
86
87
|
installPrefix,
|
|
@@ -88,30 +89,66 @@ try {
|
|
|
88
89
|
|
|
89
90
|
const installedPackage = path.join(installPrefix, "node_modules", packageName);
|
|
90
91
|
if (!fs.existsSync(installedPackage)) throw new Error(`packed package was not installed at ${installedPackage}`);
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
"@earendil-works/pi-agent-core": path.join(nodeModules, "@earendil-works/pi-agent-core"),
|
|
96
|
-
"@earendil-works/pi-ai": path.join(nodeModules, "@earendil-works/pi-ai"),
|
|
97
|
-
"@earendil-works/pi-coding-agent": path.join(nodeModules, "@earendil-works/pi-coding-agent"),
|
|
98
|
-
"@earendil-works/pi-tui": path.join(nodeModules, "@earendil-works/pi-tui"),
|
|
99
|
-
// Audit 2026-09-06: the @tintinweb/pi-subagents alias was stale
|
|
100
|
-
// tintinweb-era drift — nothing imports the scoped name and the
|
|
101
|
-
// package depends on unscoped pi-subagents (which no extension
|
|
102
|
-
// imports directly, so no alias is needed).
|
|
103
|
-
// TypeBox exposes only an ESM `exports` entry; point Jiti at that
|
|
104
|
-
// concrete module because the disposable install intentionally omits
|
|
105
|
-
// peer dependencies.
|
|
106
|
-
typebox: path.join(nodeModules, "typebox", "build", "index.mjs"),
|
|
107
|
-
},
|
|
108
|
-
});
|
|
92
|
+
// Load the packed extension with the disposable install's own peer tree.
|
|
93
|
+
// Never alias imports back to this checkout: that masks a published peer
|
|
94
|
+
// declaration or an artifact-only module-resolution failure.
|
|
95
|
+
const jiti = createJiti(pathToFileURL(installedPackage).href, { moduleCache: false });
|
|
109
96
|
|
|
110
97
|
const activate = await jiti.import(path.join(installedPackage, "extensions/loops/goal.ts"), { default: true });
|
|
111
98
|
if (typeof activate !== "function") throw new Error("packed extension entry did not export a default activation function");
|
|
112
99
|
const auditor = await jiti.import(path.join(installedPackage, "extensions/goal-loop-auditor-process.ts"));
|
|
113
100
|
if (typeof auditor.resolveWorkerCommand !== "function") throw new Error("packed auditor process did not expose its worker command resolver");
|
|
114
101
|
if (auditor.resolveWorkerCommand("/usr/bin/node") !== "/usr/bin/node") throw new Error("packed auditor resolver returned an unexpected command");
|
|
102
|
+
const packedLauncher = await import(pathToFileURL(path.join(installedPackage, "scripts/goal-auditor-launch.mjs")).href);
|
|
103
|
+
if (typeof packedLauncher.buildAuditorPiSpawnSpec !== "function") throw new Error("packed launcher did not load");
|
|
104
|
+
if (typeof packedLauncher.renameWithWindowsRetry !== "function") throw new Error("packed launcher exports are incomplete");
|
|
105
|
+
|
|
106
|
+
// Tar-list presence is not worker coverage. Start the shipped worker from
|
|
107
|
+
// the installed tree with a tiny RPC stub, then require its real result.json
|
|
108
|
+
// protocol to complete within the smoke timeout.
|
|
109
|
+
const stableJson = (value) => {
|
|
110
|
+
if (value === null || typeof value !== "object") return JSON.stringify(value);
|
|
111
|
+
if (Array.isArray(value)) return `[${value.map(stableJson).join(",")}]`;
|
|
112
|
+
return `{${Object.keys(value).sort().map((key) => `${JSON.stringify(key)}:${stableJson(value[key])}`).join(",")}}`;
|
|
113
|
+
};
|
|
114
|
+
const attemptId = "packed-worker-probe";
|
|
115
|
+
const workerProbe = path.join(workspace, attemptId);
|
|
116
|
+
fs.mkdirSync(workerProbe, { recursive: true });
|
|
117
|
+
const request = {
|
|
118
|
+
protocolVersion: 1,
|
|
119
|
+
attemptId,
|
|
120
|
+
cwd: repoRoot,
|
|
121
|
+
prompt: "packed worker smoke probe",
|
|
122
|
+
model: "packed-probe/model",
|
|
123
|
+
thinkingLevel: "minimal",
|
|
124
|
+
};
|
|
125
|
+
request.requestHash = createHash("sha256").update(stableJson(request), "utf8").digest("hex");
|
|
126
|
+
fs.writeFileSync(path.join(workerProbe, "request.json"), `${JSON.stringify(request)}\n`);
|
|
127
|
+
fs.writeFileSync(path.join(workerProbe, "lock"), "{}\n");
|
|
128
|
+
const piStub = path.join(workspace, "pi-rpc-stub.mjs");
|
|
129
|
+
fs.writeFileSync(piStub, [
|
|
130
|
+
"#!/usr/bin/env node",
|
|
131
|
+
"process.stdout.write(JSON.stringify({type: 'message_update', assistantMessageEvent: {type: 'text_delta', delta: '<approved/>'}}) + '\\n');",
|
|
132
|
+
"process.stdout.write(JSON.stringify({type: 'agent_settled'}) + '\\n');",
|
|
133
|
+
].join("\n"));
|
|
134
|
+
fs.chmodSync(piStub, 0o755);
|
|
135
|
+
const workerPath = path.join(installedPackage, "scripts/goal-auditor-worker.mjs");
|
|
136
|
+
// The worker's bounded process-tree cleanup enumerates its own process
|
|
137
|
+
// group. Keep this probe in a detached group so a direct smoke invocation
|
|
138
|
+
// cannot signal the release gate's parent shell while terminating its RPC
|
|
139
|
+
// stub; the production launcher already supplies the same isolation.
|
|
140
|
+
execFileSync(process.execPath, [workerPath, "--job-dir", workerProbe], {
|
|
141
|
+
cwd: installedPackage,
|
|
142
|
+
detached: true,
|
|
143
|
+
env: { ...process.env, GLLA_PI_BINARY: piStub, GLLA_AUDITOR_STALL_MS: "1000", GLLA_AUDITOR_EOF_EXIT_GRACE_MS: "100" },
|
|
144
|
+
encoding: "utf8",
|
|
145
|
+
timeout: 15_000,
|
|
146
|
+
stdio: ["ignore", "pipe", "pipe"],
|
|
147
|
+
});
|
|
148
|
+
const workerResult = JSON.parse(fs.readFileSync(path.join(workerProbe, "result.json"), "utf8"));
|
|
149
|
+
if (workerResult.ok !== true || workerResult.output !== "<approved/>") throw new Error("packed worker did not complete its RPC probe");
|
|
150
|
+
console.log("OK: packed launcher loaded and worker completed its bounded RPC probe");
|
|
151
|
+
|
|
115
152
|
// Audit 2026-09-13: presence is not loadability — run the packed skill
|
|
116
153
|
// through Pi's own loader against the INSTALLED tree (the source-tree
|
|
117
154
|
// check in release-contract.test.ts cannot catch tarball-only defects).
|