pi-goal-list-loop-audit 0.35.4 → 0.35.13
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +7666 -0
- package/INSTALL.md +377 -0
- package/README.md +26 -13
- package/docs/DESIGN.md +31 -4
- package/docs/INDEX.md +32 -7
- package/examples/example-objective.md +119 -0
- package/extensions/faulty-objective-recovery.ts +13 -2
- package/extensions/goal-commands.ts +50 -6
- package/extensions/goal-heartbeat.ts +25 -21
- package/extensions/goal-loop-auditor-process.ts +184 -3
- package/extensions/goal-loop-core.ts +108 -15
- package/extensions/goal-loop-display.ts +82 -0
- package/extensions/goal-loop-shield.ts +55 -0
- package/extensions/goal-recovery.ts +317 -7
- package/extensions/goal-settings.ts +31 -1
- package/extensions/loops/goal-activation.ts +74 -7
- package/extensions/loops/goal-auditor-hooks.ts +30 -55
- package/extensions/loops/goal-list-queue.ts +18 -5
- package/extensions/loops/goal-orchestrator.ts +5 -0
- package/extensions/loops/goal-runtime-globals.ts +2 -0
- package/extensions/loops/goal-session.ts +101 -21
- package/extensions/loops/goal-settings-ui.ts +122 -63
- package/extensions/loops/goal-tools.ts +110 -36
- package/extensions/loops/goal-ui.ts +59 -2
- package/extensions/main-model-recovery.ts +24 -1
- package/extensions/model-picker.ts +3 -1
- package/extensions/model-selector.ts +2 -1
- package/extensions/multi-model-picker.ts +64 -7
- package/extensions/settings-menu.ts +16 -0
- package/package.json +4 -1
- package/prompts/goal-loop-continuation.md +11 -8
- package/prompts/goal-loop-draft.md +5 -5
- package/schemas/goal.schema.json +17 -0
|
@@ -95,6 +95,8 @@ export interface Task {
|
|
|
95
95
|
status: "pending" | "in_progress" | "complete";
|
|
96
96
|
/** Optional specialist hand-off for this task. */
|
|
97
97
|
agentRole?: AgentRole;
|
|
98
|
+
/** Optional verification gate for milestone-checked tasks. */
|
|
99
|
+
verificationContract?: string;
|
|
98
100
|
subtasks?: Task[];
|
|
99
101
|
}
|
|
100
102
|
|
|
@@ -117,6 +119,7 @@ export const MAX_SUBTASKS_PER_TASK = 5;
|
|
|
117
119
|
export interface TaskProposal {
|
|
118
120
|
title: string;
|
|
119
121
|
agentRole?: AgentRole;
|
|
122
|
+
verificationContract?: string;
|
|
120
123
|
subtasks?: string[];
|
|
121
124
|
}
|
|
122
125
|
|
|
@@ -145,6 +148,7 @@ export function buildTaskList(tasks: TaskProposal[]): TaskList {
|
|
|
145
148
|
title: t.title.trim(),
|
|
146
149
|
status: "pending" as const,
|
|
147
150
|
...(t.agentRole ? { agentRole: t.agentRole } : {}),
|
|
151
|
+
...(t.verificationContract ? { verificationContract: t.verificationContract } : {}),
|
|
148
152
|
subtasks: (t.subtasks ?? []).map((s, j) => ({
|
|
149
153
|
id: `${i + 1}.${j + 1}`,
|
|
150
154
|
title: s.trim(),
|
|
@@ -651,6 +655,9 @@ export interface ListItem {
|
|
|
651
655
|
parentId?: string;
|
|
652
656
|
/** Durable link from a repair/replan queue item back to the malformed item. */
|
|
653
657
|
repairTarget?: ObjectiveRepairTarget;
|
|
658
|
+
/** Monotonic queue position persisted with the sidecar. Legacy items omit it
|
|
659
|
+
* and fall back to addedAt/id ordering during recovery. */
|
|
660
|
+
queueOrder?: number;
|
|
654
661
|
addedAt: string;
|
|
655
662
|
}
|
|
656
663
|
|
|
@@ -738,6 +745,10 @@ export interface MainModelRecovery {
|
|
|
738
745
|
attempted: string[];
|
|
739
746
|
/** Next safe probe time; persisted so reloads do not forget the wait. */
|
|
740
747
|
retryAt?: string;
|
|
748
|
+
/** Next preferred-primary health probe while a fallback is serving. */
|
|
749
|
+
primaryProbeAt?: string;
|
|
750
|
+
/** A preferred-primary switch was accepted and awaits one supervised turn. */
|
|
751
|
+
primaryProbeInFlight?: boolean;
|
|
741
752
|
/** Number of completed recovery waits; drives the bounded exponential cadence. */
|
|
742
753
|
attempts: number;
|
|
743
754
|
/** Human-readable provider failure excerpt. */
|
|
@@ -787,6 +798,8 @@ export function formatMainModelRecoveryStatus(recovery: MainModelRecovery | unde
|
|
|
787
798
|
` Skipped: ${skipped}`,
|
|
788
799
|
];
|
|
789
800
|
if (recovery.retryAt) lines.push(` Retry at: ${recovery.retryAt}`);
|
|
801
|
+
if (recovery.primaryProbeAt) lines.push(` Preferred-primary probe at: ${recovery.primaryProbeAt}`);
|
|
802
|
+
if (recovery.primaryProbeInFlight) lines.push(" Preferred-primary probe: supervised turn pending");
|
|
790
803
|
if (recovery.manualResumeRequired) lines.push(" Automatic probes: stopped; explicit resume required");
|
|
791
804
|
return lines;
|
|
792
805
|
}
|
|
@@ -839,6 +852,8 @@ export function sanitizeMainModelRecovery(value: unknown): MainModelRecovery | u
|
|
|
839
852
|
...(bounded(raw.recoveryEpisodeKey, 300) ? { recoveryEpisodeKey: bounded(raw.recoveryEpisodeKey, 300) } : {}),
|
|
840
853
|
...(Array.isArray(raw.recoveryNoticeKeys) ? { recoveryNoticeKeys: raw.recoveryNoticeKeys.filter((key): key is string => typeof key === "string").slice(-16).map((key) => key.slice(0, 300)) } : {}),
|
|
841
854
|
...(date(raw.retryAt) ? { retryAt: date(raw.retryAt) } : {}),
|
|
855
|
+
...(date(raw.primaryProbeAt) ? { primaryProbeAt: date(raw.primaryProbeAt) } : {}),
|
|
856
|
+
...(raw.primaryProbeInFlight === true ? { primaryProbeInFlight: true } : {}),
|
|
842
857
|
...(date(raw.firstFailureAt) ? { firstFailureAt: date(raw.firstFailureAt) } : {}),
|
|
843
858
|
...(date(raw.autoRetryUntil) ? { autoRetryUntil: date(raw.autoRetryUntil) } : {}),
|
|
844
859
|
...(raw.manualResumeRequired === true ? { manualResumeRequired: true } : {}),
|
|
@@ -964,16 +979,46 @@ export function queueItemPath(cwd: string, id: string): string {
|
|
|
964
979
|
return path.join(piGlaDir(cwd), "goals", `${id}.queue.json`);
|
|
965
980
|
}
|
|
966
981
|
|
|
967
|
-
export
|
|
982
|
+
export interface QueueItemWriteResult {
|
|
983
|
+
path: string;
|
|
984
|
+
wrote: boolean;
|
|
985
|
+
/** True when the persistence boundary rejected the write. A collision or
|
|
986
|
+
* symlink refusal is a normal wrote=false result, not a persistence error. */
|
|
987
|
+
failed?: boolean;
|
|
988
|
+
}
|
|
989
|
+
|
|
990
|
+
/** Write one queue sidecar atomically. The default is idempotent/no-overwrite;
|
|
991
|
+
* repair metadata updates may explicitly request an atomic replacement. All
|
|
992
|
+
* filesystem failures go through the persistence-degradation boundary and
|
|
993
|
+
* clean up the temporary file before returning to the caller. */
|
|
994
|
+
export function writeQueueItemFile(cwd: string, item: ListItem, options: { replace?: boolean } = {}): QueueItemWriteResult {
|
|
968
995
|
const file = queueItemPath(cwd, item.id);
|
|
969
|
-
|
|
970
|
-
|
|
996
|
+
const replace = options.replace === true;
|
|
997
|
+
if (!replace && fs.existsSync(file)) return { path: file, wrote: false }; // idempotent — never overwrite
|
|
998
|
+
const result = runPersistStep("writeQueueItemFile", () => {
|
|
999
|
+
if (replace) {
|
|
1000
|
+
try {
|
|
1001
|
+
if (fs.lstatSync(file).isSymbolicLink()) return { path: file, wrote: false, failed: true };
|
|
1002
|
+
} catch (err) {
|
|
1003
|
+
if ((err as NodeJS.ErrnoException).code !== "ENOENT") throw err;
|
|
1004
|
+
}
|
|
1005
|
+
}
|
|
971
1006
|
fs.mkdirSync(path.dirname(file), { recursive: true });
|
|
972
|
-
|
|
973
|
-
|
|
974
|
-
|
|
975
|
-
|
|
976
|
-
|
|
1007
|
+
const tempPath = `${file}.${process.pid}.${Date.now()}.${Math.random().toString(36).slice(2, 8)}.tmp`;
|
|
1008
|
+
try {
|
|
1009
|
+
fs.writeFileSync(tempPath, JSON.stringify({ schema: 1, type: "queue-item", ...item }), "utf-8");
|
|
1010
|
+
fs.renameSync(tempPath, file);
|
|
1011
|
+
return { path: file, wrote: true };
|
|
1012
|
+
} finally {
|
|
1013
|
+
try {
|
|
1014
|
+
if (fs.existsSync(tempPath)) fs.unlinkSync(tempPath);
|
|
1015
|
+
} catch {
|
|
1016
|
+
// The landing write already succeeded or the original error is more
|
|
1017
|
+
// useful; a later cleanup pass can remove an orphaned temp file.
|
|
1018
|
+
}
|
|
1019
|
+
}
|
|
1020
|
+
});
|
|
1021
|
+
return result ?? { path: file, wrote: false, failed: true };
|
|
977
1022
|
}
|
|
978
1023
|
|
|
979
1024
|
export function deleteQueueItemFile(cwd: string, id: string): boolean {
|
|
@@ -1022,8 +1067,8 @@ export function readQueueFromDisk(cwd: string, excludeIds: ReadonlySet<string> =
|
|
|
1022
1067
|
const dir = path.join(piGlaDir(cwd), "goals");
|
|
1023
1068
|
let names: string[];
|
|
1024
1069
|
try { names = fs.readdirSync(dir); } catch { return []; }
|
|
1025
|
-
// readdir order is filesystem-dependent;
|
|
1026
|
-
//
|
|
1070
|
+
// readdir order is filesystem-dependent; parse every sidecar first and
|
|
1071
|
+
// apply compareQueueItems below so durable queue order survives recovery.
|
|
1027
1072
|
names.sort();
|
|
1028
1073
|
const out: ListItem[] = [];
|
|
1029
1074
|
for (const name of names) {
|
|
@@ -1035,6 +1080,20 @@ export function readQueueFromDisk(cwd: string, excludeIds: ReadonlySet<string> =
|
|
|
1035
1080
|
if (!e || e.schema !== 1 || e.type !== "queue-item") continue;
|
|
1036
1081
|
if (typeof e.id !== "string" || typeof e.objective !== "string") continue;
|
|
1037
1082
|
if (excludeIds.has(e.id)) continue;
|
|
1083
|
+
const repairTarget = e.repairTarget && typeof e.repairTarget === "object"
|
|
1084
|
+
&& typeof e.repairTarget.id === "string"
|
|
1085
|
+
&& typeof e.repairTarget.objective === "string"
|
|
1086
|
+
&& typeof e.repairTarget.source === "string"
|
|
1087
|
+
&& Array.isArray(e.repairTarget.reasons)
|
|
1088
|
+
&& e.repairTarget.reasons.every((reason: unknown) => typeof reason === "string")
|
|
1089
|
+
? {
|
|
1090
|
+
id: e.repairTarget.id,
|
|
1091
|
+
objective: e.repairTarget.objective,
|
|
1092
|
+
...(typeof e.repairTarget.verificationContract === "string" ? { verificationContract: e.repairTarget.verificationContract } : {}),
|
|
1093
|
+
reasons: e.repairTarget.reasons,
|
|
1094
|
+
source: e.repairTarget.source,
|
|
1095
|
+
} satisfies ObjectiveRepairTarget
|
|
1096
|
+
: undefined;
|
|
1038
1097
|
out.push({
|
|
1039
1098
|
id: e.id,
|
|
1040
1099
|
objective: e.objective,
|
|
@@ -1044,10 +1103,12 @@ export function readQueueFromDisk(cwd: string, excludeIds: ReadonlySet<string> =
|
|
|
1044
1103
|
// v0.34.81: subtask binding round-trips from the sidecar the same way
|
|
1045
1104
|
// parallelSafe does — must be a string id matching another queue item.
|
|
1046
1105
|
...(typeof e.parentId === "string" && e.parentId ? { parentId: e.parentId } : {}),
|
|
1106
|
+
...(repairTarget ? { repairTarget } : {}),
|
|
1107
|
+
...(typeof e.queueOrder === "number" && Number.isFinite(e.queueOrder) && e.queueOrder >= 0 ? { queueOrder: e.queueOrder } : {}),
|
|
1047
1108
|
addedAt: typeof e.addedAt === "string" ? e.addedAt : new Date().toISOString(),
|
|
1048
1109
|
});
|
|
1049
1110
|
}
|
|
1050
|
-
return out;
|
|
1111
|
+
return out.sort(compareQueueItems);
|
|
1051
1112
|
}
|
|
1052
1113
|
|
|
1053
1114
|
// =================================================================
|
|
@@ -1432,6 +1493,29 @@ export function nowIso(): string {
|
|
|
1432
1493
|
return new Date().toISOString();
|
|
1433
1494
|
}
|
|
1434
1495
|
|
|
1496
|
+
/** Stable queue ordering: new sidecars use queueOrder; legacy sidecars fall
|
|
1497
|
+
* back to their durable timestamp and id instead of filesystem enumeration. */
|
|
1498
|
+
export function compareQueueItems(a: Pick<ListItem, "id" | "addedAt" | "queueOrder">, b: Pick<ListItem, "id" | "addedAt" | "queueOrder">): number {
|
|
1499
|
+
const aOrder = typeof a.queueOrder === "number" && Number.isFinite(a.queueOrder) ? a.queueOrder : undefined;
|
|
1500
|
+
const bOrder = typeof b.queueOrder === "number" && Number.isFinite(b.queueOrder) ? b.queueOrder : undefined;
|
|
1501
|
+
if (aOrder !== undefined && bOrder !== undefined && aOrder !== bOrder) return aOrder - bOrder;
|
|
1502
|
+
const added = a.addedAt.localeCompare(b.addedAt);
|
|
1503
|
+
if (added !== 0) return added;
|
|
1504
|
+
if (aOrder !== undefined && bOrder === undefined) return -1;
|
|
1505
|
+
if (aOrder === undefined && bOrder !== undefined) return 1;
|
|
1506
|
+
return a.id.localeCompare(b.id);
|
|
1507
|
+
}
|
|
1508
|
+
|
|
1509
|
+
/** Assign durable positions to newly enqueued items. Existing positions are
|
|
1510
|
+
* retained, so a reload/recovery cannot reorder a confirmed batch. */
|
|
1511
|
+
export function assignQueueOrder<T extends ListItem>(items: readonly T[], existing: readonly ListItem[] = []): T[] {
|
|
1512
|
+
let next = existing.reduce((max, item) => {
|
|
1513
|
+
const order = typeof item.queueOrder === "number" && Number.isFinite(item.queueOrder) ? item.queueOrder : -1;
|
|
1514
|
+
return Math.max(max, order);
|
|
1515
|
+
}, -1) + 1;
|
|
1516
|
+
return items.map((item) => item.queueOrder === undefined ? { ...item, queueOrder: next++ } : item);
|
|
1517
|
+
}
|
|
1518
|
+
|
|
1435
1519
|
/**
|
|
1436
1520
|
* v0.34.59: revision token — return {goalId, revision} for use as a
|
|
1437
1521
|
* (focus-token, sandbox-check) at async boundaries. The orchestrator
|
|
@@ -1547,6 +1631,9 @@ export const LONG_RUNNING_JUDGMENT_POLICY = `LONG-RUNNING JUDGMENT POLICY:
|
|
|
1547
1631
|
- Preserve the objective and verification contract as the source of truth. The default answer is the durable, maintainable root-cause fix — decide it and proceed; do not stop at a cosmetic workaround merely because it is faster.
|
|
1548
1632
|
- "Band-aid now vs do it proper" is NEVER a question: when the durable fix is the clearly best call, do it — no ask_user_question, no pause_goal, no "which do you prefer" framing. Record the choice and the reasoning in the turn.
|
|
1549
1633
|
- Use an opportunistic workaround only when the durable fix is genuinely unsafe, impossible, or blocked right now; the workaround must be reversible and testable, and its durable follow-up is recorded (ledger or comment) instead of silently treated as final.
|
|
1634
|
+
- Premium engineering standards are mandatory: code must be cleanly typed, tested, architecturally sound, and resilient across lifecycle boundaries. Never lower test standards, fake assertions, or bypass types.
|
|
1635
|
+
- Autonomous pivot strategy: if an implementation approach fails verification after 2 attempts, do not loop on the same failing line. Autonomously step back, diagnose the root invariant, and pivot to a clean alternative architecture.
|
|
1636
|
+
- Non-interruption & sensible defaults: never pause a multi-hour run for obvious choices, cosmetic naming, or non-blocking secondary questions. Pick the sensible architectural default, implement it, record the rationale, and continue. Defer non-blocking notes to the final completion summary.
|
|
1550
1637
|
- Decide autonomously through local implementation choices without interrupting the user. Ask one focused question ONLY at a genuine trade-off where the user's preference materially changes the outcome: an irreversible/destructive external action, a missing permission/credential, or two options with comparable real cost.
|
|
1551
1638
|
- In unattended mode, choose the safest contract-preserving path and continue. If no safe choice exists, raise a concrete DECIDE question with a recommended default; never ask a vague progress question or wait on a guessed provider/quota reset.`;
|
|
1552
1639
|
|
|
@@ -1722,9 +1809,13 @@ export function extractVerificationContract(raw: string): { objective: string; v
|
|
|
1722
1809
|
|
|
1723
1810
|
// Inline fallback: users write one-liners like
|
|
1724
1811
|
// "Create x.txt. Done when: grep -q ok x.txt"
|
|
1725
|
-
// where the marker is mid-line. Split at the first inline marker.
|
|
1812
|
+
// where the marker is mid-line. Split at the first inline marker. Keep
|
|
1813
|
+
// bare `verify` out of the broad marker alternative: it is also a normal
|
|
1814
|
+
// imperative (`Run the audit and verify ... Done when: ...`) and must not
|
|
1815
|
+
// truncate the objective before the actual Done when marker. `Verify:`
|
|
1816
|
+
// remains supported as the explicit short marker form.
|
|
1726
1817
|
if (!verificationContract) {
|
|
1727
|
-
const m = raw.match(/^(.*?)(?:\.|;)
|
|
1818
|
+
const m = raw.match(/^(.*?)(?:\.|;)?\s+(?:(?:done when|verified when|verification)\b[^:]*:|verify\s*:)\s*(.+)$/is);
|
|
1728
1819
|
if (m) {
|
|
1729
1820
|
objective = (m[1] ?? "").trim().replace(/[.;]\s*$/, "");
|
|
1730
1821
|
verificationContract = (m[2] ?? "").trim();
|
|
@@ -2189,10 +2280,12 @@ export function formatAuditLog(entries: AuditLogEntry[]): string {
|
|
|
2189
2280
|
// =================================================================
|
|
2190
2281
|
|
|
2191
2282
|
/** Which auditor infra errors are worth an automatic retry? User aborts
|
|
2192
|
-
* and missing-model config are NOT — retrying can't help them.
|
|
2283
|
+
* and missing-model config are NOT — retrying can't help them. Timeouts and
|
|
2284
|
+
* watchdog stalls are retriable infrastructure failures. */
|
|
2193
2285
|
export function isRetriableInfraError(error?: string): boolean {
|
|
2194
2286
|
if (!error) return false;
|
|
2195
|
-
if (/
|
|
2287
|
+
if (/^(?:Auditor (?:exceeded|stalled)|.*(?:timed?\s*out|timeout|inactivity))/i.test(error)) return true;
|
|
2288
|
+
if (/^(?:Auditor aborted\.?$|user (?:interrupt|abort)|cancelled by user)/i.test(error.trim())) return false;
|
|
2196
2289
|
if (/no (?:auditor )?model/i.test(error)) return false;
|
|
2197
2290
|
return true;
|
|
2198
2291
|
}
|
|
@@ -96,6 +96,23 @@ export interface RecentActionDisplay {
|
|
|
96
96
|
/** v0.33.0: widget extras — the refire streak plus the recent-action feed. */
|
|
97
97
|
export type GoalDisplayActivity = "active" | "awaiting-first-turn" | "working" | "busy" | "queued" | "idle";
|
|
98
98
|
|
|
99
|
+
export interface ModelProvenanceDisplay {
|
|
100
|
+
/** The primary model reference selected for the supervised work. */
|
|
101
|
+
primary?: string;
|
|
102
|
+
/** Whether the primary is an explicit pin or inherited from the session. */
|
|
103
|
+
primarySource?: "pinned" | "inherited";
|
|
104
|
+
/** Ordered backup references available to the main-model recovery policy. */
|
|
105
|
+
fallbackRefs?: string[];
|
|
106
|
+
/** References rejected by the explicit forbidden-model gate. */
|
|
107
|
+
skippedForbiddenRefs?: string[];
|
|
108
|
+
/** Model reference observed at the current main-host turn boundary. */
|
|
109
|
+
handledTurn?: string;
|
|
110
|
+
/** Model reference currently executing the detached audit, if any. */
|
|
111
|
+
handledAudit?: string;
|
|
112
|
+
/** Candidate provenance for the detached audit (setting/fallback/session). */
|
|
113
|
+
handledAuditSource?: string;
|
|
114
|
+
}
|
|
115
|
+
|
|
99
116
|
export interface WidgetExtras {
|
|
100
117
|
stalls?: number;
|
|
101
118
|
recent?: RecentActionDisplay[];
|
|
@@ -121,6 +138,8 @@ export interface WidgetExtras {
|
|
|
121
138
|
auditorProgressSignals?: boolean;
|
|
122
139
|
/** Effective global main-model backup order for truthful recovery HUDs. */
|
|
123
140
|
mainModelFallbacks?: string[];
|
|
141
|
+
/** Truthful model-selection provenance for the active goal card/footer. */
|
|
142
|
+
modelProvenance?: ModelProvenanceDisplay;
|
|
124
143
|
}
|
|
125
144
|
|
|
126
145
|
/**
|
|
@@ -181,6 +200,57 @@ function budgetFor(width: number | undefined, prefixCols: number, floor: number)
|
|
|
181
200
|
return Math.max(floor, width - WIDGET_HORIZONTAL_MARGIN - prefixCols);
|
|
182
201
|
}
|
|
183
202
|
|
|
203
|
+
function uniqueModelRefs(refs: readonly string[] | undefined): string[] {
|
|
204
|
+
const out: string[] = [];
|
|
205
|
+
const seen = new Set<string>();
|
|
206
|
+
for (const value of refs ?? []) {
|
|
207
|
+
if (typeof value !== "string") continue;
|
|
208
|
+
const ref = value.trim();
|
|
209
|
+
if (!ref || seen.has(ref.toLowerCase())) continue;
|
|
210
|
+
seen.add(ref.toLowerCase());
|
|
211
|
+
out.push(ref);
|
|
212
|
+
}
|
|
213
|
+
return out;
|
|
214
|
+
}
|
|
215
|
+
|
|
216
|
+
/**
|
|
217
|
+
* Keep model selection provenance on the same always-visible card as the
|
|
218
|
+
* goal. These are projections of observed/configured refs, not a guess from
|
|
219
|
+
* elapsed time: the primary source is explicit, forbidden refs stay visible
|
|
220
|
+
* as skipped, and the handled refs are separate from the configured order.
|
|
221
|
+
*/
|
|
222
|
+
function modelSourceLabel(source: string): string {
|
|
223
|
+
const key = source.trim().toLowerCase();
|
|
224
|
+
if (key === "setting" || key === "pinned" || key === "auditor-pin") return "pinned";
|
|
225
|
+
if (key === "session" || key === "session-fallback") return "inherited from session";
|
|
226
|
+
if (key.includes("fallback")) return "fallback";
|
|
227
|
+
return source.trim();
|
|
228
|
+
}
|
|
229
|
+
|
|
230
|
+
function modelProvenanceLines(provenance: ModelProvenanceDisplay | undefined, width?: number): string[] {
|
|
231
|
+
if (!provenance) return [];
|
|
232
|
+
const budget = budgetFor(width, 3, 60);
|
|
233
|
+
const lines: string[] = [];
|
|
234
|
+
const primary = typeof provenance.primary === "string" ? provenance.primary.trim() : "";
|
|
235
|
+
if (primary) {
|
|
236
|
+
const source = provenance.primarySource === "pinned" ? "pinned" : "inherited from session";
|
|
237
|
+
lines.push(`model: primary ${truncate(primary, budget)} · ${source}`);
|
|
238
|
+
}
|
|
239
|
+
const fallbacks = uniqueModelRefs(provenance.fallbackRefs);
|
|
240
|
+
if (fallbacks.length > 0) lines.push(`fallbacks: ${truncate(fallbacks.join(" → "), budget)}`);
|
|
241
|
+
const skipped = uniqueModelRefs(provenance.skippedForbiddenRefs);
|
|
242
|
+
if (skipped.length > 0) lines.push(`skipped forbidden: ${truncate(skipped.join(", "), budget)}`);
|
|
243
|
+
const handledTurn = typeof provenance.handledTurn === "string" ? provenance.handledTurn.trim() : "";
|
|
244
|
+
if (handledTurn) lines.push(`handled turn: ${truncate(handledTurn, budget)}`);
|
|
245
|
+
const handledAudit = typeof provenance.handledAudit === "string" ? provenance.handledAudit.trim() : "";
|
|
246
|
+
if (handledAudit) {
|
|
247
|
+
const source = provenance.handledAuditSource?.trim() ? modelSourceLabel(provenance.handledAuditSource) : "";
|
|
248
|
+
const via = source ? ` · via ${truncate(source, 24)}` : "";
|
|
249
|
+
lines.push(`handled audit: ${truncate(handledAudit, Math.max(16, budget - via.length))}${via}`);
|
|
250
|
+
}
|
|
251
|
+
return lines;
|
|
252
|
+
}
|
|
253
|
+
|
|
184
254
|
// ---- semantic colors (optional; tests call without a theme → plain strings) ----
|
|
185
255
|
|
|
186
256
|
export type DisplayColor = "accent" | "success" | "warning" | "error" | "muted" | "dim";
|
|
@@ -351,6 +421,10 @@ function auditRecoveryPending(g: Goal): boolean {
|
|
|
351
421
|
// ---- status line (one-liner, always-on) ----
|
|
352
422
|
|
|
353
423
|
export interface AuditDisplayProgress {
|
|
424
|
+
/** Model reference selected for the currently running detached attempt. */
|
|
425
|
+
model?: string;
|
|
426
|
+
/** Candidate provenance: pinned setting, fallback pin, or session fallback. */
|
|
427
|
+
via?: string;
|
|
354
428
|
currentTool?: string;
|
|
355
429
|
/** JSON-safe tool arguments from the detached worker; display only a safe target summary. */
|
|
356
430
|
currentToolArgs?: string;
|
|
@@ -1114,6 +1188,14 @@ function goalLines(g: Goal, state: State, audit: AuditDisplayProgress | null | u
|
|
|
1114
1188
|
lines.push(`${i === 0 ? "├─" : "│ "} ${paint(theme, "dim", line.replace(/^Main-model recovery: /, ""))}`);
|
|
1115
1189
|
});
|
|
1116
1190
|
}
|
|
1191
|
+
// Model provenance is a card fact, not a notification: keep it visible
|
|
1192
|
+
// across active, interrupted, auditing, and paused branches. In
|
|
1193
|
+
// particular, never replace a configured forbidden ref with "none" — the
|
|
1194
|
+
// skipped line explains why it did not handle the turn.
|
|
1195
|
+
const provenance = modelProvenanceLines(extras?.modelProvenance, width);
|
|
1196
|
+
provenance.forEach((line, i) => {
|
|
1197
|
+
lines.push(`${i === 0 ? "├─" : "│ "} ${paint(theme, "dim", line)}`);
|
|
1198
|
+
});
|
|
1117
1199
|
if (interrupted) {
|
|
1118
1200
|
const resumeCmd = isList ? "/list resume" : "/goal resume";
|
|
1119
1201
|
if (interruptedForNoStart(g)) {
|
|
@@ -129,3 +129,58 @@ export function parseAuditorVerdict(output: string): { approved: boolean; disapp
|
|
|
129
129
|
impossibleReason: impossibleMatch?.[1]?.trim().slice(0, 300) || undefined,
|
|
130
130
|
};
|
|
131
131
|
}
|
|
132
|
+
|
|
133
|
+
/**
|
|
134
|
+
* v0.35.7: Extract mechanical shell command gates from a verification contract.
|
|
135
|
+
* Captures explicit commands (e.g. `npm test`, `tsc --noEmit`, `cargo test`)
|
|
136
|
+
* for deterministic fast-fail pre-auditing before spawning the heavy LLM worker.
|
|
137
|
+
*/
|
|
138
|
+
export function extractMechanicalCheckCommands(contract: string): string[] {
|
|
139
|
+
if (!contract) return [];
|
|
140
|
+
const items = contractItems(contract);
|
|
141
|
+
const commands: string[] = [];
|
|
142
|
+
for (const item of items) {
|
|
143
|
+
const backtickMatch = /`([^`]+)`/.exec(item);
|
|
144
|
+
let candidate = backtickMatch ? backtickMatch[1]!.trim() : item.trim();
|
|
145
|
+
if (!backtickMatch) {
|
|
146
|
+
candidate = candidate.replace(/\s+(?:passes(?:\s+cleanly|\s+with\s+zero\s+errors)?|exits\s+0|returns\s+0|cleanly).*$/i, "").trim();
|
|
147
|
+
}
|
|
148
|
+
if (/^(?:npm\s+(?:test|run\s+[\w:-]+)|bun\s+(?:test|run\s+[\w:-]+)|pnpm\s+(?:test|run\s+[\w:-]+)|yarn\s+(?:test|[\w:-]+)|tsc\b|cargo\s+(?:test|check|build)|pytest\b|python\s+-m\s+unittest|go\s+test|vitest\b|jest\b|make\s+test|git\s+diff|test\s+-[a-z])/i.test(candidate)) {
|
|
149
|
+
commands.push(candidate);
|
|
150
|
+
}
|
|
151
|
+
}
|
|
152
|
+
return commands;
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
export interface MechanicalCheckResult {
|
|
156
|
+
passed: boolean;
|
|
157
|
+
failedCommand?: string;
|
|
158
|
+
output?: string;
|
|
159
|
+
exitCode?: number;
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
/**
|
|
163
|
+
* v0.35.7: Execute mechanical pre-audit checks deterministically.
|
|
164
|
+
*/
|
|
165
|
+
export function runMechanicalPreAuditChecks(cwd: string, commands: string[], timeoutMs = 60000): MechanicalCheckResult {
|
|
166
|
+
if (!commands || commands.length === 0) return { passed: true };
|
|
167
|
+
const { execSync } = require("node:child_process");
|
|
168
|
+
for (const cmd of commands) {
|
|
169
|
+
try {
|
|
170
|
+
execSync(cmd, { cwd, timeout: timeoutMs, stdio: ["ignore", "pipe", "pipe"], encoding: "utf-8" });
|
|
171
|
+
} catch (err: any) {
|
|
172
|
+
const exitCode = typeof err.status === "number" ? err.status : (typeof err.code === "number" ? err.code : 1);
|
|
173
|
+
const stdout = err.stdout ? String(err.stdout) : "";
|
|
174
|
+
const stderr = err.stderr ? String(err.stderr) : "";
|
|
175
|
+
const output = (stdout + "\n" + stderr).trim() || err.message || "Command failed";
|
|
176
|
+
return {
|
|
177
|
+
passed: false,
|
|
178
|
+
failedCommand: cmd,
|
|
179
|
+
output: output.slice(0, 4000),
|
|
180
|
+
exitCode,
|
|
181
|
+
};
|
|
182
|
+
}
|
|
183
|
+
}
|
|
184
|
+
return { passed: true };
|
|
185
|
+
}
|
|
186
|
+
|