pi-goal-list-loop-audit 0.35.4 → 0.35.13

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -95,6 +95,8 @@ export interface Task {
95
95
  status: "pending" | "in_progress" | "complete";
96
96
  /** Optional specialist hand-off for this task. */
97
97
  agentRole?: AgentRole;
98
+ /** Optional verification gate for milestone-checked tasks. */
99
+ verificationContract?: string;
98
100
  subtasks?: Task[];
99
101
  }
100
102
 
@@ -117,6 +119,7 @@ export const MAX_SUBTASKS_PER_TASK = 5;
117
119
  export interface TaskProposal {
118
120
  title: string;
119
121
  agentRole?: AgentRole;
122
+ verificationContract?: string;
120
123
  subtasks?: string[];
121
124
  }
122
125
 
@@ -145,6 +148,7 @@ export function buildTaskList(tasks: TaskProposal[]): TaskList {
145
148
  title: t.title.trim(),
146
149
  status: "pending" as const,
147
150
  ...(t.agentRole ? { agentRole: t.agentRole } : {}),
151
+ ...(t.verificationContract ? { verificationContract: t.verificationContract } : {}),
148
152
  subtasks: (t.subtasks ?? []).map((s, j) => ({
149
153
  id: `${i + 1}.${j + 1}`,
150
154
  title: s.trim(),
@@ -651,6 +655,9 @@ export interface ListItem {
651
655
  parentId?: string;
652
656
  /** Durable link from a repair/replan queue item back to the malformed item. */
653
657
  repairTarget?: ObjectiveRepairTarget;
658
+ /** Monotonic queue position persisted with the sidecar. Legacy items omit it
659
+ * and fall back to addedAt/id ordering during recovery. */
660
+ queueOrder?: number;
654
661
  addedAt: string;
655
662
  }
656
663
 
@@ -738,6 +745,10 @@ export interface MainModelRecovery {
738
745
  attempted: string[];
739
746
  /** Next safe probe time; persisted so reloads do not forget the wait. */
740
747
  retryAt?: string;
748
+ /** Next preferred-primary health probe while a fallback is serving. */
749
+ primaryProbeAt?: string;
750
+ /** A preferred-primary switch was accepted and awaits one supervised turn. */
751
+ primaryProbeInFlight?: boolean;
741
752
  /** Number of completed recovery waits; drives the bounded exponential cadence. */
742
753
  attempts: number;
743
754
  /** Human-readable provider failure excerpt. */
@@ -787,6 +798,8 @@ export function formatMainModelRecoveryStatus(recovery: MainModelRecovery | unde
787
798
  ` Skipped: ${skipped}`,
788
799
  ];
789
800
  if (recovery.retryAt) lines.push(` Retry at: ${recovery.retryAt}`);
801
+ if (recovery.primaryProbeAt) lines.push(` Preferred-primary probe at: ${recovery.primaryProbeAt}`);
802
+ if (recovery.primaryProbeInFlight) lines.push(" Preferred-primary probe: supervised turn pending");
790
803
  if (recovery.manualResumeRequired) lines.push(" Automatic probes: stopped; explicit resume required");
791
804
  return lines;
792
805
  }
@@ -839,6 +852,8 @@ export function sanitizeMainModelRecovery(value: unknown): MainModelRecovery | u
839
852
  ...(bounded(raw.recoveryEpisodeKey, 300) ? { recoveryEpisodeKey: bounded(raw.recoveryEpisodeKey, 300) } : {}),
840
853
  ...(Array.isArray(raw.recoveryNoticeKeys) ? { recoveryNoticeKeys: raw.recoveryNoticeKeys.filter((key): key is string => typeof key === "string").slice(-16).map((key) => key.slice(0, 300)) } : {}),
841
854
  ...(date(raw.retryAt) ? { retryAt: date(raw.retryAt) } : {}),
855
+ ...(date(raw.primaryProbeAt) ? { primaryProbeAt: date(raw.primaryProbeAt) } : {}),
856
+ ...(raw.primaryProbeInFlight === true ? { primaryProbeInFlight: true } : {}),
842
857
  ...(date(raw.firstFailureAt) ? { firstFailureAt: date(raw.firstFailureAt) } : {}),
843
858
  ...(date(raw.autoRetryUntil) ? { autoRetryUntil: date(raw.autoRetryUntil) } : {}),
844
859
  ...(raw.manualResumeRequired === true ? { manualResumeRequired: true } : {}),
@@ -964,16 +979,46 @@ export function queueItemPath(cwd: string, id: string): string {
964
979
  return path.join(piGlaDir(cwd), "goals", `${id}.queue.json`);
965
980
  }
966
981
 
967
- export function writeQueueItemFile(cwd: string, item: ListItem): { path: string; wrote: boolean } {
982
+ export interface QueueItemWriteResult {
983
+ path: string;
984
+ wrote: boolean;
985
+ /** True when the persistence boundary rejected the write. A collision or
986
+ * symlink refusal is a normal wrote=false result, not a persistence error. */
987
+ failed?: boolean;
988
+ }
989
+
990
+ /** Write one queue sidecar atomically. The default is idempotent/no-overwrite;
991
+ * repair metadata updates may explicitly request an atomic replacement. All
992
+ * filesystem failures go through the persistence-degradation boundary and
993
+ * clean up the temporary file before returning to the caller. */
994
+ export function writeQueueItemFile(cwd: string, item: ListItem, options: { replace?: boolean } = {}): QueueItemWriteResult {
968
995
  const file = queueItemPath(cwd, item.id);
969
- if (fs.existsSync(file)) return { path: file, wrote: false }; // idempotent — never overwrite
970
- try {
996
+ const replace = options.replace === true;
997
+ if (!replace && fs.existsSync(file)) return { path: file, wrote: false }; // idempotent — never overwrite
998
+ const result = runPersistStep("writeQueueItemFile", () => {
999
+ if (replace) {
1000
+ try {
1001
+ if (fs.lstatSync(file).isSymbolicLink()) return { path: file, wrote: false, failed: true };
1002
+ } catch (err) {
1003
+ if ((err as NodeJS.ErrnoException).code !== "ENOENT") throw err;
1004
+ }
1005
+ }
971
1006
  fs.mkdirSync(path.dirname(file), { recursive: true });
972
- } catch { /* ensureDirs on the goals/ path handles this */ }
973
- const tempPath = `${file}.${process.pid}.${Date.now()}.${Math.random().toString(36).slice(2, 8)}.tmp`;
974
- fs.writeFileSync(tempPath, JSON.stringify({ schema: 1, type: "queue-item", ...item }), "utf-8");
975
- fs.renameSync(tempPath, file);
976
- return { path: file, wrote: true };
1007
+ const tempPath = `${file}.${process.pid}.${Date.now()}.${Math.random().toString(36).slice(2, 8)}.tmp`;
1008
+ try {
1009
+ fs.writeFileSync(tempPath, JSON.stringify({ schema: 1, type: "queue-item", ...item }), "utf-8");
1010
+ fs.renameSync(tempPath, file);
1011
+ return { path: file, wrote: true };
1012
+ } finally {
1013
+ try {
1014
+ if (fs.existsSync(tempPath)) fs.unlinkSync(tempPath);
1015
+ } catch {
1016
+ // The landing write already succeeded or the original error is more
1017
+ // useful; a later cleanup pass can remove an orphaned temp file.
1018
+ }
1019
+ }
1020
+ });
1021
+ return result ?? { path: file, wrote: false, failed: true };
977
1022
  }
978
1023
 
979
1024
  export function deleteQueueItemFile(cwd: string, id: string): boolean {
@@ -1022,8 +1067,8 @@ export function readQueueFromDisk(cwd: string, excludeIds: ReadonlySet<string> =
1022
1067
  const dir = path.join(piGlaDir(cwd), "goals");
1023
1068
  let names: string[];
1024
1069
  try { names = fs.readdirSync(dir); } catch { return []; }
1025
- // readdir order is filesystem-dependent; queue semantics and the public
1026
- // reader contract are id order, so normalize before parsing sidecars.
1070
+ // readdir order is filesystem-dependent; parse every sidecar first and
1071
+ // apply compareQueueItems below so durable queue order survives recovery.
1027
1072
  names.sort();
1028
1073
  const out: ListItem[] = [];
1029
1074
  for (const name of names) {
@@ -1035,6 +1080,20 @@ export function readQueueFromDisk(cwd: string, excludeIds: ReadonlySet<string> =
1035
1080
  if (!e || e.schema !== 1 || e.type !== "queue-item") continue;
1036
1081
  if (typeof e.id !== "string" || typeof e.objective !== "string") continue;
1037
1082
  if (excludeIds.has(e.id)) continue;
1083
+ const repairTarget = e.repairTarget && typeof e.repairTarget === "object"
1084
+ && typeof e.repairTarget.id === "string"
1085
+ && typeof e.repairTarget.objective === "string"
1086
+ && typeof e.repairTarget.source === "string"
1087
+ && Array.isArray(e.repairTarget.reasons)
1088
+ && e.repairTarget.reasons.every((reason: unknown) => typeof reason === "string")
1089
+ ? {
1090
+ id: e.repairTarget.id,
1091
+ objective: e.repairTarget.objective,
1092
+ ...(typeof e.repairTarget.verificationContract === "string" ? { verificationContract: e.repairTarget.verificationContract } : {}),
1093
+ reasons: e.repairTarget.reasons,
1094
+ source: e.repairTarget.source,
1095
+ } satisfies ObjectiveRepairTarget
1096
+ : undefined;
1038
1097
  out.push({
1039
1098
  id: e.id,
1040
1099
  objective: e.objective,
@@ -1044,10 +1103,12 @@ export function readQueueFromDisk(cwd: string, excludeIds: ReadonlySet<string> =
1044
1103
  // v0.34.81: subtask binding round-trips from the sidecar the same way
1045
1104
  // parallelSafe does — must be a string id matching another queue item.
1046
1105
  ...(typeof e.parentId === "string" && e.parentId ? { parentId: e.parentId } : {}),
1106
+ ...(repairTarget ? { repairTarget } : {}),
1107
+ ...(typeof e.queueOrder === "number" && Number.isFinite(e.queueOrder) && e.queueOrder >= 0 ? { queueOrder: e.queueOrder } : {}),
1047
1108
  addedAt: typeof e.addedAt === "string" ? e.addedAt : new Date().toISOString(),
1048
1109
  });
1049
1110
  }
1050
- return out;
1111
+ return out.sort(compareQueueItems);
1051
1112
  }
1052
1113
 
1053
1114
  // =================================================================
@@ -1432,6 +1493,29 @@ export function nowIso(): string {
1432
1493
  return new Date().toISOString();
1433
1494
  }
1434
1495
 
1496
+ /** Stable queue ordering: new sidecars use queueOrder; legacy sidecars fall
1497
+ * back to their durable timestamp and id instead of filesystem enumeration. */
1498
+ export function compareQueueItems(a: Pick<ListItem, "id" | "addedAt" | "queueOrder">, b: Pick<ListItem, "id" | "addedAt" | "queueOrder">): number {
1499
+ const aOrder = typeof a.queueOrder === "number" && Number.isFinite(a.queueOrder) ? a.queueOrder : undefined;
1500
+ const bOrder = typeof b.queueOrder === "number" && Number.isFinite(b.queueOrder) ? b.queueOrder : undefined;
1501
+ if (aOrder !== undefined && bOrder !== undefined && aOrder !== bOrder) return aOrder - bOrder;
1502
+ const added = a.addedAt.localeCompare(b.addedAt);
1503
+ if (added !== 0) return added;
1504
+ if (aOrder !== undefined && bOrder === undefined) return -1;
1505
+ if (aOrder === undefined && bOrder !== undefined) return 1;
1506
+ return a.id.localeCompare(b.id);
1507
+ }
1508
+
1509
+ /** Assign durable positions to newly enqueued items. Existing positions are
1510
+ * retained, so a reload/recovery cannot reorder a confirmed batch. */
1511
+ export function assignQueueOrder<T extends ListItem>(items: readonly T[], existing: readonly ListItem[] = []): T[] {
1512
+ let next = existing.reduce((max, item) => {
1513
+ const order = typeof item.queueOrder === "number" && Number.isFinite(item.queueOrder) ? item.queueOrder : -1;
1514
+ return Math.max(max, order);
1515
+ }, -1) + 1;
1516
+ return items.map((item) => item.queueOrder === undefined ? { ...item, queueOrder: next++ } : item);
1517
+ }
1518
+
1435
1519
  /**
1436
1520
  * v0.34.59: revision token — return {goalId, revision} for use as a
1437
1521
  * (focus-token, sandbox-check) at async boundaries. The orchestrator
@@ -1547,6 +1631,9 @@ export const LONG_RUNNING_JUDGMENT_POLICY = `LONG-RUNNING JUDGMENT POLICY:
1547
1631
  - Preserve the objective and verification contract as the source of truth. The default answer is the durable, maintainable root-cause fix — decide it and proceed; do not stop at a cosmetic workaround merely because it is faster.
1548
1632
  - "Band-aid now vs do it proper" is NEVER a question: when the durable fix is the clearly best call, do it — no ask_user_question, no pause_goal, no "which do you prefer" framing. Record the choice and the reasoning in the turn.
1549
1633
  - Use an opportunistic workaround only when the durable fix is genuinely unsafe, impossible, or blocked right now; the workaround must be reversible and testable, and its durable follow-up is recorded (ledger or comment) instead of silently treated as final.
1634
+ - Premium engineering standards are mandatory: code must be cleanly typed, tested, architecturally sound, and resilient across lifecycle boundaries. Never lower test standards, fake assertions, or bypass types.
1635
+ - Autonomous pivot strategy: if an implementation approach fails verification after 2 attempts, do not loop on the same failing line. Autonomously step back, diagnose the root invariant, and pivot to a clean alternative architecture.
1636
+ - Non-interruption & sensible defaults: never pause a multi-hour run for obvious choices, cosmetic naming, or non-blocking secondary questions. Pick the sensible architectural default, implement it, record the rationale, and continue. Defer non-blocking notes to the final completion summary.
1550
1637
  - Decide autonomously through local implementation choices without interrupting the user. Ask one focused question ONLY at a genuine trade-off where the user's preference materially changes the outcome: an irreversible/destructive external action, a missing permission/credential, or two options with comparable real cost.
1551
1638
  - In unattended mode, choose the safest contract-preserving path and continue. If no safe choice exists, raise a concrete DECIDE question with a recommended default; never ask a vague progress question or wait on a guessed provider/quota reset.`;
1552
1639
 
@@ -1722,9 +1809,13 @@ export function extractVerificationContract(raw: string): { objective: string; v
1722
1809
 
1723
1810
  // Inline fallback: users write one-liners like
1724
1811
  // "Create x.txt. Done when: grep -q ok x.txt"
1725
- // where the marker is mid-line. Split at the first inline marker.
1812
+ // where the marker is mid-line. Split at the first inline marker. Keep
1813
+ // bare `verify` out of the broad marker alternative: it is also a normal
1814
+ // imperative (`Run the audit and verify ... Done when: ...`) and must not
1815
+ // truncate the objective before the actual Done when marker. `Verify:`
1816
+ // remains supported as the explicit short marker form.
1726
1817
  if (!verificationContract) {
1727
- const m = raw.match(/^(.*?)(?:\.|;)??\s+(?:done when|verified when|verify|verification)\b[^:]*:\s*(.+)$/is);
1818
+ const m = raw.match(/^(.*?)(?:\.|;)?\s+(?:(?:done when|verified when|verification)\b[^:]*:|verify\s*:)\s*(.+)$/is);
1728
1819
  if (m) {
1729
1820
  objective = (m[1] ?? "").trim().replace(/[.;]\s*$/, "");
1730
1821
  verificationContract = (m[2] ?? "").trim();
@@ -2189,10 +2280,12 @@ export function formatAuditLog(entries: AuditLogEntry[]): string {
2189
2280
  // =================================================================
2190
2281
 
2191
2282
  /** Which auditor infra errors are worth an automatic retry? User aborts
2192
- * and missing-model config are NOT — retrying can't help them. */
2283
+ * and missing-model config are NOT — retrying can't help them. Timeouts and
2284
+ * watchdog stalls are retriable infrastructure failures. */
2193
2285
  export function isRetriableInfraError(error?: string): boolean {
2194
2286
  if (!error) return false;
2195
- if (/aborted/i.test(error)) return false;
2287
+ if (/^(?:Auditor (?:exceeded|stalled)|.*(?:timed?\s*out|timeout|inactivity))/i.test(error)) return true;
2288
+ if (/^(?:Auditor aborted\.?$|user (?:interrupt|abort)|cancelled by user)/i.test(error.trim())) return false;
2196
2289
  if (/no (?:auditor )?model/i.test(error)) return false;
2197
2290
  return true;
2198
2291
  }
@@ -96,6 +96,23 @@ export interface RecentActionDisplay {
96
96
  /** v0.33.0: widget extras — the refire streak plus the recent-action feed. */
97
97
  export type GoalDisplayActivity = "active" | "awaiting-first-turn" | "working" | "busy" | "queued" | "idle";
98
98
 
99
+ export interface ModelProvenanceDisplay {
100
+ /** The primary model reference selected for the supervised work. */
101
+ primary?: string;
102
+ /** Whether the primary is an explicit pin or inherited from the session. */
103
+ primarySource?: "pinned" | "inherited";
104
+ /** Ordered backup references available to the main-model recovery policy. */
105
+ fallbackRefs?: string[];
106
+ /** References rejected by the explicit forbidden-model gate. */
107
+ skippedForbiddenRefs?: string[];
108
+ /** Model reference observed at the current main-host turn boundary. */
109
+ handledTurn?: string;
110
+ /** Model reference currently executing the detached audit, if any. */
111
+ handledAudit?: string;
112
+ /** Candidate provenance for the detached audit (setting/fallback/session). */
113
+ handledAuditSource?: string;
114
+ }
115
+
99
116
  export interface WidgetExtras {
100
117
  stalls?: number;
101
118
  recent?: RecentActionDisplay[];
@@ -121,6 +138,8 @@ export interface WidgetExtras {
121
138
  auditorProgressSignals?: boolean;
122
139
  /** Effective global main-model backup order for truthful recovery HUDs. */
123
140
  mainModelFallbacks?: string[];
141
+ /** Truthful model-selection provenance for the active goal card/footer. */
142
+ modelProvenance?: ModelProvenanceDisplay;
124
143
  }
125
144
 
126
145
  /**
@@ -181,6 +200,57 @@ function budgetFor(width: number | undefined, prefixCols: number, floor: number)
181
200
  return Math.max(floor, width - WIDGET_HORIZONTAL_MARGIN - prefixCols);
182
201
  }
183
202
 
203
+ function uniqueModelRefs(refs: readonly string[] | undefined): string[] {
204
+ const out: string[] = [];
205
+ const seen = new Set<string>();
206
+ for (const value of refs ?? []) {
207
+ if (typeof value !== "string") continue;
208
+ const ref = value.trim();
209
+ if (!ref || seen.has(ref.toLowerCase())) continue;
210
+ seen.add(ref.toLowerCase());
211
+ out.push(ref);
212
+ }
213
+ return out;
214
+ }
215
+
216
+ /**
217
+ * Keep model selection provenance on the same always-visible card as the
218
+ * goal. These are projections of observed/configured refs, not a guess from
219
+ * elapsed time: the primary source is explicit, forbidden refs stay visible
220
+ * as skipped, and the handled refs are separate from the configured order.
221
+ */
222
+ function modelSourceLabel(source: string): string {
223
+ const key = source.trim().toLowerCase();
224
+ if (key === "setting" || key === "pinned" || key === "auditor-pin") return "pinned";
225
+ if (key === "session" || key === "session-fallback") return "inherited from session";
226
+ if (key.includes("fallback")) return "fallback";
227
+ return source.trim();
228
+ }
229
+
230
+ function modelProvenanceLines(provenance: ModelProvenanceDisplay | undefined, width?: number): string[] {
231
+ if (!provenance) return [];
232
+ const budget = budgetFor(width, 3, 60);
233
+ const lines: string[] = [];
234
+ const primary = typeof provenance.primary === "string" ? provenance.primary.trim() : "";
235
+ if (primary) {
236
+ const source = provenance.primarySource === "pinned" ? "pinned" : "inherited from session";
237
+ lines.push(`model: primary ${truncate(primary, budget)} · ${source}`);
238
+ }
239
+ const fallbacks = uniqueModelRefs(provenance.fallbackRefs);
240
+ if (fallbacks.length > 0) lines.push(`fallbacks: ${truncate(fallbacks.join(" → "), budget)}`);
241
+ const skipped = uniqueModelRefs(provenance.skippedForbiddenRefs);
242
+ if (skipped.length > 0) lines.push(`skipped forbidden: ${truncate(skipped.join(", "), budget)}`);
243
+ const handledTurn = typeof provenance.handledTurn === "string" ? provenance.handledTurn.trim() : "";
244
+ if (handledTurn) lines.push(`handled turn: ${truncate(handledTurn, budget)}`);
245
+ const handledAudit = typeof provenance.handledAudit === "string" ? provenance.handledAudit.trim() : "";
246
+ if (handledAudit) {
247
+ const source = provenance.handledAuditSource?.trim() ? modelSourceLabel(provenance.handledAuditSource) : "";
248
+ const via = source ? ` · via ${truncate(source, 24)}` : "";
249
+ lines.push(`handled audit: ${truncate(handledAudit, Math.max(16, budget - via.length))}${via}`);
250
+ }
251
+ return lines;
252
+ }
253
+
184
254
  // ---- semantic colors (optional; tests call without a theme → plain strings) ----
185
255
 
186
256
  export type DisplayColor = "accent" | "success" | "warning" | "error" | "muted" | "dim";
@@ -351,6 +421,10 @@ function auditRecoveryPending(g: Goal): boolean {
351
421
  // ---- status line (one-liner, always-on) ----
352
422
 
353
423
  export interface AuditDisplayProgress {
424
+ /** Model reference selected for the currently running detached attempt. */
425
+ model?: string;
426
+ /** Candidate provenance: pinned setting, fallback pin, or session fallback. */
427
+ via?: string;
354
428
  currentTool?: string;
355
429
  /** JSON-safe tool arguments from the detached worker; display only a safe target summary. */
356
430
  currentToolArgs?: string;
@@ -1114,6 +1188,14 @@ function goalLines(g: Goal, state: State, audit: AuditDisplayProgress | null | u
1114
1188
  lines.push(`${i === 0 ? "├─" : "│ "} ${paint(theme, "dim", line.replace(/^Main-model recovery: /, ""))}`);
1115
1189
  });
1116
1190
  }
1191
+ // Model provenance is a card fact, not a notification: keep it visible
1192
+ // across active, interrupted, auditing, and paused branches. In
1193
+ // particular, never replace a configured forbidden ref with "none" — the
1194
+ // skipped line explains why it did not handle the turn.
1195
+ const provenance = modelProvenanceLines(extras?.modelProvenance, width);
1196
+ provenance.forEach((line, i) => {
1197
+ lines.push(`${i === 0 ? "├─" : "│ "} ${paint(theme, "dim", line)}`);
1198
+ });
1117
1199
  if (interrupted) {
1118
1200
  const resumeCmd = isList ? "/list resume" : "/goal resume";
1119
1201
  if (interruptedForNoStart(g)) {
@@ -129,3 +129,58 @@ export function parseAuditorVerdict(output: string): { approved: boolean; disapp
129
129
  impossibleReason: impossibleMatch?.[1]?.trim().slice(0, 300) || undefined,
130
130
  };
131
131
  }
132
+
133
+ /**
134
+ * v0.35.7: Extract mechanical shell command gates from a verification contract.
135
+ * Captures explicit commands (e.g. `npm test`, `tsc --noEmit`, `cargo test`)
136
+ * for deterministic fast-fail pre-auditing before spawning the heavy LLM worker.
137
+ */
138
+ export function extractMechanicalCheckCommands(contract: string): string[] {
139
+ if (!contract) return [];
140
+ const items = contractItems(contract);
141
+ const commands: string[] = [];
142
+ for (const item of items) {
143
+ const backtickMatch = /`([^`]+)`/.exec(item);
144
+ let candidate = backtickMatch ? backtickMatch[1]!.trim() : item.trim();
145
+ if (!backtickMatch) {
146
+ candidate = candidate.replace(/\s+(?:passes(?:\s+cleanly|\s+with\s+zero\s+errors)?|exits\s+0|returns\s+0|cleanly).*$/i, "").trim();
147
+ }
148
+ if (/^(?:npm\s+(?:test|run\s+[\w:-]+)|bun\s+(?:test|run\s+[\w:-]+)|pnpm\s+(?:test|run\s+[\w:-]+)|yarn\s+(?:test|[\w:-]+)|tsc\b|cargo\s+(?:test|check|build)|pytest\b|python\s+-m\s+unittest|go\s+test|vitest\b|jest\b|make\s+test|git\s+diff|test\s+-[a-z])/i.test(candidate)) {
149
+ commands.push(candidate);
150
+ }
151
+ }
152
+ return commands;
153
+ }
154
+
155
+ export interface MechanicalCheckResult {
156
+ passed: boolean;
157
+ failedCommand?: string;
158
+ output?: string;
159
+ exitCode?: number;
160
+ }
161
+
162
+ /**
163
+ * v0.35.7: Execute mechanical pre-audit checks deterministically.
164
+ */
165
+ export function runMechanicalPreAuditChecks(cwd: string, commands: string[], timeoutMs = 60000): MechanicalCheckResult {
166
+ if (!commands || commands.length === 0) return { passed: true };
167
+ const { execSync } = require("node:child_process");
168
+ for (const cmd of commands) {
169
+ try {
170
+ execSync(cmd, { cwd, timeout: timeoutMs, stdio: ["ignore", "pipe", "pipe"], encoding: "utf-8" });
171
+ } catch (err: any) {
172
+ const exitCode = typeof err.status === "number" ? err.status : (typeof err.code === "number" ? err.code : 1);
173
+ const stdout = err.stdout ? String(err.stdout) : "";
174
+ const stderr = err.stderr ? String(err.stderr) : "";
175
+ const output = (stdout + "\n" + stderr).trim() || err.message || "Command failed";
176
+ return {
177
+ passed: false,
178
+ failedCommand: cmd,
179
+ output: output.slice(0, 4000),
180
+ exitCode,
181
+ };
182
+ }
183
+ }
184
+ return { passed: true };
185
+ }
186
+