jules-orchestrator-kit 0.74.0 → 0.75.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,36 +1,211 @@
1
1
  /**
2
- * Trojan Source (CVE-2021-42574) BiDi override detection.
2
+ * Unicode security detection for diffs and source text.
3
+ *
4
+ * Covers Trojan Source BiDi overrides (CVE-2021-42574), invisible/zero-width
5
+ * obfuscation characters, Unicode Plane 14 tags, and mixed-script confusable
6
+ * identifiers (Latin mixed with vetted Cyrillic/Greek lookalikes).
3
7
  *
4
- * Split out of src/security.mjs (P05). A Unicode directional override can make
5
- * the line an agent reads differ from the line the parser executes, so the
6
- * check runs on every added line of every non-markdown file before anything
7
- * else looks at the diff.
8
+ * Split out of src/security.mjs (P05). A Unicode directional override — or an
9
+ * invisible filler mid-token — can make the line an agent reads differ from
10
+ * the line the parser executes, so these checks run on every added line of
11
+ * every non-markdown file before anything else looks at the diff.
12
+ *
13
+ * Zero third-party deps; Node ESM builtins + the shared confusable table from
14
+ * secret-scanner.mjs only.
8
15
  */
9
16
 
10
- export function checkTrojanSource(diffOrText = "", _options = {}) {
11
- if (!diffOrText || typeof diffOrText !== "string") return { ok: true, violations: [] };
17
+ import { CONFUSABLE_TO_ASCII } from "./secret-scanner.mjs";
12
18
 
13
- const violations = [];
14
- const bidiRegex = /[\u202A\u202B\u202C\u202D\u202E\u2066\u2067\u2068\u2069\u061C\u200E\u200F]/;
19
+ /**
20
+ * BiDi directional controls — the CVE-2021-42574 Trojan Source set (12 chars).
21
+ * Exported so scanDiff can locate the offending line without duplicating the set.
22
+ */
23
+ export const BIDI_CONTROL_REGEX =
24
+ /[\u202A\u202B\u202C\u202D\u202E\u2066\u2067\u2068\u2069\u061C\u200E\u200F]/;
25
+
26
+ /**
27
+ * Invisible ZW / Hangul fillers used to obfuscate identifiers on added lines:
28
+ * U+00AD, U+200B–U+200D, U+2060–U+2064, U+FEFF, U+034F, U+3164, U+FFA0,
29
+ * U+115F, U+1160.
30
+ */
31
+ export const INVISIBLE_OBFUSCATION_REGEX =
32
+ /[\u00AD\u200B-\u200D\u2060-\u2064\uFEFF\u034F\u3164\uFFA0\u115F\u1160]/;
33
+
34
+ /** Unicode Plane 14 language tags U+E0000–U+E007F. */
35
+ export const PLANE14_TAG_REGEX = /[\u{E0000}-\u{E007F}]/u;
36
+
37
+ /**
38
+ * Cyrillic/Greek entries from the secret-scanner confusable table.
39
+ * Latin lookalikes outside those scripts (ſ, K) are intentionally excluded —
40
+ * they are not mixed-script substitutions.
41
+ */
42
+ const MIXED_SCRIPT_CONFUSABLE_CHARS = [...CONFUSABLE_TO_ASCII.keys()].filter((ch) => {
43
+ const cp = ch.codePointAt(0);
44
+ return (cp >= 0x0370 && cp <= 0x03ff) || (cp >= 0x0400 && cp <= 0x04ff);
45
+ });
15
46
 
47
+ const CONFUSABLE_CLASS = MIXED_SCRIPT_CONFUSABLE_CHARS.map(
48
+ (ch) => `\\u${ch.codePointAt(0).toString(16).padStart(4, "0")}`,
49
+ ).join("");
50
+
51
+ /** Identifier-like tokens built from ASCII id chars and vetted confusables. */
52
+ export const MIXED_SCRIPT_TOKEN_REGEX = new RegExp(
53
+ `[A-Za-z0-9_${CONFUSABLE_CLASS}]+`,
54
+ "g",
55
+ );
56
+
57
+ const HAS_LATIN_ID = /[A-Za-z0-9_]/;
58
+ const HAS_CONFUSABLE = new RegExp(`[${CONFUSABLE_CLASS}]`);
59
+
60
+ /**
61
+ * True when a line contains an identifier token that mixes Latin [A-Za-z0-9_]
62
+ * with at least one Cyrillic/Greek confusable (e.g. Cyrillic i or a in an ASCII token).
63
+ * Pure Cyrillic / Greek / multilingual text without intra-token Latin mixing
64
+ * does not match.
65
+ *
66
+ * @param {string} line
67
+ * @returns {boolean}
68
+ */
69
+ export function hasMixedScriptConfusable(line) {
70
+ if (!line || typeof line !== "string") return false;
71
+ MIXED_SCRIPT_TOKEN_REGEX.lastIndex = 0;
72
+ let match;
73
+ while ((match = MIXED_SCRIPT_TOKEN_REGEX.exec(line)) !== null) {
74
+ const token = match[0];
75
+ if (HAS_LATIN_ID.test(token) && HAS_CONFUSABLE.test(token)) return true;
76
+ }
77
+ return false;
78
+ }
79
+
80
+ /**
81
+ * Select the lines to scan: added lines of a unified diff, or every line of
82
+ * plain text.
83
+ *
84
+ * @param {string} diffOrText
85
+ * @returns {string[]}
86
+ */
87
+ function targetLines(diffOrText) {
16
88
  const lines = diffOrText.split("\n");
17
- // Only filter diff syntax if we actually start with typical diff headers
89
+ // Only filter diff syntax if we actually start with typical diff headers.
18
90
  // This avoids treating text containing "+++ b/" as a diff mistakenly.
19
- const isDiff = diffOrText.startsWith("--- a/") || diffOrText.startsWith("+++ b/") || diffOrText.includes("\n+++ b/");
91
+ const isDiff =
92
+ diffOrText.startsWith("--- a/") ||
93
+ diffOrText.startsWith("+++ b/") ||
94
+ diffOrText.includes("\n+++ b/");
20
95
 
21
- const targetLines = lines.filter((line) => {
96
+ return lines.filter((line) => {
22
97
  if (isDiff) {
23
98
  return line.startsWith("+") && !line.startsWith("+++");
24
99
  }
25
100
  return true;
26
101
  });
102
+ }
27
103
 
28
- for (const line of targetLines) {
29
- if (bidiRegex.test(line)) {
30
- violations.push({ reason: "Trojan Source BiDi override detected: contains invisible directional control characters (CVE-2021-42574)." });
104
+ /**
105
+ * Trojan Source (CVE-2021-42574) BiDi override detection.
106
+ * Backward-compatible: returns `{ ok, violations }` and only reports BiDi
107
+ * controls. Broader Unicode checks live in {@link checkUnicodeSecurity}.
108
+ *
109
+ * @param {string} [diffOrText=""]
110
+ * @param {object} [_options={}]
111
+ * @returns {{ ok: boolean, violations: Array<{ reason: string }> }}
112
+ */
113
+ export function checkTrojanSource(diffOrText = "", _options = {}) {
114
+ if (!diffOrText || typeof diffOrText !== "string") return { ok: true, violations: [] };
115
+
116
+ const violations = [];
117
+ for (const line of targetLines(diffOrText)) {
118
+ if (BIDI_CONTROL_REGEX.test(line)) {
119
+ violations.push({
120
+ reason:
121
+ "Trojan Source BiDi override detected: contains invisible directional control characters (CVE-2021-42574).",
122
+ });
31
123
  break; // One violation is enough for the file/diff block
32
124
  }
33
125
  }
34
126
 
35
127
  return { ok: violations.length === 0, violations };
36
128
  }
129
+
130
+ /**
131
+ * Full Unicode security pass over a diff or plain text.
132
+ *
133
+ * Categories (at most one finding each):
134
+ * - TROJAN_SOURCE_DETECTED — BiDi controls (CVE-2021-42574)
135
+ * - UNICODE_OBFUSCATION_DETECTED — ZW/Hangul fillers or Plane 14 tags
136
+ * - MIXED_SCRIPT_CONFUSABLE_DETECTED — Latin+confusable identifier tokens
137
+ *
138
+ * @param {string} [diffOrText=""]
139
+ * @param {object} [_options={}]
140
+ * @returns {{
141
+ * ok: boolean,
142
+ * violations: Array<{ reason: string, type: string }>
143
+ * }}
144
+ */
145
+ export function checkUnicodeSecurity(diffOrText = "", _options = {}) {
146
+ if (!diffOrText || typeof diffOrText !== "string") return { ok: true, violations: [] };
147
+
148
+ const violations = [];
149
+ let sawBidi = false;
150
+ let sawObfuscation = false;
151
+ let sawMixed = false;
152
+
153
+ for (const line of targetLines(diffOrText)) {
154
+ if (!sawBidi && BIDI_CONTROL_REGEX.test(line)) {
155
+ sawBidi = true;
156
+ violations.push({
157
+ type: "TROJAN_SOURCE_DETECTED",
158
+ reason:
159
+ "Trojan Source BiDi override detected: contains invisible directional control characters (CVE-2021-42574).",
160
+ });
161
+ }
162
+ if (
163
+ !sawObfuscation &&
164
+ (INVISIBLE_OBFUSCATION_REGEX.test(line) || PLANE14_TAG_REGEX.test(line))
165
+ ) {
166
+ sawObfuscation = true;
167
+ violations.push({
168
+ type: "UNICODE_OBFUSCATION_DETECTED",
169
+ reason:
170
+ "Unicode obfuscation detected: invisible zero-width, Hangul filler, or Plane 14 tag characters on an added line.",
171
+ });
172
+ }
173
+ if (!sawMixed && hasMixedScriptConfusable(line)) {
174
+ sawMixed = true;
175
+ violations.push({
176
+ type: "MIXED_SCRIPT_CONFUSABLE_DETECTED",
177
+ reason:
178
+ "Mixed-script confusable identifier detected: Latin characters combined with Cyrillic/Greek lookalikes in one token.",
179
+ });
180
+ }
181
+ if (sawBidi && sawObfuscation && sawMixed) break;
182
+ }
183
+
184
+ return { ok: violations.length === 0, violations };
185
+ }
186
+
187
+ /**
188
+ * Locate the first added-line number that triggers a given Unicode finding type.
189
+ * Shared by scanDiff so it does not re-embed detection regexes.
190
+ *
191
+ * @param {Array<{ text: string, no: number|null }>} lines
192
+ * @param {string} type
193
+ * @returns {number|null}
194
+ */
195
+ export function locateUnicodeFindingLine(lines, type) {
196
+ if (!Array.isArray(lines)) return null;
197
+ for (const l of lines) {
198
+ if (!l || typeof l.text !== "string") continue;
199
+ const text = l.text;
200
+ let hit = false;
201
+ if (type === "TROJAN_SOURCE_DETECTED") {
202
+ hit = BIDI_CONTROL_REGEX.test(text);
203
+ } else if (type === "UNICODE_OBFUSCATION_DETECTED") {
204
+ hit = INVISIBLE_OBFUSCATION_REGEX.test(text) || PLANE14_TAG_REGEX.test(text);
205
+ } else if (type === "MIXED_SCRIPT_CONFUSABLE_DETECTED") {
206
+ hit = hasMixedScriptConfusable(text);
207
+ }
208
+ if (hit) return l.no ?? null;
209
+ }
210
+ return null;
211
+ }
package/src/config.mjs CHANGED
@@ -34,7 +34,7 @@ const DEFAULTS = {
34
34
  // use: an identical exfiltration job placed in `.gitlab-ci.yml` was approved
35
35
  // where `.github/workflows/x.yml` was rejected. A safety gate that is only
36
36
  // safe on GitHub is not a safety gate.
37
- const CI_DEFINITIONS = [
37
+ export const CI_DEFINITIONS = [
38
38
  ".github/**",
39
39
  ".gitlab-ci.yml",
40
40
  "**/.gitlab-ci.yml",
@@ -1019,9 +1019,11 @@ export function resolveTrustedPolicy(root = process.cwd(), baseRef = null, mode
1019
1019
 
1020
1020
  let trustedConfigRaw = null;
1021
1021
  let trustedJulesRaw = null;
1022
+ let trustedProtectedRaw = null;
1022
1023
  try {
1023
1024
  trustedConfigRaw = showFromOrigin(root, base, ".agent/config.yml");
1024
1025
  trustedJulesRaw = showFromOrigin(root, base, ".agent/jules.yml");
1026
+ trustedProtectedRaw = showFromOrigin(root, base, ".agent/protected-paths.json");
1025
1027
  } catch (_) {
1026
1028
  // If base branch cannot be resolved or show fails, let changedFiles report code 1
1027
1029
  }
@@ -1035,7 +1037,21 @@ export function resolveTrustedPolicy(root = process.cwd(), baseRef = null, mode
1035
1037
  if (raw) parsed = parseYaml(raw) || {};
1036
1038
  } catch (_) {}
1037
1039
 
1040
+ let trustedProtectedPaths = [];
1041
+ if (trustedProtectedRaw) {
1042
+ try {
1043
+ const parsedProtected = JSON.parse(trustedProtectedRaw);
1044
+ if (Array.isArray(parsedProtected.protected)) {
1045
+ trustedProtectedPaths = parsedProtected.protected.filter((p) => typeof p === "string" && p);
1046
+ }
1047
+ } catch (_) {}
1048
+ }
1049
+
1038
1050
  const trustedScope = normalizeScope(parsed);
1051
+ if (trustedProtectedPaths.length > 0) {
1052
+ trustedScope.protect = dedupe([...trustedScope.protect, ...trustedProtectedPaths]);
1053
+ trustedScope.protectedPatterns = trustedProtectedPaths;
1054
+ }
1039
1055
  const trustedDiffKb = Number(parsed.limits?.diff_kb || parsed.limits?.diffKb) || 75;
1040
1056
 
1041
1057
  const rawSetup = parsed.setup_cmd ?? parsed.verify?.setup;
@@ -1100,6 +1116,7 @@ export function resolveTrustedPolicy(root = process.cwd(), baseRef = null, mode
1100
1116
  // Inspect any proposed scaffold in this change (F06)
1101
1117
  let proposedConfig = null;
1102
1118
  let proposedJules = null;
1119
+ let proposedProtectedPaths = [];
1103
1120
 
1104
1121
  try {
1105
1122
  if (mode === "committed") {
@@ -1107,11 +1124,29 @@ export function resolveTrustedPolicy(root = process.cwd(), baseRef = null, mode
1107
1124
  if (c) proposedConfig = parseYaml(c);
1108
1125
  const j = showFromOrigin(root, "HEAD", ".agent/jules.yml");
1109
1126
  if (j) proposedJules = parseYaml(j);
1127
+ const p = showFromOrigin(root, "HEAD", ".agent/protected-paths.json");
1128
+ if (p) {
1129
+ try {
1130
+ const parsedP = JSON.parse(p);
1131
+ if (Array.isArray(parsedP.protected)) {
1132
+ proposedProtectedPaths = parsedP.protected.filter((x) => typeof x === "string" && x);
1133
+ }
1134
+ } catch (_) {}
1135
+ }
1110
1136
  } else {
1111
1137
  const configPath = join(root, ".agent/config.yml");
1112
1138
  if (existsSync(configPath)) proposedConfig = parseYaml(readFileSync(configPath, "utf-8"));
1113
1139
  const julesPath = join(root, ".agent/jules.yml");
1114
1140
  if (existsSync(julesPath)) proposedJules = parseYaml(readFileSync(julesPath, "utf-8"));
1141
+ const protectedPath = join(root, ".agent/protected-paths.json");
1142
+ if (existsSync(protectedPath)) {
1143
+ try {
1144
+ const parsedP = JSON.parse(readFileSync(protectedPath, "utf-8"));
1145
+ if (Array.isArray(parsedP.protected)) {
1146
+ proposedProtectedPaths = parsedP.protected.filter((x) => typeof x === "string" && x);
1147
+ }
1148
+ } catch (_) {}
1149
+ }
1115
1150
  }
1116
1151
  } catch (_) {}
1117
1152
 
@@ -1182,6 +1217,10 @@ export function resolveTrustedPolicy(root = process.cwd(), baseRef = null, mode
1182
1217
 
1183
1218
  const stages = explicitStages ?? profilePlan?.stages ?? null;
1184
1219
  const trustedScope = normalizeScope(pConfig);
1220
+ if (proposedProtectedPaths.length > 0) {
1221
+ trustedScope.protect = dedupe([...trustedScope.protect, ...proposedProtectedPaths]);
1222
+ trustedScope.protectedPatterns = proposedProtectedPaths;
1223
+ }
1185
1224
  const trustedLimits = {
1186
1225
  diffKb: Number(pConfig.limits?.diff_kb || pConfig.limits?.diffKb) || 75,
1187
1226
  };
package/src/engine.mjs CHANGED
@@ -1,4 +1,4 @@
1
- import { loadConfig, resolveTrustedPolicy } from "./config.mjs";
1
+ import { loadConfig, resolveTrustedPolicy, normalizePath } from "./config.mjs";
2
2
  import { isTestPath } from "./test-paths.mjs";
3
3
  import { isPlaceholderTestScript, isSrcLayout } from "./stack-detector.mjs";
4
4
  import { checkCollectionFloor } from "./ops/test-collection.mjs";
@@ -10,9 +10,9 @@ import { withBudget, appendLedger, getQueueDir, ensureDir, rollbackBudgetReserva
10
10
  import { resolveDailyLimit, recordObservedCeiling, isDailyQuotaRejection, resolveAmbientIdentity } from "./budget.mjs";
11
11
  import { sanitizeUntrustedData, buildAgentEnvelope } from "./prompt-guard.mjs";
12
12
  import { recordVerifyRun, readVerifyRuns, flakyVerdict } from "./flaky-ledger.mjs";
13
- import fs, { readdirSync, readFileSync, renameSync, existsSync } from "node:fs";
13
+ import fs, { readdirSync, readFileSync, writeFileSync, renameSync, existsSync } from "node:fs";
14
14
  import { join, basename } from "node:path";
15
- import { createHash } from "node:crypto";
15
+ import { createHash, randomUUID } from "node:crypto";
16
16
  import { appendTelemetryBestEffort as appendTelemetry } from "./telemetry.mjs";
17
17
 
18
18
  import { spawn } from "node:child_process";
@@ -207,6 +207,7 @@ export async function gate(opts = {}) {
207
207
  process.env.JULES_ALLOW_COMMAND_FILE_CHANGES === "1" ||
208
208
  process.env.AGENT_ALLOW_COMMAND_FILE_CHANGES === "true" ||
209
209
  process.env.AGENT_ALLOW_COMMAND_FILE_CHANGES === "1",
210
+ protectedPatterns: trustedScope.protectedPatterns,
210
211
  });
211
212
  // Report the violation against the link the change actually introduced, not
212
213
  // against a path the diff never names — the operator has to be able to find it.
@@ -336,8 +337,18 @@ export async function gate(opts = {}) {
336
337
  // failures stop reaching the exit code — the gate then sees exit 0 and
337
338
  // approves a change whose tests failed. src/perf.mjs already strips these for
338
339
  // the same reason; the gate, which is the one that decides, did not.
340
+ // Ambient PR waiver labels and HEAD_SHA must also be stripped so verification
341
+ // suites and child audits run hermetically without inherited waivers.
339
342
  for (const key of Object.keys(testEnv)) {
340
- if (key.startsWith("NODE_TEST_") || key.startsWith("NODE_CHANNEL_")) delete testEnv[key];
343
+ if (
344
+ key.startsWith("NODE_TEST_") ||
345
+ key.startsWith("NODE_CHANNEL_") ||
346
+ key === "PR_LABELS" ||
347
+ key === "HEAD_SHA" ||
348
+ key.startsWith("JULES_ALLOW_")
349
+ ) {
350
+ delete testEnv[key];
351
+ }
341
352
  }
342
353
 
343
354
  let flakyVerdictResult = null;
@@ -815,12 +826,69 @@ export async function repair(failure, opts = {}) {
815
826
  threshold: 2,
816
827
  });
817
828
 
829
+ const repairId = `repair-${Date.now()}-${randomUUID().slice(0, 8)}`;
830
+ let repairDir = null;
831
+
818
832
  let currentFailure = failure;
819
833
  const initialFingerprint = fingerprintFailureState(currentFailure, root);
820
834
  breaker.observe(initialFingerprint);
821
835
  whackAMole.recordTestOutcome(currentFailure.stderr || currentFailure.message);
822
836
  let extraPromptDirective = null;
823
837
 
838
+ const persistFailedAttempt = (n, diff, diagnostics) => {
839
+ const relDir = normalizePath(join(".agent/state/repairs", repairId));
840
+ const absDir = join(root, relDir);
841
+ ensureDir(absDir);
842
+ const patchName = `attempt-${n}.patch`;
843
+ const diagName = `attempt-${n}-diagnostics.json`;
844
+ const patchRel = normalizePath(join(relDir, patchName));
845
+ const diagRel = normalizePath(join(relDir, diagName));
846
+ writeFileSync(join(absDir, patchName), redactSecrets(String(diff || "")), "utf-8");
847
+ const payload = {
848
+ ...diagnostics,
849
+ patchPath: patchRel,
850
+ diagnosticsPath: diagRel,
851
+ };
852
+ writeFileSync(join(absDir, diagName), JSON.stringify(payload, null, 2) + "\n", "utf-8");
853
+ repairDir = relDir;
854
+ return payload;
855
+ };
856
+
857
+ const captureAttemptDiff = () => {
858
+ try {
859
+ // Working-tree mode retains what the repair attempt left on disk; the
860
+ // default committed mode only shows the last commit and would hide
861
+ // uncommitted agent edits that operators need to review.
862
+ const base = config.baseBranch || config.base || "main";
863
+ return diffText(root, base, "working-tree") || "";
864
+ } catch (_) {
865
+ return "";
866
+ }
867
+ };
868
+
869
+ const buildGateDiagnostics = (gateRes) => {
870
+ const failingPhase = (gateRes.phases || []).find((p) => p && p.ok === false) || null;
871
+ const verifyPhase = (gateRes.phases || []).find((p) => p && p.phase === "verify") || null;
872
+ const failureInfo = verifyPhase?.failure || null;
873
+ const testResult = verifyPhase?.testResult || null;
874
+ const stderr = failureInfo?.stderr || testResult?.stderr || gateRes.error || "";
875
+ const stdout = failureInfo?.stdout || testResult?.stdout || "";
876
+ return {
877
+ code: gateRes.code ?? null,
878
+ phase: failingPhase?.phase || "verify",
879
+ command: failureInfo?.command || testResult?.command || null,
880
+ exitCode: failureInfo?.exitCode ?? testResult?.status ?? null,
881
+ stderr: redactSecrets(String(stderr).slice(0, 8000)),
882
+ stdout: redactSecrets(String(stdout).slice(0, 4000)),
883
+ messages: Array.isArray(failureInfo?.diagnostics) ? failureInfo.diagnostics.map((d) => redactSecrets(String(d))) : [],
884
+ error: redactSecrets(
885
+ String(gateRes.error || failureInfo?.stderr || testResult?.stderr || "Gate re-verification failed")
886
+ .split("\n")[0]
887
+ .slice(0, 500)
888
+ ),
889
+ };
890
+ };
891
+
824
892
  for (let n = 1; n <= maxRetries; n++) {
825
893
  if (progressBus && progressToken) {
826
894
  progressBus.reportProgress(progressToken, Math.round((n / maxRetries) * 100), 100, `OODA Repair attempt ${n}/${maxRetries}`);
@@ -847,7 +915,14 @@ export async function repair(failure, opts = {}) {
847
915
  const retryAfterMs = err.retryAfterMs || 60000;
848
916
  const backoffSec = Math.ceil(retryAfterMs / 1000);
849
917
  console.warn(`[PROVIDER_INFRASTRUCTURE_FAILURE] ${err.name}: ${err.message}. Recommended backoff: ${backoffSec}s.`);
850
- attempts.push({ n, ok: false, error: err.message, providerError: true, retryAfterMs });
918
+ attempts.push({
919
+ n,
920
+ ok: false,
921
+ phase: "dispatch",
922
+ error: err.message,
923
+ providerError: true,
924
+ retryAfterMs,
925
+ });
851
926
  appendTelemetry(root, "ooda_repair_attempt", { attempt: n, ok: false, error: err.message, providerError: true });
852
927
  return {
853
928
  ok: false,
@@ -856,9 +931,10 @@ export async function repair(failure, opts = {}) {
856
931
  error: err.message,
857
932
  retryAfterMs,
858
933
  providerError: true,
934
+ repairDir,
859
935
  };
860
936
  }
861
- attempts.push({ n, ok: false, error: err.message });
937
+ attempts.push({ n, ok: false, phase: "dispatch", error: err.message });
862
938
  appendTelemetry(root, "ooda_repair_attempt", { attempt: n, ok: false, error: err.message });
863
939
  break;
864
940
  }
@@ -898,17 +974,21 @@ export async function repair(failure, opts = {}) {
898
974
  }
899
975
  }
900
976
 
901
- attempts.push({
902
- n,
903
- session,
904
- ok: true,
905
- poll: pollVerdict ? { status: pollVerdict.status, terminal: pollVerdict.terminal } : null,
906
- });
907
- appendTelemetry(root, "ooda_repair_attempt", { attempt: n, ok: true });
977
+ const poll = pollVerdict ? { status: pollVerdict.status, terminal: pollVerdict.terminal } : null;
908
978
 
909
979
  // Re-verify after repair attempt
910
980
  const gateRes = await gate({ root, config, fix: false, progressBus, progressToken });
981
+ const attemptDiff = captureAttemptDiff();
911
982
  if (gateRes.ok) {
983
+ attempts.push({
984
+ n,
985
+ session,
986
+ ok: true,
987
+ verified: true,
988
+ diff: attemptDiff,
989
+ poll,
990
+ });
991
+ appendTelemetry(root, "ooda_repair_attempt", { attempt: n, ok: true, verified: true });
912
992
  breaker.reset();
913
993
  whackAMole.reset();
914
994
  recordRemediation(root, {
@@ -917,9 +997,27 @@ export async function repair(failure, opts = {}) {
917
997
  remediationHint: `Resolved on repair attempt #${n}`,
918
998
  targetFiles: currentFailure.targetFiles || [],
919
999
  });
920
- return { ok: true, attempts, finalStatus: "PASSED" };
1000
+ return { ok: true, attempts, finalStatus: "PASSED", repairDir };
921
1001
  }
922
1002
 
1003
+ const diagnostics = persistFailedAttempt(n, attemptDiff, buildGateDiagnostics(gateRes));
1004
+ attempts.push({
1005
+ n,
1006
+ session,
1007
+ ok: false,
1008
+ verified: false,
1009
+ diff: attemptDiff,
1010
+ diagnostics,
1011
+ poll,
1012
+ });
1013
+ appendTelemetry(root, "ooda_repair_attempt", {
1014
+ attempt: n,
1015
+ ok: false,
1016
+ verified: false,
1017
+ code: gateRes.code ?? null,
1018
+ phase: diagnostics.phase,
1019
+ });
1020
+
923
1021
  currentFailure = gateRes.phases.find((p) => p.phase === "verify")?.testResult || failure;
924
1022
  const currentFingerprint = fingerprintFailureState(currentFailure, root);
925
1023
 
@@ -943,6 +1041,7 @@ export async function repair(failure, opts = {}) {
943
1041
  finalStatus: "DETERMINISTIC_REGRESSION",
944
1042
  reason: `Identical failure state fingerprint (${currentFingerprint}) observed during attempt #${n}`,
945
1043
  fingerprint: currentFingerprint,
1044
+ repairDir,
946
1045
  };
947
1046
  }
948
1047
  }
@@ -963,39 +1062,66 @@ export async function repair(failure, opts = {}) {
963
1062
  });
964
1063
  } catch (_) {}
965
1064
 
966
- return { ok: false, attempts, finalStatus: "OODA_EXHAUSTED" };
1065
+ return { ok: false, attempts, finalStatus: "OODA_EXHAUSTED", repairDir };
967
1066
 
968
1067
  }
969
1068
 
970
1069
  /**
971
- * Polls an async provider for terminal session state (COMPLETED / FAILED) before re-verification.
972
- */
973
- /**
974
- * Evaluates whether a task's verification oracle or goal is already satisfied on the current working tree.
975
- * Prevents redundant session dispatch and API budget burning.
1070
+ * A passing generic test suite does not prove that a requested feature exists.
1071
+ * Only an explicit goal-specific check may suppress dispatch. This is advisory:
1072
+ * the supplied command must return 0 exactly when the requested state exists.
976
1073
  */
977
1074
  export async function checkTaskPremise(task = {}, opts = {}) {
978
1075
  const root = opts.config?._root || opts.root || process.cwd();
979
- const verifyCmd = task.verifyCmd || task.verify;
980
- if (!verifyCmd) {
981
- return { satisfied: false, reason: "No verification oracle specified for pre-flight premise check." };
1076
+ const goalCheck = task.goalCheck ?? opts.goalCheck;
1077
+ const evidence = { revision: null, goalCheckSha256: null, exitCode: null, durationMs: null };
1078
+ const unknown = (reasonCode, reason) => ({ satisfied: false, status: "UNKNOWN", reasonCode, reason, evidence });
1079
+ if (typeof goalCheck !== "string" || !goalCheck.trim()) {
1080
+ return unknown("NO_GOAL_CHECK", "No explicit goal-specific check supplied; a passing verification suite is not proof that the task is complete.");
982
1081
  }
983
-
984
- const { runVerificationProbe } = await import("./wizard-oracle.mjs");
985
- const probe = await runVerificationProbe(verifyCmd, root, { timeoutMs: opts.timeoutMs || 30_000 });
986
- if (probe.ok) {
987
- return {
988
- satisfied: true,
989
- reason: `Verification oracle '${verifyCmd}' already passes cleanly with exit code 0 on base branch.`,
990
- durationMs: probe.durationMs,
991
- };
1082
+ const command = goalCheck.trim();
1083
+ evidence.goalCheckSha256 = sha256(command);
1084
+ if (isPlaceholderTestScript(command)) {
1085
+ return unknown("PLACEHOLDER_GOAL_CHECK", "Goal check is a no-op or incapable of establishing the requested state.");
1086
+ }
1087
+ const genericCommands = [
1088
+ task.verifyCmd, task.verify,
1089
+ opts.config?.verify?.test, opts.config?.verify?.unit,
1090
+ opts.config?.verify?.lint, opts.config?.verify?.build,
1091
+ ].filter((value) => typeof value === "string").map((value) => value.trim());
1092
+ if (genericCommands.includes(command)) {
1093
+ return unknown("GENERIC_VERIFY_IS_NOT_GOAL_PROOF", "Goal check matches the generic verification command; provide a separate objective-specific condition.");
992
1094
  }
993
1095
 
994
- return {
995
- satisfied: false,
996
- reason: `Verification oracle '${verifyCmd}' failed (exit ${probe.code}), proving task need.`,
997
- durationMs: probe.durationMs,
1096
+ // The check runs against the working tree. A clean tree and stable HEAD are
1097
+ // prerequisites for attributing a successful result to a committed revision.
1098
+ const snapshot = () => {
1099
+ try {
1100
+ const head = runCmd(["git", "rev-parse", "HEAD"], { cwd: root, ignoreError: true, timeout: 5000 });
1101
+ const state = runCmd(["git", "status", "--porcelain", "--untracked-files=all"], { cwd: root, ignoreError: true, timeout: 5000 });
1102
+ if (head.status !== 0 || state.status !== 0 || !/^[0-9a-f]{40,64}$/.test(head.stdout.trim())) return null;
1103
+ return { revision: head.stdout.trim(), clean: !state.stdout.trim() };
1104
+ } catch (_) {
1105
+ return null;
1106
+ }
998
1107
  };
1108
+ const before = snapshot();
1109
+ if (!before) return unknown("GIT_STATE_UNAVAILABLE", "Cannot establish the Git revision and working-tree state.");
1110
+ evidence.revision = before.revision;
1111
+ if (!before.clean) return unknown("DIRTY_WORKTREE", "The working tree is not clean; goal proof cannot be attributed to the committed revision.");
1112
+
1113
+ const { runVerificationProbe } = await import("./wizard-oracle.mjs");
1114
+ const probe = await runVerificationProbe(command, root, { timeoutMs: opts.timeoutMs || 30_000 });
1115
+ evidence.exitCode = probe.code;
1116
+ evidence.durationMs = probe.durationMs;
1117
+ const after = snapshot();
1118
+ if (!after || after.revision !== before.revision || !after.clean) {
1119
+ return unknown("GIT_STATE_CHANGED", "The goal check changed the repository state or revision; its result cannot justify skipping dispatch.");
1120
+ }
1121
+ if (!probe.ok) {
1122
+ return { satisfied: false, status: "NOT_PROVEN", reasonCode: "GOAL_CHECK_NOT_SATISFIED", reason: "The goal-specific check did not establish the requested state.", evidence };
1123
+ }
1124
+ return { satisfied: true, status: "PROVEN", reasonCode: "GOAL_CHECK_SATISFIED", reason: "Explicit goal-specific check passed on an unchanged, clean Git revision.", evidence };
999
1125
  }
1000
1126
 
1001
1127
  /**
@@ -1181,17 +1307,28 @@ export async function dispatch(task = {}, opts = {}) {
1181
1307
  throw new Error(`Task prompt exceeds maximum payload limit of ${config.limits.promptKb} KB`);
1182
1308
  }
1183
1309
 
1184
- // Pre-flight idempotency premise check
1185
- if (opts.checkPremise || task.checkPremise) {
1186
- const premise = await checkTaskPremise(task, { root, config });
1187
- if (premise.satisfied) {
1188
- return {
1189
- id: "premise-already-satisfied",
1190
- status: "ALREADY_SATISFIED",
1191
- skipped: true,
1192
- reason: premise.reason,
1193
- };
1194
- }
1310
+ // Record both positive and negative decisions. A generic green test suite
1311
+ // must never be mistaken for proof that a requested task is complete.
1312
+ const premiseRequested = Boolean(opts.checkPremise || task.checkPremise);
1313
+ const premise = premiseRequested
1314
+ ? (opts.dryRun
1315
+ ? { satisfied: false, reasonCode: "DRY_RUN", reason: "Preview does not execute the goal check.", evidence: { revision: null, goalCheckSha256: null, exitCode: null, durationMs: null } }
1316
+ : await checkTaskPremise(task, { root, config }))
1317
+ : { satisfied: false, reasonCode: "PREMISE_CHECK_NOT_REQUESTED", evidence: null };
1318
+ appendTelemetry(root, "dispatch_decision", {
1319
+ taskId: task.id || task.taskId || null,
1320
+ decision: premise.satisfied ? "SKIP" : "DISPATCH",
1321
+ reasonCode: premise.reasonCode,
1322
+ evidence: premise.evidence,
1323
+ });
1324
+ if (premise.satisfied) {
1325
+ return {
1326
+ id: "premise-already-satisfied",
1327
+ status: "ALREADY_SATISFIED",
1328
+ skipped: true,
1329
+ reason: premise.reason,
1330
+ evidence: premise.evidence,
1331
+ };
1195
1332
  }
1196
1333
 
1197
1334
  // Redact secrets in prompt before dispatching