thumbgate 1.29.1 → 1.30.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (77) hide show
  1. package/.claude/commands/dashboard.md +11 -1
  2. package/.claude/commands/thumbgate-dashboard.md +23 -8
  3. package/.claude-plugin/plugin.json +1 -1
  4. package/.well-known/mcp/server-card.json +1 -1
  5. package/README.md +61 -1
  6. package/adapters/claude/.mcp.json +2 -2
  7. package/adapters/forge/forge.yaml +3 -3
  8. package/adapters/mcp/server-stdio.js +164 -7
  9. package/adapters/opencode/opencode.json +1 -1
  10. package/bin/cli.js +7 -5
  11. package/commands/dashboard.md +11 -1
  12. package/commands/thumbgate-dashboard.md +23 -8
  13. package/config/agent-outcome-monitor-thresholds.json +63 -0
  14. package/config/evals/agent-outcomes-baseline.json +17 -0
  15. package/config/evals/agent-outcomes-golden.json +412 -0
  16. package/config/evals/prompt-eval-baseline.json +23 -0
  17. package/config/mcp-allowlists.json +26 -2
  18. package/config/post-deploy-marketing-pages.json +26 -1
  19. package/config/schemas/task-outcome-receipt.schema.json +296 -0
  20. package/openapi/openapi.yaml +235 -0
  21. package/package.json +55 -11
  22. package/public/architecture.html +130 -0
  23. package/public/assets/diagrams/agent-integration.png +0 -0
  24. package/public/assets/diagrams/before-after.svg +21 -0
  25. package/public/assets/diagrams/decision.svg +36 -0
  26. package/public/assets/diagrams/feedback-pipeline.png +0 -0
  27. package/public/assets/diagrams/loop.svg +34 -0
  28. package/public/assets/diagrams/plugin-topology.png +0 -0
  29. package/public/assets/diagrams/pre-action-gate-loop.svg +59 -0
  30. package/public/assets/diagrams/stack.svg +18 -0
  31. package/public/assets/diagrams/thumbgate-architecture.png +0 -0
  32. package/public/case-studies.html +151 -0
  33. package/public/eval-scorecard.html +195 -0
  34. package/public/eval-scorecard.json +18 -0
  35. package/public/evaluations.html +168 -0
  36. package/public/index.html +6 -3
  37. package/public/numbers.html +2 -2
  38. package/public/whitepaper.html +189 -0
  39. package/scripts/activation-quickstart.js +1 -0
  40. package/scripts/agent-outcome-eval.js +130 -0
  41. package/scripts/agent-outcome-monitor.js +331 -0
  42. package/scripts/agent-reasoning-traces.js +8 -9
  43. package/scripts/async-job-runner.js +107 -13
  44. package/scripts/billing.js +3 -1
  45. package/scripts/claude-feedback-sync.js +3 -2
  46. package/scripts/cli-feedback.js +13 -7
  47. package/scripts/cross-encoder-reranker.js +3 -0
  48. package/scripts/durability/step.js +121 -12
  49. package/scripts/feedback-aggregate.js +5 -2
  50. package/scripts/feedback-loop.js +244 -182
  51. package/scripts/gates-engine.js +512 -22
  52. package/scripts/generate-case-study-outreach.js +253 -0
  53. package/scripts/generate-eval-scorecard.js +276 -0
  54. package/scripts/growth-campaigns.js +183 -0
  55. package/scripts/human-escalation.js +265 -0
  56. package/scripts/hybrid-feedback-context.js +93 -50
  57. package/scripts/jsonl-watcher.js +1 -0
  58. package/scripts/judge-reward-function.js +30 -18
  59. package/scripts/lesson-inference.js +23 -4
  60. package/scripts/lesson-retrieval.js +71 -4
  61. package/scripts/lesson-search.js +26 -3
  62. package/scripts/mcp-config.js +26 -5
  63. package/scripts/mcp-oauth.js +37 -2
  64. package/scripts/model-eval.js +308 -0
  65. package/scripts/parallel-workflow-orchestrator.js +86 -22
  66. package/scripts/prompt-eval.js +81 -4
  67. package/scripts/published-cli.js +11 -1
  68. package/scripts/refresh-proof-pack.js +261 -0
  69. package/scripts/risk-scorer.js +144 -15
  70. package/scripts/schedule-manager.js +249 -0
  71. package/scripts/statusline-local-stats.js +1 -1
  72. package/scripts/task-outcomes.js +425 -0
  73. package/scripts/thumbgate-bench.js +13 -0
  74. package/scripts/tool-contract-validator.js +287 -59
  75. package/scripts/tool-kpi-tracker.js +124 -0
  76. package/scripts/tool-registry.js +192 -1
  77. package/src/api/server.js +355 -89
@@ -0,0 +1,308 @@
1
+ #!/usr/bin/env node
2
+ 'use strict';
3
+
4
+ /**
5
+ * model-eval.js — honest evaluation primitives for ThumbGate's learned models.
6
+ *
7
+ * WHY THIS EXISTS
8
+ *
9
+ * The risk scorer reported exactly one number: `trainingAccuracy`, measured on the same rows
10
+ * it trained on. On 2026-07-28 that number was 0.820 — which sounds good until you notice the
11
+ * base rate was 0.711. A model that answers "high-risk" unconditionally scores 71.1%, so the
12
+ * headline figure was measuring an 11-point in-sample lift and presenting it as quality.
13
+ * Nothing anywhere compared against that trivial baseline, and no split existed, so the
14
+ * generalization number was not merely bad — it was unknown.
15
+ *
16
+ * Accuracy is also the wrong summary for a 71/29 split. These primitives therefore report the
17
+ * metrics that survive class imbalance (precision/recall/F1/MCC/ROC-AUC) and the ones that say
18
+ * whether a probability means anything (Brier score, expected calibration error).
19
+ *
20
+ * DETERMINISM IS A REQUIREMENT, NOT A PREFERENCE.
21
+ * Splits are derived from a content hash, never from Math.random(). The same corpus must
22
+ * produce the same split on every machine and every run, or a "quality regression" is
23
+ * indistinguishable from a reshuffle and the CI gate built on top is noise.
24
+ *
25
+ * Everything here is pure: no I/O, no clock, no global state. That is what makes it testable.
26
+ */
27
+
28
+ /** FNV-1a. Small, fast, and stable across platforms — the only properties we need. */
29
+ function hashString(text) {
30
+ let hash = 0x811c9dc5;
31
+ const value = String(text);
32
+ for (let index = 0; index < value.length; index += 1) {
33
+ hash ^= value.charCodeAt(index);
34
+ // 32-bit FNV prime multiply via shifts; Math.imul keeps this exact in JS.
35
+ hash = Math.imul(hash, 0x01000193) >>> 0;
36
+ }
37
+ return hash >>> 0;
38
+ }
39
+
40
+ /** Labels arrive as 1/-1 (AdaBoost) or 1/0 (everything else). Normalize to 1/0. */
41
+ function toBinaryLabel(label) {
42
+ return Number(label) === 1 ? 1 : 0;
43
+ }
44
+
45
+ /**
46
+ * Deterministic stratified split.
47
+ *
48
+ * Stratified because a random split of a 71/29 corpus can hand back a test fold with a
49
+ * materially different base rate, which moves the very number we are trying to measure.
50
+ * Splitting inside each class keeps the test fold's base rate close to the corpus.
51
+ *
52
+ * Returns { train, test }. Both are always non-empty when the input supports it; when a class
53
+ * is too small to contribute a test row, the split degrades to "everything is training" rather
54
+ * than silently producing a test fold with one class in it (an AUC of NaN dressed up as data).
55
+ */
56
+ function stratifiedSplit(examples, options = {}) {
57
+ const testFraction = Number(options.testFraction || 0.25);
58
+ const keyFn = options.keyFn || ((example, index) => JSON.stringify(example.features || example) + index);
59
+
60
+ // Group ACROSS classes, then assign whole groups — StratifiedGroupKFold semantics.
61
+ //
62
+ // Two bugs led here, both caught by tests/risk-model-quality.test.js:
63
+ // 1. Assigning individual rows split tied blocks, so an identical row could sit in both
64
+ // folds and the test fold scored rows the model had memorized.
65
+ // 2. Grouping per class still split duplicates, because with label noise the SAME feature
66
+ // vector can carry both labels — the two copies then landed in different class buckets
67
+ // and different folds.
68
+ // Grouping globally is the only version where "this input is in exactly one fold" holds.
69
+ const classTotals = new Map();
70
+ const groups = new Map();
71
+ examples.forEach((example, index) => {
72
+ const label = toBinaryLabel(example.label);
73
+ classTotals.set(label, (classTotals.get(label) || 0) + 1);
74
+ const hash = hashString(keyFn(example, index));
75
+ if (!groups.has(hash)) groups.set(hash, { members: [], counts: new Map() });
76
+ const group = groups.get(hash);
77
+ group.members.push(example);
78
+ group.counts.set(label, (group.counts.get(label) || 0) + 1);
79
+ });
80
+
81
+ // A single-class corpus has nothing to measure; there is no honest split of it.
82
+ if (classTotals.size < 2) return { train: examples.slice(), test: [] };
83
+
84
+ // Per-class quotas keep the test fold's base rate near the corpus even though whole groups
85
+ // move together. Without quotas, one large group could swamp the fold and shift the very
86
+ // base rate the lift is measured against.
87
+ const quotas = new Map();
88
+ for (const [label, total] of classTotals) quotas.set(label, Math.floor(total * testFraction));
89
+
90
+ const train = [];
91
+ const test = [];
92
+ const taken = new Map();
93
+ // Order by hash, not by position: input order in a JSONL log is chronological, so a
94
+ // positional split would put all recent rows in one fold and measure drift, not skill.
95
+ const ordered = [...groups.entries()].sort((left, right) => left[0] - right[0]);
96
+
97
+ for (const [, group] of ordered) {
98
+ // Take the group only if it fits ENTIRELY within the remaining quota of every class it
99
+ // contains. Checking merely "is this class still under quota" admitted the whole group and
100
+ // let it overshoot by hundreds of rows — the real corpus has a 655-row content group, so a
101
+ // nominal 25% fold could be swamped by one category and the base rate the lift is measured
102
+ // against would shift underneath it.
103
+ let fits = true;
104
+ for (const [label, count] of group.counts) {
105
+ if ((taken.get(label) || 0) + count > (quotas.get(label) || 0)) { fits = false; break; }
106
+ }
107
+ if (fits) {
108
+ test.push(...group.members);
109
+ for (const [label, count] of group.counts) taken.set(label, (taken.get(label) || 0) + count);
110
+ } else {
111
+ train.push(...group.members);
112
+ }
113
+ }
114
+
115
+ // Both classes must appear in the test fold. With a small minority class the floor()'d quota
116
+ // can be 0, which would leave every minority group in training and produce a single-class
117
+ // test fold: ROC-AUC is undefined there and precision/recall/MCC are degenerate, yet the
118
+ // report would still be marked available. Refusing to split is the honest outcome.
119
+ const testClasses = new Set(test.map((example) => toBinaryLabel(example.label)));
120
+ if (testClasses.size < 2) return { train: examples.slice(), test: [] };
121
+
122
+ // If either side collapsed, report no test fold instead of a meaningless one.
123
+ if (test.length === 0 || train.length === 0) return { train: examples.slice(), test: [] };
124
+ return { train, test };
125
+ }
126
+
127
+ /**
128
+ * Threshold metrics. `pairs` is [{ probability, label }].
129
+ * Precision/recall are reported for the positive (high-risk) class.
130
+ */
131
+ function classificationMetrics(pairs, threshold = 0.5) {
132
+ let truePositive = 0;
133
+ let falsePositive = 0;
134
+ let trueNegative = 0;
135
+ let falseNegative = 0;
136
+
137
+ for (const pair of pairs) {
138
+ const actual = toBinaryLabel(pair.label);
139
+ const predicted = Number(pair.probability) >= threshold ? 1 : 0;
140
+ if (predicted === 1 && actual === 1) truePositive += 1;
141
+ else if (predicted === 1 && actual === 0) falsePositive += 1;
142
+ else if (predicted === 0 && actual === 0) trueNegative += 1;
143
+ else falseNegative += 1;
144
+ }
145
+
146
+ const total = pairs.length || 1;
147
+ const precision = truePositive + falsePositive > 0 ? truePositive / (truePositive + falsePositive) : 0;
148
+ const recall = truePositive + falseNegative > 0 ? truePositive / (truePositive + falseNegative) : 0;
149
+ const specificity = trueNegative + falsePositive > 0 ? trueNegative / (trueNegative + falsePositive) : 0;
150
+ const f1 = precision + recall > 0 ? (2 * precision * recall) / (precision + recall) : 0;
151
+
152
+ // Matthews correlation: the summary that does not flatter a majority-class predictor.
153
+ // A constant classifier scores 0 here no matter how skewed the corpus is.
154
+ const mccDenominator = Math.sqrt(
155
+ (truePositive + falsePositive)
156
+ * (truePositive + falseNegative)
157
+ * (trueNegative + falsePositive)
158
+ * (trueNegative + falseNegative),
159
+ );
160
+ const mcc = mccDenominator > 0
161
+ ? ((truePositive * trueNegative) - (falsePositive * falseNegative)) / mccDenominator
162
+ : 0;
163
+
164
+ return {
165
+ accuracy: (truePositive + trueNegative) / total,
166
+ precision,
167
+ recall,
168
+ specificity,
169
+ f1,
170
+ mcc,
171
+ confusion: { truePositive, falsePositive, trueNegative, falseNegative },
172
+ };
173
+ }
174
+
175
+ /**
176
+ * ROC-AUC by the rank method, with tied scores sharing an average rank.
177
+ *
178
+ * Tie handling matters here specifically: a stump ensemble emits a small set of discrete
179
+ * scores, so ties are the common case rather than an edge case. Treating them by input order
180
+ * would make AUC depend on row order — a number that changes when you sort your log.
181
+ */
182
+ function rocAuc(pairs) {
183
+ const positives = pairs.filter((pair) => toBinaryLabel(pair.label) === 1).length;
184
+ const negatives = pairs.length - positives;
185
+ if (positives === 0 || negatives === 0) return null; // undefined, and saying so beats guessing
186
+
187
+ const ordered = pairs
188
+ .map((pair) => ({ score: Number(pair.probability), label: toBinaryLabel(pair.label) }))
189
+ .sort((left, right) => left.score - right.score);
190
+
191
+ const ranks = new Array(ordered.length);
192
+ let index = 0;
193
+ while (index < ordered.length) {
194
+ let end = index;
195
+ while (end + 1 < ordered.length && ordered[end + 1].score === ordered[index].score) end += 1;
196
+ // Ranks are 1-based; tied entries all take the midpoint of the block they occupy.
197
+ const averageRank = ((index + 1) + (end + 1)) / 2;
198
+ for (let position = index; position <= end; position += 1) ranks[position] = averageRank;
199
+ index = end + 1;
200
+ }
201
+
202
+ let positiveRankSum = 0;
203
+ ordered.forEach((entry, position) => {
204
+ if (entry.label === 1) positiveRankSum += ranks[position];
205
+ });
206
+
207
+ return (positiveRankSum - (positives * (positives + 1)) / 2) / (positives * negatives);
208
+ }
209
+
210
+ /** Mean squared error of the probability itself. Rewards being uncertain when uncertain. */
211
+ function brierScore(pairs) {
212
+ if (pairs.length === 0) return null;
213
+ const total = pairs.reduce((sum, pair) => {
214
+ const actual = toBinaryLabel(pair.label);
215
+ const probability = Number(pair.probability);
216
+ return sum + ((probability - actual) ** 2);
217
+ }, 0);
218
+ return total / pairs.length;
219
+ }
220
+
221
+ /**
222
+ * Expected calibration error: |predicted - observed| averaged over equal-width bins,
223
+ * weighted by bin population.
224
+ *
225
+ * This is the metric that catches a model whose ranking is fine but whose probabilities are
226
+ * meaningless — which matters because downstream gates threshold on the probability, not on
227
+ * the rank.
228
+ */
229
+ function calibration(pairs, binCount = 10) {
230
+ if (pairs.length === 0) return { expectedCalibrationError: null, bins: [] };
231
+ const bins = Array.from({ length: binCount }, () => ({ count: 0, predicted: 0, observed: 0 }));
232
+
233
+ for (const pair of pairs) {
234
+ const probability = Math.min(0.999999, Math.max(0, Number(pair.probability)));
235
+ const slot = Math.min(binCount - 1, Math.floor(probability * binCount));
236
+ bins[slot].count += 1;
237
+ bins[slot].predicted += probability;
238
+ bins[slot].observed += toBinaryLabel(pair.label);
239
+ }
240
+
241
+ let error = 0;
242
+ const report = bins.map((bin, slot) => {
243
+ if (bin.count === 0) return { bin: slot, count: 0, meanPredicted: null, meanObserved: null };
244
+ const meanPredicted = bin.predicted / bin.count;
245
+ const meanObserved = bin.observed / bin.count;
246
+ error += (bin.count / pairs.length) * Math.abs(meanPredicted - meanObserved);
247
+ return { bin: slot, count: bin.count, meanPredicted, meanObserved };
248
+ });
249
+
250
+ return { expectedCalibrationError: error, bins: report };
251
+ }
252
+
253
+ /**
254
+ * Full report for a set of (probability, label) pairs.
255
+ *
256
+ * `baselineAccuracy` and `lift` are the point of this whole module: a model is only worth
257
+ * running if it beats the constant classifier, and that comparison must appear next to the
258
+ * accuracy every single time so it can never again be quoted without it.
259
+ */
260
+ function evaluate(pairs, options = {}) {
261
+ const threshold = Number(options.threshold ?? 0.5);
262
+ const positives = pairs.filter((pair) => toBinaryLabel(pair.label) === 1).length;
263
+ const baseRate = pairs.length > 0 ? positives / pairs.length : 0;
264
+ // The trivial classifier always answers with the majority class.
265
+ const baselineAccuracy = Math.max(baseRate, 1 - baseRate);
266
+ const metrics = classificationMetrics(pairs, threshold);
267
+
268
+ return {
269
+ sampleCount: pairs.length,
270
+ positiveCount: positives,
271
+ baseRate,
272
+ baselineAccuracy,
273
+ accuracy: metrics.accuracy,
274
+ lift: metrics.accuracy - baselineAccuracy,
275
+ precision: metrics.precision,
276
+ recall: metrics.recall,
277
+ specificity: metrics.specificity,
278
+ f1: metrics.f1,
279
+ mcc: metrics.mcc,
280
+ rocAuc: rocAuc(pairs),
281
+ brierScore: brierScore(pairs),
282
+ expectedCalibrationError: calibration(pairs, options.calibrationBins || 10).expectedCalibrationError,
283
+ confusion: metrics.confusion,
284
+ };
285
+ }
286
+
287
+ /** Round every float in a report so persisted artifacts diff cleanly. */
288
+ function roundReport(report, digits = 4) {
289
+ const factor = 10 ** digits;
290
+ const rounded = {};
291
+ for (const [key, value] of Object.entries(report || {})) {
292
+ if (typeof value === 'number' && Number.isFinite(value)) rounded[key] = Math.round(value * factor) / factor;
293
+ else if (value && typeof value === 'object' && !Array.isArray(value)) rounded[key] = roundReport(value, digits);
294
+ else rounded[key] = value;
295
+ }
296
+ return rounded;
297
+ }
298
+
299
+ module.exports = {
300
+ hashString,
301
+ stratifiedSplit,
302
+ classificationMetrics,
303
+ rocAuc,
304
+ brierScore,
305
+ calibration,
306
+ evaluate,
307
+ roundReport,
308
+ };
@@ -1,18 +1,41 @@
1
1
  'use strict';
2
2
 
3
- const fs = require('fs');
4
- const path = require('path');
3
+ const crypto = require('node:crypto');
4
+ const fs = require('node:fs');
5
+ const path = require('node:path');
6
+ const { spawn } = require('node:child_process');
5
7
  const { getFeedbackPaths } = require('./feedback-loop');
6
8
  const { ensureDir } = require('./fs-utils');
7
9
  const { loadOptionalModule } = require('./private-core-boundary');
8
10
 
11
+ const RUNNER_SCRIPT_PATH = path.join(__dirname, 'async-job-runner.js');
12
+
13
+ function launchPublicManagedJob(jobSpec, options = {}) {
14
+ const publicRunner = require('./async-job-runner');
15
+ const jobId = options.jobId || jobSpec.id || `job_${Date.now()}_${crypto.randomBytes(4).toString('hex')}`;
16
+ const { jobDir } = publicRunner.getJobRuntimePaths(jobId);
17
+ ensureDir(jobDir);
18
+ const jobFilePath = path.join(jobDir, 'job.json');
19
+ const finalSpec = { ...jobSpec, id: jobId };
20
+ fs.writeFileSync(jobFilePath, `${JSON.stringify(finalSpec, null, 2)}\n`, 'utf8');
21
+ publicRunner.queueJob({ ...finalSpec, jobFilePath });
22
+ const child = spawn(process.execPath, [RUNNER_SCRIPT_PATH, `--run-file=${jobFilePath}`], {
23
+ cwd: options.cwd || process.cwd(),
24
+ env: process.env,
25
+ detached: true,
26
+ stdio: 'ignore',
27
+ });
28
+ child.unref();
29
+ return {
30
+ jobId,
31
+ jobFilePath,
32
+ launchMode: 'public-background',
33
+ pid: child.pid || null,
34
+ };
35
+ }
36
+
9
37
  const launcher = loadOptionalModule(path.join(__dirname, 'hosted-job-launcher'), () => ({
10
- launchManagedJob: () => {
11
- throw new Error('Managed jobs require ThumbGate-Core.');
12
- },
13
- resumeHostedJob: () => {
14
- throw new Error('Resuming hosted jobs requires ThumbGate-Core.');
15
- },
38
+ launchManagedJob: launchPublicManagedJob,
16
39
  }));
17
40
 
18
41
  const runner = loadOptionalModule(path.join(__dirname, 'async-job-runner'), () => ({
@@ -20,8 +43,8 @@ const runner = loadOptionalModule(path.join(__dirname, 'async-job-runner'), () =
20
43
  listJobStates: () => [],
21
44
  }));
22
45
 
23
- const { launchManagedJob, resumeHostedJob } = launcher;
24
- const { readJobState, listJobStates } = runner;
46
+ const { launchManagedJob } = launcher;
47
+ const { readJobState } = runner;
25
48
 
26
49
  const DEFAULT_CONCURRENCY = 3;
27
50
  const POLL_INTERVAL_MS = 200;
@@ -45,7 +68,7 @@ function planWorkflow(objective) {
45
68
  stages: [
46
69
  {
47
70
  name: 'secret_scan',
48
- command: 'node scripts/secret-scanner.js --json || true',
71
+ command: 'node scripts/secret-scanner.js --json',
49
72
  }
50
73
  ]
51
74
  });
@@ -55,7 +78,7 @@ function planWorkflow(objective) {
55
78
  stages: [
56
79
  {
57
80
  name: 'npm_audit',
58
- command: 'npm audit --json || true',
81
+ command: 'npm audit --json',
59
82
  }
60
83
  ]
61
84
  });
@@ -65,7 +88,7 @@ function planWorkflow(objective) {
65
88
  stages: [
66
89
  {
67
90
  name: 'credential_gate_check',
68
- command: 'node scripts/single-use-credential-gate.js plan || true',
91
+ command: 'node scripts/single-use-credential-gate.js plan',
69
92
  }
70
93
  ]
71
94
  });
@@ -76,7 +99,7 @@ function planWorkflow(objective) {
76
99
  stages: [
77
100
  {
78
101
  name: 'run_bench',
79
- command: 'npx thumbgate bench --json --min-score=90 || true',
102
+ command: 'npx thumbgate bench --json --min-score=90',
80
103
  }
81
104
  ]
82
105
  });
@@ -86,7 +109,7 @@ function planWorkflow(objective) {
86
109
  stages: [
87
110
  {
88
111
  name: 'budget_status',
89
- command: 'node scripts/budget-guard.js --status || true',
112
+ command: 'node scripts/budget-guard.js --status',
90
113
  }
91
114
  ]
92
115
  });
@@ -98,7 +121,7 @@ function planWorkflow(objective) {
98
121
  stages: [
99
122
  {
100
123
  name: 'search_fs',
101
- command: 'node scripts/filesystem-search.js --query="pretool" --limit=5 || true',
124
+ command: 'node scripts/filesystem-search.js --query="pretool" --limit=5',
102
125
  }
103
126
  ]
104
127
  });
@@ -108,7 +131,7 @@ function planWorkflow(objective) {
108
131
  stages: [
109
132
  {
110
133
  name: 'ops_integrity',
111
- command: 'node scripts/operational-integrity.js --ci || true',
134
+ command: 'node scripts/operational-integrity.js --ci',
112
135
  }
113
136
  ]
114
137
  });
@@ -119,7 +142,7 @@ function planWorkflow(objective) {
119
142
  plannedAt: nowIso(),
120
143
  subtasks: subtasks.map((task, idx) => ({
121
144
  ...task,
122
- id: `subtask_${Date.now()}_${idx}_${Math.random().toString(36).slice(2, 6)}`,
145
+ id: `subtask_${Date.now()}_${idx}_${crypto.randomBytes(3).toString('hex')}`,
123
146
  autoImprove: false,
124
147
  verificationMode: 'none',
125
148
  recordFeedback: false,
@@ -132,12 +155,14 @@ function planWorkflow(objective) {
132
155
  * Polls active jobs until all complete, then consolidates the results.
133
156
  */
134
157
  async function executeWorkflow(objective, options = {}) {
135
- const plan = planWorkflow(objective);
158
+ const plan = options.plan || planWorkflow(objective);
136
159
  const concurrency = Number(options.concurrency) || DEFAULT_CONCURRENCY;
137
160
  const timeoutMs = Number(options.timeoutMs) || 60000; // 60s timeout safety
161
+ const launchJob = options.launchManagedJob || launchManagedJob;
162
+ const getJobState = options.readJobState || readJobState;
138
163
 
139
164
  const { FEEDBACK_DIR } = getFeedbackPaths();
140
- const workflowId = `wf_${Date.now()}_${Math.random().toString(36).slice(2, 8)}`;
165
+ const workflowId = `wf_${Date.now()}_${crypto.randomBytes(4).toString('hex')}`;
141
166
  const workflowDir = path.join(FEEDBACK_DIR, 'workflows', workflowId);
142
167
  ensureDir(workflowDir);
143
168
 
@@ -145,17 +170,33 @@ async function executeWorkflow(objective, options = {}) {
145
170
  const queue = [...plan.subtasks];
146
171
  const results = [];
147
172
  const start = Date.now();
173
+ const statePath = path.join(workflowDir, 'state.json');
174
+
175
+ const persistState = (status) => {
176
+ fs.writeFileSync(statePath, `${JSON.stringify({
177
+ workflowId,
178
+ objective,
179
+ status,
180
+ updatedAt: nowIso(),
181
+ queue: queue.map((task) => ({ id: task.id, name: task.name })),
182
+ activeJobs: [...activeJobs.entries()].map(([taskId, info]) => ({ taskId, ...info })),
183
+ results,
184
+ }, null, 2)}\n`, 'utf8');
185
+ };
148
186
 
149
187
  const runNext = () => {
150
188
  while (activeJobs.size < concurrency && queue.length > 0) {
151
189
  const task = queue.shift();
152
- const launched = launchManagedJob(task, { cwd: options.cwd });
190
+ const launched = launchJob(task, { cwd: options.cwd });
153
191
  activeJobs.set(task.id, {
154
192
  jobId: launched.jobId,
155
193
  taskName: task.name,
156
194
  launchedAt: Date.now(),
195
+ pid: launched.pid || null,
196
+ launchMode: launched.launchMode || 'managed',
157
197
  });
158
198
  }
199
+ persistState('running');
159
200
  };
160
201
 
161
202
  runNext();
@@ -166,7 +207,7 @@ async function executeWorkflow(objective, options = {}) {
166
207
  let allDone = true;
167
208
 
168
209
  for (const [taskId, info] of activeJobs.entries()) {
169
- const jobState = readJobState(info.jobId);
210
+ const jobState = getJobState(info.jobId);
170
211
  if (!jobState) {
171
212
  allDone = false;
172
213
  continue;
@@ -193,11 +234,21 @@ async function executeWorkflow(objective, options = {}) {
193
234
  const elapsed = Date.now() - start;
194
235
  if (allDone && queue.length === 0) {
195
236
  clearInterval(interval);
237
+ persistState(results.every((result) => result.status === 'completed')
238
+ ? 'completed'
239
+ : 'completed_with_failures');
196
240
  resolve();
197
241
  } else if (elapsed >= timeoutMs) {
198
242
  clearInterval(interval);
199
243
  // Timeout remaining active tasks
200
244
  for (const [taskId, info] of activeJobs.entries()) {
245
+ if (info.pid) {
246
+ try {
247
+ process.kill(process.platform === 'win32' ? info.pid : -info.pid, 'SIGTERM');
248
+ } catch {
249
+ // The worker may have exited between the last poll and timeout.
250
+ }
251
+ }
201
252
  results.push({
202
253
  taskId,
203
254
  taskName: info.taskName,
@@ -206,6 +257,17 @@ async function executeWorkflow(objective, options = {}) {
206
257
  lastError: { message: `Subtask timed out after ${timeoutMs}ms`, code: 'TIMEOUT' },
207
258
  });
208
259
  }
260
+ for (const task of queue.splice(0)) {
261
+ results.push({
262
+ taskId: task.id,
263
+ taskName: task.name,
264
+ jobId: null,
265
+ status: 'timeout',
266
+ lastError: { message: `Subtask was not launched before ${timeoutMs}ms`, code: 'TIMEOUT' },
267
+ });
268
+ }
269
+ activeJobs.clear();
270
+ persistState('timed_out');
209
271
  resolve();
210
272
  }
211
273
  }, POLL_INTERVAL_MS);
@@ -234,6 +296,7 @@ async function executeWorkflow(objective, options = {}) {
234
296
  durationMs,
235
297
  reportPath,
236
298
  results,
299
+ statePath,
237
300
  };
238
301
  }
239
302
 
@@ -290,4 +353,5 @@ module.exports = {
290
353
  planWorkflow,
291
354
  executeWorkflow,
292
355
  compileWorkflowReport,
356
+ launchPublicManagedJob,
293
357
  };
@@ -143,6 +143,8 @@ function handleRejectExpectation(checks, result, expected) {
143
143
 
144
144
  const wasRejected = result.accepted === false
145
145
  || result.status === 'rejected'
146
+ || result.status === 'clarification_required'
147
+ || result.needsClarification === true
146
148
  || result.actionType === 'no-action';
147
149
  checks.push({
148
150
  criterion: 'shouldReject',
@@ -466,8 +468,9 @@ function runSuiteObject(suite, options = {}) {
466
468
  const skipped = results.filter((r) => r.status === 'skip').length;
467
469
  const totalScore = results.length > 0
468
470
  ? Math.round(results.reduce((s, r) => s + r.score, 0) / results.length)
469
- : 100;
471
+ : 0;
470
472
  const minScore = options.minScore ?? 80;
473
+ const insufficientEvidence = results.length === 0;
471
474
 
472
475
  return {
473
476
  suite: suite.name,
@@ -478,8 +481,9 @@ function runSuiteObject(suite, options = {}) {
478
481
  skipped,
479
482
  score: totalScore,
480
483
  minScore,
481
- pass: totalScore >= minScore,
482
- noCases: results.length === 0,
484
+ pass: !insufficientEvidence && totalScore >= minScore,
485
+ noCases: insufficientEvidence,
486
+ evidenceStatus: insufficientEvidence ? 'insufficient_evidence' : 'measured',
483
487
  feedbackDerived: suite.source && suite.source.type === 'feedback-log',
484
488
  generatedAt: new Date().toISOString(),
485
489
  results,
@@ -685,6 +689,7 @@ function runSuite(suitePath = DEFAULT_SUITE, options = {}) {
685
689
 
686
690
  function compareReports(currentReport, baselineReport) {
687
691
  const baselineById = new Map((baselineReport?.results || []).map((result) => [result.id, result]));
692
+ const currentById = new Map((currentReport?.results || []).map((result) => [result.id, result]));
688
693
  const regressions = [];
689
694
  const improvements = [];
690
695
 
@@ -717,12 +722,83 @@ function compareReports(currentReport, baselineReport) {
717
722
  }
718
723
  }
719
724
 
725
+ for (const baseline of baselineReport?.results || []) {
726
+ if (currentById.has(baseline.id)) continue;
727
+ regressions.push({
728
+ id: baseline.id,
729
+ baselineScore: baseline.score,
730
+ currentScore: null,
731
+ delta: null,
732
+ baselineStatus: baseline.status,
733
+ currentStatus: 'missing',
734
+ });
735
+ }
736
+
720
737
  return {
721
738
  baselineSuite: baselineReport?.suite || null,
722
739
  baselineScore: Number.isFinite(Number(baselineReport?.score)) ? Number(baselineReport.score) : null,
723
740
  scoreDelta: Number.isFinite(Number(baselineReport?.score)) ? currentReport.score - Number(baselineReport.score) : null,
724
741
  regressions,
725
742
  improvements,
743
+ baselineCases: baselineById.size,
744
+ currentCases: currentById.size,
745
+ baselineCoverageRate: baselineById.size
746
+ ? Math.round((Array.from(baselineById.keys()).filter((id) => currentById.has(id)).length / baselineById.size) * 10000) / 10000
747
+ : null,
748
+ };
749
+ }
750
+
751
+ function logSafeResult(result, index) {
752
+ const allowedStatuses = new Set(['pass', 'fail', 'error', 'skip']);
753
+ return {
754
+ case: index + 1,
755
+ status: allowedStatuses.has(result.status) ? result.status : 'error',
756
+ score: Number(result.score || 0),
757
+ passCount: Number(result.passCount || 0),
758
+ totalChecks: Number(result.totalChecks || 0),
759
+ };
760
+ }
761
+
762
+ function logSafeReport(report, suite, fromFeedback) {
763
+ const source = suite?.source || {};
764
+ const comparison = report.comparison
765
+ ? {
766
+ baselineScore: Number(report.comparison.baselineScore || 0),
767
+ scoreDelta: Number(report.comparison.scoreDelta || 0),
768
+ regressionCount: Number(report.comparison.regressions?.length || 0),
769
+ improvementCount: Number(report.comparison.improvements?.length || 0),
770
+ baselineCases: Number(report.comparison.baselineCases || 0),
771
+ currentCases: Number(report.comparison.currentCases || 0),
772
+ baselineCoverageRate: report.comparison.baselineCoverageRate,
773
+ }
774
+ : undefined;
775
+ return {
776
+ suiteType: fromFeedback ? 'feedback-derived' : 'configured',
777
+ total: Number(report.total || 0),
778
+ passed: Number(report.passed || 0),
779
+ failed: Number(report.failed || 0),
780
+ errors: Number(report.errors || 0),
781
+ skipped: Number(report.skipped || 0),
782
+ score: Number(report.score || 0),
783
+ minScore: Number(report.minScore || 0),
784
+ pass: report.pass === true,
785
+ noCases: report.noCases === true,
786
+ evidenceStatus: report.evidenceStatus === 'measured' ? 'measured' : 'insufficient_evidence',
787
+ feedbackDerived: report.feedbackDerived === true,
788
+ syntheticCount: Number(report.syntheticCount || 0),
789
+ comparison,
790
+ results: (report.results || []).map(logSafeResult),
791
+ suiteDefinition: fromFeedback
792
+ ? {
793
+ version: Number(suite?.version || 0),
794
+ source: {
795
+ type: 'feedback-log',
796
+ totalEntries: Number(source.totalEntries || 0),
797
+ selectedCases: Number(source.selectedCases || 0),
798
+ },
799
+ evaluationCount: Number(suite?.evaluations?.length || 0),
800
+ }
801
+ : undefined,
726
802
  };
727
803
  }
728
804
 
@@ -847,7 +923,7 @@ if (isCliInvocation()) {
847
923
  }
848
924
 
849
925
  if (json) {
850
- console.log(JSON.stringify({ ...report, suiteDefinition: fromFeedback ? suite : undefined }, null, 2));
926
+ console.log(JSON.stringify(logSafeReport(report, suite, fromFeedback), null, 2));
851
927
  } else {
852
928
  console.log(`\n${report.suite}`);
853
929
  console.log('='.repeat(50));
@@ -883,6 +959,7 @@ module.exports = {
883
959
  gradeOutput,
884
960
  loadSuite,
885
961
  loadReport,
962
+ logSafeReport,
886
963
  compareReports,
887
964
  readJsonl,
888
965
  runEvaluation,