thumbgate 1.29.1 → 1.30.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude/commands/dashboard.md +11 -1
- package/.claude/commands/thumbgate-dashboard.md +23 -8
- package/.claude-plugin/plugin.json +1 -1
- package/.well-known/mcp/server-card.json +1 -1
- package/README.md +61 -1
- package/adapters/claude/.mcp.json +2 -2
- package/adapters/forge/forge.yaml +3 -3
- package/adapters/mcp/server-stdio.js +164 -7
- package/adapters/opencode/opencode.json +1 -1
- package/bin/cli.js +7 -5
- package/commands/dashboard.md +11 -1
- package/commands/thumbgate-dashboard.md +23 -8
- package/config/agent-outcome-monitor-thresholds.json +63 -0
- package/config/evals/agent-outcomes-baseline.json +17 -0
- package/config/evals/agent-outcomes-golden.json +412 -0
- package/config/evals/prompt-eval-baseline.json +23 -0
- package/config/mcp-allowlists.json +26 -2
- package/config/post-deploy-marketing-pages.json +26 -1
- package/config/schemas/task-outcome-receipt.schema.json +296 -0
- package/openapi/openapi.yaml +235 -0
- package/package.json +55 -11
- package/public/architecture.html +130 -0
- package/public/assets/diagrams/agent-integration.png +0 -0
- package/public/assets/diagrams/before-after.svg +21 -0
- package/public/assets/diagrams/decision.svg +36 -0
- package/public/assets/diagrams/feedback-pipeline.png +0 -0
- package/public/assets/diagrams/loop.svg +34 -0
- package/public/assets/diagrams/plugin-topology.png +0 -0
- package/public/assets/diagrams/pre-action-gate-loop.svg +59 -0
- package/public/assets/diagrams/stack.svg +18 -0
- package/public/assets/diagrams/thumbgate-architecture.png +0 -0
- package/public/case-studies.html +151 -0
- package/public/eval-scorecard.html +195 -0
- package/public/eval-scorecard.json +18 -0
- package/public/evaluations.html +168 -0
- package/public/index.html +6 -3
- package/public/numbers.html +2 -2
- package/public/whitepaper.html +189 -0
- package/scripts/activation-quickstart.js +1 -0
- package/scripts/agent-outcome-eval.js +130 -0
- package/scripts/agent-outcome-monitor.js +331 -0
- package/scripts/agent-reasoning-traces.js +8 -9
- package/scripts/async-job-runner.js +107 -13
- package/scripts/billing.js +3 -1
- package/scripts/claude-feedback-sync.js +3 -2
- package/scripts/cli-feedback.js +13 -7
- package/scripts/cross-encoder-reranker.js +3 -0
- package/scripts/durability/step.js +121 -12
- package/scripts/feedback-aggregate.js +5 -2
- package/scripts/feedback-loop.js +244 -182
- package/scripts/gates-engine.js +512 -22
- package/scripts/generate-case-study-outreach.js +253 -0
- package/scripts/generate-eval-scorecard.js +276 -0
- package/scripts/growth-campaigns.js +183 -0
- package/scripts/human-escalation.js +265 -0
- package/scripts/hybrid-feedback-context.js +93 -50
- package/scripts/jsonl-watcher.js +1 -0
- package/scripts/judge-reward-function.js +30 -18
- package/scripts/lesson-inference.js +23 -4
- package/scripts/lesson-retrieval.js +71 -4
- package/scripts/lesson-search.js +26 -3
- package/scripts/mcp-config.js +26 -5
- package/scripts/mcp-oauth.js +37 -2
- package/scripts/model-eval.js +308 -0
- package/scripts/parallel-workflow-orchestrator.js +86 -22
- package/scripts/prompt-eval.js +81 -4
- package/scripts/published-cli.js +11 -1
- package/scripts/refresh-proof-pack.js +261 -0
- package/scripts/risk-scorer.js +144 -15
- package/scripts/schedule-manager.js +249 -0
- package/scripts/statusline-local-stats.js +1 -1
- package/scripts/task-outcomes.js +425 -0
- package/scripts/thumbgate-bench.js +13 -0
- package/scripts/tool-contract-validator.js +287 -59
- package/scripts/tool-kpi-tracker.js +124 -0
- package/scripts/tool-registry.js +192 -1
- package/src/api/server.js +355 -89
|
@@ -0,0 +1,308 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
'use strict';
|
|
3
|
+
|
|
4
|
+
/**
|
|
5
|
+
* model-eval.js — honest evaluation primitives for ThumbGate's learned models.
|
|
6
|
+
*
|
|
7
|
+
* WHY THIS EXISTS
|
|
8
|
+
*
|
|
9
|
+
* The risk scorer reported exactly one number: `trainingAccuracy`, measured on the same rows
|
|
10
|
+
* it trained on. On 2026-07-28 that number was 0.820 — which sounds good until you notice the
|
|
11
|
+
* base rate was 0.711. A model that answers "high-risk" unconditionally scores 71.1%, so the
|
|
12
|
+
* headline figure was measuring an 11-point in-sample lift and presenting it as quality.
|
|
13
|
+
* Nothing anywhere compared against that trivial baseline, and no split existed, so the
|
|
14
|
+
* generalization number was not merely bad — it was unknown.
|
|
15
|
+
*
|
|
16
|
+
* Accuracy is also the wrong summary for a 71/29 split. These primitives therefore report the
|
|
17
|
+
* metrics that survive class imbalance (precision/recall/F1/MCC/ROC-AUC) and the ones that say
|
|
18
|
+
* whether a probability means anything (Brier score, expected calibration error).
|
|
19
|
+
*
|
|
20
|
+
* DETERMINISM IS A REQUIREMENT, NOT A PREFERENCE.
|
|
21
|
+
* Splits are derived from a content hash, never from Math.random(). The same corpus must
|
|
22
|
+
* produce the same split on every machine and every run, or a "quality regression" is
|
|
23
|
+
* indistinguishable from a reshuffle and the CI gate built on top is noise.
|
|
24
|
+
*
|
|
25
|
+
* Everything here is pure: no I/O, no clock, no global state. That is what makes it testable.
|
|
26
|
+
*/
|
|
27
|
+
|
|
28
|
+
/** FNV-1a. Small, fast, and stable across platforms — the only properties we need. */
|
|
29
|
+
function hashString(text) {
|
|
30
|
+
let hash = 0x811c9dc5;
|
|
31
|
+
const value = String(text);
|
|
32
|
+
for (let index = 0; index < value.length; index += 1) {
|
|
33
|
+
hash ^= value.charCodeAt(index);
|
|
34
|
+
// 32-bit FNV prime multiply via shifts; Math.imul keeps this exact in JS.
|
|
35
|
+
hash = Math.imul(hash, 0x01000193) >>> 0;
|
|
36
|
+
}
|
|
37
|
+
return hash >>> 0;
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
/** Labels arrive as 1/-1 (AdaBoost) or 1/0 (everything else). Normalize to 1/0. */
|
|
41
|
+
function toBinaryLabel(label) {
|
|
42
|
+
return Number(label) === 1 ? 1 : 0;
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
/**
|
|
46
|
+
* Deterministic stratified split.
|
|
47
|
+
*
|
|
48
|
+
* Stratified because a random split of a 71/29 corpus can hand back a test fold with a
|
|
49
|
+
* materially different base rate, which moves the very number we are trying to measure.
|
|
50
|
+
* Splitting inside each class keeps the test fold's base rate close to the corpus.
|
|
51
|
+
*
|
|
52
|
+
* Returns { train, test }. Both are always non-empty when the input supports it; when a class
|
|
53
|
+
* is too small to contribute a test row, the split degrades to "everything is training" rather
|
|
54
|
+
* than silently producing a test fold with one class in it (an AUC of NaN dressed up as data).
|
|
55
|
+
*/
|
|
56
|
+
function stratifiedSplit(examples, options = {}) {
|
|
57
|
+
const testFraction = Number(options.testFraction || 0.25);
|
|
58
|
+
const keyFn = options.keyFn || ((example, index) => JSON.stringify(example.features || example) + index);
|
|
59
|
+
|
|
60
|
+
// Group ACROSS classes, then assign whole groups — StratifiedGroupKFold semantics.
|
|
61
|
+
//
|
|
62
|
+
// Two bugs led here, both caught by tests/risk-model-quality.test.js:
|
|
63
|
+
// 1. Assigning individual rows split tied blocks, so an identical row could sit in both
|
|
64
|
+
// folds and the test fold scored rows the model had memorized.
|
|
65
|
+
// 2. Grouping per class still split duplicates, because with label noise the SAME feature
|
|
66
|
+
// vector can carry both labels — the two copies then landed in different class buckets
|
|
67
|
+
// and different folds.
|
|
68
|
+
// Grouping globally is the only version where "this input is in exactly one fold" holds.
|
|
69
|
+
const classTotals = new Map();
|
|
70
|
+
const groups = new Map();
|
|
71
|
+
examples.forEach((example, index) => {
|
|
72
|
+
const label = toBinaryLabel(example.label);
|
|
73
|
+
classTotals.set(label, (classTotals.get(label) || 0) + 1);
|
|
74
|
+
const hash = hashString(keyFn(example, index));
|
|
75
|
+
if (!groups.has(hash)) groups.set(hash, { members: [], counts: new Map() });
|
|
76
|
+
const group = groups.get(hash);
|
|
77
|
+
group.members.push(example);
|
|
78
|
+
group.counts.set(label, (group.counts.get(label) || 0) + 1);
|
|
79
|
+
});
|
|
80
|
+
|
|
81
|
+
// A single-class corpus has nothing to measure; there is no honest split of it.
|
|
82
|
+
if (classTotals.size < 2) return { train: examples.slice(), test: [] };
|
|
83
|
+
|
|
84
|
+
// Per-class quotas keep the test fold's base rate near the corpus even though whole groups
|
|
85
|
+
// move together. Without quotas, one large group could swamp the fold and shift the very
|
|
86
|
+
// base rate the lift is measured against.
|
|
87
|
+
const quotas = new Map();
|
|
88
|
+
for (const [label, total] of classTotals) quotas.set(label, Math.floor(total * testFraction));
|
|
89
|
+
|
|
90
|
+
const train = [];
|
|
91
|
+
const test = [];
|
|
92
|
+
const taken = new Map();
|
|
93
|
+
// Order by hash, not by position: input order in a JSONL log is chronological, so a
|
|
94
|
+
// positional split would put all recent rows in one fold and measure drift, not skill.
|
|
95
|
+
const ordered = [...groups.entries()].sort((left, right) => left[0] - right[0]);
|
|
96
|
+
|
|
97
|
+
for (const [, group] of ordered) {
|
|
98
|
+
// Take the group only if it fits ENTIRELY within the remaining quota of every class it
|
|
99
|
+
// contains. Checking merely "is this class still under quota" admitted the whole group and
|
|
100
|
+
// let it overshoot by hundreds of rows — the real corpus has a 655-row content group, so a
|
|
101
|
+
// nominal 25% fold could be swamped by one category and the base rate the lift is measured
|
|
102
|
+
// against would shift underneath it.
|
|
103
|
+
let fits = true;
|
|
104
|
+
for (const [label, count] of group.counts) {
|
|
105
|
+
if ((taken.get(label) || 0) + count > (quotas.get(label) || 0)) { fits = false; break; }
|
|
106
|
+
}
|
|
107
|
+
if (fits) {
|
|
108
|
+
test.push(...group.members);
|
|
109
|
+
for (const [label, count] of group.counts) taken.set(label, (taken.get(label) || 0) + count);
|
|
110
|
+
} else {
|
|
111
|
+
train.push(...group.members);
|
|
112
|
+
}
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
// Both classes must appear in the test fold. With a small minority class the floor()'d quota
|
|
116
|
+
// can be 0, which would leave every minority group in training and produce a single-class
|
|
117
|
+
// test fold: ROC-AUC is undefined there and precision/recall/MCC are degenerate, yet the
|
|
118
|
+
// report would still be marked available. Refusing to split is the honest outcome.
|
|
119
|
+
const testClasses = new Set(test.map((example) => toBinaryLabel(example.label)));
|
|
120
|
+
if (testClasses.size < 2) return { train: examples.slice(), test: [] };
|
|
121
|
+
|
|
122
|
+
// If either side collapsed, report no test fold instead of a meaningless one.
|
|
123
|
+
if (test.length === 0 || train.length === 0) return { train: examples.slice(), test: [] };
|
|
124
|
+
return { train, test };
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
/**
|
|
128
|
+
* Threshold metrics. `pairs` is [{ probability, label }].
|
|
129
|
+
* Precision/recall are reported for the positive (high-risk) class.
|
|
130
|
+
*/
|
|
131
|
+
function classificationMetrics(pairs, threshold = 0.5) {
|
|
132
|
+
let truePositive = 0;
|
|
133
|
+
let falsePositive = 0;
|
|
134
|
+
let trueNegative = 0;
|
|
135
|
+
let falseNegative = 0;
|
|
136
|
+
|
|
137
|
+
for (const pair of pairs) {
|
|
138
|
+
const actual = toBinaryLabel(pair.label);
|
|
139
|
+
const predicted = Number(pair.probability) >= threshold ? 1 : 0;
|
|
140
|
+
if (predicted === 1 && actual === 1) truePositive += 1;
|
|
141
|
+
else if (predicted === 1 && actual === 0) falsePositive += 1;
|
|
142
|
+
else if (predicted === 0 && actual === 0) trueNegative += 1;
|
|
143
|
+
else falseNegative += 1;
|
|
144
|
+
}
|
|
145
|
+
|
|
146
|
+
const total = pairs.length || 1;
|
|
147
|
+
const precision = truePositive + falsePositive > 0 ? truePositive / (truePositive + falsePositive) : 0;
|
|
148
|
+
const recall = truePositive + falseNegative > 0 ? truePositive / (truePositive + falseNegative) : 0;
|
|
149
|
+
const specificity = trueNegative + falsePositive > 0 ? trueNegative / (trueNegative + falsePositive) : 0;
|
|
150
|
+
const f1 = precision + recall > 0 ? (2 * precision * recall) / (precision + recall) : 0;
|
|
151
|
+
|
|
152
|
+
// Matthews correlation: the summary that does not flatter a majority-class predictor.
|
|
153
|
+
// A constant classifier scores 0 here no matter how skewed the corpus is.
|
|
154
|
+
const mccDenominator = Math.sqrt(
|
|
155
|
+
(truePositive + falsePositive)
|
|
156
|
+
* (truePositive + falseNegative)
|
|
157
|
+
* (trueNegative + falsePositive)
|
|
158
|
+
* (trueNegative + falseNegative),
|
|
159
|
+
);
|
|
160
|
+
const mcc = mccDenominator > 0
|
|
161
|
+
? ((truePositive * trueNegative) - (falsePositive * falseNegative)) / mccDenominator
|
|
162
|
+
: 0;
|
|
163
|
+
|
|
164
|
+
return {
|
|
165
|
+
accuracy: (truePositive + trueNegative) / total,
|
|
166
|
+
precision,
|
|
167
|
+
recall,
|
|
168
|
+
specificity,
|
|
169
|
+
f1,
|
|
170
|
+
mcc,
|
|
171
|
+
confusion: { truePositive, falsePositive, trueNegative, falseNegative },
|
|
172
|
+
};
|
|
173
|
+
}
|
|
174
|
+
|
|
175
|
+
/**
|
|
176
|
+
* ROC-AUC by the rank method, with tied scores sharing an average rank.
|
|
177
|
+
*
|
|
178
|
+
* Tie handling matters here specifically: a stump ensemble emits a small set of discrete
|
|
179
|
+
* scores, so ties are the common case rather than an edge case. Treating them by input order
|
|
180
|
+
* would make AUC depend on row order — a number that changes when you sort your log.
|
|
181
|
+
*/
|
|
182
|
+
function rocAuc(pairs) {
|
|
183
|
+
const positives = pairs.filter((pair) => toBinaryLabel(pair.label) === 1).length;
|
|
184
|
+
const negatives = pairs.length - positives;
|
|
185
|
+
if (positives === 0 || negatives === 0) return null; // undefined, and saying so beats guessing
|
|
186
|
+
|
|
187
|
+
const ordered = pairs
|
|
188
|
+
.map((pair) => ({ score: Number(pair.probability), label: toBinaryLabel(pair.label) }))
|
|
189
|
+
.sort((left, right) => left.score - right.score);
|
|
190
|
+
|
|
191
|
+
const ranks = new Array(ordered.length);
|
|
192
|
+
let index = 0;
|
|
193
|
+
while (index < ordered.length) {
|
|
194
|
+
let end = index;
|
|
195
|
+
while (end + 1 < ordered.length && ordered[end + 1].score === ordered[index].score) end += 1;
|
|
196
|
+
// Ranks are 1-based; tied entries all take the midpoint of the block they occupy.
|
|
197
|
+
const averageRank = ((index + 1) + (end + 1)) / 2;
|
|
198
|
+
for (let position = index; position <= end; position += 1) ranks[position] = averageRank;
|
|
199
|
+
index = end + 1;
|
|
200
|
+
}
|
|
201
|
+
|
|
202
|
+
let positiveRankSum = 0;
|
|
203
|
+
ordered.forEach((entry, position) => {
|
|
204
|
+
if (entry.label === 1) positiveRankSum += ranks[position];
|
|
205
|
+
});
|
|
206
|
+
|
|
207
|
+
return (positiveRankSum - (positives * (positives + 1)) / 2) / (positives * negatives);
|
|
208
|
+
}
|
|
209
|
+
|
|
210
|
+
/** Mean squared error of the probability itself. Rewards being uncertain when uncertain. */
|
|
211
|
+
function brierScore(pairs) {
|
|
212
|
+
if (pairs.length === 0) return null;
|
|
213
|
+
const total = pairs.reduce((sum, pair) => {
|
|
214
|
+
const actual = toBinaryLabel(pair.label);
|
|
215
|
+
const probability = Number(pair.probability);
|
|
216
|
+
return sum + ((probability - actual) ** 2);
|
|
217
|
+
}, 0);
|
|
218
|
+
return total / pairs.length;
|
|
219
|
+
}
|
|
220
|
+
|
|
221
|
+
/**
|
|
222
|
+
* Expected calibration error: |predicted - observed| averaged over equal-width bins,
|
|
223
|
+
* weighted by bin population.
|
|
224
|
+
*
|
|
225
|
+
* This is the metric that catches a model whose ranking is fine but whose probabilities are
|
|
226
|
+
* meaningless — which matters because downstream gates threshold on the probability, not on
|
|
227
|
+
* the rank.
|
|
228
|
+
*/
|
|
229
|
+
function calibration(pairs, binCount = 10) {
|
|
230
|
+
if (pairs.length === 0) return { expectedCalibrationError: null, bins: [] };
|
|
231
|
+
const bins = Array.from({ length: binCount }, () => ({ count: 0, predicted: 0, observed: 0 }));
|
|
232
|
+
|
|
233
|
+
for (const pair of pairs) {
|
|
234
|
+
const probability = Math.min(0.999999, Math.max(0, Number(pair.probability)));
|
|
235
|
+
const slot = Math.min(binCount - 1, Math.floor(probability * binCount));
|
|
236
|
+
bins[slot].count += 1;
|
|
237
|
+
bins[slot].predicted += probability;
|
|
238
|
+
bins[slot].observed += toBinaryLabel(pair.label);
|
|
239
|
+
}
|
|
240
|
+
|
|
241
|
+
let error = 0;
|
|
242
|
+
const report = bins.map((bin, slot) => {
|
|
243
|
+
if (bin.count === 0) return { bin: slot, count: 0, meanPredicted: null, meanObserved: null };
|
|
244
|
+
const meanPredicted = bin.predicted / bin.count;
|
|
245
|
+
const meanObserved = bin.observed / bin.count;
|
|
246
|
+
error += (bin.count / pairs.length) * Math.abs(meanPredicted - meanObserved);
|
|
247
|
+
return { bin: slot, count: bin.count, meanPredicted, meanObserved };
|
|
248
|
+
});
|
|
249
|
+
|
|
250
|
+
return { expectedCalibrationError: error, bins: report };
|
|
251
|
+
}
|
|
252
|
+
|
|
253
|
+
/**
|
|
254
|
+
* Full report for a set of (probability, label) pairs.
|
|
255
|
+
*
|
|
256
|
+
* `baselineAccuracy` and `lift` are the point of this whole module: a model is only worth
|
|
257
|
+
* running if it beats the constant classifier, and that comparison must appear next to the
|
|
258
|
+
* accuracy every single time so it can never again be quoted without it.
|
|
259
|
+
*/
|
|
260
|
+
function evaluate(pairs, options = {}) {
|
|
261
|
+
const threshold = Number(options.threshold ?? 0.5);
|
|
262
|
+
const positives = pairs.filter((pair) => toBinaryLabel(pair.label) === 1).length;
|
|
263
|
+
const baseRate = pairs.length > 0 ? positives / pairs.length : 0;
|
|
264
|
+
// The trivial classifier always answers with the majority class.
|
|
265
|
+
const baselineAccuracy = Math.max(baseRate, 1 - baseRate);
|
|
266
|
+
const metrics = classificationMetrics(pairs, threshold);
|
|
267
|
+
|
|
268
|
+
return {
|
|
269
|
+
sampleCount: pairs.length,
|
|
270
|
+
positiveCount: positives,
|
|
271
|
+
baseRate,
|
|
272
|
+
baselineAccuracy,
|
|
273
|
+
accuracy: metrics.accuracy,
|
|
274
|
+
lift: metrics.accuracy - baselineAccuracy,
|
|
275
|
+
precision: metrics.precision,
|
|
276
|
+
recall: metrics.recall,
|
|
277
|
+
specificity: metrics.specificity,
|
|
278
|
+
f1: metrics.f1,
|
|
279
|
+
mcc: metrics.mcc,
|
|
280
|
+
rocAuc: rocAuc(pairs),
|
|
281
|
+
brierScore: brierScore(pairs),
|
|
282
|
+
expectedCalibrationError: calibration(pairs, options.calibrationBins || 10).expectedCalibrationError,
|
|
283
|
+
confusion: metrics.confusion,
|
|
284
|
+
};
|
|
285
|
+
}
|
|
286
|
+
|
|
287
|
+
/** Round every float in a report so persisted artifacts diff cleanly. */
|
|
288
|
+
function roundReport(report, digits = 4) {
|
|
289
|
+
const factor = 10 ** digits;
|
|
290
|
+
const rounded = {};
|
|
291
|
+
for (const [key, value] of Object.entries(report || {})) {
|
|
292
|
+
if (typeof value === 'number' && Number.isFinite(value)) rounded[key] = Math.round(value * factor) / factor;
|
|
293
|
+
else if (value && typeof value === 'object' && !Array.isArray(value)) rounded[key] = roundReport(value, digits);
|
|
294
|
+
else rounded[key] = value;
|
|
295
|
+
}
|
|
296
|
+
return rounded;
|
|
297
|
+
}
|
|
298
|
+
|
|
299
|
+
module.exports = {
|
|
300
|
+
hashString,
|
|
301
|
+
stratifiedSplit,
|
|
302
|
+
classificationMetrics,
|
|
303
|
+
rocAuc,
|
|
304
|
+
brierScore,
|
|
305
|
+
calibration,
|
|
306
|
+
evaluate,
|
|
307
|
+
roundReport,
|
|
308
|
+
};
|
|
@@ -1,18 +1,41 @@
|
|
|
1
1
|
'use strict';
|
|
2
2
|
|
|
3
|
-
const
|
|
4
|
-
const
|
|
3
|
+
const crypto = require('node:crypto');
|
|
4
|
+
const fs = require('node:fs');
|
|
5
|
+
const path = require('node:path');
|
|
6
|
+
const { spawn } = require('node:child_process');
|
|
5
7
|
const { getFeedbackPaths } = require('./feedback-loop');
|
|
6
8
|
const { ensureDir } = require('./fs-utils');
|
|
7
9
|
const { loadOptionalModule } = require('./private-core-boundary');
|
|
8
10
|
|
|
11
|
+
const RUNNER_SCRIPT_PATH = path.join(__dirname, 'async-job-runner.js');
|
|
12
|
+
|
|
13
|
+
function launchPublicManagedJob(jobSpec, options = {}) {
|
|
14
|
+
const publicRunner = require('./async-job-runner');
|
|
15
|
+
const jobId = options.jobId || jobSpec.id || `job_${Date.now()}_${crypto.randomBytes(4).toString('hex')}`;
|
|
16
|
+
const { jobDir } = publicRunner.getJobRuntimePaths(jobId);
|
|
17
|
+
ensureDir(jobDir);
|
|
18
|
+
const jobFilePath = path.join(jobDir, 'job.json');
|
|
19
|
+
const finalSpec = { ...jobSpec, id: jobId };
|
|
20
|
+
fs.writeFileSync(jobFilePath, `${JSON.stringify(finalSpec, null, 2)}\n`, 'utf8');
|
|
21
|
+
publicRunner.queueJob({ ...finalSpec, jobFilePath });
|
|
22
|
+
const child = spawn(process.execPath, [RUNNER_SCRIPT_PATH, `--run-file=${jobFilePath}`], {
|
|
23
|
+
cwd: options.cwd || process.cwd(),
|
|
24
|
+
env: process.env,
|
|
25
|
+
detached: true,
|
|
26
|
+
stdio: 'ignore',
|
|
27
|
+
});
|
|
28
|
+
child.unref();
|
|
29
|
+
return {
|
|
30
|
+
jobId,
|
|
31
|
+
jobFilePath,
|
|
32
|
+
launchMode: 'public-background',
|
|
33
|
+
pid: child.pid || null,
|
|
34
|
+
};
|
|
35
|
+
}
|
|
36
|
+
|
|
9
37
|
const launcher = loadOptionalModule(path.join(__dirname, 'hosted-job-launcher'), () => ({
|
|
10
|
-
launchManagedJob:
|
|
11
|
-
throw new Error('Managed jobs require ThumbGate-Core.');
|
|
12
|
-
},
|
|
13
|
-
resumeHostedJob: () => {
|
|
14
|
-
throw new Error('Resuming hosted jobs requires ThumbGate-Core.');
|
|
15
|
-
},
|
|
38
|
+
launchManagedJob: launchPublicManagedJob,
|
|
16
39
|
}));
|
|
17
40
|
|
|
18
41
|
const runner = loadOptionalModule(path.join(__dirname, 'async-job-runner'), () => ({
|
|
@@ -20,8 +43,8 @@ const runner = loadOptionalModule(path.join(__dirname, 'async-job-runner'), () =
|
|
|
20
43
|
listJobStates: () => [],
|
|
21
44
|
}));
|
|
22
45
|
|
|
23
|
-
const { launchManagedJob
|
|
24
|
-
const { readJobState
|
|
46
|
+
const { launchManagedJob } = launcher;
|
|
47
|
+
const { readJobState } = runner;
|
|
25
48
|
|
|
26
49
|
const DEFAULT_CONCURRENCY = 3;
|
|
27
50
|
const POLL_INTERVAL_MS = 200;
|
|
@@ -45,7 +68,7 @@ function planWorkflow(objective) {
|
|
|
45
68
|
stages: [
|
|
46
69
|
{
|
|
47
70
|
name: 'secret_scan',
|
|
48
|
-
command: 'node scripts/secret-scanner.js --json
|
|
71
|
+
command: 'node scripts/secret-scanner.js --json',
|
|
49
72
|
}
|
|
50
73
|
]
|
|
51
74
|
});
|
|
@@ -55,7 +78,7 @@ function planWorkflow(objective) {
|
|
|
55
78
|
stages: [
|
|
56
79
|
{
|
|
57
80
|
name: 'npm_audit',
|
|
58
|
-
command: 'npm audit --json
|
|
81
|
+
command: 'npm audit --json',
|
|
59
82
|
}
|
|
60
83
|
]
|
|
61
84
|
});
|
|
@@ -65,7 +88,7 @@ function planWorkflow(objective) {
|
|
|
65
88
|
stages: [
|
|
66
89
|
{
|
|
67
90
|
name: 'credential_gate_check',
|
|
68
|
-
command: 'node scripts/single-use-credential-gate.js plan
|
|
91
|
+
command: 'node scripts/single-use-credential-gate.js plan',
|
|
69
92
|
}
|
|
70
93
|
]
|
|
71
94
|
});
|
|
@@ -76,7 +99,7 @@ function planWorkflow(objective) {
|
|
|
76
99
|
stages: [
|
|
77
100
|
{
|
|
78
101
|
name: 'run_bench',
|
|
79
|
-
command: 'npx thumbgate bench --json --min-score=90
|
|
102
|
+
command: 'npx thumbgate bench --json --min-score=90',
|
|
80
103
|
}
|
|
81
104
|
]
|
|
82
105
|
});
|
|
@@ -86,7 +109,7 @@ function planWorkflow(objective) {
|
|
|
86
109
|
stages: [
|
|
87
110
|
{
|
|
88
111
|
name: 'budget_status',
|
|
89
|
-
command: 'node scripts/budget-guard.js --status
|
|
112
|
+
command: 'node scripts/budget-guard.js --status',
|
|
90
113
|
}
|
|
91
114
|
]
|
|
92
115
|
});
|
|
@@ -98,7 +121,7 @@ function planWorkflow(objective) {
|
|
|
98
121
|
stages: [
|
|
99
122
|
{
|
|
100
123
|
name: 'search_fs',
|
|
101
|
-
command: 'node scripts/filesystem-search.js --query="pretool" --limit=5
|
|
124
|
+
command: 'node scripts/filesystem-search.js --query="pretool" --limit=5',
|
|
102
125
|
}
|
|
103
126
|
]
|
|
104
127
|
});
|
|
@@ -108,7 +131,7 @@ function planWorkflow(objective) {
|
|
|
108
131
|
stages: [
|
|
109
132
|
{
|
|
110
133
|
name: 'ops_integrity',
|
|
111
|
-
command: 'node scripts/operational-integrity.js --ci
|
|
134
|
+
command: 'node scripts/operational-integrity.js --ci',
|
|
112
135
|
}
|
|
113
136
|
]
|
|
114
137
|
});
|
|
@@ -119,7 +142,7 @@ function planWorkflow(objective) {
|
|
|
119
142
|
plannedAt: nowIso(),
|
|
120
143
|
subtasks: subtasks.map((task, idx) => ({
|
|
121
144
|
...task,
|
|
122
|
-
id: `subtask_${Date.now()}_${idx}_${
|
|
145
|
+
id: `subtask_${Date.now()}_${idx}_${crypto.randomBytes(3).toString('hex')}`,
|
|
123
146
|
autoImprove: false,
|
|
124
147
|
verificationMode: 'none',
|
|
125
148
|
recordFeedback: false,
|
|
@@ -132,12 +155,14 @@ function planWorkflow(objective) {
|
|
|
132
155
|
* Polls active jobs until all complete, then consolidates the results.
|
|
133
156
|
*/
|
|
134
157
|
async function executeWorkflow(objective, options = {}) {
|
|
135
|
-
const plan = planWorkflow(objective);
|
|
158
|
+
const plan = options.plan || planWorkflow(objective);
|
|
136
159
|
const concurrency = Number(options.concurrency) || DEFAULT_CONCURRENCY;
|
|
137
160
|
const timeoutMs = Number(options.timeoutMs) || 60000; // 60s timeout safety
|
|
161
|
+
const launchJob = options.launchManagedJob || launchManagedJob;
|
|
162
|
+
const getJobState = options.readJobState || readJobState;
|
|
138
163
|
|
|
139
164
|
const { FEEDBACK_DIR } = getFeedbackPaths();
|
|
140
|
-
const workflowId = `wf_${Date.now()}_${
|
|
165
|
+
const workflowId = `wf_${Date.now()}_${crypto.randomBytes(4).toString('hex')}`;
|
|
141
166
|
const workflowDir = path.join(FEEDBACK_DIR, 'workflows', workflowId);
|
|
142
167
|
ensureDir(workflowDir);
|
|
143
168
|
|
|
@@ -145,17 +170,33 @@ async function executeWorkflow(objective, options = {}) {
|
|
|
145
170
|
const queue = [...plan.subtasks];
|
|
146
171
|
const results = [];
|
|
147
172
|
const start = Date.now();
|
|
173
|
+
const statePath = path.join(workflowDir, 'state.json');
|
|
174
|
+
|
|
175
|
+
const persistState = (status) => {
|
|
176
|
+
fs.writeFileSync(statePath, `${JSON.stringify({
|
|
177
|
+
workflowId,
|
|
178
|
+
objective,
|
|
179
|
+
status,
|
|
180
|
+
updatedAt: nowIso(),
|
|
181
|
+
queue: queue.map((task) => ({ id: task.id, name: task.name })),
|
|
182
|
+
activeJobs: [...activeJobs.entries()].map(([taskId, info]) => ({ taskId, ...info })),
|
|
183
|
+
results,
|
|
184
|
+
}, null, 2)}\n`, 'utf8');
|
|
185
|
+
};
|
|
148
186
|
|
|
149
187
|
const runNext = () => {
|
|
150
188
|
while (activeJobs.size < concurrency && queue.length > 0) {
|
|
151
189
|
const task = queue.shift();
|
|
152
|
-
const launched =
|
|
190
|
+
const launched = launchJob(task, { cwd: options.cwd });
|
|
153
191
|
activeJobs.set(task.id, {
|
|
154
192
|
jobId: launched.jobId,
|
|
155
193
|
taskName: task.name,
|
|
156
194
|
launchedAt: Date.now(),
|
|
195
|
+
pid: launched.pid || null,
|
|
196
|
+
launchMode: launched.launchMode || 'managed',
|
|
157
197
|
});
|
|
158
198
|
}
|
|
199
|
+
persistState('running');
|
|
159
200
|
};
|
|
160
201
|
|
|
161
202
|
runNext();
|
|
@@ -166,7 +207,7 @@ async function executeWorkflow(objective, options = {}) {
|
|
|
166
207
|
let allDone = true;
|
|
167
208
|
|
|
168
209
|
for (const [taskId, info] of activeJobs.entries()) {
|
|
169
|
-
const jobState =
|
|
210
|
+
const jobState = getJobState(info.jobId);
|
|
170
211
|
if (!jobState) {
|
|
171
212
|
allDone = false;
|
|
172
213
|
continue;
|
|
@@ -193,11 +234,21 @@ async function executeWorkflow(objective, options = {}) {
|
|
|
193
234
|
const elapsed = Date.now() - start;
|
|
194
235
|
if (allDone && queue.length === 0) {
|
|
195
236
|
clearInterval(interval);
|
|
237
|
+
persistState(results.every((result) => result.status === 'completed')
|
|
238
|
+
? 'completed'
|
|
239
|
+
: 'completed_with_failures');
|
|
196
240
|
resolve();
|
|
197
241
|
} else if (elapsed >= timeoutMs) {
|
|
198
242
|
clearInterval(interval);
|
|
199
243
|
// Timeout remaining active tasks
|
|
200
244
|
for (const [taskId, info] of activeJobs.entries()) {
|
|
245
|
+
if (info.pid) {
|
|
246
|
+
try {
|
|
247
|
+
process.kill(process.platform === 'win32' ? info.pid : -info.pid, 'SIGTERM');
|
|
248
|
+
} catch {
|
|
249
|
+
// The worker may have exited between the last poll and timeout.
|
|
250
|
+
}
|
|
251
|
+
}
|
|
201
252
|
results.push({
|
|
202
253
|
taskId,
|
|
203
254
|
taskName: info.taskName,
|
|
@@ -206,6 +257,17 @@ async function executeWorkflow(objective, options = {}) {
|
|
|
206
257
|
lastError: { message: `Subtask timed out after ${timeoutMs}ms`, code: 'TIMEOUT' },
|
|
207
258
|
});
|
|
208
259
|
}
|
|
260
|
+
for (const task of queue.splice(0)) {
|
|
261
|
+
results.push({
|
|
262
|
+
taskId: task.id,
|
|
263
|
+
taskName: task.name,
|
|
264
|
+
jobId: null,
|
|
265
|
+
status: 'timeout',
|
|
266
|
+
lastError: { message: `Subtask was not launched before ${timeoutMs}ms`, code: 'TIMEOUT' },
|
|
267
|
+
});
|
|
268
|
+
}
|
|
269
|
+
activeJobs.clear();
|
|
270
|
+
persistState('timed_out');
|
|
209
271
|
resolve();
|
|
210
272
|
}
|
|
211
273
|
}, POLL_INTERVAL_MS);
|
|
@@ -234,6 +296,7 @@ async function executeWorkflow(objective, options = {}) {
|
|
|
234
296
|
durationMs,
|
|
235
297
|
reportPath,
|
|
236
298
|
results,
|
|
299
|
+
statePath,
|
|
237
300
|
};
|
|
238
301
|
}
|
|
239
302
|
|
|
@@ -290,4 +353,5 @@ module.exports = {
|
|
|
290
353
|
planWorkflow,
|
|
291
354
|
executeWorkflow,
|
|
292
355
|
compileWorkflowReport,
|
|
356
|
+
launchPublicManagedJob,
|
|
293
357
|
};
|
package/scripts/prompt-eval.js
CHANGED
|
@@ -143,6 +143,8 @@ function handleRejectExpectation(checks, result, expected) {
|
|
|
143
143
|
|
|
144
144
|
const wasRejected = result.accepted === false
|
|
145
145
|
|| result.status === 'rejected'
|
|
146
|
+
|| result.status === 'clarification_required'
|
|
147
|
+
|| result.needsClarification === true
|
|
146
148
|
|| result.actionType === 'no-action';
|
|
147
149
|
checks.push({
|
|
148
150
|
criterion: 'shouldReject',
|
|
@@ -466,8 +468,9 @@ function runSuiteObject(suite, options = {}) {
|
|
|
466
468
|
const skipped = results.filter((r) => r.status === 'skip').length;
|
|
467
469
|
const totalScore = results.length > 0
|
|
468
470
|
? Math.round(results.reduce((s, r) => s + r.score, 0) / results.length)
|
|
469
|
-
:
|
|
471
|
+
: 0;
|
|
470
472
|
const minScore = options.minScore ?? 80;
|
|
473
|
+
const insufficientEvidence = results.length === 0;
|
|
471
474
|
|
|
472
475
|
return {
|
|
473
476
|
suite: suite.name,
|
|
@@ -478,8 +481,9 @@ function runSuiteObject(suite, options = {}) {
|
|
|
478
481
|
skipped,
|
|
479
482
|
score: totalScore,
|
|
480
483
|
minScore,
|
|
481
|
-
pass: totalScore >= minScore,
|
|
482
|
-
noCases:
|
|
484
|
+
pass: !insufficientEvidence && totalScore >= minScore,
|
|
485
|
+
noCases: insufficientEvidence,
|
|
486
|
+
evidenceStatus: insufficientEvidence ? 'insufficient_evidence' : 'measured',
|
|
483
487
|
feedbackDerived: suite.source && suite.source.type === 'feedback-log',
|
|
484
488
|
generatedAt: new Date().toISOString(),
|
|
485
489
|
results,
|
|
@@ -685,6 +689,7 @@ function runSuite(suitePath = DEFAULT_SUITE, options = {}) {
|
|
|
685
689
|
|
|
686
690
|
function compareReports(currentReport, baselineReport) {
|
|
687
691
|
const baselineById = new Map((baselineReport?.results || []).map((result) => [result.id, result]));
|
|
692
|
+
const currentById = new Map((currentReport?.results || []).map((result) => [result.id, result]));
|
|
688
693
|
const regressions = [];
|
|
689
694
|
const improvements = [];
|
|
690
695
|
|
|
@@ -717,12 +722,83 @@ function compareReports(currentReport, baselineReport) {
|
|
|
717
722
|
}
|
|
718
723
|
}
|
|
719
724
|
|
|
725
|
+
for (const baseline of baselineReport?.results || []) {
|
|
726
|
+
if (currentById.has(baseline.id)) continue;
|
|
727
|
+
regressions.push({
|
|
728
|
+
id: baseline.id,
|
|
729
|
+
baselineScore: baseline.score,
|
|
730
|
+
currentScore: null,
|
|
731
|
+
delta: null,
|
|
732
|
+
baselineStatus: baseline.status,
|
|
733
|
+
currentStatus: 'missing',
|
|
734
|
+
});
|
|
735
|
+
}
|
|
736
|
+
|
|
720
737
|
return {
|
|
721
738
|
baselineSuite: baselineReport?.suite || null,
|
|
722
739
|
baselineScore: Number.isFinite(Number(baselineReport?.score)) ? Number(baselineReport.score) : null,
|
|
723
740
|
scoreDelta: Number.isFinite(Number(baselineReport?.score)) ? currentReport.score - Number(baselineReport.score) : null,
|
|
724
741
|
regressions,
|
|
725
742
|
improvements,
|
|
743
|
+
baselineCases: baselineById.size,
|
|
744
|
+
currentCases: currentById.size,
|
|
745
|
+
baselineCoverageRate: baselineById.size
|
|
746
|
+
? Math.round((Array.from(baselineById.keys()).filter((id) => currentById.has(id)).length / baselineById.size) * 10000) / 10000
|
|
747
|
+
: null,
|
|
748
|
+
};
|
|
749
|
+
}
|
|
750
|
+
|
|
751
|
+
function logSafeResult(result, index) {
|
|
752
|
+
const allowedStatuses = new Set(['pass', 'fail', 'error', 'skip']);
|
|
753
|
+
return {
|
|
754
|
+
case: index + 1,
|
|
755
|
+
status: allowedStatuses.has(result.status) ? result.status : 'error',
|
|
756
|
+
score: Number(result.score || 0),
|
|
757
|
+
passCount: Number(result.passCount || 0),
|
|
758
|
+
totalChecks: Number(result.totalChecks || 0),
|
|
759
|
+
};
|
|
760
|
+
}
|
|
761
|
+
|
|
762
|
+
function logSafeReport(report, suite, fromFeedback) {
|
|
763
|
+
const source = suite?.source || {};
|
|
764
|
+
const comparison = report.comparison
|
|
765
|
+
? {
|
|
766
|
+
baselineScore: Number(report.comparison.baselineScore || 0),
|
|
767
|
+
scoreDelta: Number(report.comparison.scoreDelta || 0),
|
|
768
|
+
regressionCount: Number(report.comparison.regressions?.length || 0),
|
|
769
|
+
improvementCount: Number(report.comparison.improvements?.length || 0),
|
|
770
|
+
baselineCases: Number(report.comparison.baselineCases || 0),
|
|
771
|
+
currentCases: Number(report.comparison.currentCases || 0),
|
|
772
|
+
baselineCoverageRate: report.comparison.baselineCoverageRate,
|
|
773
|
+
}
|
|
774
|
+
: undefined;
|
|
775
|
+
return {
|
|
776
|
+
suiteType: fromFeedback ? 'feedback-derived' : 'configured',
|
|
777
|
+
total: Number(report.total || 0),
|
|
778
|
+
passed: Number(report.passed || 0),
|
|
779
|
+
failed: Number(report.failed || 0),
|
|
780
|
+
errors: Number(report.errors || 0),
|
|
781
|
+
skipped: Number(report.skipped || 0),
|
|
782
|
+
score: Number(report.score || 0),
|
|
783
|
+
minScore: Number(report.minScore || 0),
|
|
784
|
+
pass: report.pass === true,
|
|
785
|
+
noCases: report.noCases === true,
|
|
786
|
+
evidenceStatus: report.evidenceStatus === 'measured' ? 'measured' : 'insufficient_evidence',
|
|
787
|
+
feedbackDerived: report.feedbackDerived === true,
|
|
788
|
+
syntheticCount: Number(report.syntheticCount || 0),
|
|
789
|
+
comparison,
|
|
790
|
+
results: (report.results || []).map(logSafeResult),
|
|
791
|
+
suiteDefinition: fromFeedback
|
|
792
|
+
? {
|
|
793
|
+
version: Number(suite?.version || 0),
|
|
794
|
+
source: {
|
|
795
|
+
type: 'feedback-log',
|
|
796
|
+
totalEntries: Number(source.totalEntries || 0),
|
|
797
|
+
selectedCases: Number(source.selectedCases || 0),
|
|
798
|
+
},
|
|
799
|
+
evaluationCount: Number(suite?.evaluations?.length || 0),
|
|
800
|
+
}
|
|
801
|
+
: undefined,
|
|
726
802
|
};
|
|
727
803
|
}
|
|
728
804
|
|
|
@@ -847,7 +923,7 @@ if (isCliInvocation()) {
|
|
|
847
923
|
}
|
|
848
924
|
|
|
849
925
|
if (json) {
|
|
850
|
-
console.log(JSON.stringify(
|
|
926
|
+
console.log(JSON.stringify(logSafeReport(report, suite, fromFeedback), null, 2));
|
|
851
927
|
} else {
|
|
852
928
|
console.log(`\n${report.suite}`);
|
|
853
929
|
console.log('='.repeat(50));
|
|
@@ -883,6 +959,7 @@ module.exports = {
|
|
|
883
959
|
gradeOutput,
|
|
884
960
|
loadSuite,
|
|
885
961
|
loadReport,
|
|
962
|
+
logSafeReport,
|
|
886
963
|
compareReports,
|
|
887
964
|
readJsonl,
|
|
888
965
|
runEvaluation,
|