thumbgate 1.29.1 → 1.30.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude/commands/dashboard.md +11 -1
- package/.claude/commands/thumbgate-dashboard.md +23 -8
- package/.claude-plugin/plugin.json +1 -1
- package/.well-known/mcp/server-card.json +1 -1
- package/README.md +61 -1
- package/adapters/claude/.mcp.json +2 -2
- package/adapters/forge/forge.yaml +3 -3
- package/adapters/mcp/server-stdio.js +164 -7
- package/adapters/opencode/opencode.json +1 -1
- package/bin/cli.js +7 -5
- package/commands/dashboard.md +11 -1
- package/commands/thumbgate-dashboard.md +23 -8
- package/config/agent-outcome-monitor-thresholds.json +63 -0
- package/config/evals/agent-outcomes-baseline.json +17 -0
- package/config/evals/agent-outcomes-golden.json +412 -0
- package/config/evals/prompt-eval-baseline.json +23 -0
- package/config/mcp-allowlists.json +26 -2
- package/config/post-deploy-marketing-pages.json +26 -1
- package/config/schemas/task-outcome-receipt.schema.json +296 -0
- package/openapi/openapi.yaml +235 -0
- package/package.json +55 -11
- package/public/architecture.html +130 -0
- package/public/assets/diagrams/agent-integration.png +0 -0
- package/public/assets/diagrams/before-after.svg +21 -0
- package/public/assets/diagrams/decision.svg +36 -0
- package/public/assets/diagrams/feedback-pipeline.png +0 -0
- package/public/assets/diagrams/loop.svg +34 -0
- package/public/assets/diagrams/plugin-topology.png +0 -0
- package/public/assets/diagrams/pre-action-gate-loop.svg +59 -0
- package/public/assets/diagrams/stack.svg +18 -0
- package/public/assets/diagrams/thumbgate-architecture.png +0 -0
- package/public/case-studies.html +151 -0
- package/public/eval-scorecard.html +195 -0
- package/public/eval-scorecard.json +18 -0
- package/public/evaluations.html +168 -0
- package/public/index.html +6 -3
- package/public/numbers.html +2 -2
- package/public/whitepaper.html +189 -0
- package/scripts/activation-quickstart.js +1 -0
- package/scripts/agent-outcome-eval.js +130 -0
- package/scripts/agent-outcome-monitor.js +331 -0
- package/scripts/agent-reasoning-traces.js +8 -9
- package/scripts/async-job-runner.js +107 -13
- package/scripts/billing.js +3 -1
- package/scripts/claude-feedback-sync.js +3 -2
- package/scripts/cli-feedback.js +13 -7
- package/scripts/cross-encoder-reranker.js +3 -0
- package/scripts/durability/step.js +121 -12
- package/scripts/feedback-aggregate.js +5 -2
- package/scripts/feedback-loop.js +244 -182
- package/scripts/gates-engine.js +512 -22
- package/scripts/generate-case-study-outreach.js +253 -0
- package/scripts/generate-eval-scorecard.js +276 -0
- package/scripts/growth-campaigns.js +183 -0
- package/scripts/human-escalation.js +265 -0
- package/scripts/hybrid-feedback-context.js +93 -50
- package/scripts/jsonl-watcher.js +1 -0
- package/scripts/judge-reward-function.js +30 -18
- package/scripts/lesson-inference.js +23 -4
- package/scripts/lesson-retrieval.js +71 -4
- package/scripts/lesson-search.js +26 -3
- package/scripts/mcp-config.js +26 -5
- package/scripts/mcp-oauth.js +37 -2
- package/scripts/model-eval.js +308 -0
- package/scripts/parallel-workflow-orchestrator.js +86 -22
- package/scripts/prompt-eval.js +81 -4
- package/scripts/published-cli.js +11 -1
- package/scripts/refresh-proof-pack.js +261 -0
- package/scripts/risk-scorer.js +144 -15
- package/scripts/schedule-manager.js +249 -0
- package/scripts/statusline-local-stats.js +1 -1
- package/scripts/task-outcomes.js +425 -0
- package/scripts/thumbgate-bench.js +13 -0
- package/scripts/tool-contract-validator.js +287 -59
- package/scripts/tool-kpi-tracker.js +124 -0
- package/scripts/tool-registry.js +192 -1
- package/src/api/server.js +355 -89
package/scripts/published-cli.js
CHANGED
|
@@ -13,6 +13,14 @@ function runtimePrefixDir(prefixDir) {
|
|
|
13
13
|
return prefixDir || path.join(os.homedir(), '.thumbgate', 'runtime');
|
|
14
14
|
}
|
|
15
15
|
|
|
16
|
+
// For GENERATED SHELL COMMANDS only. Shell command strings land in shared, committed config
|
|
17
|
+
// (.mcp.json entries, hook command lines), so expanding os.homedir() at generation time bakes
|
|
18
|
+
// the generating machine's home into files other machines execute — /Users/alice/.thumbgate
|
|
19
|
+
// fails with a permission error on bob's machine. shellQuote uses double quotes, so a literal
|
|
20
|
+
// $HOME expands at RUNTIME on whichever machine runs the command. Non-shell consumers
|
|
21
|
+
// (execFileSync paths) must keep using runtimePrefixDir, which returns a real filesystem path.
|
|
22
|
+
const SHELL_RUNTIME_PREFIX = '$HOME/.thumbgate/runtime';
|
|
23
|
+
|
|
16
24
|
function installedRuntimeBin(prefixDir) {
|
|
17
25
|
return path.join(runtimePrefixDir(prefixDir), 'node_modules', '.bin', 'thumbgate');
|
|
18
26
|
}
|
|
@@ -32,7 +40,9 @@ function publishedCliArgs(pkgVersion, commandArgs = [], options = {}) {
|
|
|
32
40
|
}
|
|
33
41
|
|
|
34
42
|
function publishedCliShellCommand(pkgVersion, commandArgs = [], options = {}) {
|
|
35
|
-
|
|
43
|
+
// Default to the runtime-expanded $HOME form; an explicit options.prefixDir (tests,
|
|
44
|
+
// throwaway prefixes) is honoured verbatim.
|
|
45
|
+
const prefixDir = options.prefixDir || SHELL_RUNTIME_PREFIX;
|
|
36
46
|
const runtimeBin = installedRuntimeBin(prefixDir);
|
|
37
47
|
const escapedArgs = commandArgs.map(shellQuote).join(' ');
|
|
38
48
|
const fastPath = `[ -x ${shellQuote(runtimeBin)} ] && exec ${shellQuote(runtimeBin)}${escapedArgs ? ` ${escapedArgs}` : ''}`;
|
|
@@ -0,0 +1,261 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
'use strict';
|
|
3
|
+
|
|
4
|
+
/**
|
|
5
|
+
* refresh-proof-pack.js — regenerate (or check) the public evaluation scorecard.
|
|
6
|
+
*
|
|
7
|
+
* Cadence (repo policy): GitHub Actions schedule is limited to CodeQL. Noncritical
|
|
8
|
+
* loops run via workflow_dispatch or local LaunchAgent:
|
|
9
|
+
* npm run proof-pack:refresh # write public/eval-scorecard.html
|
|
10
|
+
* npm run proof-pack:refresh:check # CI gate: metrics must still match
|
|
11
|
+
* npm run proof-pack:schedule # install daily local LaunchAgent
|
|
12
|
+
*
|
|
13
|
+
* Isolation: generation goes through generate-eval-scorecard → thumbgate-bench
|
|
14
|
+
* isolated runtime (strict enforcement pinned).
|
|
15
|
+
*/
|
|
16
|
+
|
|
17
|
+
const fs = require('node:fs');
|
|
18
|
+
const path = require('node:path');
|
|
19
|
+
const { spawnSync } = require('node:child_process');
|
|
20
|
+
|
|
21
|
+
const PROJECT_ROOT = path.resolve(__dirname, '..');
|
|
22
|
+
const SCORECARD_HTML = path.join(PROJECT_ROOT, 'public', 'eval-scorecard.html');
|
|
23
|
+
const SCORECARD_JSON = path.join(PROJECT_ROOT, 'public', 'eval-scorecard.json');
|
|
24
|
+
|
|
25
|
+
function parseArgs(argv = process.argv.slice(2)) {
|
|
26
|
+
const args = {
|
|
27
|
+
write: false,
|
|
28
|
+
check: false,
|
|
29
|
+
json: false,
|
|
30
|
+
help: false,
|
|
31
|
+
minScore: 90,
|
|
32
|
+
};
|
|
33
|
+
for (const arg of argv) {
|
|
34
|
+
if (arg === '--write') args.write = true;
|
|
35
|
+
else if (arg === '--check') args.check = true;
|
|
36
|
+
else if (arg === '--json') args.json = true;
|
|
37
|
+
else if (arg === '--help' || arg === '-h') args.help = true;
|
|
38
|
+
else if (arg.startsWith('--min-score=')) {
|
|
39
|
+
args.minScore = Number(arg.slice('--min-score='.length));
|
|
40
|
+
}
|
|
41
|
+
}
|
|
42
|
+
if (!args.write && !args.check) {
|
|
43
|
+
// Default to write for operator cadence runs.
|
|
44
|
+
args.write = true;
|
|
45
|
+
}
|
|
46
|
+
return args;
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
function printHelp() {
|
|
50
|
+
console.log(`Usage: node scripts/refresh-proof-pack.js [--write] [--check] [--json] [--min-score=90]
|
|
51
|
+
|
|
52
|
+
--write Regenerate public/eval-scorecard.html (+ .json sidecar)
|
|
53
|
+
--check Fail if committed scorecard metrics diverge from a fresh bench run
|
|
54
|
+
--json Print machine-readable summary to stdout
|
|
55
|
+
--min-score=N Minimum composite score (default 90)
|
|
56
|
+
`);
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
function extractMetricsFromHtml(html) {
|
|
60
|
+
const metrics = {};
|
|
61
|
+
// Prefer JSON-LD Dataset variableMeasured
|
|
62
|
+
const ldMatch = html.match(/<script type="application\/ld\+json">([\s\S]*?)<\/script>/);
|
|
63
|
+
if (ldMatch) {
|
|
64
|
+
try {
|
|
65
|
+
const ld = JSON.parse(ldMatch[1]);
|
|
66
|
+
const vars = Array.isArray(ld.variableMeasured) ? ld.variableMeasured : [];
|
|
67
|
+
for (const item of vars) {
|
|
68
|
+
if (item && item.name != null) metrics[item.name] = item.value;
|
|
69
|
+
}
|
|
70
|
+
} catch {
|
|
71
|
+
// fall through to regex
|
|
72
|
+
}
|
|
73
|
+
}
|
|
74
|
+
const scoreMatch = html.match(/composite score <strong>([^<]+)<\/strong>/i)
|
|
75
|
+
|| html.match(/composite score[^0-9]*([0-9]+)/i);
|
|
76
|
+
if (scoreMatch && metrics.score == null) {
|
|
77
|
+
metrics.score = Number(scoreMatch[1]);
|
|
78
|
+
}
|
|
79
|
+
const passMatch = html.match(/Overall:\s*<span class="(good|bad)">(PASSED|FAILED)<\/span>/i);
|
|
80
|
+
if (passMatch) {
|
|
81
|
+
metrics.passedLabel = passMatch[2].toUpperCase();
|
|
82
|
+
}
|
|
83
|
+
return metrics;
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
function normalizeMetrics(metrics = {}) {
|
|
87
|
+
const keys = [
|
|
88
|
+
'score',
|
|
89
|
+
'taskSuccessRate',
|
|
90
|
+
'unsafeActionRate',
|
|
91
|
+
'blockedUnsafeRate',
|
|
92
|
+
'capabilityRate',
|
|
93
|
+
'falseBlockRate',
|
|
94
|
+
'replayStability',
|
|
95
|
+
];
|
|
96
|
+
const out = {};
|
|
97
|
+
for (const key of keys) {
|
|
98
|
+
if (metrics[key] == null || metrics[key] === '') continue;
|
|
99
|
+
const n = Number(metrics[key]);
|
|
100
|
+
out[key] = Number.isFinite(n) ? Number(n.toFixed(4)) : metrics[key];
|
|
101
|
+
}
|
|
102
|
+
return out;
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
function metricsEqual(a, b, options = {}) {
|
|
106
|
+
const left = normalizeMetrics(a);
|
|
107
|
+
const right = normalizeMetrics(b);
|
|
108
|
+
// When comparing a committed HTML extract to a fresh bench report, only
|
|
109
|
+
// assert keys present on the committed side (JSON-LD may omit some rates).
|
|
110
|
+
const keys = options.keys
|
|
111
|
+
|| (options.committedOnly
|
|
112
|
+
? Object.keys(left)
|
|
113
|
+
: [...new Set([...Object.keys(left), ...Object.keys(right)])]);
|
|
114
|
+
const diffs = [];
|
|
115
|
+
for (const key of keys) {
|
|
116
|
+
if (left[key] !== right[key]) {
|
|
117
|
+
diffs.push({ key, committed: left[key], fresh: right[key] });
|
|
118
|
+
}
|
|
119
|
+
}
|
|
120
|
+
return { equal: diffs.length === 0, diffs, left, right };
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
function runFreshBench() {
|
|
124
|
+
const { generate, runBench } = require('./generate-eval-scorecard');
|
|
125
|
+
// Prefer direct bench for metrics; generate for write path.
|
|
126
|
+
let report;
|
|
127
|
+
try {
|
|
128
|
+
report = runBench();
|
|
129
|
+
} catch {
|
|
130
|
+
// generate also runs the bench
|
|
131
|
+
report = null;
|
|
132
|
+
}
|
|
133
|
+
return { generate, report };
|
|
134
|
+
}
|
|
135
|
+
|
|
136
|
+
function buildSidecar(report, version, nowIso) {
|
|
137
|
+
const metrics = report.metrics || report;
|
|
138
|
+
return {
|
|
139
|
+
generatedAt: nowIso,
|
|
140
|
+
version,
|
|
141
|
+
sourcePath: report.sourcePath || 'bench/thumbgate-bench.json',
|
|
142
|
+
passed: report.passed !== false,
|
|
143
|
+
isolatedRuntime: report.isolatedRuntime !== false,
|
|
144
|
+
metrics: normalizeMetrics(metrics),
|
|
145
|
+
scenarioCount: Array.isArray(report.scenarios) ? report.scenarios.length : null,
|
|
146
|
+
proofUrl: 'https://thumbgate.ai/eval-scorecard',
|
|
147
|
+
};
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
function refreshWrite(options = {}) {
|
|
151
|
+
const { generate } = require('./generate-eval-scorecard');
|
|
152
|
+
const now = options.now instanceof Date ? options.now : new Date();
|
|
153
|
+
const result = generate({
|
|
154
|
+
now,
|
|
155
|
+
outputPath: options.outputPath || SCORECARD_HTML,
|
|
156
|
+
});
|
|
157
|
+
const version = options.version || require(path.join(PROJECT_ROOT, 'package.json')).version;
|
|
158
|
+
const sidecar = buildSidecar(result.report, version, now.toISOString());
|
|
159
|
+
const sidecarPath = options.sidecarPath || SCORECARD_JSON;
|
|
160
|
+
fs.writeFileSync(sidecarPath, `${JSON.stringify(sidecar, null, 2)}\n`, 'utf8');
|
|
161
|
+
return {
|
|
162
|
+
mode: 'write',
|
|
163
|
+
htmlPath: result.outPath,
|
|
164
|
+
sidecarPath,
|
|
165
|
+
passed: result.report.passed !== false,
|
|
166
|
+
metrics: sidecar.metrics,
|
|
167
|
+
score: sidecar.metrics.score,
|
|
168
|
+
};
|
|
169
|
+
}
|
|
170
|
+
|
|
171
|
+
function refreshCheck(options = {}) {
|
|
172
|
+
const htmlPath = options.htmlPath || SCORECARD_HTML;
|
|
173
|
+
if (!fs.existsSync(htmlPath)) {
|
|
174
|
+
throw new Error(`Missing committed scorecard: ${htmlPath}`);
|
|
175
|
+
}
|
|
176
|
+
const committedHtml = fs.readFileSync(htmlPath, 'utf8');
|
|
177
|
+
const committed = normalizeMetrics(extractMetricsFromHtml(committedHtml));
|
|
178
|
+
|
|
179
|
+
const { runBench } = require('./generate-eval-scorecard');
|
|
180
|
+
const report = options.report || runBench();
|
|
181
|
+
const fresh = normalizeMetrics(report.metrics || report);
|
|
182
|
+
const comparison = metricsEqual(committed, fresh, { committedOnly: true });
|
|
183
|
+
const score = Number(fresh.score ?? committed.score);
|
|
184
|
+
const minScore = options.minScore ?? 90;
|
|
185
|
+
const scoreOk = Number.isFinite(score) && score >= minScore;
|
|
186
|
+
const passed = report.passed !== false && scoreOk && comparison.equal;
|
|
187
|
+
|
|
188
|
+
return {
|
|
189
|
+
mode: 'check',
|
|
190
|
+
passed,
|
|
191
|
+
scoreOk,
|
|
192
|
+
metricsMatch: comparison.equal,
|
|
193
|
+
diffs: comparison.diffs,
|
|
194
|
+
committed,
|
|
195
|
+
fresh,
|
|
196
|
+
minScore,
|
|
197
|
+
reportPassed: report.passed !== false,
|
|
198
|
+
};
|
|
199
|
+
}
|
|
200
|
+
|
|
201
|
+
function main(argv = process.argv.slice(2)) {
|
|
202
|
+
const args = parseArgs(argv);
|
|
203
|
+
if (args.help) {
|
|
204
|
+
printHelp();
|
|
205
|
+
return 0;
|
|
206
|
+
}
|
|
207
|
+
|
|
208
|
+
let summary;
|
|
209
|
+
if (args.check) {
|
|
210
|
+
summary = refreshCheck({ minScore: args.minScore });
|
|
211
|
+
} else {
|
|
212
|
+
summary = refreshWrite();
|
|
213
|
+
if (Number(summary.score) < args.minScore || summary.passed === false) {
|
|
214
|
+
summary.checkFailed = true;
|
|
215
|
+
}
|
|
216
|
+
}
|
|
217
|
+
|
|
218
|
+
if (args.json) {
|
|
219
|
+
console.log(JSON.stringify(summary, null, 2));
|
|
220
|
+
} else if (summary.mode === 'write') {
|
|
221
|
+
console.log(
|
|
222
|
+
`Proof pack scorecard written: ${summary.htmlPath} (score=${summary.score}, passed=${summary.passed})`,
|
|
223
|
+
);
|
|
224
|
+
console.log(`Sidecar: ${summary.sidecarPath}`);
|
|
225
|
+
} else {
|
|
226
|
+
console.log(
|
|
227
|
+
`Proof pack check: metricsMatch=${summary.metricsMatch} scoreOk=${summary.scoreOk} reportPassed=${summary.reportPassed}`,
|
|
228
|
+
);
|
|
229
|
+
if (summary.diffs.length) {
|
|
230
|
+
for (const diff of summary.diffs) {
|
|
231
|
+
console.log(` drift ${diff.key}: committed=${diff.committed} fresh=${diff.fresh}`);
|
|
232
|
+
}
|
|
233
|
+
}
|
|
234
|
+
}
|
|
235
|
+
|
|
236
|
+
if (summary.mode === 'check' && !summary.passed) return 1;
|
|
237
|
+
if (summary.mode === 'write' && summary.checkFailed) return 1;
|
|
238
|
+
return 0;
|
|
239
|
+
}
|
|
240
|
+
|
|
241
|
+
if (path.resolve(process.argv[1] || '') === path.resolve(__filename)) {
|
|
242
|
+
try {
|
|
243
|
+
process.exitCode = main();
|
|
244
|
+
} catch (err) {
|
|
245
|
+
console.error(err.message || err);
|
|
246
|
+
process.exitCode = 1;
|
|
247
|
+
}
|
|
248
|
+
}
|
|
249
|
+
|
|
250
|
+
module.exports = {
|
|
251
|
+
parseArgs,
|
|
252
|
+
extractMetricsFromHtml,
|
|
253
|
+
normalizeMetrics,
|
|
254
|
+
metricsEqual,
|
|
255
|
+
refreshWrite,
|
|
256
|
+
refreshCheck,
|
|
257
|
+
buildSidecar,
|
|
258
|
+
main,
|
|
259
|
+
SCORECARD_HTML,
|
|
260
|
+
SCORECARD_JSON,
|
|
261
|
+
};
|
package/scripts/risk-scorer.js
CHANGED
|
@@ -5,6 +5,7 @@ const fs = require('fs');
|
|
|
5
5
|
const path = require('path');
|
|
6
6
|
const { resolveFeedbackDir: resolveSharedFeedbackDir } = require('./feedback-paths');
|
|
7
7
|
const { requireLearnedModelsEntitlement } = require('./entitlement');
|
|
8
|
+
const { stratifiedSplit, evaluate, roundReport } = require('./model-eval');
|
|
8
9
|
|
|
9
10
|
const PROJECT_ROOT = path.join(__dirname, '..');
|
|
10
11
|
const DEFAULT_FEEDBACK_DIR = resolveSharedFeedbackDir();
|
|
@@ -187,10 +188,16 @@ function stumpPredict(value, threshold, polarity) {
|
|
|
187
188
|
return decision * polarity;
|
|
188
189
|
}
|
|
189
190
|
|
|
190
|
-
function findBestWeakLearner(examples, weights, featureNames) {
|
|
191
|
+
function findBestWeakLearner(examples, weights, featureNames, options = {}) {
|
|
192
|
+
const usage = options.usage || null;
|
|
193
|
+
const maxPerFeature = Number(options.maxPerFeature || Infinity);
|
|
191
194
|
let best = null;
|
|
192
195
|
|
|
193
196
|
featureNames.forEach((feature) => {
|
|
197
|
+
// Diversity bound: once a feature has been split on maxPerFeature times, later rounds must
|
|
198
|
+
// find signal elsewhere or stop. Off by default (Infinity) so existing behaviour is bit
|
|
199
|
+
// identical unless a caller opts in.
|
|
200
|
+
if (usage && Number.isFinite(maxPerFeature) && (usage.get(feature) || 0) >= maxPerFeature) return;
|
|
194
201
|
const values = examples.map((example) => example.features[feature]);
|
|
195
202
|
const thresholds = candidateThresholds(values);
|
|
196
203
|
thresholds.forEach((threshold) => {
|
|
@@ -268,18 +275,14 @@ function buildPatternSummary(rows) {
|
|
|
268
275
|
};
|
|
269
276
|
}
|
|
270
277
|
|
|
271
|
-
|
|
272
|
-
|
|
273
|
-
|
|
274
|
-
|
|
275
|
-
|
|
276
|
-
|
|
277
|
-
|
|
278
|
-
|
|
279
|
-
label: deriveTargetRisk(row) === 1 ? 1 : -1,
|
|
280
|
-
features: extractFeatureMap(row, registry),
|
|
281
|
-
}));
|
|
282
|
-
|
|
278
|
+
/**
|
|
279
|
+
* Fit the ensemble on a given set of examples.
|
|
280
|
+
*
|
|
281
|
+
* Extracted from trainRiskModel so that the held-out probe below is fitted by IDENTICAL code.
|
|
282
|
+
* If the probe were trained by a separate path, its score would describe a model we do not
|
|
283
|
+
* ship, and the resulting "generalization" number would be fiction.
|
|
284
|
+
*/
|
|
285
|
+
function fitBoostedModel(examples, registry, options = {}) {
|
|
283
286
|
const model = {
|
|
284
287
|
version: 1,
|
|
285
288
|
algorithm: 'adaboost-stumps',
|
|
@@ -292,7 +295,7 @@ function trainRiskModel(rows, options = {}) {
|
|
|
292
295
|
featureRegistry: registry,
|
|
293
296
|
featureNames: examples[0] ? Object.keys(examples[0].features) : [],
|
|
294
297
|
learners: [],
|
|
295
|
-
patterns:
|
|
298
|
+
patterns: options.patterns || { tags: [], domains: [], skills: [] },
|
|
296
299
|
metrics: {
|
|
297
300
|
trainingAccuracy: 0,
|
|
298
301
|
rounds: 0,
|
|
@@ -307,9 +310,16 @@ function trainRiskModel(rows, options = {}) {
|
|
|
307
310
|
|
|
308
311
|
let weights = normalizeWeights(Array(examples.length).fill(1));
|
|
309
312
|
const rounds = Math.max(1, Math.min(12, Number(options.rounds || 8)));
|
|
313
|
+
// Boosting is free to pick the same feature every round. On the real corpus it did exactly
|
|
314
|
+
// that — six of eight stumps split on `recentTrend`, so an "ensemble" was in practice a
|
|
315
|
+
// one-feature model of how the session had been going lately. maxPerFeature bounds that.
|
|
316
|
+
// Default is Infinity (historical behaviour); enabling it is an evidence-based decision made
|
|
317
|
+
// by comparing held-out lift, not an assumption. See docs/ML-EVALUATION.md.
|
|
318
|
+
const maxPerFeature = Number(options.maxPerFeature || Infinity);
|
|
319
|
+
const usage = new Map();
|
|
310
320
|
|
|
311
321
|
for (let round = 0; round < rounds; round += 1) {
|
|
312
|
-
const learner = findBestWeakLearner(examples, weights, model.featureNames);
|
|
322
|
+
const learner = findBestWeakLearner(examples, weights, model.featureNames, { usage, maxPerFeature });
|
|
313
323
|
if (!learner) break;
|
|
314
324
|
|
|
315
325
|
const clippedError = Math.min(Math.max(learner.error, 1e-6), 1 - 1e-6);
|
|
@@ -322,6 +332,7 @@ function trainRiskModel(rows, options = {}) {
|
|
|
322
332
|
polarity: learner.polarity,
|
|
323
333
|
alpha: Math.round(alpha * 1000) / 1000,
|
|
324
334
|
});
|
|
335
|
+
usage.set(learner.feature, (usage.get(learner.feature) || 0) + 1);
|
|
325
336
|
|
|
326
337
|
weights = normalizeWeights(weights.map((weight, index) => (
|
|
327
338
|
weight * Math.exp(-alpha * examples[index].label * learner.predictions[index])
|
|
@@ -334,6 +345,119 @@ function trainRiskModel(rows, options = {}) {
|
|
|
334
345
|
return model;
|
|
335
346
|
}
|
|
336
347
|
|
|
348
|
+
/** Score a fitted model over examples as (probability, label) pairs for model-eval. */
|
|
349
|
+
function scorePairs(model, examples) {
|
|
350
|
+
return examples.map((example) => ({
|
|
351
|
+
probability: predictRisk(model, example.row).probability,
|
|
352
|
+
label: example.label === 1 ? 1 : 0,
|
|
353
|
+
}));
|
|
354
|
+
}
|
|
355
|
+
|
|
356
|
+
/**
|
|
357
|
+
* Held-out estimate of what this training procedure generalizes to.
|
|
358
|
+
*
|
|
359
|
+
* Split on CONTENT, not position or an RNG. Content hashing gives two properties we need:
|
|
360
|
+
* the split is reproducible on every machine, and near-duplicate rows (which this corpus has
|
|
361
|
+
* plenty of, since similar actions recur) land in the SAME fold instead of leaking an answer
|
|
362
|
+
* from train into test and inflating the score.
|
|
363
|
+
*/
|
|
364
|
+
function holdoutEvaluation(examples, registry, options = {}) {
|
|
365
|
+
const { train, test } = stratifiedSplit(examples, {
|
|
366
|
+
testFraction: Number(options.testFraction || 0.25),
|
|
367
|
+
// splitSalt exists so the SAME corpus can be re-split many ways for repeated-resampling
|
|
368
|
+
// validation. A single split of a few hundred rows cannot distinguish a real improvement
|
|
369
|
+
// from sampling noise, and picking a configuration on one split is how you overfit the
|
|
370
|
+
// validation set itself.
|
|
371
|
+
// Key on the extracted feature vector ALONE. The model observes nothing else, so two rows
|
|
372
|
+
// with identical features are the same input to it and must share a fold. An earlier
|
|
373
|
+
// version also mixed in the raw `context` string, which split rows whose different prose
|
|
374
|
+
// maps to identical features — recreating the very leakage this splitter exists to stop.
|
|
375
|
+
keyFn: options.groupKeyFn
|
|
376
|
+
? (example) => JSON.stringify([options.splitSalt || '', options.groupKeyFn(example)])
|
|
377
|
+
: (example) => JSON.stringify([options.splitSalt || '', example.features]),
|
|
378
|
+
});
|
|
379
|
+
|
|
380
|
+
// Saying "not measurable" is the honest output for a corpus too small or too one-sided to
|
|
381
|
+
// hold anything out. Reporting a number here would be worse than reporting nothing.
|
|
382
|
+
if (test.length === 0 || train.length < 6) {
|
|
383
|
+
return { available: false, reason: 'corpus-too-small-or-single-class' };
|
|
384
|
+
}
|
|
385
|
+
|
|
386
|
+
// REBUILD THE VOCABULARY FROM THE TRAINING FOLD ONLY.
|
|
387
|
+
//
|
|
388
|
+
// buildFeatureRegistry picks top tags and skills by frequency. Deriving it from the whole
|
|
389
|
+
// corpus lets held-out rows decide which features exist — a transductive fit. The probe would
|
|
390
|
+
// see vocabulary chosen with knowledge of the test fold, and the "held-out" number would then
|
|
391
|
+
// describe a procedure we never run in production. Registry and features come from train only.
|
|
392
|
+
const foldRegistry = buildFeatureRegistry(train.map((example) => example.row), options);
|
|
393
|
+
const trainRefit = train.map((example) => ({
|
|
394
|
+
row: example.row,
|
|
395
|
+
label: example.label,
|
|
396
|
+
features: extractFeatureMap(example.row, foldRegistry),
|
|
397
|
+
}));
|
|
398
|
+
|
|
399
|
+
const probe = fitBoostedModel(trainRefit, foldRegistry, { ...options, patterns: undefined });
|
|
400
|
+
// Test rows are scored via predictRisk, which extracts features using the probe's OWN
|
|
401
|
+
// registry — so the test fold is judged under exactly the vocabulary the probe learned.
|
|
402
|
+
const report = evaluate(scorePairs(probe, test));
|
|
403
|
+
return {
|
|
404
|
+
available: true,
|
|
405
|
+
trainCount: train.length,
|
|
406
|
+
testCount: test.length,
|
|
407
|
+
...roundReport(report),
|
|
408
|
+
};
|
|
409
|
+
}
|
|
410
|
+
|
|
411
|
+
function trainRiskModel(rows, options = {}) {
|
|
412
|
+
requireLearnedModelsEntitlement({
|
|
413
|
+
...(options.entitlement || {}),
|
|
414
|
+
label: 'risk-scorer AdaBoost training',
|
|
415
|
+
});
|
|
416
|
+
const registry = buildFeatureRegistry(rows, options);
|
|
417
|
+
const examples = rows.map((row) => ({
|
|
418
|
+
row,
|
|
419
|
+
label: deriveTargetRisk(row) === 1 ? 1 : -1,
|
|
420
|
+
features: extractFeatureMap(row, registry),
|
|
421
|
+
}));
|
|
422
|
+
|
|
423
|
+
const model = fitBoostedModel(examples, registry, {
|
|
424
|
+
...options,
|
|
425
|
+
patterns: buildPatternSummary(rows),
|
|
426
|
+
});
|
|
427
|
+
|
|
428
|
+
// The shipped model is fitted on everything — that is the right thing to deploy. The probe
|
|
429
|
+
// above estimates what that procedure generalizes to. Both numbers are recorded, and
|
|
430
|
+
// `inSample` is labelled as such so it can never again be quoted as if it were quality.
|
|
431
|
+
model.metrics.inSample = roundReport(evaluate(scorePairs(model, examples)));
|
|
432
|
+
if (options.skipHoldout !== true) {
|
|
433
|
+
// TWO held-out estimates, because they answer different questions and only reporting the
|
|
434
|
+
// friendlier one would repeat the original sin of this file.
|
|
435
|
+
//
|
|
436
|
+
// holdout — IID split on the full feature vector. "Does this work on new
|
|
437
|
+
// rows of the kinds we have seen?" Measured 2026-07-28: +0.091
|
|
438
|
+
// lift, AUC 0.884, 12/12 resamples beat baseline.
|
|
439
|
+
//
|
|
440
|
+
// holdoutNovelContext — split by coarse content group, so whole action categories are
|
|
441
|
+
// absent from training. "Does this work on kinds of actions we
|
|
442
|
+
// have never seen?" Measured: NEGATIVE lift.
|
|
443
|
+
//
|
|
444
|
+
// For a firewall the second question is the one that matters most — novel attacks are by
|
|
445
|
+
// definition unfamiliar — so the pessimistic number is recorded next to the optimistic one
|
|
446
|
+
// permanently, and docs/ML-EVALUATION.md explains the gap.
|
|
447
|
+
model.metrics.holdout = holdoutEvaluation(examples, registry, options);
|
|
448
|
+
model.metrics.holdoutNovelContext = holdoutEvaluation(examples, registry, {
|
|
449
|
+
...options,
|
|
450
|
+
groupKeyFn: (example) => JSON.stringify([
|
|
451
|
+
example.row && example.row.context,
|
|
452
|
+
example.row && example.row.domain,
|
|
453
|
+
example.row && example.row.skill,
|
|
454
|
+
example.row && example.row.targetTags,
|
|
455
|
+
]),
|
|
456
|
+
});
|
|
457
|
+
}
|
|
458
|
+
return model;
|
|
459
|
+
}
|
|
460
|
+
|
|
337
461
|
function rawScore(model, row) {
|
|
338
462
|
if (!model || !model.featureRegistry) {
|
|
339
463
|
return 0;
|
|
@@ -457,6 +581,11 @@ module.exports = {
|
|
|
457
581
|
trainAndPersistRiskModel,
|
|
458
582
|
trainRiskModel,
|
|
459
583
|
getRiskSummary,
|
|
584
|
+
// Exported for the evaluation harness (scripts/eval-risk-model.js) and its tests: measuring
|
|
585
|
+
// this model requires fitting it on a fold, which requires reaching the fit step directly.
|
|
586
|
+
fitBoostedModel,
|
|
587
|
+
holdoutEvaluation,
|
|
588
|
+
scorePairs,
|
|
460
589
|
};
|
|
461
590
|
|
|
462
591
|
if (require.main === module) {
|