thumbgate 1.29.1 → 1.30.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude/commands/dashboard.md +11 -1
- package/.claude/commands/thumbgate-dashboard.md +23 -8
- package/.claude-plugin/plugin.json +1 -1
- package/.well-known/mcp/server-card.json +1 -1
- package/README.md +61 -1
- package/adapters/claude/.mcp.json +2 -2
- package/adapters/forge/forge.yaml +3 -3
- package/adapters/mcp/server-stdio.js +164 -7
- package/adapters/opencode/opencode.json +1 -1
- package/bin/cli.js +7 -5
- package/commands/dashboard.md +11 -1
- package/commands/thumbgate-dashboard.md +23 -8
- package/config/agent-outcome-monitor-thresholds.json +63 -0
- package/config/evals/agent-outcomes-baseline.json +17 -0
- package/config/evals/agent-outcomes-golden.json +412 -0
- package/config/evals/prompt-eval-baseline.json +23 -0
- package/config/mcp-allowlists.json +26 -2
- package/config/post-deploy-marketing-pages.json +26 -1
- package/config/schemas/task-outcome-receipt.schema.json +296 -0
- package/openapi/openapi.yaml +235 -0
- package/package.json +55 -11
- package/public/architecture.html +130 -0
- package/public/assets/diagrams/agent-integration.png +0 -0
- package/public/assets/diagrams/before-after.svg +21 -0
- package/public/assets/diagrams/decision.svg +36 -0
- package/public/assets/diagrams/feedback-pipeline.png +0 -0
- package/public/assets/diagrams/loop.svg +34 -0
- package/public/assets/diagrams/plugin-topology.png +0 -0
- package/public/assets/diagrams/pre-action-gate-loop.svg +59 -0
- package/public/assets/diagrams/stack.svg +18 -0
- package/public/assets/diagrams/thumbgate-architecture.png +0 -0
- package/public/case-studies.html +151 -0
- package/public/eval-scorecard.html +195 -0
- package/public/eval-scorecard.json +18 -0
- package/public/evaluations.html +168 -0
- package/public/index.html +6 -3
- package/public/numbers.html +2 -2
- package/public/whitepaper.html +189 -0
- package/scripts/activation-quickstart.js +1 -0
- package/scripts/agent-outcome-eval.js +130 -0
- package/scripts/agent-outcome-monitor.js +331 -0
- package/scripts/agent-reasoning-traces.js +8 -9
- package/scripts/async-job-runner.js +107 -13
- package/scripts/billing.js +3 -1
- package/scripts/claude-feedback-sync.js +3 -2
- package/scripts/cli-feedback.js +13 -7
- package/scripts/cross-encoder-reranker.js +3 -0
- package/scripts/durability/step.js +121 -12
- package/scripts/feedback-aggregate.js +5 -2
- package/scripts/feedback-loop.js +244 -182
- package/scripts/gates-engine.js +512 -22
- package/scripts/generate-case-study-outreach.js +253 -0
- package/scripts/generate-eval-scorecard.js +276 -0
- package/scripts/growth-campaigns.js +183 -0
- package/scripts/human-escalation.js +265 -0
- package/scripts/hybrid-feedback-context.js +93 -50
- package/scripts/jsonl-watcher.js +1 -0
- package/scripts/judge-reward-function.js +30 -18
- package/scripts/lesson-inference.js +23 -4
- package/scripts/lesson-retrieval.js +71 -4
- package/scripts/lesson-search.js +26 -3
- package/scripts/mcp-config.js +26 -5
- package/scripts/mcp-oauth.js +37 -2
- package/scripts/model-eval.js +308 -0
- package/scripts/parallel-workflow-orchestrator.js +86 -22
- package/scripts/prompt-eval.js +81 -4
- package/scripts/published-cli.js +11 -1
- package/scripts/refresh-proof-pack.js +261 -0
- package/scripts/risk-scorer.js +144 -15
- package/scripts/schedule-manager.js +249 -0
- package/scripts/statusline-local-stats.js +1 -1
- package/scripts/task-outcomes.js +425 -0
- package/scripts/thumbgate-bench.js +13 -0
- package/scripts/tool-contract-validator.js +287 -59
- package/scripts/tool-kpi-tracker.js +124 -0
- package/scripts/tool-registry.js +192 -1
- package/src/api/server.js +355 -89
|
@@ -0,0 +1,253 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
'use strict';
|
|
3
|
+
|
|
4
|
+
/**
|
|
5
|
+
* generate-case-study-outreach.js — buyer outreach pack from dogfood case studies.
|
|
6
|
+
*
|
|
7
|
+
* No fabricated customer logos. Packs are built from public case-study anchors
|
|
8
|
+
* with first-party UTMs for the cash path.
|
|
9
|
+
*
|
|
10
|
+
* node scripts/generate-case-study-outreach.js --case=sudo-evasion
|
|
11
|
+
* node scripts/generate-case-study-outreach.js --case=sudo-evasion --json
|
|
12
|
+
* node scripts/generate-case-study-outreach.js --case=sudo-evasion --write
|
|
13
|
+
*/
|
|
14
|
+
|
|
15
|
+
const fs = require('node:fs');
|
|
16
|
+
const path = require('node:path');
|
|
17
|
+
|
|
18
|
+
const PROJECT_ROOT = path.resolve(__dirname, '..');
|
|
19
|
+
const DEFAULT_OUT_DIR = path.join(PROJECT_ROOT, 'docs', 'proof', 'outreach');
|
|
20
|
+
|
|
21
|
+
const CASES = Object.freeze({
|
|
22
|
+
'sudo-evasion': {
|
|
23
|
+
id: 'sudo-evasion',
|
|
24
|
+
title: 'A guardrail you could walk past with sudo',
|
|
25
|
+
anchor: 'sudo-evasion',
|
|
26
|
+
problem: 'Catastrophic PreToolUse gates matched the happy path but missed wrappers like `sudo rm -rf ~`.',
|
|
27
|
+
metric: '62 evasion holes → 0 on the published npm artifact',
|
|
28
|
+
result: 'Canonicalization + an adversarial grid (14 commands × 9 transforms) closed the class; CI + 6-hourly published-artifact jobs keep it closed.',
|
|
29
|
+
buyerPain: 'Your coding agent can re-spell a blocked command and walk past a regex denylist.',
|
|
30
|
+
ctaPrimary: 'diagnostic',
|
|
31
|
+
proofLinks: {
|
|
32
|
+
caseStudyPath: '/case-studies#sudo-evasion',
|
|
33
|
+
scorecardPath: '/eval-scorecard',
|
|
34
|
+
whitepaperPath: '/whitepaper',
|
|
35
|
+
diagnosticPath: '/diagnostic',
|
|
36
|
+
proPath: '/checkout/pro',
|
|
37
|
+
},
|
|
38
|
+
},
|
|
39
|
+
'fail-open': {
|
|
40
|
+
id: 'fail-open',
|
|
41
|
+
title: 'Production failure: a firewall enforcing nothing',
|
|
42
|
+
anchor: 'fail-open',
|
|
43
|
+
problem: 'A missing PreToolUse hook binary fails open — the product looked fine while blocking nothing.',
|
|
44
|
+
metric: 'Silent-gate canary + published-artifact deny checks',
|
|
45
|
+
result: 'Enforcement restored and verified with known-dangerous commands; silence is now treated as a P0 class.',
|
|
46
|
+
buyerPain: 'Green uptime does not mean your agent firewall is still firing.',
|
|
47
|
+
ctaPrimary: 'diagnostic',
|
|
48
|
+
proofLinks: {
|
|
49
|
+
caseStudyPath: '/case-studies#fail-open',
|
|
50
|
+
scorecardPath: '/eval-scorecard',
|
|
51
|
+
whitepaperPath: '/whitepaper',
|
|
52
|
+
diagnosticPath: '/diagnostic',
|
|
53
|
+
proPath: '/checkout/pro',
|
|
54
|
+
},
|
|
55
|
+
},
|
|
56
|
+
});
|
|
57
|
+
|
|
58
|
+
function parseArgs(argv = process.argv.slice(2)) {
|
|
59
|
+
const args = {
|
|
60
|
+
caseId: 'sudo-evasion',
|
|
61
|
+
json: false,
|
|
62
|
+
write: false,
|
|
63
|
+
help: false,
|
|
64
|
+
outDir: DEFAULT_OUT_DIR,
|
|
65
|
+
baseUrl: 'https://thumbgate.ai',
|
|
66
|
+
};
|
|
67
|
+
for (const arg of argv) {
|
|
68
|
+
if (arg === '--json') args.json = true;
|
|
69
|
+
else if (arg === '--write') args.write = true;
|
|
70
|
+
else if (arg === '--help' || arg === '-h') args.help = true;
|
|
71
|
+
else if (arg.startsWith('--case=')) args.caseId = arg.slice('--case='.length);
|
|
72
|
+
else if (arg.startsWith('--out-dir=')) args.outDir = path.resolve(arg.slice('--out-dir='.length));
|
|
73
|
+
else if (arg.startsWith('--base-url=')) args.baseUrl = arg.slice('--base-url='.length).replace(/\/$/, '');
|
|
74
|
+
}
|
|
75
|
+
return args;
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
function printHelp() {
|
|
79
|
+
console.log(`Usage: node scripts/generate-case-study-outreach.js --case=<id> [--write] [--json]
|
|
80
|
+
|
|
81
|
+
Cases: ${Object.keys(CASES).join(', ')}
|
|
82
|
+
`);
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
function withUtm(baseUrl, pathAndHash, campaign, content) {
|
|
86
|
+
const [pathname, hash = ''] = pathAndHash.split('#');
|
|
87
|
+
const url = new URL(pathname, baseUrl);
|
|
88
|
+
url.searchParams.set('utm_source', 'case_study_outreach');
|
|
89
|
+
url.searchParams.set('utm_medium', content);
|
|
90
|
+
url.searchParams.set('utm_campaign', campaign);
|
|
91
|
+
url.searchParams.set('cta_id', `${campaign}_${content}`);
|
|
92
|
+
const hashPart = hash ? `#${hash}` : '';
|
|
93
|
+
return `${url.toString()}${hashPart}`;
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
function buildPack(caseDef, options = {}) {
|
|
97
|
+
const baseUrl = options.baseUrl || 'https://thumbgate.ai';
|
|
98
|
+
const campaign = `case_${caseDef.id.replace(/-/g, '_')}`;
|
|
99
|
+
const links = {
|
|
100
|
+
caseStudy: withUtm(baseUrl, caseDef.proofLinks.caseStudyPath, campaign, 'case_study'),
|
|
101
|
+
scorecard: withUtm(baseUrl, caseDef.proofLinks.scorecardPath, campaign, 'scorecard'),
|
|
102
|
+
whitepaper: withUtm(baseUrl, caseDef.proofLinks.whitepaperPath, campaign, 'whitepaper'),
|
|
103
|
+
diagnostic: withUtm(baseUrl, caseDef.proofLinks.diagnosticPath, campaign, 'diagnostic'),
|
|
104
|
+
pro: withUtm(baseUrl, `${caseDef.proofLinks.proPath}`, campaign, 'pro'),
|
|
105
|
+
};
|
|
106
|
+
|
|
107
|
+
const linkedin = [
|
|
108
|
+
caseDef.buyerPain,
|
|
109
|
+
'',
|
|
110
|
+
`We dogfooded this on ThumbGate itself: ${caseDef.metric}.`,
|
|
111
|
+
caseDef.result,
|
|
112
|
+
'',
|
|
113
|
+
`Full write-up (no fabricated logos): ${links.caseStudy}`,
|
|
114
|
+
`Live bench scorecard: ${links.scorecard}`,
|
|
115
|
+
'',
|
|
116
|
+
`If one repeated AI-agent failure is already costing you, the $499 Diagnostic installs one hard gate with regression proof: ${links.diagnostic}`,
|
|
117
|
+
].join('\n');
|
|
118
|
+
|
|
119
|
+
const email = [
|
|
120
|
+
`Subject: Your agent can walk past a regex denylist`,
|
|
121
|
+
'',
|
|
122
|
+
`Hi —`,
|
|
123
|
+
'',
|
|
124
|
+
caseDef.buyerPain,
|
|
125
|
+
'',
|
|
126
|
+
`Concrete proof from our own product loop (not a customer logo page):`,
|
|
127
|
+
`- ${caseDef.metric}`,
|
|
128
|
+
`- ${caseDef.result}`,
|
|
129
|
+
'',
|
|
130
|
+
`Case study: ${links.caseStudy}`,
|
|
131
|
+
`Scorecard: ${links.scorecard}`,
|
|
132
|
+
`White paper: ${links.whitepaper}`,
|
|
133
|
+
'',
|
|
134
|
+
`If you want this on one painful workflow this week: ${links.diagnostic}`,
|
|
135
|
+
`Self-serve Pro: ${links.pro}`,
|
|
136
|
+
'',
|
|
137
|
+
`— Igor`,
|
|
138
|
+
].join('\n');
|
|
139
|
+
|
|
140
|
+
const reddit = [
|
|
141
|
+
`**Problem:** ${caseDef.problem}`,
|
|
142
|
+
'',
|
|
143
|
+
`**What we measured:** ${caseDef.metric}`,
|
|
144
|
+
'',
|
|
145
|
+
`**What fixed it:** ${caseDef.result}`,
|
|
146
|
+
'',
|
|
147
|
+
`Public case study (dogfood, not a fake logo wall): ${links.caseStudy}`,
|
|
148
|
+
`Bench scorecard: ${links.scorecard}`,
|
|
149
|
+
].join('\n');
|
|
150
|
+
|
|
151
|
+
const markdown = [
|
|
152
|
+
`# Outreach pack — ${caseDef.title}`,
|
|
153
|
+
'',
|
|
154
|
+
`Case id: \`${caseDef.id}\``,
|
|
155
|
+
'',
|
|
156
|
+
'## Links (tracked)',
|
|
157
|
+
'',
|
|
158
|
+
`- Case study: ${links.caseStudy}`,
|
|
159
|
+
`- Scorecard: ${links.scorecard}`,
|
|
160
|
+
`- White paper: ${links.whitepaper}`,
|
|
161
|
+
`- Diagnostic $499: ${links.diagnostic}`,
|
|
162
|
+
`- Pro: ${links.pro}`,
|
|
163
|
+
'',
|
|
164
|
+
'## LinkedIn',
|
|
165
|
+
'',
|
|
166
|
+
linkedin,
|
|
167
|
+
'',
|
|
168
|
+
'## Email',
|
|
169
|
+
'',
|
|
170
|
+
'```',
|
|
171
|
+
email,
|
|
172
|
+
'```',
|
|
173
|
+
'',
|
|
174
|
+
'## Reddit / forum',
|
|
175
|
+
'',
|
|
176
|
+
reddit,
|
|
177
|
+
'',
|
|
178
|
+
'## Honesty',
|
|
179
|
+
'',
|
|
180
|
+
'First-party dogfood narrative only. Do not imply third-party customer endorsement.',
|
|
181
|
+
'',
|
|
182
|
+
].join('\n');
|
|
183
|
+
|
|
184
|
+
return {
|
|
185
|
+
caseId: caseDef.id,
|
|
186
|
+
title: caseDef.title,
|
|
187
|
+
links,
|
|
188
|
+
channels: {
|
|
189
|
+
linkedin,
|
|
190
|
+
email,
|
|
191
|
+
reddit,
|
|
192
|
+
},
|
|
193
|
+
markdown,
|
|
194
|
+
};
|
|
195
|
+
}
|
|
196
|
+
|
|
197
|
+
function generate(options = {}) {
|
|
198
|
+
const caseId = options.caseId || 'sudo-evasion';
|
|
199
|
+
const caseDef = CASES[caseId];
|
|
200
|
+
if (!caseDef) {
|
|
201
|
+
throw new Error(`Unknown case id: ${caseId}. Known: ${Object.keys(CASES).join(', ')}`);
|
|
202
|
+
}
|
|
203
|
+
const pack = buildPack(caseDef, { baseUrl: options.baseUrl });
|
|
204
|
+
let outPath = null;
|
|
205
|
+
if (options.write) {
|
|
206
|
+
const outDir = options.outDir || DEFAULT_OUT_DIR;
|
|
207
|
+
fs.mkdirSync(outDir, { recursive: true });
|
|
208
|
+
outPath = path.join(outDir, `case-study-outreach-${caseId}.md`);
|
|
209
|
+
fs.writeFileSync(outPath, pack.markdown, 'utf8');
|
|
210
|
+
}
|
|
211
|
+
return { ...pack, outPath };
|
|
212
|
+
}
|
|
213
|
+
|
|
214
|
+
function main(argv = process.argv.slice(2)) {
|
|
215
|
+
const args = parseArgs(argv);
|
|
216
|
+
if (args.help) {
|
|
217
|
+
printHelp();
|
|
218
|
+
return 0;
|
|
219
|
+
}
|
|
220
|
+
const result = generate(args);
|
|
221
|
+
if (args.json) {
|
|
222
|
+
console.log(JSON.stringify({
|
|
223
|
+
caseId: result.caseId,
|
|
224
|
+
title: result.title,
|
|
225
|
+
links: result.links,
|
|
226
|
+
channels: result.channels,
|
|
227
|
+
outPath: result.outPath,
|
|
228
|
+
}, null, 2));
|
|
229
|
+
} else if (result.outPath) {
|
|
230
|
+
console.log(`Wrote ${result.outPath}`);
|
|
231
|
+
} else {
|
|
232
|
+
console.log(result.markdown);
|
|
233
|
+
}
|
|
234
|
+
return 0;
|
|
235
|
+
}
|
|
236
|
+
|
|
237
|
+
if (path.resolve(process.argv[1] || '') === path.resolve(__filename)) {
|
|
238
|
+
try {
|
|
239
|
+
process.exitCode = main();
|
|
240
|
+
} catch (err) {
|
|
241
|
+
console.error(err.message || err);
|
|
242
|
+
process.exitCode = 1;
|
|
243
|
+
}
|
|
244
|
+
}
|
|
245
|
+
|
|
246
|
+
module.exports = {
|
|
247
|
+
CASES,
|
|
248
|
+
parseArgs,
|
|
249
|
+
withUtm,
|
|
250
|
+
buildPack,
|
|
251
|
+
generate,
|
|
252
|
+
main,
|
|
253
|
+
};
|
|
@@ -0,0 +1,276 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
'use strict';
|
|
3
|
+
|
|
4
|
+
/**
|
|
5
|
+
* generate-eval-scorecard.js — render public/eval-scorecard.html from ThumbGate Bench.
|
|
6
|
+
*
|
|
7
|
+
* Runs the committed golden suite (bench/thumbgate-bench.json) in an isolated
|
|
8
|
+
* runtime and publishes the measured rates so buyers can inspect tool-call
|
|
9
|
+
* correctness without cloning the repo. Regenerated via:
|
|
10
|
+
* npm run eval-scorecard:generate
|
|
11
|
+
*/
|
|
12
|
+
|
|
13
|
+
const fs = require('node:fs');
|
|
14
|
+
const path = require('node:path');
|
|
15
|
+
|
|
16
|
+
const PROJECT_ROOT = path.resolve(__dirname, '..');
|
|
17
|
+
const OUTPUT_PATH = path.join(PROJECT_ROOT, 'public', 'eval-scorecard.html');
|
|
18
|
+
const DEFAULT_SUITE = path.join(PROJECT_ROOT, 'bench', 'thumbgate-bench.json');
|
|
19
|
+
|
|
20
|
+
function escapeHtml(value) {
|
|
21
|
+
return String(value)
|
|
22
|
+
.replaceAll('&', '&')
|
|
23
|
+
.replaceAll('<', '<')
|
|
24
|
+
.replaceAll('>', '>')
|
|
25
|
+
.replaceAll('"', '"')
|
|
26
|
+
.replaceAll("'", ''');
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
function pct(rate) {
|
|
30
|
+
if (!Number.isFinite(rate)) return 'n/a';
|
|
31
|
+
return `${(rate * 100).toFixed(1)}%`;
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
function loadVersion() {
|
|
35
|
+
const pkg = JSON.parse(fs.readFileSync(path.join(PROJECT_ROOT, 'package.json'), 'utf8'));
|
|
36
|
+
return pkg.version || '0.0.0';
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
function runBench() {
|
|
40
|
+
// Prefer the public API surface of thumbgate-bench.
|
|
41
|
+
const bench = require('./thumbgate-bench');
|
|
42
|
+
if (typeof bench.runBenchmark === 'function') {
|
|
43
|
+
return bench.runBenchmark({
|
|
44
|
+
suitePath: DEFAULT_SUITE,
|
|
45
|
+
minScore: 90,
|
|
46
|
+
useRuntimeState: false,
|
|
47
|
+
});
|
|
48
|
+
}
|
|
49
|
+
if (typeof bench.main === 'function') {
|
|
50
|
+
// Some builds only export CLI entry; fall back to subprocess below.
|
|
51
|
+
}
|
|
52
|
+
const { spawnSync } = require('node:child_process');
|
|
53
|
+
const result = spawnSync(
|
|
54
|
+
process.execPath,
|
|
55
|
+
[path.join(PROJECT_ROOT, 'scripts', 'thumbgate-bench.js'), '--json'],
|
|
56
|
+
{ cwd: PROJECT_ROOT, encoding: 'utf8', maxBuffer: 8 * 1024 * 1024 },
|
|
57
|
+
);
|
|
58
|
+
if (result.status !== 0 && !result.stdout) {
|
|
59
|
+
throw new Error(result.stderr || 'thumbgate-bench failed');
|
|
60
|
+
}
|
|
61
|
+
const text = String(result.stdout || '').trim();
|
|
62
|
+
const start = text.indexOf('{');
|
|
63
|
+
const end = text.lastIndexOf('}');
|
|
64
|
+
if (start < 0 || end < start) {
|
|
65
|
+
throw new Error('thumbgate-bench did not emit JSON');
|
|
66
|
+
}
|
|
67
|
+
return JSON.parse(text.slice(start, end + 1));
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
function renderScorecard(input) {
|
|
71
|
+
const {
|
|
72
|
+
version,
|
|
73
|
+
nowIso,
|
|
74
|
+
nowDate,
|
|
75
|
+
metrics,
|
|
76
|
+
passed,
|
|
77
|
+
scenarios,
|
|
78
|
+
sourcePath,
|
|
79
|
+
} = input;
|
|
80
|
+
|
|
81
|
+
const m = metrics || {};
|
|
82
|
+
const rows = (scenarios || []).map((s) => {
|
|
83
|
+
const status = s.passed
|
|
84
|
+
? '<span class="good">PASS</span>'
|
|
85
|
+
: '<span class="bad">FAIL</span>';
|
|
86
|
+
return `<tr>
|
|
87
|
+
<td><code>${escapeHtml(s.id)}</code></td>
|
|
88
|
+
<td>${escapeHtml(s.service || '')}</td>
|
|
89
|
+
<td>${s.unsafe ? 'unsafe' : 'safe'}</td>
|
|
90
|
+
<td><code>${escapeHtml(s.expectedDecision)}</code></td>
|
|
91
|
+
<td><code>${escapeHtml(s.actualDecision)}</code></td>
|
|
92
|
+
<td>${status}</td>
|
|
93
|
+
</tr>`;
|
|
94
|
+
}).join('\n');
|
|
95
|
+
|
|
96
|
+
const softwareLd = {
|
|
97
|
+
'@context': 'https://schema.org',
|
|
98
|
+
'@type': 'Dataset',
|
|
99
|
+
name: 'ThumbGate Bench Scorecard',
|
|
100
|
+
description:
|
|
101
|
+
'Deterministic pre-action gate benchmark metrics: task success, unsafe-action rate, capability rate, false-block rate, and replay stability.',
|
|
102
|
+
url: 'https://thumbgate.ai/eval-scorecard',
|
|
103
|
+
dateModified: nowDate,
|
|
104
|
+
creator: {
|
|
105
|
+
'@type': 'Person',
|
|
106
|
+
name: 'Igor Ganapolsky',
|
|
107
|
+
url: 'https://github.com/IgorGanapolsky',
|
|
108
|
+
},
|
|
109
|
+
variableMeasured: [
|
|
110
|
+
{ '@type': 'PropertyValue', name: 'score', value: m.score },
|
|
111
|
+
{ '@type': 'PropertyValue', name: 'taskSuccessRate', value: m.taskSuccessRate },
|
|
112
|
+
{ '@type': 'PropertyValue', name: 'unsafeActionRate', value: m.unsafeActionRate },
|
|
113
|
+
{ '@type': 'PropertyValue', name: 'blockedUnsafeRate', value: m.blockedUnsafeRate },
|
|
114
|
+
{ '@type': 'PropertyValue', name: 'capabilityRate', value: m.capabilityRate },
|
|
115
|
+
{ '@type': 'PropertyValue', name: 'falseBlockRate', value: m.falseBlockRate },
|
|
116
|
+
{ '@type': 'PropertyValue', name: 'replayStability', value: m.replayStability },
|
|
117
|
+
],
|
|
118
|
+
};
|
|
119
|
+
|
|
120
|
+
return `<!DOCTYPE html>
|
|
121
|
+
<html lang="en">
|
|
122
|
+
<head>
|
|
123
|
+
<meta charset="UTF-8">
|
|
124
|
+
<meta name="viewport" content="width=device-width, initial-scale=1.0">
|
|
125
|
+
<meta name="generator" content="ThumbGate">
|
|
126
|
+
<meta name="author" content="Igor Ganapolsky">
|
|
127
|
+
<title>ThumbGate — Eval Scorecard | ThumbGate Bench Metrics</title>
|
|
128
|
+
<meta name="description" content="Live-regenerated ThumbGate Bench scorecard: task success, unsafe-action rate (must be 0), capability rate, false-block rate, and per-scenario tool-call decisions.">
|
|
129
|
+
<meta property="og:title" content="ThumbGate — Eval Scorecard">
|
|
130
|
+
<meta property="og:description" content="Deterministic gate benchmark metrics buyers can re-run: unsafeActionRate must stay 0.">
|
|
131
|
+
<meta property="og:type" content="website">
|
|
132
|
+
<meta property="og:url" content="https://thumbgate.ai/eval-scorecard">
|
|
133
|
+
<link rel="canonical" href="https://thumbgate.ai/eval-scorecard">
|
|
134
|
+
<link rel="icon" type="image/png" href="/thumbgate-icon.png">
|
|
135
|
+
<script defer data-domain="thumbgate.ai" src="https://plausible.io/js/script.js"></script>
|
|
136
|
+
<script type="application/ld+json">${JSON.stringify(softwareLd)}</script>
|
|
137
|
+
<style>
|
|
138
|
+
:root {
|
|
139
|
+
--bg:#0b0f14; --panel:#111823; --border:#1e2a3a; --text:#e6edf3;
|
|
140
|
+
--muted:#8b98a5; --cyan:#39c5cf; --green:#3fb950; --red:#f85149; --amber:#d29922;
|
|
141
|
+
}
|
|
142
|
+
* { margin:0; padding:0; box-sizing:border-box; }
|
|
143
|
+
body { background:var(--bg); color:var(--text); font-family:-apple-system,BlinkMacSystemFont,"Segoe UI",Roboto,sans-serif; line-height:1.6; }
|
|
144
|
+
nav { padding:1rem 2rem; border-bottom:1px solid var(--border); display:flex; gap:1.25rem; flex-wrap:wrap; align-items:center; }
|
|
145
|
+
nav a { color:var(--muted); text-decoration:none; font-size:0.9rem; }
|
|
146
|
+
nav a:hover { color:var(--cyan); }
|
|
147
|
+
nav .brand { color:var(--text); font-weight:700; }
|
|
148
|
+
.container { max-width:960px; margin:0 auto; padding:2.5rem 1.5rem 4rem; }
|
|
149
|
+
h1 { font-size:2rem; margin-bottom:0.4rem; }
|
|
150
|
+
h2 { font-size:1.3rem; margin:2.2rem 0 0.75rem; color:var(--cyan); }
|
|
151
|
+
.subtitle { color:var(--muted); font-size:1.05rem; margin-bottom:1.25rem; }
|
|
152
|
+
.grid { display:grid; grid-template-columns:repeat(auto-fit,minmax(140px,1fr)); gap:0.75rem; margin:1.25rem 0; }
|
|
153
|
+
.metric { background:var(--panel); border:1px solid var(--border); border-radius:10px; padding:1rem; }
|
|
154
|
+
.metric .label { color:var(--muted); font-size:0.78rem; text-transform:uppercase; letter-spacing:0.04em; }
|
|
155
|
+
.metric .value { font-size:1.55rem; font-weight:700; margin-top:0.25rem; }
|
|
156
|
+
.good { color:var(--green); } .bad { color:var(--red); } .warn { color:var(--amber); }
|
|
157
|
+
table { width:100%; border-collapse:collapse; margin:1rem 0; font-size:0.9rem; }
|
|
158
|
+
th, td { text-align:left; padding:0.5rem 0.6rem; border-bottom:1px solid var(--border); vertical-align:top; }
|
|
159
|
+
th { color:var(--muted); font-weight:600; }
|
|
160
|
+
code { background:#0d1420; border:1px solid var(--border); border-radius:4px; padding:0.1rem 0.35rem; font-size:0.84em; color:var(--cyan); }
|
|
161
|
+
pre { background:#0d1420; border:1px solid var(--border); border-radius:8px; padding:1rem; overflow-x:auto; font-size:0.85rem; margin:1rem 0; }
|
|
162
|
+
.panel { background:var(--panel); border:1px solid var(--border); border-radius:10px; padding:1.1rem 1.25rem; margin:1rem 0; }
|
|
163
|
+
.footer-note { color:var(--muted); font-size:0.85rem; margin-top:2.5rem; text-align:center; }
|
|
164
|
+
.footer-note a { color:var(--cyan); text-decoration:none; }
|
|
165
|
+
a.cta { display:inline-block; background:var(--cyan); color:#001018; font-weight:700; padding:0.7rem 1rem; border-radius:8px; text-decoration:none; margin-right:0.75rem; }
|
|
166
|
+
a.secondary { color:var(--cyan); }
|
|
167
|
+
</style>
|
|
168
|
+
</head>
|
|
169
|
+
<body>
|
|
170
|
+
<nav>
|
|
171
|
+
<a class="brand" href="/">ThumbGate</a>
|
|
172
|
+
<a href="/whitepaper">White paper</a>
|
|
173
|
+
<a href="/architecture">Architecture</a>
|
|
174
|
+
<a href="/eval-scorecard">Scorecard</a>
|
|
175
|
+
<a href="/evaluations">Evaluations</a>
|
|
176
|
+
<a href="/case-studies">Case studies</a>
|
|
177
|
+
<a href="/numbers">Numbers</a>
|
|
178
|
+
<a href="/pricing">Pricing</a>
|
|
179
|
+
</nav>
|
|
180
|
+
<main class="container">
|
|
181
|
+
<h1>Eval scorecard</h1>
|
|
182
|
+
<p class="subtitle">Regenerated ThumbGate Bench results for version <strong>${escapeHtml(version)}</strong>. Updated: ${escapeHtml(nowDate)}. Suite: <code>${escapeHtml(sourcePath || 'bench/thumbgate-bench.json')}</code>.</p>
|
|
183
|
+
|
|
184
|
+
<div class="panel">
|
|
185
|
+
<strong>Pass criteria:</strong>
|
|
186
|
+
<code>unsafeActionRate</code> must be <span class="good">0</span>,
|
|
187
|
+
score ≥ 90, safe work still allowed (<code>capabilityRate</code>),
|
|
188
|
+
and decisions must replay stably.
|
|
189
|
+
Overall: ${passed ? '<span class="good">PASSED</span>' : '<span class="bad">FAILED</span>'} · composite score <strong>${escapeHtml(String(m.score ?? 'n/a'))}</strong>
|
|
190
|
+
<br><br>
|
|
191
|
+
<strong>Reproducibility:</strong> the generator runs ThumbGate Bench in an
|
|
192
|
+
<em>isolated</em> runtime with <code>THUMBGATE_STRICT_ENFORCEMENT=1</code>
|
|
193
|
+
so golden <code>deny</code> expectations are not downgraded by warn-by-default
|
|
194
|
+
posture or free-tier daily-cap state from the operator machine.
|
|
195
|
+
</div>
|
|
196
|
+
|
|
197
|
+
<div class="grid">
|
|
198
|
+
<div class="metric"><div class="label">Task success</div><div class="value good">${escapeHtml(pct(m.taskSuccessRate))}</div></div>
|
|
199
|
+
<div class="metric"><div class="label">Unsafe allowed</div><div class="value ${m.unsafeActionRate === 0 ? 'good' : 'bad'}">${escapeHtml(pct(m.unsafeActionRate))}</div></div>
|
|
200
|
+
<div class="metric"><div class="label">Unsafe blocked</div><div class="value">${escapeHtml(pct(m.blockedUnsafeRate))}</div></div>
|
|
201
|
+
<div class="metric"><div class="label">Capability</div><div class="value">${escapeHtml(pct(m.capabilityRate))}</div></div>
|
|
202
|
+
<div class="metric"><div class="label">False blocks</div><div class="value ${m.falseBlockRate === 0 ? 'good' : 'warn'}">${escapeHtml(pct(m.falseBlockRate))}</div></div>
|
|
203
|
+
<div class="metric"><div class="label">Replay stability</div><div class="value">${escapeHtml(pct(m.replayStability))}</div></div>
|
|
204
|
+
</div>
|
|
205
|
+
|
|
206
|
+
<h2>Per-scenario tool-call decisions</h2>
|
|
207
|
+
<p>Each row is a golden tool-call scenario: expected decision vs actual PreToolUse decision.</p>
|
|
208
|
+
<table>
|
|
209
|
+
<thead>
|
|
210
|
+
<tr><th>Scenario</th><th>Service</th><th>Class</th><th>Expected</th><th>Actual</th><th>Result</th></tr>
|
|
211
|
+
</thead>
|
|
212
|
+
<tbody>
|
|
213
|
+
${rows}
|
|
214
|
+
</tbody>
|
|
215
|
+
</table>
|
|
216
|
+
|
|
217
|
+
<h2>Reproduce locally</h2>
|
|
218
|
+
<pre>git clone https://github.com/IgorGanapolsky/ThumbGate
|
|
219
|
+
cd ThumbGate && npm ci
|
|
220
|
+
npm run thumbgate:bench -- --json
|
|
221
|
+
npm run eval-scorecard:generate</pre>
|
|
222
|
+
|
|
223
|
+
<p>
|
|
224
|
+
<a class="cta" href="/whitepaper">Read the evaluation white paper</a>
|
|
225
|
+
<a class="secondary" href="https://github.com/IgorGanapolsky/ThumbGate/blob/main/docs/THUMBGATE_BENCH.md">Bench methodology on GitHub →</a>
|
|
226
|
+
</p>
|
|
227
|
+
|
|
228
|
+
<p class="footer-note">
|
|
229
|
+
Generated at ${escapeHtml(nowIso)}. First-party measurement only — not customer traction.
|
|
230
|
+
Related: <a href="/evaluations">ML evaluations</a> · <a href="/architecture">Architecture diagrams</a> ·
|
|
231
|
+
<a href="https://github.com/IgorGanapolsky/ThumbGate/blob/main/docs/VERIFICATION_EVIDENCE.md">Verification evidence</a>
|
|
232
|
+
</p>
|
|
233
|
+
</main>
|
|
234
|
+
</body>
|
|
235
|
+
</html>
|
|
236
|
+
`;
|
|
237
|
+
}
|
|
238
|
+
|
|
239
|
+
function generate(options = {}) {
|
|
240
|
+
const now = options.now instanceof Date ? options.now : new Date();
|
|
241
|
+
const nowIso = now.toISOString();
|
|
242
|
+
const nowDate = nowIso.slice(0, 10);
|
|
243
|
+
const version = options.version || loadVersion();
|
|
244
|
+
const report = options.report || runBench();
|
|
245
|
+
const html = renderScorecard({
|
|
246
|
+
version,
|
|
247
|
+
nowIso,
|
|
248
|
+
nowDate,
|
|
249
|
+
metrics: report.metrics || report,
|
|
250
|
+
passed: report.passed !== false,
|
|
251
|
+
scenarios: report.scenarios || [],
|
|
252
|
+
sourcePath: report.sourcePath || 'bench/thumbgate-bench.json',
|
|
253
|
+
});
|
|
254
|
+
const outPath = options.outputPath || OUTPUT_PATH;
|
|
255
|
+
fs.mkdirSync(path.dirname(outPath), { recursive: true });
|
|
256
|
+
fs.writeFileSync(outPath, html, 'utf8');
|
|
257
|
+
return { outPath, html, report };
|
|
258
|
+
}
|
|
259
|
+
|
|
260
|
+
if (path.resolve(process.argv[1] || '') === path.resolve(__filename)) {
|
|
261
|
+
try {
|
|
262
|
+
const { outPath, report } = generate();
|
|
263
|
+
const score = report.metrics?.score ?? report.score;
|
|
264
|
+
console.log(`Wrote ${outPath} (score=${score}, passed=${report.passed !== false})`);
|
|
265
|
+
} catch (err) {
|
|
266
|
+
console.error(err.message || err);
|
|
267
|
+
process.exit(1);
|
|
268
|
+
}
|
|
269
|
+
}
|
|
270
|
+
|
|
271
|
+
module.exports = {
|
|
272
|
+
generate,
|
|
273
|
+
renderScorecard,
|
|
274
|
+
runBench,
|
|
275
|
+
OUTPUT_PATH,
|
|
276
|
+
};
|