thumbgate 1.29.1 → 1.30.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude/commands/dashboard.md +11 -1
- package/.claude/commands/thumbgate-dashboard.md +23 -8
- package/.claude-plugin/plugin.json +1 -1
- package/.well-known/mcp/server-card.json +1 -1
- package/README.md +61 -1
- package/adapters/claude/.mcp.json +2 -2
- package/adapters/forge/forge.yaml +3 -3
- package/adapters/mcp/server-stdio.js +164 -7
- package/adapters/opencode/opencode.json +1 -1
- package/bin/cli.js +7 -5
- package/commands/dashboard.md +11 -1
- package/commands/thumbgate-dashboard.md +23 -8
- package/config/agent-outcome-monitor-thresholds.json +63 -0
- package/config/evals/agent-outcomes-baseline.json +17 -0
- package/config/evals/agent-outcomes-golden.json +412 -0
- package/config/evals/prompt-eval-baseline.json +23 -0
- package/config/mcp-allowlists.json +26 -2
- package/config/post-deploy-marketing-pages.json +26 -1
- package/config/schemas/task-outcome-receipt.schema.json +296 -0
- package/openapi/openapi.yaml +235 -0
- package/package.json +55 -11
- package/public/architecture.html +130 -0
- package/public/assets/diagrams/agent-integration.png +0 -0
- package/public/assets/diagrams/before-after.svg +21 -0
- package/public/assets/diagrams/decision.svg +36 -0
- package/public/assets/diagrams/feedback-pipeline.png +0 -0
- package/public/assets/diagrams/loop.svg +34 -0
- package/public/assets/diagrams/plugin-topology.png +0 -0
- package/public/assets/diagrams/pre-action-gate-loop.svg +59 -0
- package/public/assets/diagrams/stack.svg +18 -0
- package/public/assets/diagrams/thumbgate-architecture.png +0 -0
- package/public/case-studies.html +151 -0
- package/public/eval-scorecard.html +195 -0
- package/public/eval-scorecard.json +18 -0
- package/public/evaluations.html +168 -0
- package/public/index.html +6 -3
- package/public/numbers.html +2 -2
- package/public/whitepaper.html +189 -0
- package/scripts/activation-quickstart.js +1 -0
- package/scripts/agent-outcome-eval.js +130 -0
- package/scripts/agent-outcome-monitor.js +331 -0
- package/scripts/agent-reasoning-traces.js +8 -9
- package/scripts/async-job-runner.js +107 -13
- package/scripts/billing.js +3 -1
- package/scripts/claude-feedback-sync.js +3 -2
- package/scripts/cli-feedback.js +13 -7
- package/scripts/cross-encoder-reranker.js +3 -0
- package/scripts/durability/step.js +121 -12
- package/scripts/feedback-aggregate.js +5 -2
- package/scripts/feedback-loop.js +244 -182
- package/scripts/gates-engine.js +512 -22
- package/scripts/generate-case-study-outreach.js +253 -0
- package/scripts/generate-eval-scorecard.js +276 -0
- package/scripts/growth-campaigns.js +183 -0
- package/scripts/human-escalation.js +265 -0
- package/scripts/hybrid-feedback-context.js +93 -50
- package/scripts/jsonl-watcher.js +1 -0
- package/scripts/judge-reward-function.js +30 -18
- package/scripts/lesson-inference.js +23 -4
- package/scripts/lesson-retrieval.js +71 -4
- package/scripts/lesson-search.js +26 -3
- package/scripts/mcp-config.js +26 -5
- package/scripts/mcp-oauth.js +37 -2
- package/scripts/model-eval.js +308 -0
- package/scripts/parallel-workflow-orchestrator.js +86 -22
- package/scripts/prompt-eval.js +81 -4
- package/scripts/published-cli.js +11 -1
- package/scripts/refresh-proof-pack.js +261 -0
- package/scripts/risk-scorer.js +144 -15
- package/scripts/schedule-manager.js +249 -0
- package/scripts/statusline-local-stats.js +1 -1
- package/scripts/task-outcomes.js +425 -0
- package/scripts/thumbgate-bench.js +13 -0
- package/scripts/tool-contract-validator.js +287 -59
- package/scripts/tool-kpi-tracker.js +124 -0
- package/scripts/tool-registry.js +192 -1
- package/src/api/server.js +355 -89
package/scripts/tool-registry.js
CHANGED
|
@@ -60,8 +60,93 @@ const GOAL_CONTRACT_SCHEMA = {
|
|
|
60
60
|
},
|
|
61
61
|
};
|
|
62
62
|
|
|
63
|
+
const TASK_OUTCOME_INPUT_SCHEMA = {
|
|
64
|
+
type: 'object',
|
|
65
|
+
additionalProperties: false,
|
|
66
|
+
required: ['taskId', 'goal', 'status', 'verification', 'toolCalls', 'policy', 'efficiency'],
|
|
67
|
+
properties: {
|
|
68
|
+
taskId: { type: 'string', minLength: 1 },
|
|
69
|
+
taskType: { type: 'string' },
|
|
70
|
+
goal: { type: 'string', minLength: 1 },
|
|
71
|
+
expectedOutcome: { type: 'string' },
|
|
72
|
+
status: { type: 'string', enum: ['completed', 'failed', 'partial', 'escalated'] },
|
|
73
|
+
verification: {
|
|
74
|
+
type: 'object',
|
|
75
|
+
additionalProperties: false,
|
|
76
|
+
required: ['performed', 'passed', 'evidence'],
|
|
77
|
+
properties: {
|
|
78
|
+
performed: { type: 'boolean' },
|
|
79
|
+
passed: { type: 'boolean' },
|
|
80
|
+
verifier: { type: 'string' },
|
|
81
|
+
method: { type: 'string' },
|
|
82
|
+
evidence: { type: 'array', items: { type: 'string' } },
|
|
83
|
+
unsupportedClaims: { type: 'integer', minimum: 0 },
|
|
84
|
+
},
|
|
85
|
+
},
|
|
86
|
+
toolCalls: {
|
|
87
|
+
type: 'array',
|
|
88
|
+
items: {
|
|
89
|
+
type: 'object',
|
|
90
|
+
additionalProperties: false,
|
|
91
|
+
required: ['name', 'contractValid', 'allowed', 'succeeded', 'attempts'],
|
|
92
|
+
properties: {
|
|
93
|
+
name: { type: 'string' },
|
|
94
|
+
contractValid: { type: 'boolean' },
|
|
95
|
+
allowed: { type: 'boolean' },
|
|
96
|
+
succeeded: { type: 'boolean' },
|
|
97
|
+
attempts: { type: 'integer', minimum: 1 },
|
|
98
|
+
latencyMs: { type: 'number', minimum: 0 },
|
|
99
|
+
costUsd: { type: 'number', minimum: 0 },
|
|
100
|
+
sideEffect: { type: 'boolean' },
|
|
101
|
+
idempotencyKey: { type: 'string' },
|
|
102
|
+
duplicateSideEffect: { type: 'boolean' },
|
|
103
|
+
},
|
|
104
|
+
},
|
|
105
|
+
},
|
|
106
|
+
policy: {
|
|
107
|
+
type: 'object',
|
|
108
|
+
additionalProperties: false,
|
|
109
|
+
required: ['violations', 'unsafeEscapes', 'falseBlocks'],
|
|
110
|
+
properties: {
|
|
111
|
+
violations: { type: 'integer', minimum: 0 },
|
|
112
|
+
unsafeEscapes: { type: 'integer', minimum: 0 },
|
|
113
|
+
falseBlocks: { type: 'integer', minimum: 0 },
|
|
114
|
+
},
|
|
115
|
+
},
|
|
116
|
+
failure: { type: 'object', additionalProperties: true },
|
|
117
|
+
escalation: { type: 'object', additionalProperties: true },
|
|
118
|
+
efficiency: {
|
|
119
|
+
type: 'object',
|
|
120
|
+
additionalProperties: false,
|
|
121
|
+
required: ['latencyMs', 'costUsd'],
|
|
122
|
+
properties: {
|
|
123
|
+
latencyMs: { type: 'number', minimum: 0 },
|
|
124
|
+
costUsd: { type: 'number', minimum: 0 },
|
|
125
|
+
firstAttempt: { type: 'boolean' },
|
|
126
|
+
},
|
|
127
|
+
},
|
|
128
|
+
businessOutcome: { type: 'object', additionalProperties: true },
|
|
129
|
+
traceId: { type: 'string' },
|
|
130
|
+
idempotencyKey: { type: 'string' },
|
|
131
|
+
versions: { type: 'object', additionalProperties: true },
|
|
132
|
+
metadata: { type: 'object', additionalProperties: true },
|
|
133
|
+
},
|
|
134
|
+
};
|
|
135
|
+
|
|
136
|
+
const MEMORY_SCOPE_SCHEMA = {
|
|
137
|
+
type: 'object',
|
|
138
|
+
additionalProperties: false,
|
|
139
|
+
required: ['entityId', 'projectId', 'processId', 'sessionId'],
|
|
140
|
+
properties: {
|
|
141
|
+
entityId: { type: 'string', minLength: 1 },
|
|
142
|
+
projectId: { type: 'string', minLength: 1 },
|
|
143
|
+
processId: { type: 'string', minLength: 1 },
|
|
144
|
+
sessionId: { type: 'string', minLength: 1 },
|
|
145
|
+
},
|
|
146
|
+
};
|
|
147
|
+
|
|
63
148
|
const TOOLS = [
|
|
64
|
-
|
|
149
|
+
destructiveTool({
|
|
65
150
|
name: 'capture_feedback',
|
|
66
151
|
description: 'Capture an up/down signal plus one line of why. Vague feedback is logged, then returned with a clarification prompt instead of memory promotion.',
|
|
67
152
|
inputSchema: {
|
|
@@ -145,6 +230,9 @@ const TOOLS = [
|
|
|
145
230
|
limit: { type: 'number', description: 'Maximum results to return (default 10)' },
|
|
146
231
|
category: { type: 'string', enum: ['error', 'learning', 'preference'] },
|
|
147
232
|
tags: { type: 'array', items: { type: 'string' }, description: 'Require all tags to be present on a lesson' },
|
|
233
|
+
scope: MEMORY_SCOPE_SCHEMA,
|
|
234
|
+
requireScope: { type: 'boolean', description: 'Fail closed unless a complete four-field scope is supplied.' },
|
|
235
|
+
includeShared: { type: 'boolean', description: 'Include explicitly shared memories with scoped results. Defaults true.' },
|
|
148
236
|
},
|
|
149
237
|
},
|
|
150
238
|
}),
|
|
@@ -157,6 +245,9 @@ const TOOLS = [
|
|
|
157
245
|
toolName: { type: 'string', description: 'The tool being called (e.g., Bash, Edit, Read)' },
|
|
158
246
|
actionContext: { type: 'string', description: 'Description of what the tool call is doing' },
|
|
159
247
|
maxResults: { type: 'number', description: 'Max lessons to return (default 5)' },
|
|
248
|
+
scope: MEMORY_SCOPE_SCHEMA,
|
|
249
|
+
requireScope: { type: 'boolean', description: 'Fail closed unless a complete four-field scope is supplied.' },
|
|
250
|
+
includeShared: { type: 'boolean', description: 'Include explicitly shared memories with scoped results. Defaults true.' },
|
|
160
251
|
},
|
|
161
252
|
required: ['toolName'],
|
|
162
253
|
},
|
|
@@ -831,6 +922,10 @@ const TOOLS = [
|
|
|
831
922
|
items: { type: 'string' },
|
|
832
923
|
description: 'Optional protected-file globs that require explicit approval before editing or publishing',
|
|
833
924
|
},
|
|
925
|
+
ttlMs: {
|
|
926
|
+
type: 'number',
|
|
927
|
+
description: 'Optional lease length in milliseconds. With it the scope becomes time-bounded authority (e.g. 90000 for "write under ./src for 90 seconds") and FAILS CLOSED on expiry: a lapsed lease authorises nothing until renewed. Omit for a permanent scope. Clamped to 60s..24h.',
|
|
928
|
+
},
|
|
834
929
|
workflowContract: {
|
|
835
930
|
type: 'object',
|
|
836
931
|
description: 'Optional deterministic workflow run contract. Supports workflowId, allowedBranches, blockedActions, requiredEvidence, and completionGate.',
|
|
@@ -969,6 +1064,102 @@ const TOOLS = [
|
|
|
969
1064
|
},
|
|
970
1065
|
},
|
|
971
1066
|
}),
|
|
1067
|
+
destructiveTool({
|
|
1068
|
+
name: 'record_task_outcome',
|
|
1069
|
+
title: 'Record Verified Task Outcome',
|
|
1070
|
+
description: 'Record an idempotent task-level outcome with verification evidence, tool correctness, policy behavior, latency, cost, and business KPI movement. A completed response without evidence is recorded as not working.',
|
|
1071
|
+
inputSchema: TASK_OUTCOME_INPUT_SCHEMA,
|
|
1072
|
+
outputSchema: {
|
|
1073
|
+
type: 'object',
|
|
1074
|
+
additionalProperties: false,
|
|
1075
|
+
required: ['recorded', 'duplicate', 'receipt'],
|
|
1076
|
+
properties: {
|
|
1077
|
+
recorded: { type: 'boolean' },
|
|
1078
|
+
duplicate: { type: 'boolean' },
|
|
1079
|
+
receipt: { type: 'object', additionalProperties: true },
|
|
1080
|
+
},
|
|
1081
|
+
},
|
|
1082
|
+
}),
|
|
1083
|
+
readOnlyTool({
|
|
1084
|
+
name: 'get_task_outcomes',
|
|
1085
|
+
title: 'Get Task Outcomes',
|
|
1086
|
+
description: 'Read a task outcome by taskId or the most recent evidence-backed outcome receipts.',
|
|
1087
|
+
inputSchema: {
|
|
1088
|
+
type: 'object',
|
|
1089
|
+
additionalProperties: false,
|
|
1090
|
+
properties: {
|
|
1091
|
+
taskId: { type: 'string' },
|
|
1092
|
+
limit: { type: 'integer', minimum: 1, maximum: 100 },
|
|
1093
|
+
},
|
|
1094
|
+
},
|
|
1095
|
+
}),
|
|
1096
|
+
readOnlyTool({
|
|
1097
|
+
name: 'get_agent_outcome_metrics',
|
|
1098
|
+
title: 'Get Agent Outcome Metrics',
|
|
1099
|
+
description: 'Compute transparent task, tool, safety, escalation, latency, cost, and business metrics from recorded task outcomes. Empty data returns insufficient_evidence.',
|
|
1100
|
+
inputSchema: {
|
|
1101
|
+
type: 'object',
|
|
1102
|
+
additionalProperties: false,
|
|
1103
|
+
properties: {},
|
|
1104
|
+
},
|
|
1105
|
+
outputSchema: {
|
|
1106
|
+
type: 'object',
|
|
1107
|
+
additionalProperties: true,
|
|
1108
|
+
required: ['generatedAt', 'sampleSize', 'evidenceStatus', 'task', 'tools', 'safety', 'escalation', 'efficiency', 'businessOutcomes'],
|
|
1109
|
+
properties: {
|
|
1110
|
+
generatedAt: { type: 'string', format: 'date-time' },
|
|
1111
|
+
sampleSize: { type: 'integer', minimum: 0 },
|
|
1112
|
+
evidenceStatus: { type: 'string', enum: ['measured', 'insufficient_evidence'] },
|
|
1113
|
+
task: { type: 'object', additionalProperties: true },
|
|
1114
|
+
tools: { type: 'object', additionalProperties: true },
|
|
1115
|
+
safety: { type: 'object', additionalProperties: true },
|
|
1116
|
+
escalation: { type: 'object', additionalProperties: true },
|
|
1117
|
+
efficiency: { type: 'object', additionalProperties: true },
|
|
1118
|
+
businessOutcomes: { type: 'array' },
|
|
1119
|
+
},
|
|
1120
|
+
},
|
|
1121
|
+
}),
|
|
1122
|
+
destructiveTool({
|
|
1123
|
+
name: 'request_human_escalation',
|
|
1124
|
+
title: 'Request Human Escalation',
|
|
1125
|
+
description: 'Create an idempotent, expiring human-review request with requester identity and evidence. Agents cannot approve their own requests.',
|
|
1126
|
+
inputSchema: {
|
|
1127
|
+
type: 'object',
|
|
1128
|
+
additionalProperties: false,
|
|
1129
|
+
required: ['taskId', 'reason', 'requester', 'evidence'],
|
|
1130
|
+
properties: {
|
|
1131
|
+
taskId: { type: 'string', minLength: 1 },
|
|
1132
|
+
reason: { type: 'string', minLength: 1 },
|
|
1133
|
+
severity: { type: 'string', enum: ['low', 'medium', 'high', 'critical'] },
|
|
1134
|
+
requester: {
|
|
1135
|
+
type: 'object',
|
|
1136
|
+
additionalProperties: false,
|
|
1137
|
+
required: ['id', 'kind'],
|
|
1138
|
+
properties: {
|
|
1139
|
+
id: { type: 'string', minLength: 1 },
|
|
1140
|
+
kind: { type: 'string', enum: ['agent', 'service', 'human'] },
|
|
1141
|
+
displayName: { type: 'string' },
|
|
1142
|
+
},
|
|
1143
|
+
},
|
|
1144
|
+
evidence: { type: 'array', minItems: 1, items: { type: 'string', minLength: 1 } },
|
|
1145
|
+
ttlMs: { type: 'number', minimum: 1 },
|
|
1146
|
+
idempotencyKey: { type: 'string' },
|
|
1147
|
+
},
|
|
1148
|
+
},
|
|
1149
|
+
}),
|
|
1150
|
+
readOnlyTool({
|
|
1151
|
+
name: 'list_human_escalations',
|
|
1152
|
+
title: 'List Human Escalations',
|
|
1153
|
+
description: 'List auditable human-escalation state. Approval decisions are deliberately absent from the agent tool surface.',
|
|
1154
|
+
inputSchema: {
|
|
1155
|
+
type: 'object',
|
|
1156
|
+
additionalProperties: false,
|
|
1157
|
+
properties: {
|
|
1158
|
+
status: { type: 'string', enum: ['pending', 'approved', 'rejected', 'cancelled', 'expired'] },
|
|
1159
|
+
limit: { type: 'integer', minimum: 1, maximum: 100 },
|
|
1160
|
+
},
|
|
1161
|
+
},
|
|
1162
|
+
}),
|
|
972
1163
|
readOnlyTool({
|
|
973
1164
|
name: 'verify_claim',
|
|
974
1165
|
description: 'Check whether a claim has enough tracked evidence before the agent asserts it.',
|