thumbgate 1.29.1 → 1.30.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (77) hide show
  1. package/.claude/commands/dashboard.md +11 -1
  2. package/.claude/commands/thumbgate-dashboard.md +23 -8
  3. package/.claude-plugin/plugin.json +1 -1
  4. package/.well-known/mcp/server-card.json +1 -1
  5. package/README.md +61 -1
  6. package/adapters/claude/.mcp.json +2 -2
  7. package/adapters/forge/forge.yaml +3 -3
  8. package/adapters/mcp/server-stdio.js +164 -7
  9. package/adapters/opencode/opencode.json +1 -1
  10. package/bin/cli.js +7 -5
  11. package/commands/dashboard.md +11 -1
  12. package/commands/thumbgate-dashboard.md +23 -8
  13. package/config/agent-outcome-monitor-thresholds.json +63 -0
  14. package/config/evals/agent-outcomes-baseline.json +17 -0
  15. package/config/evals/agent-outcomes-golden.json +412 -0
  16. package/config/evals/prompt-eval-baseline.json +23 -0
  17. package/config/mcp-allowlists.json +26 -2
  18. package/config/post-deploy-marketing-pages.json +26 -1
  19. package/config/schemas/task-outcome-receipt.schema.json +296 -0
  20. package/openapi/openapi.yaml +235 -0
  21. package/package.json +55 -11
  22. package/public/architecture.html +130 -0
  23. package/public/assets/diagrams/agent-integration.png +0 -0
  24. package/public/assets/diagrams/before-after.svg +21 -0
  25. package/public/assets/diagrams/decision.svg +36 -0
  26. package/public/assets/diagrams/feedback-pipeline.png +0 -0
  27. package/public/assets/diagrams/loop.svg +34 -0
  28. package/public/assets/diagrams/plugin-topology.png +0 -0
  29. package/public/assets/diagrams/pre-action-gate-loop.svg +59 -0
  30. package/public/assets/diagrams/stack.svg +18 -0
  31. package/public/assets/diagrams/thumbgate-architecture.png +0 -0
  32. package/public/case-studies.html +151 -0
  33. package/public/eval-scorecard.html +195 -0
  34. package/public/eval-scorecard.json +18 -0
  35. package/public/evaluations.html +168 -0
  36. package/public/index.html +6 -3
  37. package/public/numbers.html +2 -2
  38. package/public/whitepaper.html +189 -0
  39. package/scripts/activation-quickstart.js +1 -0
  40. package/scripts/agent-outcome-eval.js +130 -0
  41. package/scripts/agent-outcome-monitor.js +331 -0
  42. package/scripts/agent-reasoning-traces.js +8 -9
  43. package/scripts/async-job-runner.js +107 -13
  44. package/scripts/billing.js +3 -1
  45. package/scripts/claude-feedback-sync.js +3 -2
  46. package/scripts/cli-feedback.js +13 -7
  47. package/scripts/cross-encoder-reranker.js +3 -0
  48. package/scripts/durability/step.js +121 -12
  49. package/scripts/feedback-aggregate.js +5 -2
  50. package/scripts/feedback-loop.js +244 -182
  51. package/scripts/gates-engine.js +512 -22
  52. package/scripts/generate-case-study-outreach.js +253 -0
  53. package/scripts/generate-eval-scorecard.js +276 -0
  54. package/scripts/growth-campaigns.js +183 -0
  55. package/scripts/human-escalation.js +265 -0
  56. package/scripts/hybrid-feedback-context.js +93 -50
  57. package/scripts/jsonl-watcher.js +1 -0
  58. package/scripts/judge-reward-function.js +30 -18
  59. package/scripts/lesson-inference.js +23 -4
  60. package/scripts/lesson-retrieval.js +71 -4
  61. package/scripts/lesson-search.js +26 -3
  62. package/scripts/mcp-config.js +26 -5
  63. package/scripts/mcp-oauth.js +37 -2
  64. package/scripts/model-eval.js +308 -0
  65. package/scripts/parallel-workflow-orchestrator.js +86 -22
  66. package/scripts/prompt-eval.js +81 -4
  67. package/scripts/published-cli.js +11 -1
  68. package/scripts/refresh-proof-pack.js +261 -0
  69. package/scripts/risk-scorer.js +144 -15
  70. package/scripts/schedule-manager.js +249 -0
  71. package/scripts/statusline-local-stats.js +1 -1
  72. package/scripts/task-outcomes.js +425 -0
  73. package/scripts/thumbgate-bench.js +13 -0
  74. package/scripts/tool-contract-validator.js +287 -59
  75. package/scripts/tool-kpi-tracker.js +124 -0
  76. package/scripts/tool-registry.js +192 -1
  77. package/src/api/server.js +355 -89
@@ -60,8 +60,93 @@ const GOAL_CONTRACT_SCHEMA = {
60
60
  },
61
61
  };
62
62
 
63
+ const TASK_OUTCOME_INPUT_SCHEMA = {
64
+ type: 'object',
65
+ additionalProperties: false,
66
+ required: ['taskId', 'goal', 'status', 'verification', 'toolCalls', 'policy', 'efficiency'],
67
+ properties: {
68
+ taskId: { type: 'string', minLength: 1 },
69
+ taskType: { type: 'string' },
70
+ goal: { type: 'string', minLength: 1 },
71
+ expectedOutcome: { type: 'string' },
72
+ status: { type: 'string', enum: ['completed', 'failed', 'partial', 'escalated'] },
73
+ verification: {
74
+ type: 'object',
75
+ additionalProperties: false,
76
+ required: ['performed', 'passed', 'evidence'],
77
+ properties: {
78
+ performed: { type: 'boolean' },
79
+ passed: { type: 'boolean' },
80
+ verifier: { type: 'string' },
81
+ method: { type: 'string' },
82
+ evidence: { type: 'array', items: { type: 'string' } },
83
+ unsupportedClaims: { type: 'integer', minimum: 0 },
84
+ },
85
+ },
86
+ toolCalls: {
87
+ type: 'array',
88
+ items: {
89
+ type: 'object',
90
+ additionalProperties: false,
91
+ required: ['name', 'contractValid', 'allowed', 'succeeded', 'attempts'],
92
+ properties: {
93
+ name: { type: 'string' },
94
+ contractValid: { type: 'boolean' },
95
+ allowed: { type: 'boolean' },
96
+ succeeded: { type: 'boolean' },
97
+ attempts: { type: 'integer', minimum: 1 },
98
+ latencyMs: { type: 'number', minimum: 0 },
99
+ costUsd: { type: 'number', minimum: 0 },
100
+ sideEffect: { type: 'boolean' },
101
+ idempotencyKey: { type: 'string' },
102
+ duplicateSideEffect: { type: 'boolean' },
103
+ },
104
+ },
105
+ },
106
+ policy: {
107
+ type: 'object',
108
+ additionalProperties: false,
109
+ required: ['violations', 'unsafeEscapes', 'falseBlocks'],
110
+ properties: {
111
+ violations: { type: 'integer', minimum: 0 },
112
+ unsafeEscapes: { type: 'integer', minimum: 0 },
113
+ falseBlocks: { type: 'integer', minimum: 0 },
114
+ },
115
+ },
116
+ failure: { type: 'object', additionalProperties: true },
117
+ escalation: { type: 'object', additionalProperties: true },
118
+ efficiency: {
119
+ type: 'object',
120
+ additionalProperties: false,
121
+ required: ['latencyMs', 'costUsd'],
122
+ properties: {
123
+ latencyMs: { type: 'number', minimum: 0 },
124
+ costUsd: { type: 'number', minimum: 0 },
125
+ firstAttempt: { type: 'boolean' },
126
+ },
127
+ },
128
+ businessOutcome: { type: 'object', additionalProperties: true },
129
+ traceId: { type: 'string' },
130
+ idempotencyKey: { type: 'string' },
131
+ versions: { type: 'object', additionalProperties: true },
132
+ metadata: { type: 'object', additionalProperties: true },
133
+ },
134
+ };
135
+
136
+ const MEMORY_SCOPE_SCHEMA = {
137
+ type: 'object',
138
+ additionalProperties: false,
139
+ required: ['entityId', 'projectId', 'processId', 'sessionId'],
140
+ properties: {
141
+ entityId: { type: 'string', minLength: 1 },
142
+ projectId: { type: 'string', minLength: 1 },
143
+ processId: { type: 'string', minLength: 1 },
144
+ sessionId: { type: 'string', minLength: 1 },
145
+ },
146
+ };
147
+
63
148
  const TOOLS = [
64
- readOnlyTool({
149
+ destructiveTool({
65
150
  name: 'capture_feedback',
66
151
  description: 'Capture an up/down signal plus one line of why. Vague feedback is logged, then returned with a clarification prompt instead of memory promotion.',
67
152
  inputSchema: {
@@ -145,6 +230,9 @@ const TOOLS = [
145
230
  limit: { type: 'number', description: 'Maximum results to return (default 10)' },
146
231
  category: { type: 'string', enum: ['error', 'learning', 'preference'] },
147
232
  tags: { type: 'array', items: { type: 'string' }, description: 'Require all tags to be present on a lesson' },
233
+ scope: MEMORY_SCOPE_SCHEMA,
234
+ requireScope: { type: 'boolean', description: 'Fail closed unless a complete four-field scope is supplied.' },
235
+ includeShared: { type: 'boolean', description: 'Include explicitly shared memories with scoped results. Defaults true.' },
148
236
  },
149
237
  },
150
238
  }),
@@ -157,6 +245,9 @@ const TOOLS = [
157
245
  toolName: { type: 'string', description: 'The tool being called (e.g., Bash, Edit, Read)' },
158
246
  actionContext: { type: 'string', description: 'Description of what the tool call is doing' },
159
247
  maxResults: { type: 'number', description: 'Max lessons to return (default 5)' },
248
+ scope: MEMORY_SCOPE_SCHEMA,
249
+ requireScope: { type: 'boolean', description: 'Fail closed unless a complete four-field scope is supplied.' },
250
+ includeShared: { type: 'boolean', description: 'Include explicitly shared memories with scoped results. Defaults true.' },
160
251
  },
161
252
  required: ['toolName'],
162
253
  },
@@ -831,6 +922,10 @@ const TOOLS = [
831
922
  items: { type: 'string' },
832
923
  description: 'Optional protected-file globs that require explicit approval before editing or publishing',
833
924
  },
925
+ ttlMs: {
926
+ type: 'number',
927
+ description: 'Optional lease length in milliseconds. With it the scope becomes time-bounded authority (e.g. 90000 for "write under ./src for 90 seconds") and FAILS CLOSED on expiry: a lapsed lease authorises nothing until renewed. Omit for a permanent scope. Clamped to 60s..24h.',
928
+ },
834
929
  workflowContract: {
835
930
  type: 'object',
836
931
  description: 'Optional deterministic workflow run contract. Supports workflowId, allowedBranches, blockedActions, requiredEvidence, and completionGate.',
@@ -969,6 +1064,102 @@ const TOOLS = [
969
1064
  },
970
1065
  },
971
1066
  }),
1067
+ destructiveTool({
1068
+ name: 'record_task_outcome',
1069
+ title: 'Record Verified Task Outcome',
1070
+ description: 'Record an idempotent task-level outcome with verification evidence, tool correctness, policy behavior, latency, cost, and business KPI movement. A completed response without evidence is recorded as not working.',
1071
+ inputSchema: TASK_OUTCOME_INPUT_SCHEMA,
1072
+ outputSchema: {
1073
+ type: 'object',
1074
+ additionalProperties: false,
1075
+ required: ['recorded', 'duplicate', 'receipt'],
1076
+ properties: {
1077
+ recorded: { type: 'boolean' },
1078
+ duplicate: { type: 'boolean' },
1079
+ receipt: { type: 'object', additionalProperties: true },
1080
+ },
1081
+ },
1082
+ }),
1083
+ readOnlyTool({
1084
+ name: 'get_task_outcomes',
1085
+ title: 'Get Task Outcomes',
1086
+ description: 'Read a task outcome by taskId or the most recent evidence-backed outcome receipts.',
1087
+ inputSchema: {
1088
+ type: 'object',
1089
+ additionalProperties: false,
1090
+ properties: {
1091
+ taskId: { type: 'string' },
1092
+ limit: { type: 'integer', minimum: 1, maximum: 100 },
1093
+ },
1094
+ },
1095
+ }),
1096
+ readOnlyTool({
1097
+ name: 'get_agent_outcome_metrics',
1098
+ title: 'Get Agent Outcome Metrics',
1099
+ description: 'Compute transparent task, tool, safety, escalation, latency, cost, and business metrics from recorded task outcomes. Empty data returns insufficient_evidence.',
1100
+ inputSchema: {
1101
+ type: 'object',
1102
+ additionalProperties: false,
1103
+ properties: {},
1104
+ },
1105
+ outputSchema: {
1106
+ type: 'object',
1107
+ additionalProperties: true,
1108
+ required: ['generatedAt', 'sampleSize', 'evidenceStatus', 'task', 'tools', 'safety', 'escalation', 'efficiency', 'businessOutcomes'],
1109
+ properties: {
1110
+ generatedAt: { type: 'string', format: 'date-time' },
1111
+ sampleSize: { type: 'integer', minimum: 0 },
1112
+ evidenceStatus: { type: 'string', enum: ['measured', 'insufficient_evidence'] },
1113
+ task: { type: 'object', additionalProperties: true },
1114
+ tools: { type: 'object', additionalProperties: true },
1115
+ safety: { type: 'object', additionalProperties: true },
1116
+ escalation: { type: 'object', additionalProperties: true },
1117
+ efficiency: { type: 'object', additionalProperties: true },
1118
+ businessOutcomes: { type: 'array' },
1119
+ },
1120
+ },
1121
+ }),
1122
+ destructiveTool({
1123
+ name: 'request_human_escalation',
1124
+ title: 'Request Human Escalation',
1125
+ description: 'Create an idempotent, expiring human-review request with requester identity and evidence. Agents cannot approve their own requests.',
1126
+ inputSchema: {
1127
+ type: 'object',
1128
+ additionalProperties: false,
1129
+ required: ['taskId', 'reason', 'requester', 'evidence'],
1130
+ properties: {
1131
+ taskId: { type: 'string', minLength: 1 },
1132
+ reason: { type: 'string', minLength: 1 },
1133
+ severity: { type: 'string', enum: ['low', 'medium', 'high', 'critical'] },
1134
+ requester: {
1135
+ type: 'object',
1136
+ additionalProperties: false,
1137
+ required: ['id', 'kind'],
1138
+ properties: {
1139
+ id: { type: 'string', minLength: 1 },
1140
+ kind: { type: 'string', enum: ['agent', 'service', 'human'] },
1141
+ displayName: { type: 'string' },
1142
+ },
1143
+ },
1144
+ evidence: { type: 'array', minItems: 1, items: { type: 'string', minLength: 1 } },
1145
+ ttlMs: { type: 'number', minimum: 1 },
1146
+ idempotencyKey: { type: 'string' },
1147
+ },
1148
+ },
1149
+ }),
1150
+ readOnlyTool({
1151
+ name: 'list_human_escalations',
1152
+ title: 'List Human Escalations',
1153
+ description: 'List auditable human-escalation state. Approval decisions are deliberately absent from the agent tool surface.',
1154
+ inputSchema: {
1155
+ type: 'object',
1156
+ additionalProperties: false,
1157
+ properties: {
1158
+ status: { type: 'string', enum: ['pending', 'approved', 'rejected', 'cancelled', 'expired'] },
1159
+ limit: { type: 'integer', minimum: 1, maximum: 100 },
1160
+ },
1161
+ },
1162
+ }),
972
1163
  readOnlyTool({
973
1164
  name: 'verify_claim',
974
1165
  description: 'Check whether a claim has enough tracked evidence before the agent asserts it.',