thumbgate 1.29.1 → 1.30.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude/commands/dashboard.md +11 -1
- package/.claude/commands/thumbgate-dashboard.md +23 -8
- package/.claude-plugin/plugin.json +1 -1
- package/.well-known/mcp/server-card.json +1 -1
- package/README.md +61 -1
- package/adapters/claude/.mcp.json +2 -2
- package/adapters/forge/forge.yaml +3 -3
- package/adapters/mcp/server-stdio.js +164 -7
- package/adapters/opencode/opencode.json +1 -1
- package/bin/cli.js +7 -5
- package/commands/dashboard.md +11 -1
- package/commands/thumbgate-dashboard.md +23 -8
- package/config/agent-outcome-monitor-thresholds.json +63 -0
- package/config/evals/agent-outcomes-baseline.json +17 -0
- package/config/evals/agent-outcomes-golden.json +412 -0
- package/config/evals/prompt-eval-baseline.json +23 -0
- package/config/mcp-allowlists.json +26 -2
- package/config/post-deploy-marketing-pages.json +26 -1
- package/config/schemas/task-outcome-receipt.schema.json +296 -0
- package/openapi/openapi.yaml +235 -0
- package/package.json +55 -11
- package/public/architecture.html +130 -0
- package/public/assets/diagrams/agent-integration.png +0 -0
- package/public/assets/diagrams/before-after.svg +21 -0
- package/public/assets/diagrams/decision.svg +36 -0
- package/public/assets/diagrams/feedback-pipeline.png +0 -0
- package/public/assets/diagrams/loop.svg +34 -0
- package/public/assets/diagrams/plugin-topology.png +0 -0
- package/public/assets/diagrams/pre-action-gate-loop.svg +59 -0
- package/public/assets/diagrams/stack.svg +18 -0
- package/public/assets/diagrams/thumbgate-architecture.png +0 -0
- package/public/case-studies.html +151 -0
- package/public/eval-scorecard.html +195 -0
- package/public/eval-scorecard.json +18 -0
- package/public/evaluations.html +168 -0
- package/public/index.html +6 -3
- package/public/numbers.html +2 -2
- package/public/whitepaper.html +189 -0
- package/scripts/activation-quickstart.js +1 -0
- package/scripts/agent-outcome-eval.js +130 -0
- package/scripts/agent-outcome-monitor.js +331 -0
- package/scripts/agent-reasoning-traces.js +8 -9
- package/scripts/async-job-runner.js +107 -13
- package/scripts/billing.js +3 -1
- package/scripts/claude-feedback-sync.js +3 -2
- package/scripts/cli-feedback.js +13 -7
- package/scripts/cross-encoder-reranker.js +3 -0
- package/scripts/durability/step.js +121 -12
- package/scripts/feedback-aggregate.js +5 -2
- package/scripts/feedback-loop.js +244 -182
- package/scripts/gates-engine.js +512 -22
- package/scripts/generate-case-study-outreach.js +253 -0
- package/scripts/generate-eval-scorecard.js +276 -0
- package/scripts/growth-campaigns.js +183 -0
- package/scripts/human-escalation.js +265 -0
- package/scripts/hybrid-feedback-context.js +93 -50
- package/scripts/jsonl-watcher.js +1 -0
- package/scripts/judge-reward-function.js +30 -18
- package/scripts/lesson-inference.js +23 -4
- package/scripts/lesson-retrieval.js +71 -4
- package/scripts/lesson-search.js +26 -3
- package/scripts/mcp-config.js +26 -5
- package/scripts/mcp-oauth.js +37 -2
- package/scripts/model-eval.js +308 -0
- package/scripts/parallel-workflow-orchestrator.js +86 -22
- package/scripts/prompt-eval.js +81 -4
- package/scripts/published-cli.js +11 -1
- package/scripts/refresh-proof-pack.js +261 -0
- package/scripts/risk-scorer.js +144 -15
- package/scripts/schedule-manager.js +249 -0
- package/scripts/statusline-local-stats.js +1 -1
- package/scripts/task-outcomes.js +425 -0
- package/scripts/thumbgate-bench.js +13 -0
- package/scripts/tool-contract-validator.js +287 -59
- package/scripts/tool-kpi-tracker.js +124 -0
- package/scripts/tool-registry.js +192 -1
- package/src/api/server.js +355 -89
|
@@ -1,6 +1,183 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
2
|
'use strict';
|
|
3
3
|
|
|
4
|
+
const MARKETING_AGENT_CAMPAIGN_ID = 'marketing_agent_governance_20260727';
|
|
5
|
+
const CAMPAIGN_ALIASES = Object.freeze({
|
|
6
|
+
mg27: MARKETING_AGENT_CAMPAIGN_ID,
|
|
7
|
+
});
|
|
8
|
+
const CAMPAIGN_BUYER_ORIGIN = 'https://thumbgate-production.up.railway.app';
|
|
9
|
+
|
|
10
|
+
function campaignChannel(
|
|
11
|
+
channel,
|
|
12
|
+
permalink,
|
|
13
|
+
medium,
|
|
14
|
+
content = null,
|
|
15
|
+
campaignId = MARKETING_AGENT_CAMPAIGN_ID
|
|
16
|
+
) {
|
|
17
|
+
const buyerUrl = new URL('/go/pro', CAMPAIGN_BUYER_ORIGIN);
|
|
18
|
+
buyerUrl.searchParams.set('utm_source', channel);
|
|
19
|
+
buyerUrl.searchParams.set('utm_medium', medium);
|
|
20
|
+
buyerUrl.searchParams.set('utm_campaign', campaignId);
|
|
21
|
+
if (content) buyerUrl.searchParams.set('utm_content', content);
|
|
22
|
+
return Object.freeze({
|
|
23
|
+
channel,
|
|
24
|
+
status: 'LIVE',
|
|
25
|
+
permalink,
|
|
26
|
+
trackedBuyerUrl: buyerUrl.toString(),
|
|
27
|
+
});
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
const MARKETING_AGENT_CAMPAIGN = Object.freeze({
|
|
31
|
+
campaignId: MARKETING_AGENT_CAMPAIGN_ID,
|
|
32
|
+
aliases: ['mg27'],
|
|
33
|
+
episode: {
|
|
34
|
+
title: 'Marketing Agents Are Too Good Now',
|
|
35
|
+
url: 'https://www.youtube.com/watch?v=U2hogriGmEw',
|
|
36
|
+
},
|
|
37
|
+
channels: [
|
|
38
|
+
campaignChannel(
|
|
39
|
+
'linkedin',
|
|
40
|
+
'https://www.linkedin.com/feed/update/urn:li:share:7487654549785128960/',
|
|
41
|
+
'organic_social',
|
|
42
|
+
'episode_response'
|
|
43
|
+
),
|
|
44
|
+
campaignChannel(
|
|
45
|
+
'hashnode',
|
|
46
|
+
'https://ai-agent-blog-12345.hashnode.dev/your-marketing-agent-can-publish-and-pause-ads-who-gates-the-write',
|
|
47
|
+
'organic_article',
|
|
48
|
+
'episode_deep_dive'
|
|
49
|
+
),
|
|
50
|
+
campaignChannel(
|
|
51
|
+
'bluesky',
|
|
52
|
+
'https://bsky.app/profile/iganapolsky.bsky.social/post/3mro3mkmrzc2y',
|
|
53
|
+
'social',
|
|
54
|
+
null,
|
|
55
|
+
'mg27'
|
|
56
|
+
),
|
|
57
|
+
campaignChannel(
|
|
58
|
+
'threads',
|
|
59
|
+
'https://www.threads.com/@igorganapolsky/post/DbUMBzXDlj8',
|
|
60
|
+
'social',
|
|
61
|
+
null,
|
|
62
|
+
'mg27'
|
|
63
|
+
),
|
|
64
|
+
campaignChannel(
|
|
65
|
+
'instagram',
|
|
66
|
+
'https://www.instagram.com/igorganapolsky/p/DbUNjFsDQz3/',
|
|
67
|
+
'organic_social',
|
|
68
|
+
'episode_card'
|
|
69
|
+
),
|
|
70
|
+
campaignChannel(
|
|
71
|
+
'reddit',
|
|
72
|
+
'https://www.reddit.com/r/SideProject/comments/1v8i0it/i_built_a_preaction_firewall_for_ai_agents_that/',
|
|
73
|
+
'organic_social',
|
|
74
|
+
'sideproject_build'
|
|
75
|
+
),
|
|
76
|
+
campaignChannel(
|
|
77
|
+
'youtube',
|
|
78
|
+
'https://www.youtube.com/post/UgkxERIbGUvSgCkGQ_dx2W0nbTl5_abcF17O',
|
|
79
|
+
'community_post',
|
|
80
|
+
'episode_response'
|
|
81
|
+
),
|
|
82
|
+
],
|
|
83
|
+
});
|
|
84
|
+
|
|
85
|
+
function normalizeCampaignId(value) {
|
|
86
|
+
const normalized = String(value || '').trim();
|
|
87
|
+
if (!normalized) return null;
|
|
88
|
+
return CAMPAIGN_ALIASES[normalized] || normalized;
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
function campaignAttributionKeys(campaign = MARKETING_AGENT_CAMPAIGN) {
|
|
92
|
+
return [...new Set([
|
|
93
|
+
campaign.campaignId,
|
|
94
|
+
...(campaign.aliases || []),
|
|
95
|
+
].map((value) => String(value || '').trim()).filter(Boolean))];
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
function validateCampaignChannel(entry, campaign, seenChannels, seenSources) {
|
|
99
|
+
const issues = [];
|
|
100
|
+
const channel = String(entry.channel || '').trim().toLowerCase();
|
|
101
|
+
if (!channel || seenChannels.has(channel)) {
|
|
102
|
+
issues.push(`duplicate_or_missing_channel:${channel || 'unknown'}`);
|
|
103
|
+
}
|
|
104
|
+
seenChannels.add(channel);
|
|
105
|
+
|
|
106
|
+
if (entry.status !== 'LIVE') {
|
|
107
|
+
issues.push(`channel_not_live:${channel || 'unknown'}`);
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
let permalink;
|
|
111
|
+
let trackedBuyerUrl;
|
|
112
|
+
try {
|
|
113
|
+
permalink = new URL(entry.permalink);
|
|
114
|
+
trackedBuyerUrl = new URL(entry.trackedBuyerUrl);
|
|
115
|
+
} catch {
|
|
116
|
+
issues.push(`invalid_url:${channel || 'unknown'}`);
|
|
117
|
+
return issues;
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
if (permalink.protocol !== 'https:') {
|
|
121
|
+
issues.push(`permalink_not_https:${channel}`);
|
|
122
|
+
}
|
|
123
|
+
if (
|
|
124
|
+
trackedBuyerUrl.protocol !== 'https:'
|
|
125
|
+
|| trackedBuyerUrl.origin !== CAMPAIGN_BUYER_ORIGIN
|
|
126
|
+
|| trackedBuyerUrl.pathname !== '/go/pro'
|
|
127
|
+
) {
|
|
128
|
+
issues.push(`buyer_path_not_canonical:${channel}`);
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
const source = trackedBuyerUrl.searchParams.get('utm_source');
|
|
132
|
+
if (source !== channel || seenSources.has(source)) {
|
|
133
|
+
issues.push(`source_mismatch_or_duplicate:${channel}`);
|
|
134
|
+
}
|
|
135
|
+
seenSources.add(source);
|
|
136
|
+
|
|
137
|
+
if (
|
|
138
|
+
normalizeCampaignId(trackedBuyerUrl.searchParams.get('utm_campaign'))
|
|
139
|
+
!== campaign.campaignId
|
|
140
|
+
) {
|
|
141
|
+
issues.push(`campaign_mismatch:${channel}`);
|
|
142
|
+
}
|
|
143
|
+
if (!trackedBuyerUrl.searchParams.get('utm_medium')) {
|
|
144
|
+
issues.push(`missing_medium:${channel}`);
|
|
145
|
+
}
|
|
146
|
+
return issues;
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
function validateMarketingAgentCampaign(campaign = MARKETING_AGENT_CAMPAIGN) {
|
|
150
|
+
const issues = [];
|
|
151
|
+
const channels = Array.isArray(campaign.channels) ? campaign.channels : [];
|
|
152
|
+
const seenChannels = new Set();
|
|
153
|
+
const seenSources = new Set();
|
|
154
|
+
|
|
155
|
+
if (normalizeCampaignId(campaign.campaignId) !== MARKETING_AGENT_CAMPAIGN_ID) {
|
|
156
|
+
issues.push('campaign_id_must_be_canonical');
|
|
157
|
+
}
|
|
158
|
+
if (campaign.episode?.url !== 'https://www.youtube.com/watch?v=U2hogriGmEw') {
|
|
159
|
+
issues.push('episode_url_mismatch');
|
|
160
|
+
}
|
|
161
|
+
if (channels.length !== 7) {
|
|
162
|
+
issues.push('expected_seven_channels');
|
|
163
|
+
}
|
|
164
|
+
for (const entry of channels) {
|
|
165
|
+
issues.push(...validateCampaignChannel(
|
|
166
|
+
entry,
|
|
167
|
+
campaign,
|
|
168
|
+
seenChannels,
|
|
169
|
+
seenSources
|
|
170
|
+
));
|
|
171
|
+
}
|
|
172
|
+
|
|
173
|
+
return {
|
|
174
|
+
ok: issues.length === 0,
|
|
175
|
+
issues,
|
|
176
|
+
channelCount: channels.length,
|
|
177
|
+
campaignId: campaign.campaignId,
|
|
178
|
+
};
|
|
179
|
+
}
|
|
180
|
+
|
|
4
181
|
function buildCreatorGrowthCampaign(input = {}) {
|
|
5
182
|
const appUrl = input.appUrl || 'https://thumbgate-production.up.railway.app';
|
|
6
183
|
const webinarTitle = input.webinarTitle || 'Stop AI Agents From Repeating Expensive Mistakes';
|
|
@@ -45,5 +222,11 @@ function buildCreatorGrowthCampaign(input = {}) {
|
|
|
45
222
|
}
|
|
46
223
|
|
|
47
224
|
module.exports = {
|
|
225
|
+
CAMPAIGN_ALIASES,
|
|
226
|
+
MARKETING_AGENT_CAMPAIGN,
|
|
227
|
+
MARKETING_AGENT_CAMPAIGN_ID,
|
|
48
228
|
buildCreatorGrowthCampaign,
|
|
229
|
+
campaignAttributionKeys,
|
|
230
|
+
normalizeCampaignId,
|
|
231
|
+
validateMarketingAgentCampaign,
|
|
49
232
|
};
|
|
@@ -0,0 +1,265 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
'use strict';
|
|
3
|
+
|
|
4
|
+
/**
|
|
5
|
+
* Human Escalation Queue
|
|
6
|
+
*
|
|
7
|
+
* Agents may request or inspect escalation, but approval decisions require an
|
|
8
|
+
* explicit human actor identity distinct from the requesting agent. Events are
|
|
9
|
+
* append-only so every transition remains auditable.
|
|
10
|
+
*/
|
|
11
|
+
|
|
12
|
+
const crypto = require('node:crypto');
|
|
13
|
+
const fs = require('node:fs');
|
|
14
|
+
const path = require('node:path');
|
|
15
|
+
const { getFeedbackPaths } = require('./feedback-paths');
|
|
16
|
+
|
|
17
|
+
const ESCALATIONS_FILE = 'human-escalations.jsonl';
|
|
18
|
+
const MAX_TTL_MS = 7 * 24 * 60 * 60 * 1000;
|
|
19
|
+
const DEFAULT_TTL_MS = 24 * 60 * 60 * 1000;
|
|
20
|
+
const SEVERITIES = new Set(['low', 'medium', 'high', 'critical']);
|
|
21
|
+
const DECISIONS = new Set(['approved', 'rejected', 'cancelled']);
|
|
22
|
+
|
|
23
|
+
function getEscalationsPath(options = {}) {
|
|
24
|
+
return path.join(getFeedbackPaths(options).FEEDBACK_DIR, ESCALATIONS_FILE);
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
function requestEscalation(input = {}, options = {}) {
|
|
28
|
+
const now = options.now || new Date();
|
|
29
|
+
const taskId = requiredString(input.taskId, 'taskId');
|
|
30
|
+
const reason = requiredString(input.reason, 'reason');
|
|
31
|
+
const requester = requiredIdentity(input.requester, 'requester');
|
|
32
|
+
const evidence = stringArray(input.evidence);
|
|
33
|
+
if (evidence.length === 0) throw escalationError('evidence must contain at least one item');
|
|
34
|
+
const severity = input.severity || 'medium';
|
|
35
|
+
if (!SEVERITIES.has(severity)) throw escalationError(`severity must be one of ${Array.from(SEVERITIES).join(', ')}`);
|
|
36
|
+
const ttlMs = Math.min(MAX_TTL_MS, Math.max(1, finiteNumber(input.ttlMs, DEFAULT_TTL_MS)));
|
|
37
|
+
const idempotencyKey = requiredString(input.idempotencyKey || taskId, 'idempotencyKey');
|
|
38
|
+
const existing = listEscalations(options).find((entry) => entry.idempotencyKey === idempotencyKey);
|
|
39
|
+
|
|
40
|
+
const request = {
|
|
41
|
+
escalationId: input.escalationId || `esc_${crypto.randomUUID()}`,
|
|
42
|
+
idempotencyKey,
|
|
43
|
+
taskId,
|
|
44
|
+
reason,
|
|
45
|
+
severity,
|
|
46
|
+
requester,
|
|
47
|
+
evidence,
|
|
48
|
+
requestedAt: now.toISOString(),
|
|
49
|
+
expiresAt: new Date(now.getTime() + ttlMs).toISOString(),
|
|
50
|
+
status: 'pending',
|
|
51
|
+
eventType: 'requested',
|
|
52
|
+
};
|
|
53
|
+
request.eventHash = eventHash(request);
|
|
54
|
+
|
|
55
|
+
if (existing) {
|
|
56
|
+
if (eventComparableHash(existing) !== eventComparableHash(request)) {
|
|
57
|
+
const error = escalationError(`conflicting request for idempotency key '${idempotencyKey}'`);
|
|
58
|
+
error.code = 'THUMBGATE_IDEMPOTENCY_CONFLICT';
|
|
59
|
+
throw error;
|
|
60
|
+
}
|
|
61
|
+
return { recorded: false, duplicate: true, escalation: existing };
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
appendEvent(request, options);
|
|
65
|
+
return { recorded: true, duplicate: false, escalation: request };
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
function decideEscalation(input = {}, options = {}) {
|
|
69
|
+
const escalationId = requiredString(input.escalationId, 'escalationId');
|
|
70
|
+
const decision = requiredString(input.decision, 'decision');
|
|
71
|
+
if (!DECISIONS.has(decision)) throw escalationError(`decision must be one of ${Array.from(DECISIONS).join(', ')}`);
|
|
72
|
+
if (Object.hasOwn(input, 'actor')) {
|
|
73
|
+
throw escalationError('decision actor is derived from the authenticated reviewer and must not be supplied by the caller');
|
|
74
|
+
}
|
|
75
|
+
const actor = requiredIdentity(options.authenticatedActor, 'authenticatedActor');
|
|
76
|
+
if (actor.kind !== 'human') throw escalationError('authenticatedActor.kind must be human');
|
|
77
|
+
const reason = requiredString(input.reason, 'reason');
|
|
78
|
+
const current = getEscalation(escalationId, options);
|
|
79
|
+
if (!current) throw escalationError(`unknown escalation '${escalationId}'`);
|
|
80
|
+
if (current.status !== 'pending') throw escalationError(`escalation '${escalationId}' is already ${current.status}`);
|
|
81
|
+
if (sameIdentity(current.requester, actor)) throw escalationError('requester cannot decide their own escalation');
|
|
82
|
+
|
|
83
|
+
const now = options.now || new Date();
|
|
84
|
+
const event = {
|
|
85
|
+
escalationId,
|
|
86
|
+
taskId: current.taskId,
|
|
87
|
+
status: decision,
|
|
88
|
+
eventType: 'decided',
|
|
89
|
+
decision,
|
|
90
|
+
actor,
|
|
91
|
+
reason,
|
|
92
|
+
decidedAt: now.toISOString(),
|
|
93
|
+
};
|
|
94
|
+
event.eventHash = eventHash(event);
|
|
95
|
+
appendEvent(event, options);
|
|
96
|
+
return { recorded: true, escalation: { ...current, ...event } };
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
function listEscalations(options = {}) {
|
|
100
|
+
const events = readEvents(options);
|
|
101
|
+
const byId = new Map();
|
|
102
|
+
for (const event of events) {
|
|
103
|
+
const current = byId.get(event.escalationId) || {};
|
|
104
|
+
byId.set(event.escalationId, { ...current, ...event });
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
const nowMs = (options.now || new Date()).getTime();
|
|
108
|
+
const rows = Array.from(byId.values()).map((entry) => {
|
|
109
|
+
if (entry.status === 'pending' && Date.parse(entry.expiresAt) <= nowMs) {
|
|
110
|
+
return { ...entry, status: 'expired' };
|
|
111
|
+
}
|
|
112
|
+
return entry;
|
|
113
|
+
});
|
|
114
|
+
const status = options.status;
|
|
115
|
+
return rows
|
|
116
|
+
.filter((entry) => !status || entry.status === status)
|
|
117
|
+
.sort((a, b) => Date.parse(b.requestedAt || b.decidedAt) - Date.parse(a.requestedAt || a.decidedAt));
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
function getEscalation(escalationId, options = {}) {
|
|
121
|
+
return listEscalations(options).find((entry) => entry.escalationId === escalationId) || null;
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
function calculateEscalationMetrics(escalations = [], now = new Date()) {
|
|
125
|
+
const rows = escalations.filter(Boolean);
|
|
126
|
+
const decided = rows.filter((entry) => ['approved', 'rejected'].includes(entry.status));
|
|
127
|
+
const decisionLatencies = decided
|
|
128
|
+
.map((entry) => Date.parse(entry.decidedAt) - Date.parse(entry.requestedAt))
|
|
129
|
+
.filter((value) => Number.isFinite(value) && value >= 0);
|
|
130
|
+
const overdue = rows.filter((entry) => entry.status === 'pending' && Date.parse(entry.expiresAt) <= now.getTime());
|
|
131
|
+
return {
|
|
132
|
+
generatedAt: now.toISOString(),
|
|
133
|
+
sampleSize: rows.length,
|
|
134
|
+
evidenceStatus: rows.length ? 'measured' : 'insufficient_evidence',
|
|
135
|
+
pending: rows.filter((entry) => entry.status === 'pending').length,
|
|
136
|
+
approved: rows.filter((entry) => entry.status === 'approved').length,
|
|
137
|
+
rejected: rows.filter((entry) => entry.status === 'rejected').length,
|
|
138
|
+
expired: rows.filter((entry) => entry.status === 'expired').length,
|
|
139
|
+
overdue: overdue.length,
|
|
140
|
+
medianDecisionLatencyMs: percentile(decisionLatencies, 0.5),
|
|
141
|
+
p95DecisionLatencyMs: percentile(decisionLatencies, 0.95),
|
|
142
|
+
};
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
function readEvents(options = {}) {
|
|
146
|
+
const inputPath = options.inputPath ? path.resolve(options.inputPath) : getEscalationsPath(options);
|
|
147
|
+
let raw = '';
|
|
148
|
+
try {
|
|
149
|
+
raw = fs.readFileSync(inputPath, 'utf8');
|
|
150
|
+
} catch {
|
|
151
|
+
return [];
|
|
152
|
+
}
|
|
153
|
+
return raw.split('\n').map((line) => line.trim()).filter(Boolean).flatMap((line) => {
|
|
154
|
+
try {
|
|
155
|
+
return [JSON.parse(line)];
|
|
156
|
+
} catch {
|
|
157
|
+
return [];
|
|
158
|
+
}
|
|
159
|
+
});
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
function appendEvent(event, options) {
|
|
163
|
+
const outputPath = getEscalationsPath(options);
|
|
164
|
+
fs.mkdirSync(path.dirname(outputPath), { recursive: true });
|
|
165
|
+
fs.appendFileSync(outputPath, `${JSON.stringify(event)}\n`, 'utf8');
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
function requiredIdentity(value, field) {
|
|
169
|
+
if (!value || typeof value !== 'object') throw escalationError(`${field} identity is required`);
|
|
170
|
+
const identity = {
|
|
171
|
+
id: requiredString(value.id, `${field}.id`),
|
|
172
|
+
kind: requiredString(value.kind, `${field}.kind`),
|
|
173
|
+
};
|
|
174
|
+
const displayName = optionalString(value.displayName);
|
|
175
|
+
if (displayName) identity.displayName = displayName;
|
|
176
|
+
return identity;
|
|
177
|
+
}
|
|
178
|
+
|
|
179
|
+
function requiredString(value, field) {
|
|
180
|
+
const clean = String(value ?? '').trim();
|
|
181
|
+
if (!clean) throw escalationError(`${field} is required`);
|
|
182
|
+
return clean;
|
|
183
|
+
}
|
|
184
|
+
|
|
185
|
+
function optionalString(value) {
|
|
186
|
+
const clean = String(value ?? '').trim();
|
|
187
|
+
return clean || undefined;
|
|
188
|
+
}
|
|
189
|
+
|
|
190
|
+
function stringArray(value) {
|
|
191
|
+
return Array.isArray(value) ? value.map((entry) => String(entry).trim()).filter(Boolean) : [];
|
|
192
|
+
}
|
|
193
|
+
|
|
194
|
+
function finiteNumber(value, fallback) {
|
|
195
|
+
const number = Number(value);
|
|
196
|
+
return Number.isFinite(number) ? number : fallback;
|
|
197
|
+
}
|
|
198
|
+
|
|
199
|
+
function sameIdentity(a, b) {
|
|
200
|
+
return a?.kind === b?.kind && a?.id === b?.id;
|
|
201
|
+
}
|
|
202
|
+
|
|
203
|
+
function eventHash(event) {
|
|
204
|
+
return crypto.createHash('sha256').update(stableStringify(event)).digest('hex');
|
|
205
|
+
}
|
|
206
|
+
|
|
207
|
+
function eventComparableHash(event) {
|
|
208
|
+
const comparable = {
|
|
209
|
+
idempotencyKey: event.idempotencyKey,
|
|
210
|
+
taskId: event.taskId,
|
|
211
|
+
reason: event.reason,
|
|
212
|
+
severity: event.severity,
|
|
213
|
+
requester: event.requester,
|
|
214
|
+
evidence: event.evidence,
|
|
215
|
+
};
|
|
216
|
+
return crypto.createHash('sha256').update(stableStringify(comparable)).digest('hex');
|
|
217
|
+
}
|
|
218
|
+
|
|
219
|
+
function stableStringify(value) {
|
|
220
|
+
if (!value || typeof value !== 'object') return JSON.stringify(value);
|
|
221
|
+
if (Array.isArray(value)) return `[${value.map(stableStringify).join(',')}]`;
|
|
222
|
+
const keys = Object.keys(value).sort((left, right) => left.localeCompare(right));
|
|
223
|
+
const properties = keys.map((key) => [
|
|
224
|
+
JSON.stringify(key),
|
|
225
|
+
stableStringify(value[key]),
|
|
226
|
+
].join(':'));
|
|
227
|
+
return ['{', properties.join(','), '}'].join('');
|
|
228
|
+
}
|
|
229
|
+
|
|
230
|
+
function percentile(values, quantile) {
|
|
231
|
+
if (!values.length) return null;
|
|
232
|
+
const sorted = [...values].sort((a, b) => a - b);
|
|
233
|
+
const index = Math.max(0, Math.ceil(sorted.length * quantile) - 1);
|
|
234
|
+
return sorted[index];
|
|
235
|
+
}
|
|
236
|
+
|
|
237
|
+
function escalationError(message) {
|
|
238
|
+
const error = new Error(`Invalid human escalation: ${message}`);
|
|
239
|
+
error.code = 'THUMBGATE_ESCALATION_INVALID';
|
|
240
|
+
return error;
|
|
241
|
+
}
|
|
242
|
+
|
|
243
|
+
function isCliInvocation() {
|
|
244
|
+
return Boolean(process.argv[1]) && path.resolve(process.argv[1]) === __filename;
|
|
245
|
+
}
|
|
246
|
+
|
|
247
|
+
if (isCliInvocation()) {
|
|
248
|
+
const command = process.argv[2] || 'metrics';
|
|
249
|
+
const escalations = listEscalations();
|
|
250
|
+
if (command === 'list') console.log(JSON.stringify(escalations, null, 2));
|
|
251
|
+
else if (command === 'metrics') console.log(JSON.stringify(calculateEscalationMetrics(escalations), null, 2));
|
|
252
|
+
else {
|
|
253
|
+
console.error('Usage: human-escalation.js [list|metrics]');
|
|
254
|
+
process.exitCode = 1;
|
|
255
|
+
}
|
|
256
|
+
}
|
|
257
|
+
|
|
258
|
+
module.exports = {
|
|
259
|
+
calculateEscalationMetrics,
|
|
260
|
+
decideEscalation,
|
|
261
|
+
getEscalation,
|
|
262
|
+
getEscalationsPath,
|
|
263
|
+
listEscalations,
|
|
264
|
+
requestEscalation,
|
|
265
|
+
};
|
|
@@ -151,13 +151,14 @@ function isHookPromptEnvelope(context) {
|
|
|
151
151
|
parsed.transcriptPath
|
|
152
152
|
)
|
|
153
153
|
);
|
|
154
|
-
} catch
|
|
154
|
+
} catch {
|
|
155
|
+
// Not JSON — by definition not a hook envelope.
|
|
155
156
|
return false;
|
|
156
157
|
}
|
|
157
158
|
}
|
|
158
159
|
|
|
159
160
|
function patternContext(entry) {
|
|
160
|
-
const context = entry
|
|
161
|
+
const context = entry?.context ? String(entry.context) : '';
|
|
161
162
|
if (!context) return '';
|
|
162
163
|
const hasExplicitFeedback = Boolean(
|
|
163
164
|
entry.whatWentWrong ||
|
|
@@ -189,48 +190,6 @@ function isAutomatedFeedback(entry) {
|
|
|
189
190
|
}
|
|
190
191
|
|
|
191
192
|
|
|
192
|
-
function isHookPromptEnvelope(context) {
|
|
193
|
-
if (!context || typeof context !== 'string') return false;
|
|
194
|
-
try {
|
|
195
|
-
const parsed = JSON.parse(context);
|
|
196
|
-
if (!parsed || typeof parsed !== 'object' || Array.isArray(parsed)) return false;
|
|
197
|
-
return Boolean(
|
|
198
|
-
parsed.prompt &&
|
|
199
|
-
(
|
|
200
|
-
parsed.hookEventName ||
|
|
201
|
-
parsed.hook_event_name ||
|
|
202
|
-
parsed.workspaceRoot ||
|
|
203
|
-
parsed.workspace_root ||
|
|
204
|
-
parsed.session_id ||
|
|
205
|
-
parsed.sessionId ||
|
|
206
|
-
parsed.transcript_path ||
|
|
207
|
-
parsed.transcriptPath
|
|
208
|
-
)
|
|
209
|
-
);
|
|
210
|
-
} catch (_) {
|
|
211
|
-
return false;
|
|
212
|
-
}
|
|
213
|
-
}
|
|
214
|
-
|
|
215
|
-
function patternContext(entry) {
|
|
216
|
-
const context = entry && entry.context ? String(entry.context) : '';
|
|
217
|
-
if (!context) return '';
|
|
218
|
-
const hasExplicitFeedback = Boolean(
|
|
219
|
-
entry.whatWentWrong ||
|
|
220
|
-
entry.what_went_wrong ||
|
|
221
|
-
entry.whatToChange ||
|
|
222
|
-
entry.what_to_change ||
|
|
223
|
-
entry.failureType ||
|
|
224
|
-
(Array.isArray(entry.tags) && entry.tags.length > 0) ||
|
|
225
|
-
entry.structuredRule
|
|
226
|
-
);
|
|
227
|
-
if (isHookPromptEnvelope(context) && !hasExplicitFeedback) return '';
|
|
228
|
-
if (isHookPromptEnvelope(context) && hasExplicitFeedback) {
|
|
229
|
-
return '';
|
|
230
|
-
}
|
|
231
|
-
return context;
|
|
232
|
-
}
|
|
233
|
-
|
|
234
193
|
/**
|
|
235
194
|
* Extract ms from a timestamp value. Returns 0 on failure.
|
|
236
195
|
*/
|
|
@@ -494,14 +453,94 @@ function buildAdditionalContext(state, constraints, maxChars) {
|
|
|
494
453
|
* @param {string[]} words - keyword list from a pattern
|
|
495
454
|
* @returns {boolean}
|
|
496
455
|
*/
|
|
456
|
+
// Callers hand us the pending action in several shapes: a plain command string, an object,
|
|
457
|
+
// or a JSON envelope like {"toolName":…,"command":…,"filePath":…,"affectedFiles":[…]}.
|
|
458
|
+
// Matching over the raw JSON meant the envelope's own KEY NAMES were part of the haystack,
|
|
459
|
+
// so the tokens "files", "command", "tool", "name" and "path" were present on every single
|
|
460
|
+
// evaluation. With a two-hit block threshold, any guard whose keywords included two such
|
|
461
|
+
// common words blocked every action regardless of what that action was. Match on the VALUES
|
|
462
|
+
// only — the guard should key on the action, never on how we happened to serialize it.
|
|
463
|
+
// normalize() runs text through sanitizeFeedbackText(), which exists to reject hook
|
|
464
|
+
// TRANSPORT PAYLOADS and path-dominated blobs from human FEEDBACK. That is the wrong filter
|
|
465
|
+
// for the pending action we are matching against: a real action is frequently just a command
|
|
466
|
+
// plus a list of file paths, which sanitizeFeedbackText() discards wholesale as a "path blob",
|
|
467
|
+
// leaving an empty haystack and silently matching nothing. Apply only the redactions here.
|
|
468
|
+
function normalizeActionText(text) {
|
|
469
|
+
if (!text || typeof text !== 'string') return '';
|
|
470
|
+
return text
|
|
471
|
+
.replace(/\/Users\/[^\s/]+/g, '/Users/redacted')
|
|
472
|
+
.replace(/:\d{4,5}\b/g, ':PORT')
|
|
473
|
+
.toLowerCase()
|
|
474
|
+
.trim();
|
|
475
|
+
}
|
|
476
|
+
|
|
477
|
+
function buildMatchHaystack(input) {
|
|
478
|
+
if (input == null) return '';
|
|
479
|
+
let value = input;
|
|
480
|
+
if (typeof value === 'string') {
|
|
481
|
+
const trimmed = value.trim();
|
|
482
|
+
if (!(trimmed.startsWith('{') || trimmed.startsWith('['))) return value;
|
|
483
|
+
try {
|
|
484
|
+
value = JSON.parse(trimmed);
|
|
485
|
+
} catch {
|
|
486
|
+
return value;
|
|
487
|
+
}
|
|
488
|
+
}
|
|
489
|
+
if (typeof value !== 'object') return String(value);
|
|
490
|
+
|
|
491
|
+
const parts = [];
|
|
492
|
+
const seen = new Set();
|
|
493
|
+
const walk = (node, depth) => {
|
|
494
|
+
if (node == null || depth > 6) return;
|
|
495
|
+
if (typeof node === 'object') {
|
|
496
|
+
if (seen.has(node)) return;
|
|
497
|
+
seen.add(node);
|
|
498
|
+
for (const child of Array.isArray(node) ? node : Object.values(node)) walk(child, depth + 1);
|
|
499
|
+
return;
|
|
500
|
+
}
|
|
501
|
+
if (typeof node === 'boolean') return; // "true"/"false" are structure, not content
|
|
502
|
+
parts.push(String(node));
|
|
503
|
+
};
|
|
504
|
+
walk(value, 0);
|
|
505
|
+
return parts.join(' ');
|
|
506
|
+
}
|
|
507
|
+
|
|
508
|
+
// A guard word is "specific" when it is a compound identifier — keywords() preserves `-` and
|
|
509
|
+
// `_`, so a token like "generated-cache" or "tool_registry" survives intact and is almost
|
|
510
|
+
// always lifted from a real command, path or symbol rather than from prose. One such token
|
|
511
|
+
// is strong evidence on its own.
|
|
512
|
+
//
|
|
513
|
+
// Deliberately NOT keyed on length: ordinary English words ("deployment", "permission",
|
|
514
|
+
// "everything") are long but common, and letting one of them carry a block on its own would
|
|
515
|
+
// over-block. Those still require a second corroborating hit.
|
|
516
|
+
function isSpecificKeyword(word) {
|
|
517
|
+
return /[-_]/.test(word);
|
|
518
|
+
}
|
|
519
|
+
|
|
520
|
+
// Whole-word matching: a bare includes() let "app" hit "apps/", "application" and "happen".
|
|
521
|
+
// Boundaries are non-alphanumerics, so path and punctuation separators still delimit tokens
|
|
522
|
+
// (`src/jobs/queue.js` matches the word "jobs").
|
|
523
|
+
function containsWholeWord(haystack, word) {
|
|
524
|
+
const escaped = String(word).replace(/[.*+?^${}()|[\]\\]/g, String.raw`\$&`);
|
|
525
|
+
try {
|
|
526
|
+
return new RegExp(`(?:^|[^a-z0-9])${escaped}(?:[^a-z0-9]|$)`, 'i').test(haystack);
|
|
527
|
+
} catch {
|
|
528
|
+
return haystack.includes(word);
|
|
529
|
+
}
|
|
530
|
+
}
|
|
531
|
+
|
|
497
532
|
function hasTwoKeywordHits(normalizedInput, words) {
|
|
498
533
|
if (!normalizedInput || !words || words.length === 0) return false;
|
|
499
534
|
let hits = 0;
|
|
535
|
+
const seen = new Set();
|
|
500
536
|
for (const word of words) {
|
|
501
|
-
if (
|
|
502
|
-
|
|
503
|
-
|
|
504
|
-
|
|
537
|
+
if (!word || seen.has(word)) continue;
|
|
538
|
+
seen.add(word);
|
|
539
|
+
if (!containsWholeWord(normalizedInput, word)) continue;
|
|
540
|
+
// A specific compound/long token carries a match on its own; generic words need two.
|
|
541
|
+
if (isSpecificKeyword(word)) return true;
|
|
542
|
+
hits++;
|
|
543
|
+
if (hits >= 2) return true;
|
|
505
544
|
}
|
|
506
545
|
return false;
|
|
507
546
|
}
|
|
@@ -614,7 +653,7 @@ function evaluateCompiledGuards(artifact, toolName, toolInput) {
|
|
|
614
653
|
return { mode: 'allow', reason: '', source: 'compiled' };
|
|
615
654
|
}
|
|
616
655
|
|
|
617
|
-
const normInput =
|
|
656
|
+
const normInput = normalizeActionText(buildMatchHaystack(toolInput));
|
|
618
657
|
const normTool = (toolName || '').toLowerCase();
|
|
619
658
|
|
|
620
659
|
for (const guard of artifact.guards) {
|
|
@@ -656,7 +695,7 @@ function evaluateCompiledGuards(artifact, toolName, toolInput) {
|
|
|
656
695
|
* @returns {{ mode: string, reason: string, source: string }}
|
|
657
696
|
*/
|
|
658
697
|
function evaluatePretoolFromState(state, toolName, toolInput) {
|
|
659
|
-
const normInput =
|
|
698
|
+
const normInput = normalizeActionText(buildMatchHaystack(toolInput));
|
|
660
699
|
const normTool = (toolName || '').toLowerCase();
|
|
661
700
|
|
|
662
701
|
for (const pattern of state.recurringNegativePatterns || []) {
|
|
@@ -824,6 +863,10 @@ module.exports = {
|
|
|
824
863
|
keywords,
|
|
825
864
|
hashText,
|
|
826
865
|
hasTwoKeywordHits,
|
|
866
|
+
buildMatchHaystack,
|
|
867
|
+
normalizeActionText,
|
|
868
|
+
isSpecificKeyword,
|
|
869
|
+
containsWholeWord,
|
|
827
870
|
readJsonl,
|
|
828
871
|
getHybridPaths,
|
|
829
872
|
PATHS,
|
package/scripts/jsonl-watcher.js
CHANGED
|
@@ -71,6 +71,7 @@ function ingestEntry(entry) {
|
|
|
71
71
|
whatWorked: entry.whatWorked || undefined,
|
|
72
72
|
tags: [...(entry.tags || []), WATCHER_SOURCE_TAG, `bridged-from:${entry.source}`],
|
|
73
73
|
skill: entry.skill || undefined,
|
|
74
|
+
reviewOrigin: entry.reviewOrigin || 'imported',
|
|
74
75
|
});
|
|
75
76
|
|
|
76
77
|
return result;
|