scrapeloop-mcp 0.7.0 → 0.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +39 -18
- package/package.json +23 -7
- package/src/index.js +642 -78
- package/src/mutation-policies.js +232 -0
package/src/index.js
CHANGED
|
@@ -1,27 +1,28 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
2
|
/**
|
|
3
|
-
* scrapeloop-mcp
|
|
3
|
+
* scrapeloop-mcp: Model Context Protocol server for Scrapeloop.
|
|
4
4
|
*
|
|
5
5
|
* A thin wrapper over the Scrapeloop REST API (/api/v1). Authenticates with a
|
|
6
6
|
* per-workspace API key (Settings → API access). Credit-gated actions surface
|
|
7
7
|
* 402/403 bodies verbatim so the model can self-correct or prompt a top-up.
|
|
8
8
|
*
|
|
9
|
-
*
|
|
10
|
-
*
|
|
11
|
-
*
|
|
12
|
-
* user already is, then walk the manifest's `steps` in order.
|
|
9
|
+
* Read get_capabilities, get_setup_status and get_playbook first. Use this
|
|
10
|
+
* workspace's saved answers for setup and later tasks, following the live
|
|
11
|
+
* manifest steps and each tool's approval and retry rules.
|
|
13
12
|
*
|
|
14
13
|
* Env:
|
|
15
|
-
* SCRAPELOOP_API_KEY required
|
|
16
|
-
* SCRAPELOOP_API_URL optional
|
|
14
|
+
* SCRAPELOOP_API_KEY required: sl_live_… key
|
|
15
|
+
* SCRAPELOOP_API_URL optional, defaults to https://api.scrapeloop.com
|
|
17
16
|
*/
|
|
18
17
|
import { Server } from '@modelcontextprotocol/sdk/server/index.js';
|
|
19
18
|
import { StdioServerTransport } from '@modelcontextprotocol/sdk/server/stdio.js';
|
|
20
19
|
import { randomUUID } from 'node:crypto';
|
|
20
|
+
import { AsyncLocalStorage } from 'node:async_hooks';
|
|
21
21
|
import {
|
|
22
22
|
CallToolRequestSchema,
|
|
23
23
|
ListToolsRequestSchema,
|
|
24
24
|
} from '@modelcontextprotocol/sdk/types.js';
|
|
25
|
+
import { MUTATION_POLICIES, RETRY_CONTRACT_COPY } from './mutation-policies.js';
|
|
25
26
|
|
|
26
27
|
const API_KEY = process.env.SCRAPELOOP_API_KEY;
|
|
27
28
|
const BASE = (process.env.SCRAPELOOP_API_URL || 'https://api.scrapeloop.com').replace(/\/$/, '');
|
|
@@ -29,28 +30,57 @@ const TIMEOUT_MS = Number(process.env.SCRAPELOOP_TIMEOUT_MS) || 30000;
|
|
|
29
30
|
const MAX_RETRIES = 3;
|
|
30
31
|
const MAX_TABLE_CELL_WRITES = 500;
|
|
31
32
|
const MAX_TABLE_CELL_REQUEST_BYTES = 256 * 1024;
|
|
33
|
+
// Native Smart Table tool keys, mirroring apps/api/lib/native_tool_columns.json.
|
|
34
|
+
// This package ships standalone, so it cannot read that file; the pairing is
|
|
35
|
+
// asserted in apps/web/lib/native-tool-columns.test.ts. Drift fails SAFE: the
|
|
36
|
+
// API rejects an unknown key and returns available_tools, and
|
|
37
|
+
// get_table_native_tools always serves the live catalog.
|
|
38
|
+
// verify_catchall is deliberately absent: its correctness needs a sibling Verify
|
|
39
|
+
// email column that has already RUN and returned Catch All or Unknown, which one
|
|
40
|
+
// create call cannot establish. Customers use verify_catchall (per address) or
|
|
41
|
+
// the in-app add-column menu. See docs/roadmap/mcp-native-tool-columns.md.
|
|
42
|
+
const NATIVE_TABLE_TOOLS = Object.freeze([
|
|
43
|
+
'verify_email',
|
|
44
|
+
'verify_phone',
|
|
45
|
+
'email_finder',
|
|
46
|
+
'business_email_finder',
|
|
47
|
+
]);
|
|
32
48
|
const TRACE_HEADER = 'X-Scrapeloop-Trace-Id';
|
|
33
49
|
const SAFE_TRACE_ID = /^[A-Za-z0-9_-]{8,80}$/;
|
|
34
50
|
const STARTUP_WARNINGS = new Set();
|
|
51
|
+
const RETRY_BASE_MS = Number(process.env.SCRAPELOOP_RETRY_BASE_MS) || 500;
|
|
52
|
+
const toolCallContext = new AsyncLocalStorage();
|
|
35
53
|
|
|
36
54
|
const sleep = (ms) => new Promise((r) => setTimeout(r, ms));
|
|
37
55
|
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
56
|
+
const uncertainResult = ({ context, method, path, upstreamStatus, reconciliation }) => {
|
|
57
|
+
const policy = context?.policy || {};
|
|
58
|
+
const args = context?.args || {};
|
|
59
|
+
const match = Object.fromEntries(
|
|
60
|
+
(policy.match_fields || [])
|
|
61
|
+
.filter((field) => args[field] !== undefined)
|
|
62
|
+
.map((field) => [field, args[field]]),
|
|
63
|
+
);
|
|
44
64
|
return {
|
|
45
65
|
ok: false,
|
|
46
|
-
status,
|
|
66
|
+
status: upstreamStatus || 0,
|
|
47
67
|
error: {
|
|
48
|
-
|
|
68
|
+
type: 'uncertain_result',
|
|
49
69
|
uncertain_result: true,
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
70
|
+
detail: `The ${method} ${path} result is unknown. The request may have committed before the response was lost.`,
|
|
71
|
+
tool: context?.toolName,
|
|
72
|
+
operation: `${method} ${path}`,
|
|
73
|
+
...(args.idempotency_key ? { idempotency_key: args.idempotency_key } : {}),
|
|
74
|
+
inspect_with: policy.inspect_with || [],
|
|
75
|
+
match_fields: match,
|
|
76
|
+
...(reconciliation ? { reconciliation } : {}),
|
|
77
|
+
...(reconciliation?.retry_guidance
|
|
78
|
+
? { retry_guidance: reconciliation.retry_guidance }
|
|
79
|
+
: {}),
|
|
80
|
+
guidance:
|
|
81
|
+
reconciliation?.guidance ||
|
|
82
|
+
reconciliation?.retry_guidance ||
|
|
83
|
+
'Inspect the listed read tools and stable fields before making any new mutation attempt.',
|
|
54
84
|
},
|
|
55
85
|
};
|
|
56
86
|
};
|
|
@@ -69,34 +99,65 @@ async function api(method, path, body, options = {}) {
|
|
|
69
99
|
}
|
|
70
100
|
const fetcher = options.fetcher || fetch;
|
|
71
101
|
const sleepFn = options.sleepFn || sleep;
|
|
72
|
-
const maxRetries = options.maxRetries ?? MAX_RETRIES;
|
|
73
102
|
const shouldRetryResponse =
|
|
74
103
|
options.shouldRetryResponse || ((status) => status === 429 || status >= 500);
|
|
75
|
-
const retryDelay = options.retryDelay || ((attempt) =>
|
|
104
|
+
const retryDelay = options.retryDelay || ((attempt) => RETRY_BASE_MS * 2 ** attempt);
|
|
76
105
|
const traceId = safeTraceId(options.traceId);
|
|
106
|
+
const verb = method.toUpperCase();
|
|
107
|
+
const context = toolCallContext.getStore();
|
|
108
|
+
const policy = context?.policy;
|
|
109
|
+
if (verb !== 'GET' && (!policy || policy.method !== verb)) {
|
|
110
|
+
return {
|
|
111
|
+
ok: false,
|
|
112
|
+
status: 500,
|
|
113
|
+
error: {
|
|
114
|
+
detail: `MCP mutation policy missing or mismatched for ${context?.toolName || 'unknown tool'} (${verb} ${path}).`,
|
|
115
|
+
},
|
|
116
|
+
};
|
|
117
|
+
}
|
|
118
|
+
if (policy?.retry === 'idempotency_key' && !context?.args?.idempotency_key) {
|
|
119
|
+
return {
|
|
120
|
+
ok: false,
|
|
121
|
+
status: 400,
|
|
122
|
+
error: {
|
|
123
|
+
detail: `${context.toolName} requires idempotency_key. Generate one stable UUID and reuse it for every retry of this exact operation.`,
|
|
124
|
+
},
|
|
125
|
+
};
|
|
126
|
+
}
|
|
127
|
+
const canRetry = verb === 'GET'
|
|
128
|
+
|| policy?.effect === 'read'
|
|
129
|
+
|| policy?.retry === 'idempotent'
|
|
130
|
+
|| policy?.retry === 'idempotency_key';
|
|
131
|
+
const maxRetries = canRetry ? (options.maxRetries ?? MAX_RETRIES) : 0;
|
|
132
|
+
const headers = {
|
|
133
|
+
Authorization: `Bearer ${API_KEY}`,
|
|
134
|
+
'Content-Type': 'application/json',
|
|
135
|
+
...(traceId ? { [TRACE_HEADER]: traceId } : {}),
|
|
136
|
+
...(policy?.retry === 'idempotency_key'
|
|
137
|
+
? { 'Idempotency-Key': context.args.idempotency_key }
|
|
138
|
+
: {}),
|
|
139
|
+
};
|
|
77
140
|
let lastErr;
|
|
78
141
|
for (let attempt = 0; attempt <= maxRetries; attempt++) {
|
|
79
142
|
let res;
|
|
80
143
|
try {
|
|
81
144
|
res = await fetcher(`${BASE}/api/v1${path}`, {
|
|
82
|
-
method,
|
|
83
|
-
headers
|
|
84
|
-
Authorization: `Bearer ${API_KEY}`,
|
|
85
|
-
'Content-Type': 'application/json',
|
|
86
|
-
...(traceId ? { [TRACE_HEADER]: traceId } : {}),
|
|
87
|
-
},
|
|
145
|
+
method: verb,
|
|
146
|
+
headers,
|
|
88
147
|
body: body === undefined ? undefined : JSON.stringify(body),
|
|
89
148
|
signal: AbortSignal.timeout(TIMEOUT_MS),
|
|
90
149
|
});
|
|
91
150
|
} catch (e) {
|
|
92
|
-
// Network error / timeout — retry a few times, then surface a clean error.
|
|
93
151
|
lastErr = e;
|
|
94
152
|
if (attempt < maxRetries) {
|
|
95
153
|
await sleepFn(retryDelay(attempt));
|
|
96
154
|
continue;
|
|
97
155
|
}
|
|
156
|
+
if (policy?.effect === 'mutation') {
|
|
157
|
+
return uncertainResult({ context, method: verb, path, reconciliation: options.reconciliation });
|
|
158
|
+
}
|
|
98
159
|
const timedOut = e && (e.name === 'TimeoutError' || e.name === 'AbortError');
|
|
99
|
-
|
|
160
|
+
return {
|
|
100
161
|
ok: false,
|
|
101
162
|
status: 0,
|
|
102
163
|
error: { detail: `Could not reach Scrapeloop (${timedOut ? `timeout after ${TIMEOUT_MS}ms` : String(e)}).` },
|
|
@@ -104,12 +165,10 @@ async function api(method, path, body, options = {}) {
|
|
|
104
165
|
? { trace_id: traceId, network_error: timedOut ? 'timeout' : 'transport' }
|
|
105
166
|
: {}),
|
|
106
167
|
};
|
|
107
|
-
return options.reconciliation
|
|
108
|
-
? uncertainMutationFailure(0, failure.error, options.reconciliation)
|
|
109
|
-
: failure;
|
|
110
168
|
}
|
|
111
169
|
// Retry transient upstream failures (rate limit / server errors).
|
|
112
|
-
|
|
170
|
+
const transient = shouldRetryResponse(res.status);
|
|
171
|
+
if (transient && attempt < maxRetries) {
|
|
113
172
|
const retryAfter = Number(res.headers.get('retry-after')) * 1000;
|
|
114
173
|
const delay =
|
|
115
174
|
options.respectRetryAfter !== false && retryAfter > 0
|
|
@@ -126,9 +185,18 @@ async function api(method, path, body, options = {}) {
|
|
|
126
185
|
data = { raw: text };
|
|
127
186
|
}
|
|
128
187
|
if (!res.ok) {
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
188
|
+
if (
|
|
189
|
+
transient
|
|
190
|
+
&& policy?.effect === 'mutation'
|
|
191
|
+
&& !(res.status === 429 && canRetry)
|
|
192
|
+
) {
|
|
193
|
+
return uncertainResult({
|
|
194
|
+
context,
|
|
195
|
+
method: verb,
|
|
196
|
+
path,
|
|
197
|
+
upstreamStatus: res.status,
|
|
198
|
+
reconciliation: options.reconciliation,
|
|
199
|
+
});
|
|
132
200
|
}
|
|
133
201
|
return {
|
|
134
202
|
ok: false,
|
|
@@ -144,6 +212,9 @@ async function api(method, path, body, options = {}) {
|
|
|
144
212
|
...(options.captureTrace ? { trace_id: responseTraceId(res, traceId) } : {}),
|
|
145
213
|
};
|
|
146
214
|
}
|
|
215
|
+
if (policy?.effect === 'mutation') {
|
|
216
|
+
return uncertainResult({ context, method: verb, path, reconciliation: options.reconciliation });
|
|
217
|
+
}
|
|
147
218
|
return {
|
|
148
219
|
ok: false,
|
|
149
220
|
status: 0,
|
|
@@ -281,13 +352,34 @@ const runScopeBody = (a) => ({
|
|
|
281
352
|
...(a.only_failed !== undefined ? { only_failed: !!a.only_failed } : {}),
|
|
282
353
|
});
|
|
283
354
|
|
|
355
|
+
// Plan choices mirror the app wizard. Explicit scope and limits prevent a
|
|
356
|
+
// missing answer from silently becoming a country-wide paid plan.
|
|
357
|
+
const PLAN_FIELDS = {
|
|
358
|
+
name: S, kind: S, integration_id: S, country: S,
|
|
359
|
+
granularity: { ...S, enum: ['country', 'state', 'city', 'postal_code'] },
|
|
360
|
+
locations: ARR(S), location_items: ARR(O), categories: ARR(S),
|
|
361
|
+
min_population: { type: 'integer', minimum: 0 }, scrape_config: O,
|
|
362
|
+
run_mode: { ...S, enum: ['all_now', 'over_time'] }, spread_evenly: B,
|
|
363
|
+
monthly_max_leads: { type: 'integer', minimum: 1 },
|
|
364
|
+
monthly_lead_ceiling: { type: 'integer', minimum: 1 },
|
|
365
|
+
monthly_max_usd: { ...N, exclusiveMinimum: 0 },
|
|
366
|
+
monthly_max_credits: { type: 'integer', minimum: 1 },
|
|
367
|
+
spend_buffer_usd: { ...N, minimum: 0 },
|
|
368
|
+
max_leads_per_chunk: { type: 'integer', minimum: 0 },
|
|
369
|
+
chunk_interval_minutes: { type: 'integer', minimum: 0 },
|
|
370
|
+
chunk_order: { ...S, enum: ['sequential', 'largest_first', 'balanced'] },
|
|
371
|
+
rescrape_covered: B, min_age_days: { type: 'integer', minimum: 1 },
|
|
372
|
+
rescrape_after_days: { type: 'integer', minimum: 1 },
|
|
373
|
+
};
|
|
374
|
+
const PLAN_REQUIRED = ['name', 'kind', 'country', 'granularity', 'locations', 'categories', 'run_mode'];
|
|
375
|
+
|
|
284
376
|
// --- Tool registry: name → { def, run } -------------------------------------
|
|
285
377
|
const TOOLS = {
|
|
286
378
|
// ── Discovery / status ────────────────────────────────────────────────
|
|
287
379
|
get_capabilities: {
|
|
288
380
|
def: {
|
|
289
381
|
description:
|
|
290
|
-
'Fetch the live Scrapeloop capability manifest: scrapers (+ their config schemas), enrichment primitives & presets, cleaners, strategy tags, vendors, scopes, filter enums, and the ordered
|
|
382
|
+
'Fetch the live Scrapeloop capability manifest: scrapers (+ their config schemas), enrichment primitives & presets, cleaners, strategy tags, vendors, scopes, filter enums, and the ordered Playbook setup and run steps with guidance. CALL THIS FIRST and whenever unsure what is available: it reflects the workspace as it is right now.',
|
|
291
383
|
inputSchema: obj({}),
|
|
292
384
|
},
|
|
293
385
|
run: () => api('GET', '/manifest'),
|
|
@@ -295,7 +387,7 @@ const TOOLS = {
|
|
|
295
387
|
get_setup_status: {
|
|
296
388
|
def: {
|
|
297
389
|
description:
|
|
298
|
-
'Onboarding checklist booleans (integration connected, a lead revealed, a verify run, a
|
|
390
|
+
'Onboarding checklist booleans (Playbook saved, integration connected, a lead revealed, a verify run, a Table created) so you know what is already done and where to resume.',
|
|
299
391
|
inputSchema: obj({}),
|
|
300
392
|
},
|
|
301
393
|
run: () => api('GET', '/setup/status'),
|
|
@@ -345,7 +437,7 @@ const TOOLS = {
|
|
|
345
437
|
},
|
|
346
438
|
list_credentials: {
|
|
347
439
|
def: {
|
|
348
|
-
description: "List an integration's stored credentials (masked
|
|
440
|
+
description: "List an integration's stored credentials (masked: keys are never returned).",
|
|
349
441
|
inputSchema: obj({ integration_id: S }, ['integration_id']),
|
|
350
442
|
},
|
|
351
443
|
run: (a) =>
|
|
@@ -385,7 +477,7 @@ const TOOLS = {
|
|
|
385
477
|
list_scrapers: {
|
|
386
478
|
def: {
|
|
387
479
|
description:
|
|
388
|
-
'List available scrapers and the JSON Schema for each config.
|
|
480
|
+
'List available scrapers, whether each requires a credential, and the JSON Schema for each config. Reuse saved Playbook choices and ask only for missing fields. Read each plan capability before choosing preview_scrape_plan or a one-off estimate_scrape.',
|
|
389
481
|
inputSchema: obj({}),
|
|
390
482
|
},
|
|
391
483
|
run: () => api('GET', '/scrapers'),
|
|
@@ -413,14 +505,16 @@ const TOOLS = {
|
|
|
413
505
|
submit_scrape: {
|
|
414
506
|
def: {
|
|
415
507
|
description:
|
|
416
|
-
'Submit a scrape job.
|
|
508
|
+
'Submit a scrape job. Optionally pass table_id from get_tables to add results to an existing writable static Table; omit it to keep results in All leads only. Paid scrapers spend vendor credits (BYOK) or lead credits (managed: 1 credit per NEW lead, deduped free), so confirm paid work with the user first. For every paid scraper, pass a positive hard_max_cost_usd equal to or above the estimate after the user confirms that ceiling; paid work will not start without it. A scraper with requires_credential=false is free and needs no integration_id, credential_id, or hard maximum. billing: "auto" (default) uses the workspace key when one exists, else Scrapeloop\'s managed key; "byok"/"managed" force the mode. integration_id is required for normal BYOK, optional for managed, and omitted for credential-free sources. Returns 402 for budget or credit limits and 409 when rescrape confirmation is needed; pass confirm_rescrape:true to proceed past a coverage conflict.',
|
|
417
509
|
inputSchema: obj(
|
|
418
510
|
{
|
|
419
511
|
kind: S,
|
|
420
512
|
integration_id: S,
|
|
421
513
|
config: O,
|
|
514
|
+
table_id: S,
|
|
422
515
|
credential_id: S,
|
|
423
516
|
confirm_rescrape: B,
|
|
517
|
+
hard_max_cost_usd: { ...N, exclusiveMinimum: 0 },
|
|
424
518
|
billing: { ...S, enum: ['auto', 'byok', 'managed'] },
|
|
425
519
|
},
|
|
426
520
|
['kind', 'config'],
|
|
@@ -431,23 +525,62 @@ const TOOLS = {
|
|
|
431
525
|
kind: a.kind,
|
|
432
526
|
integration_id: a.integration_id ?? null,
|
|
433
527
|
config: a.config || {},
|
|
528
|
+
...(a.table_id ? { target_list_id: a.table_id } : {}),
|
|
434
529
|
...(a.credential_id ? { credential_id: a.credential_id } : {}),
|
|
435
530
|
...(a.confirm_rescrape ? { confirm_rescrape: true } : {}),
|
|
531
|
+
...(a.hard_max_cost_usd !== undefined ? { hard_max_cost_usd: a.hard_max_cost_usd } : {}),
|
|
436
532
|
...(a.billing ? { billing: a.billing } : {}),
|
|
437
533
|
}),
|
|
438
534
|
},
|
|
535
|
+
get_scrape_plan_territories: {
|
|
536
|
+
def: {
|
|
537
|
+
description: 'Read places a scraper can cover before previewing a plan. Use returned values exactly. For Outscraper, pass a returned parent value to read its child cities.',
|
|
538
|
+
inputSchema: obj({ kind: S, country: S, granularity: S, min_population: N, parent: S }, ['kind', 'country']),
|
|
539
|
+
},
|
|
540
|
+
run: (a) => api('GET', `/plans/territories?${new URLSearchParams(Object.entries(a).filter(([, value]) => value !== undefined)).toString()}`),
|
|
541
|
+
},
|
|
542
|
+
preview_scrape_plan: {
|
|
543
|
+
def: {
|
|
544
|
+
description: 'Preview all-now or over-time scraping with the same limits as the app. No scrape or credential is created. Read the Playbook and scraper capabilities first. Set a lead or credit cap for credit-priced work, or a dollar cap for dollar-priced work. all_now limits apply once; over_time limits reset each month. Show costs and coverage warnings before creating.',
|
|
545
|
+
inputSchema: obj(PLAN_FIELDS, PLAN_REQUIRED),
|
|
546
|
+
},
|
|
547
|
+
run: (a) => api('POST', '/plans/preview', a),
|
|
548
|
+
},
|
|
549
|
+
create_scrape_plan: {
|
|
550
|
+
def: {
|
|
551
|
+
description: 'Create and start an approved scrape plan into an existing writable Table. This can spend credits or vendor money. Preview the exact choices first and stay within the approved limit and time period. Pass explicit territory values and matching billing limits. Reuse the same idempotency_key after a lost reply, inspect list_scrape_plans and never create a second plan to escape an uncertain result. Over-time limits reset monthly; a total one-time approval does not authorize recurring spend.',
|
|
552
|
+
inputSchema: obj({ ...PLAN_FIELDS, target_list_id: { ...S, format: 'uuid' } }, [...PLAN_REQUIRED, 'target_list_id']),
|
|
553
|
+
},
|
|
554
|
+
run: ({ idempotency_key: _key, ...body }) => api('POST', '/plans', body),
|
|
555
|
+
},
|
|
556
|
+
list_scrape_plans: {
|
|
557
|
+
def: { description: 'Read this workspace scrape plans, progress and spending. Use after an uncertain plan create.', inputSchema: obj({}) },
|
|
558
|
+
run: () => api('GET', '/plans'),
|
|
559
|
+
},
|
|
560
|
+
get_scrape_plan: {
|
|
561
|
+
def: { description: 'Read one scrape plan, its saved scope, limits, progress and spending.', inputSchema: obj({ plan_id: { ...S, format: 'uuid' } }, ['plan_id']) },
|
|
562
|
+
run: (a) => api('GET', `/plans/${enc(a.plan_id)}`),
|
|
563
|
+
},
|
|
564
|
+
pause_scrape_plan: {
|
|
565
|
+
def: { description: 'Pause new work from a scrape plan. A job already running may still finish. Inspect the plan and its jobs after pausing.', inputSchema: obj({ plan_id: { ...S, format: 'uuid' } }, ['plan_id']) },
|
|
566
|
+
run: (a) => api('POST', `/plans/${enc(a.plan_id)}/pause`),
|
|
567
|
+
},
|
|
439
568
|
list_jobs: {
|
|
440
569
|
def: { description: 'List recent scrape/enrich jobs in the workspace.', inputSchema: obj({}) },
|
|
441
570
|
run: () => api('GET', '/jobs'),
|
|
442
571
|
},
|
|
443
572
|
get_job: {
|
|
444
|
-
def: { description: 'Get one job with its execution steps (poll this for scrape progress).', inputSchema: obj({ job_id: S }, ['job_id']) },
|
|
573
|
+
def: { description: 'Get one job with its execution steps (poll this for scrape progress). A scrape job carries `delivery`: how many rows were found, kept, new, already yours, and added to its Table, why rows were left out, and whether adding them to the Table worked.', inputSchema: obj({ job_id: S }, ['job_id']) },
|
|
445
574
|
run: (a) => api('GET', `/jobs/${enc(a.job_id)}`),
|
|
446
575
|
},
|
|
447
576
|
cancel_job: {
|
|
448
|
-
def: { description: '
|
|
577
|
+
def: { description: 'Request cancellation of a queued or running job. An active external vendor task remains nonterminal while the worker aborts it and settles final partial usage, then becomes cancelled.', inputSchema: obj({ job_id: S }, ['job_id']) },
|
|
449
578
|
run: (a) => api('POST', `/jobs/${enc(a.job_id)}/cancel`),
|
|
450
579
|
},
|
|
580
|
+
attach_job_to_table: {
|
|
581
|
+
def: { description: 'Add a finished scrape job\'s leads to its Table again. Use it only when get_job shows `delivery.attach.state` as "failed" and the job is done, and not when `delivery.attach.code` is "table_missing" (that Table was deleted). Rows already in the Table are skipped, so repeating it never adds a row twice. Returns `added`, `in_table`, and the updated `delivery`.', inputSchema: obj({ job_id: S }, ['job_id']) },
|
|
582
|
+
run: (a) => api('POST', `/jobs/${enc(a.job_id)}/attach-to-table`),
|
|
583
|
+
},
|
|
451
584
|
|
|
452
585
|
// ── Lead database ─────────────────────────────────────────────────────
|
|
453
586
|
search_leads: {
|
|
@@ -472,7 +605,7 @@ const TOOLS = {
|
|
|
472
605
|
},
|
|
473
606
|
reveal_lead: {
|
|
474
607
|
def: {
|
|
475
|
-
description: "Reveal a lead's email by global_person_id. SPENDS 1 lead credit (idempotent
|
|
608
|
+
description: "Reveal a lead's email by global_person_id. SPENDS 1 lead credit (idempotent: already-revealed leads are free).",
|
|
476
609
|
inputSchema: obj({ global_person_id: S }, ['global_person_id']),
|
|
477
610
|
},
|
|
478
611
|
run: (a) => api('POST', '/leads/reveal', { global_person_id: a.global_person_id }),
|
|
@@ -617,7 +750,7 @@ const TOOLS = {
|
|
|
617
750
|
verify_catchall: {
|
|
618
751
|
def: {
|
|
619
752
|
description:
|
|
620
|
-
'Resolve a mailbox on a catch-all domain with
|
|
753
|
+
'Resolve a mailbox on a catch-all domain with Scrapeloop managed verification (5 verify credits). Optional and off unless chosen. It can resolve some unknown addresses but does not promise more replies. Estimate the eligible set and stay within the approved credit budget.',
|
|
621
754
|
inputSchema: obj({ email: S }, ['email']),
|
|
622
755
|
},
|
|
623
756
|
run: (a) => api('POST', '/verify/catchall', { email: a.email }),
|
|
@@ -676,7 +809,7 @@ const TOOLS = {
|
|
|
676
809
|
},
|
|
677
810
|
run_enrich: {
|
|
678
811
|
def: {
|
|
679
|
-
description: 'Enqueue a bulk enrichment run. SPENDS vendor credits
|
|
812
|
+
description: 'Enqueue a bulk enrichment run. SPENDS vendor credits: estimate + confirm first.',
|
|
680
813
|
inputSchema: obj({ preset_slug: S, preset_id: S, lead_ids: ARR(S), filter_query: O }),
|
|
681
814
|
},
|
|
682
815
|
run: (a) =>
|
|
@@ -707,6 +840,24 @@ const TOOLS = {
|
|
|
707
840
|
...(a.cache_settings ? { cache_settings: a.cache_settings } : {}),
|
|
708
841
|
}),
|
|
709
842
|
},
|
|
843
|
+
get_playbook: {
|
|
844
|
+
def: { description: "Read this workspace's saved Playbook and niches before setting up work.", inputSchema: obj({}) },
|
|
845
|
+
run: () => api('GET', '/playbook'),
|
|
846
|
+
},
|
|
847
|
+
update_playbook: {
|
|
848
|
+
def: {
|
|
849
|
+
description: 'Save defaults for this workspace only. Saves the sections you provide and keeps other answers. Never starts a scrape or spends credits.',
|
|
850
|
+
inputSchema: obj({ sections: { type: 'object', properties: Object.fromEntries(['channels', 'scrape', 'tables', 'checks', 'routing', 'campaigns'].map((key) => [key, { type: 'object' }])), additionalProperties: false } }, ['sections']),
|
|
851
|
+
},
|
|
852
|
+
run: (a) => api('PUT', '/playbook', a.sections),
|
|
853
|
+
},
|
|
854
|
+
set_playbook_niches: {
|
|
855
|
+
def: {
|
|
856
|
+
description: 'Save chosen niches in tiers 1, 2 or 3. A null tier removes a niche without deleting its category. Set replace to true only to replace all chosen niches.',
|
|
857
|
+
inputSchema: obj({ niches: ARR(obj({ gcid: { type: ['string', 'null'] }, name: { type: ['string', 'null'] }, tier: { type: ['integer', 'null'], enum: [1, 2, 3, null] } })), replace: B }, ['niches']),
|
|
858
|
+
},
|
|
859
|
+
run: (a) => api('PUT', '/playbook/niches', a),
|
|
860
|
+
},
|
|
710
861
|
get_ai_context: {
|
|
711
862
|
def: {
|
|
712
863
|
description: 'Get the workspace AI context (company description, ICP, buyer personas) that seeds every AI feature.',
|
|
@@ -761,7 +912,7 @@ const TOOLS = {
|
|
|
761
912
|
list_senders: {
|
|
762
913
|
def: {
|
|
763
914
|
description:
|
|
764
|
-
'List the sender vendors Scrapeloop supports (instantly, …) with each one\'s capabilities (webhooks, lead lists, async moves, bulk upload). Catalog read
|
|
915
|
+
'List the sender vendors Scrapeloop supports (instantly, …) with each one\'s capabilities (webhooks, lead lists, async moves, bulk upload). Catalog read: use it to know what a connected sender can do before configuring feed/offload.',
|
|
765
916
|
inputSchema: obj({}),
|
|
766
917
|
},
|
|
767
918
|
run: () => api('GET', '/senders'),
|
|
@@ -769,7 +920,7 @@ const TOOLS = {
|
|
|
769
920
|
list_sender_campaigns: {
|
|
770
921
|
def: {
|
|
771
922
|
description:
|
|
772
|
-
"List a sender credential's campaigns at the vendor (vendor-generic: works for any connected sender key, same shape as list_instantly_campaigns
|
|
923
|
+
"List a sender credential's campaigns at the vendor (vendor-generic: works for any connected sender key, same shape as list_instantly_campaigns; use it to pick external_list_id).",
|
|
773
924
|
inputSchema: obj({ credential_id: S }, ['credential_id']),
|
|
774
925
|
},
|
|
775
926
|
run: (a) => api('GET', `/senders/${enc(a.credential_id)}/campaigns`),
|
|
@@ -806,6 +957,10 @@ const TOOLS = {
|
|
|
806
957
|
def: { description: 'List campaigns with per-campaign stats and the bound list name.', inputSchema: obj({}) },
|
|
807
958
|
run: () => api('GET', '/campaigns'),
|
|
808
959
|
},
|
|
960
|
+
get_campaign_setup_options: {
|
|
961
|
+
def: { description: 'Read this workspace\'s channel choices and defaults for new campaign drafts. Explicit create choices win. Reading options changes nothing and starts no campaign.', inputSchema: obj({}) },
|
|
962
|
+
run: () => api('GET', '/campaigns/options'),
|
|
963
|
+
},
|
|
809
964
|
preview_add_leads_to_campaign: {
|
|
810
965
|
def: {
|
|
811
966
|
description:
|
|
@@ -833,21 +988,32 @@ const TOOLS = {
|
|
|
833
988
|
create_campaign: {
|
|
834
989
|
def: {
|
|
835
990
|
description:
|
|
836
|
-
'Create a campaign
|
|
991
|
+
'Create a draft campaign connected to a sender and a Scrapeloop Table or segment. Read get_campaign_setup_options first. Absent choices use this workspace Playbook; explicit values win. GoHighLevel SMS needs a source Table and external_list_id set to a workflow ID or contacts-only. Nothing starts until activate_campaign. Ask the user before Start.',
|
|
837
992
|
inputSchema: obj(
|
|
838
993
|
{
|
|
839
994
|
name: S,
|
|
995
|
+
sender_options: { ...O, description: "GoHighLevel SMS choices: workflow_id, tags, require_verified_mobile, voip_counts_as_mobile, missing_fields (ask, create, never)." },
|
|
996
|
+
icon: { type: ["string", "null"], description: "One or two emoji, or null to clear." },
|
|
840
997
|
source_list_id: S,
|
|
841
998
|
segment_id: S,
|
|
842
999
|
sender_credential_id: S,
|
|
843
1000
|
external_list_id: S,
|
|
844
1001
|
instantly_campaign_name: S,
|
|
845
1002
|
feed_mode: { ...S, enum: ['off', 'auto_add', 'drip'] },
|
|
1003
|
+
allow_risky_emails: B,
|
|
1004
|
+
offload_enabled: B,
|
|
1005
|
+
offload_no_reply_days: N,
|
|
1006
|
+
offload_to_instantly: B,
|
|
1007
|
+
offload_to_scrapeloop: B,
|
|
1008
|
+
offload_scrapeloop_list_id: S,
|
|
846
1009
|
drip_strategy: { ...S, enum: ['target_active', 'fixed_daily'] },
|
|
847
1010
|
drip_daily_count: N,
|
|
848
1011
|
target_active_count: N,
|
|
849
1012
|
cooldown_days: N,
|
|
850
1013
|
resting_period_days: N,
|
|
1014
|
+
recycle_mode: { ...S, enum: ['off', 'after_cooldown'] },
|
|
1015
|
+
prioritize_fresh: B,
|
|
1016
|
+
bounce_instantly_list_id: S,
|
|
851
1017
|
},
|
|
852
1018
|
['name', 'sender_credential_id', 'external_list_id'],
|
|
853
1019
|
),
|
|
@@ -856,7 +1022,10 @@ const TOOLS = {
|
|
|
856
1022
|
const body = { name: a.name, sender_credential_id: a.sender_credential_id, external_list_id: a.external_list_id };
|
|
857
1023
|
for (const k of [
|
|
858
1024
|
'source_list_id', 'segment_id', 'instantly_campaign_name', 'feed_mode', 'drip_strategy',
|
|
1025
|
+
'allow_risky_emails', 'offload_enabled', 'offload_no_reply_days', 'offload_to_instantly',
|
|
1026
|
+
'offload_to_scrapeloop', 'offload_scrapeloop_list_id',
|
|
859
1027
|
'drip_daily_count', 'target_active_count', 'cooldown_days', 'resting_period_days',
|
|
1028
|
+
'recycle_mode', 'prioritize_fresh', 'bounce_instantly_list_id', 'sender_options', 'icon',
|
|
860
1029
|
]) {
|
|
861
1030
|
if (a[k] !== undefined && a[k] !== null) body[k] = a[k];
|
|
862
1031
|
}
|
|
@@ -864,19 +1033,19 @@ const TOOLS = {
|
|
|
864
1033
|
},
|
|
865
1034
|
},
|
|
866
1035
|
update_campaign: {
|
|
867
|
-
def: { description: 'Patch a campaign (feed_mode, target_active_count, cooldown_days, external_list_id, …).', inputSchema: obj({ campaign_id: S, patch: O }, ['campaign_id', 'patch']) },
|
|
1036
|
+
def: { description: 'Patch a campaign (feed_mode, target_active_count, cooldown_days, external_list_id, …). Switching feed_mode on for an already active campaign runs the same check as campaign_readiness_check: any fail answers 422 with the report under detail.readiness.', inputSchema: obj({ campaign_id: S, patch: { ...O, description: "Campaign settings to change.", properties: { sender_options: { ...O, description: "GoHighLevel SMS settings. Only supplied keys change." }, icon: { type: ["string", "null"], description: "One or two emoji, or null to clear." } }, additionalProperties: true } }, ['campaign_id', 'patch']) },
|
|
868
1037
|
run: (a) => api('PATCH', `/campaigns/${enc(a.campaign_id)}`, a.patch || {}),
|
|
869
1038
|
},
|
|
870
1039
|
duplicate_campaign: {
|
|
871
1040
|
def: {
|
|
872
1041
|
description:
|
|
873
|
-
'Duplicate a campaign: clones ALL configuration (list/segment binding, sender credential, Instantly binding, feed/drip/offload settings, field mapping) into a new draft named "Copy of …". Created SAFE regardless of the source
|
|
1042
|
+
'Duplicate a campaign: clones ALL configuration (list/segment binding, sender credential, Instantly binding, feed/drip/offload settings, field mapping) into a new draft named "Copy of …". Created SAFE regardless of the source: feed_mode is forced to off and require_approval to true, so the copy never feeds leads to Instantly until explicitly enabled. Leads/stats/ledger are NOT copied.',
|
|
874
1043
|
inputSchema: obj({ campaign_id: S }, ['campaign_id']),
|
|
875
1044
|
},
|
|
876
1045
|
run: (a) => api('POST', `/campaigns/${enc(a.campaign_id)}/duplicate`),
|
|
877
1046
|
},
|
|
878
1047
|
activate_campaign: {
|
|
879
|
-
def: { description: 'Activate a campaign
|
|
1048
|
+
def: { description: 'Activate a campaign: STARTS the feed/send loop. Run campaign_readiness_check first; any fail makes this answer 422 with the report under detail.readiness. Confirm with the user first.', inputSchema: obj({ campaign_id: S }, ['campaign_id']) },
|
|
880
1049
|
run: (a) => api('POST', `/campaigns/${enc(a.campaign_id)}/activate`),
|
|
881
1050
|
},
|
|
882
1051
|
pause_campaign: {
|
|
@@ -891,6 +1060,53 @@ const TOOLS = {
|
|
|
891
1060
|
def: { description: "A campaign's ledger stats + best-effort live Instantly analytics.", inputSchema: obj({ campaign_id: S }, ['campaign_id']) },
|
|
892
1061
|
run: (a) => api('GET', `/campaigns/${enc(a.campaign_id)}/stats`),
|
|
893
1062
|
},
|
|
1063
|
+
get_campaign_field_mapping: {
|
|
1064
|
+
def: {
|
|
1065
|
+
description:
|
|
1066
|
+
'What a campaign sends to the sender and what each detail is called there. Returns every name the sequence uses, every source that could fill it (with how full each one is over the first 500 rows), and the obvious matches. Read this when a merge variable comes out blank in the emails.',
|
|
1067
|
+
inputSchema: obj({ campaign_id: S }, ['campaign_id']),
|
|
1068
|
+
},
|
|
1069
|
+
run: (a) => api('GET', `/campaigns/${enc(a.campaign_id)}/field-mapping`),
|
|
1070
|
+
},
|
|
1071
|
+
create_sender_fields: {
|
|
1072
|
+
def: {
|
|
1073
|
+
description: 'Create missing GoHighLevel SMS contact fields only after the user agrees. Existing fields are reused. With remember never, save the choice and create nothing. Returns requested names mapped to actual field keys; use those keys when saving the mapping.',
|
|
1074
|
+
inputSchema: obj({ campaign_id: S, names: ARR(S), remember: { type: ['string', 'null'], enum: ['create', 'never', null] } }, ['campaign_id', 'names']),
|
|
1075
|
+
},
|
|
1076
|
+
run: (a) => api('POST', `/campaigns/${enc(a.campaign_id)}/sender-fields`, { names: a.names, remember: a.remember ?? null }),
|
|
1077
|
+
},
|
|
1078
|
+
set_campaign_field_mapping: {
|
|
1079
|
+
def: {
|
|
1080
|
+
description:
|
|
1081
|
+
'Set what a campaign sends. `mapping` is {name the sender receives: where it comes from}, using lead.<column>, send.<name>, col.<table column key> or custom.<key>. `identity` may override first_name, last_name and company only, never email. `only_mapped` true means the mapping is the complete list and nothing else is sent. Saving also records that a person reviewed it, which a campaign needs before its feed can be switched on.',
|
|
1082
|
+
inputSchema: obj(
|
|
1083
|
+
{ campaign_id: S, mapping: { type: 'object' }, identity: { type: 'object' }, only_mapped: B },
|
|
1084
|
+
['campaign_id'],
|
|
1085
|
+
),
|
|
1086
|
+
},
|
|
1087
|
+
run: (a) =>
|
|
1088
|
+
api('PUT', `/campaigns/${enc(a.campaign_id)}/field-mapping`, {
|
|
1089
|
+
mapping: a.mapping || {},
|
|
1090
|
+
identity: a.identity || {},
|
|
1091
|
+
only_mapped: a.only_mapped !== false,
|
|
1092
|
+
}),
|
|
1093
|
+
},
|
|
1094
|
+
campaign_delivery_coverage: {
|
|
1095
|
+
def: {
|
|
1096
|
+
description:
|
|
1097
|
+
'What the live sequence asks for versus what the feed will actually supply. Names any merge variable that nothing fills, which goes out blank in the middle of a sentence, and any that is only filled on some leads.',
|
|
1098
|
+
inputSchema: obj({ campaign_id: S }, ['campaign_id']),
|
|
1099
|
+
},
|
|
1100
|
+
run: (a) => api('GET', `/campaigns/${enc(a.campaign_id)}/delivery-coverage`),
|
|
1101
|
+
},
|
|
1102
|
+
campaign_readiness_check: {
|
|
1103
|
+
def: {
|
|
1104
|
+
description:
|
|
1105
|
+
'Is this campaign ready to start? Run it before activate_campaign. Returns ready, sender_known, sampled and a list of checks, each with a stable key, a level (pass, warn, fail or info), one plain sentence and a fix. Any fail means activate_campaign will refuse with 422 and the same report under detail.readiness. A warn does not block: read it out to the user before they confirm. It judges the leads the feed sends next (not the ones already in the campaign). With feed_mode off nothing is uploaded, so the lead and mailbox checks only warn until the feed is switched on.',
|
|
1106
|
+
inputSchema: obj({ campaign_id: S }, ['campaign_id']),
|
|
1107
|
+
},
|
|
1108
|
+
run: (a) => api('GET', `/campaigns/${enc(a.campaign_id)}/readiness`),
|
|
1109
|
+
},
|
|
894
1110
|
get_sender_config: {
|
|
895
1111
|
def: { description: "Get an Instantly credential's sender config (daily cap, send accounts, …).", inputSchema: obj({ credential_id: S }, ['credential_id']) },
|
|
896
1112
|
run: (a) => api('GET', `/sender_configs/${enc(a.credential_id)}`),
|
|
@@ -943,6 +1159,35 @@ const TOOLS = {
|
|
|
943
1159
|
},
|
|
944
1160
|
run: () => api('GET', '/workbench/tree'),
|
|
945
1161
|
},
|
|
1162
|
+
get_pipeline: {
|
|
1163
|
+
def: {
|
|
1164
|
+
description: 'Read the saved result and progress for a pipeline. Use its operation_id after an interrupted create.',
|
|
1165
|
+
inputSchema: obj({ operation_id: { ...S, format: 'uuid' } }, ['operation_id']),
|
|
1166
|
+
},
|
|
1167
|
+
run: (a) => api('GET', `/pipelines/${enc(a.operation_id)}`),
|
|
1168
|
+
},
|
|
1169
|
+
preview_pipeline: {
|
|
1170
|
+
def: {
|
|
1171
|
+
description: 'Preview an empty pipeline from this workspace Playbook, including Tables, columns, checks, rules and future-row work. Creates nothing. Review this before create_pipeline.',
|
|
1172
|
+
inputSchema: obj({}),
|
|
1173
|
+
},
|
|
1174
|
+
run: (a) => api('POST', '/pipelines/preview', a),
|
|
1175
|
+
},
|
|
1176
|
+
create_pipeline: {
|
|
1177
|
+
def: {
|
|
1178
|
+
description: 'Create an empty pipeline from this workspace Playbook. Raw keeps originals. Rules start on, and chosen columns run on future rows. Creation starts no scrape, sends no messages and activates no campaign. Read get_playbook first. Reuse the same operation_id and choices to resume a partial create. Campaign bindings require campaigns:write and feeds off.',
|
|
1179
|
+
inputSchema: obj({
|
|
1180
|
+
operation_id: { ...S, format: 'uuid' }, name: S, market: S,
|
|
1181
|
+
niche_gcids: ARR(S), mode: { ...S, enum: ['auto', 'simple', 'advanced'] },
|
|
1182
|
+
channels: ARR({ ...S, enum: ['email', 'sms', 'linkedin'] }),
|
|
1183
|
+
checks: { ...O, description: 'Overrides for saved checks, using the Playbook checks keys.' },
|
|
1184
|
+
stage_mode: { ...S, enum: ['move', 'copy'] }, carry_values: B,
|
|
1185
|
+
raw_setup: { ...O, description: 'Optional Raw columns, packs and layout. Same shape as create_table table_setup.' },
|
|
1186
|
+
bindings: { ...O, description: 'Optional email, sms or linkedin campaign IDs. Existing campaign feeds must be off.' },
|
|
1187
|
+
}, ['operation_id', 'name']),
|
|
1188
|
+
},
|
|
1189
|
+
run: (a) => api('POST', '/pipelines', a),
|
|
1190
|
+
},
|
|
946
1191
|
create_workbook: {
|
|
947
1192
|
def: {
|
|
948
1193
|
description:
|
|
@@ -955,7 +1200,7 @@ const TOOLS = {
|
|
|
955
1200
|
def: {
|
|
956
1201
|
description:
|
|
957
1202
|
'Add one tab to a workbook. Pass list_id to attach an existing standalone Table, or new_table_name to create a new empty Table and attach it. Pass exactly one.',
|
|
958
|
-
inputSchema: obj({ workbook_id: S, list_id: S, new_table_name: S }, ['workbook_id']),
|
|
1203
|
+
inputSchema: obj({ workbook_id: S, list_id: S, new_table_name: S, operation_id: { ...S, format: 'uuid' }, table_setup: { ...O, description: 'New Tables only. Same saved-column, pack and layout choices as create_table.' } }, ['workbook_id']),
|
|
959
1204
|
},
|
|
960
1205
|
run: (a) => {
|
|
961
1206
|
if (!!a.list_id === !!a.new_table_name) {
|
|
@@ -968,6 +1213,8 @@ const TOOLS = {
|
|
|
968
1213
|
return api('POST', `/workbooks/${enc(a.workbook_id)}/tables`, {
|
|
969
1214
|
...(a.list_id ? { list_id: a.list_id } : {}),
|
|
970
1215
|
...(a.new_table_name ? { new_table_name: a.new_table_name } : {}),
|
|
1216
|
+
...(a.operation_id ? { operation_id: a.operation_id } : {}),
|
|
1217
|
+
...(a.table_setup !== undefined ? { table_setup: a.table_setup } : {}),
|
|
971
1218
|
});
|
|
972
1219
|
},
|
|
973
1220
|
},
|
|
@@ -1059,16 +1306,33 @@ const TOOLS = {
|
|
|
1059
1306
|
},
|
|
1060
1307
|
run: (a) => api('GET', `/tables${qs({ archived: a.archived })}`),
|
|
1061
1308
|
},
|
|
1309
|
+
get_table_setup_options: {
|
|
1310
|
+
def: {
|
|
1311
|
+
description: 'Read this workspace\'s saved Table columns, layout and available column packs. Read-only; starts no work.',
|
|
1312
|
+
inputSchema: obj({}),
|
|
1313
|
+
},
|
|
1314
|
+
run: () => api('GET', '/playbook/tables/options'),
|
|
1315
|
+
},
|
|
1062
1316
|
create_table: {
|
|
1063
1317
|
def: {
|
|
1064
1318
|
description:
|
|
1065
|
-
'Create a
|
|
1319
|
+
'Create a Scrapeloop Table. Regular Tables start with this workspace\'s saved columns and layout. Use table_setup to change columns, choose packs or rename labels. Creation never runs columns or spends credits. Smart Tables use filter_expression, sort_expression, visible_columns and is_shared, and cannot use table_setup. Retry with the same operation_id; different inputs return 409.',
|
|
1066
1320
|
inputSchema: obj(
|
|
1067
1321
|
{
|
|
1068
1322
|
name: S,
|
|
1069
1323
|
operation_id: { ...S, format: 'uuid', description: 'Optional retry key from a prior ambiguous call.' },
|
|
1070
1324
|
description: S,
|
|
1071
1325
|
kind: { ...S, enum: ['static', 'smart'] },
|
|
1326
|
+
table_setup: {
|
|
1327
|
+
...O,
|
|
1328
|
+
description: 'Regular Tables only. Read get_table_setup_options first. columns replaces saved columns (empty list means none); packs adds named packs; identity_column_state replaces the saved layout. All created columns start with auto-run off.',
|
|
1329
|
+
properties: {
|
|
1330
|
+
columns: ARR({ ...O, properties: { kind: { ...S, enum: ['native', 'preset'] }, key: S, label: S, config: O }, required: ['kind', 'key', 'label'] }),
|
|
1331
|
+
packs: ARR(S),
|
|
1332
|
+
identity_column_state: O,
|
|
1333
|
+
},
|
|
1334
|
+
additionalProperties: false,
|
|
1335
|
+
},
|
|
1072
1336
|
filter_expression: { ...O, description: 'Smart Tables only.' },
|
|
1073
1337
|
sort_expression: { ...O, description: 'Smart Tables only.' },
|
|
1074
1338
|
visible_columns: { ...ARR(S), description: 'Smart Tables only.' },
|
|
@@ -1096,6 +1360,7 @@ const TOOLS = {
|
|
|
1096
1360
|
operation_id: a.operation_id || randomUUID(),
|
|
1097
1361
|
...(a.description !== undefined ? { description: a.description } : {}),
|
|
1098
1362
|
...(a.kind !== undefined ? { kind: a.kind } : {}),
|
|
1363
|
+
...(a.table_setup !== undefined ? { table_setup: a.table_setup } : {}),
|
|
1099
1364
|
...(a.filter_expression !== undefined ? { filter_expression: a.filter_expression } : {}),
|
|
1100
1365
|
...(a.sort_expression !== undefined ? { sort_expression: a.sort_expression } : {}),
|
|
1101
1366
|
...(a.visible_columns !== undefined ? { visible_columns: a.visible_columns } : {}),
|
|
@@ -1207,6 +1472,156 @@ const TOOLS = {
|
|
|
1207
1472
|
});
|
|
1208
1473
|
},
|
|
1209
1474
|
},
|
|
1475
|
+
create_table_preset_column: {
|
|
1476
|
+
def: {
|
|
1477
|
+
description:
|
|
1478
|
+
'Attach one shipped system preset or active workspace preset as a real enrichment column. operation_id is required and must be reused for every retry. auto_run defaults false and queues nothing. If auto_run would backfill existing rows, the API returns an exact 409 estimate first; confirm that estimate with confirm_auto_run and confirmed_estimated_cost_usd before retrying. The Table master auto-run switch and budget caps remain fail-closed. Only preset-documented overrides are accepted.',
|
|
1479
|
+
inputSchema: {
|
|
1480
|
+
...obj(
|
|
1481
|
+
{
|
|
1482
|
+
table_id: S,
|
|
1483
|
+
preset_slug: S,
|
|
1484
|
+
preset_id: { ...S, format: 'uuid' },
|
|
1485
|
+
label: S,
|
|
1486
|
+
auto_run: B,
|
|
1487
|
+
confirm_auto_run: B,
|
|
1488
|
+
confirmed_estimated_cost_usd: { ...N, minimum: 0 },
|
|
1489
|
+
overrides: O,
|
|
1490
|
+
operation_id: {
|
|
1491
|
+
...S,
|
|
1492
|
+
format: 'uuid',
|
|
1493
|
+
description: 'Required stable retry key for this logical preset-column create.',
|
|
1494
|
+
},
|
|
1495
|
+
},
|
|
1496
|
+
['table_id', 'operation_id'],
|
|
1497
|
+
),
|
|
1498
|
+
oneOf: [
|
|
1499
|
+
{ required: ['preset_slug'], not: { required: ['preset_id'] } },
|
|
1500
|
+
{ required: ['preset_id'], not: { required: ['preset_slug'] } },
|
|
1501
|
+
],
|
|
1502
|
+
},
|
|
1503
|
+
},
|
|
1504
|
+
run: async (a) => {
|
|
1505
|
+
if (!a.table_id || !a.operation_id) {
|
|
1506
|
+
return {
|
|
1507
|
+
ok: false,
|
|
1508
|
+
status: 400,
|
|
1509
|
+
error: { detail: 'table_id and a stable operation_id are required.' },
|
|
1510
|
+
};
|
|
1511
|
+
}
|
|
1512
|
+
if (!!a.preset_slug === !!a.preset_id) {
|
|
1513
|
+
return {
|
|
1514
|
+
ok: false,
|
|
1515
|
+
status: 400,
|
|
1516
|
+
error: { detail: 'Pass exactly one of preset_slug or preset_id.' },
|
|
1517
|
+
};
|
|
1518
|
+
}
|
|
1519
|
+
const requestBody = {
|
|
1520
|
+
operation_id: a.operation_id,
|
|
1521
|
+
...(a.preset_slug !== undefined ? { preset_slug: a.preset_slug } : {}),
|
|
1522
|
+
...(a.preset_id !== undefined ? { preset_id: a.preset_id } : {}),
|
|
1523
|
+
...(a.label !== undefined ? { label: a.label } : {}),
|
|
1524
|
+
...(a.auto_run !== undefined ? { auto_run: a.auto_run } : {}),
|
|
1525
|
+
...(a.confirm_auto_run !== undefined ? { confirm_auto_run: a.confirm_auto_run } : {}),
|
|
1526
|
+
...(a.confirmed_estimated_cost_usd !== undefined
|
|
1527
|
+
? { confirmed_estimated_cost_usd: a.confirmed_estimated_cost_usd }
|
|
1528
|
+
: {}),
|
|
1529
|
+
...(a.overrides !== undefined ? { overrides: a.overrides } : {}),
|
|
1530
|
+
};
|
|
1531
|
+
const path = `/tables/${enc(a.table_id)}/preset-columns`;
|
|
1532
|
+
return api('POST', path, requestBody, {
|
|
1533
|
+
reconciliation: {
|
|
1534
|
+
operation_id: a.operation_id,
|
|
1535
|
+
retry_with: {
|
|
1536
|
+
tool: 'create_table_preset_column',
|
|
1537
|
+
arguments: { ...a, operation_id: a.operation_id },
|
|
1538
|
+
},
|
|
1539
|
+
retry_guidance:
|
|
1540
|
+
`Retry create_table_preset_column with the same operation_id ${a.operation_id}. Never substitute a new key for this logical create.`,
|
|
1541
|
+
},
|
|
1542
|
+
});
|
|
1543
|
+
},
|
|
1544
|
+
},
|
|
1545
|
+
get_table_native_tools: {
|
|
1546
|
+
def: {
|
|
1547
|
+
description:
|
|
1548
|
+
'List the first-party Scrapeloop tools a Smart Table column can be built on, with their per-row credit price, documented options, and the legacy preset slugs each one supersedes. Read this when create_table_preset_column refuses a slug as superseded, or before create_table_native_column, so the tool key and options are exact rather than guessed.',
|
|
1549
|
+
inputSchema: obj({}, []),
|
|
1550
|
+
},
|
|
1551
|
+
run: () => api('GET', '/tables/native-tools'),
|
|
1552
|
+
},
|
|
1553
|
+
create_table_native_column: {
|
|
1554
|
+
def: {
|
|
1555
|
+
description:
|
|
1556
|
+
`Attach one native Scrapeloop tool as a real Smart Table column: ${NATIVE_TABLE_TOOLS.join(', ')}. These are the first-party tools that superseded four legacy presets, so they are addressed by tool key here rather than by preset slug through create_table_preset_column (which refuses a superseded slug for a new column). Each bills MANAGED VERIFICATION CREDITS per row, never dollars and never the customer's own vendor keys: verify_email reserves 1 credit per selected row, verify_phone reserves 5, and each email finder reserves 3. Settlement charges only fresh vendor checks and returns the unused reservation; missing inputs, cached rows, and skipped rows are free. Creating a column is always free. auto_run defaults false and queues nothing; when auto_run would backfill existing rows the API returns a 409 estimate whose cost.total_credits is the real charge, and you confirm it with confirm_auto_run plus confirmed_credits (CREDITS, not dollars - estimated_cost_usd is our vendor cost and is far smaller than what the workspace is billed). An auto_run that exceeds the workspace verify-credit balance returns 402 with {credit_kind, required, available} before anything is created. The Table master auto-run switch and workspace budget caps stay fail-closed. options are per tool and only the documented names are accepted - call get_table_native_tools for the current list. Owner email finder needs an owner first name plus the company website; point options.owner_name_column at an Owner name finder column to source the name, which auto_run requires. Business email finder is a SEPARATE column - create it with its own call and set options.owner_email_column to run it behind an owner finder. The Catch-all verifier is NOT creatable here: it needs a Verify email column that has already RUN and returned Catch All or Unknown, so use verify_catchall per address, or add the column from the Table's add-column menu.`,
|
|
1557
|
+
inputSchema: obj(
|
|
1558
|
+
{
|
|
1559
|
+
table_id: S,
|
|
1560
|
+
tool: { ...S, enum: NATIVE_TABLE_TOOLS },
|
|
1561
|
+
label: { ...S, description: 'Column label. Defaults to the tool name.' },
|
|
1562
|
+
auto_run: B,
|
|
1563
|
+
confirm_auto_run: B,
|
|
1564
|
+
confirmed_credits: {
|
|
1565
|
+
...N,
|
|
1566
|
+
minimum: 0,
|
|
1567
|
+
description:
|
|
1568
|
+
'The 409 estimate\'s cost.total_credits, echoed back exactly. Verification credits, not dollars.',
|
|
1569
|
+
},
|
|
1570
|
+
options: {
|
|
1571
|
+
...O,
|
|
1572
|
+
description:
|
|
1573
|
+
'Tool-specific options. verify_phone: phone_col, default_region. email_finder: owner_name_column. business_email_finder: owner_email_column. An undocumented name is rejected, never ignored.',
|
|
1574
|
+
},
|
|
1575
|
+
operation_id: {
|
|
1576
|
+
...S,
|
|
1577
|
+
format: 'uuid',
|
|
1578
|
+
description: 'Required stable retry key for this logical native-column create.',
|
|
1579
|
+
},
|
|
1580
|
+
},
|
|
1581
|
+
['table_id', 'tool', 'operation_id'],
|
|
1582
|
+
),
|
|
1583
|
+
},
|
|
1584
|
+
run: async (a) => {
|
|
1585
|
+
if (!a.table_id || !a.tool || !a.operation_id) {
|
|
1586
|
+
return {
|
|
1587
|
+
ok: false,
|
|
1588
|
+
status: 400,
|
|
1589
|
+
error: { detail: 'table_id, tool, and a stable operation_id are required.' },
|
|
1590
|
+
};
|
|
1591
|
+
}
|
|
1592
|
+
if (!NATIVE_TABLE_TOOLS.includes(a.tool)) {
|
|
1593
|
+
return {
|
|
1594
|
+
ok: false,
|
|
1595
|
+
status: 400,
|
|
1596
|
+
error: {
|
|
1597
|
+
detail: `Unknown native tool '${a.tool}'. Expected one of ${NATIVE_TABLE_TOOLS.join(', ')}. Call get_table_native_tools for the live catalog.`,
|
|
1598
|
+
},
|
|
1599
|
+
};
|
|
1600
|
+
}
|
|
1601
|
+
const requestBody = {
|
|
1602
|
+
tool: a.tool,
|
|
1603
|
+
operation_id: a.operation_id,
|
|
1604
|
+
...(a.label !== undefined ? { label: a.label } : {}),
|
|
1605
|
+
...(a.auto_run !== undefined ? { auto_run: a.auto_run } : {}),
|
|
1606
|
+
...(a.confirm_auto_run !== undefined ? { confirm_auto_run: a.confirm_auto_run } : {}),
|
|
1607
|
+
...(a.confirmed_credits !== undefined
|
|
1608
|
+
? { confirmed_credits: a.confirmed_credits }
|
|
1609
|
+
: {}),
|
|
1610
|
+
...(a.options !== undefined ? { options: a.options } : {}),
|
|
1611
|
+
};
|
|
1612
|
+
return api('POST', `/tables/${enc(a.table_id)}/native-columns`, requestBody, {
|
|
1613
|
+
reconciliation: {
|
|
1614
|
+
operation_id: a.operation_id,
|
|
1615
|
+
retry_with: {
|
|
1616
|
+
tool: 'create_table_native_column',
|
|
1617
|
+
arguments: { ...a, operation_id: a.operation_id },
|
|
1618
|
+
},
|
|
1619
|
+
retry_guidance:
|
|
1620
|
+
`Retry create_table_native_column with the same operation_id ${a.operation_id}. Never substitute a new key for this logical create.`,
|
|
1621
|
+
},
|
|
1622
|
+
});
|
|
1623
|
+
},
|
|
1624
|
+
},
|
|
1210
1625
|
set_table_cells: {
|
|
1211
1626
|
def: {
|
|
1212
1627
|
description:
|
|
@@ -1311,6 +1726,55 @@ const TOOLS = {
|
|
|
1311
1726
|
});
|
|
1312
1727
|
},
|
|
1313
1728
|
},
|
|
1729
|
+
list_apify_actors: {
|
|
1730
|
+
def: {
|
|
1731
|
+
description:
|
|
1732
|
+
'List the APPROVED Apify actors you may run/import from. Arbitrary Apify actors are NOT allowed: only vetted actors (website content, contact info with add-ons off, google search with add-ons off) can be used, and LinkedIn / Facebook-group / social-graph actors are always rejected. Returns [{actor_id, add_ons_allowed}]. Use this before preview_apify_import to pick a permitted actor_id.',
|
|
1733
|
+
inputSchema: obj({}),
|
|
1734
|
+
},
|
|
1735
|
+
run: () => api('GET', '/apify/actors'),
|
|
1736
|
+
},
|
|
1737
|
+
preview_apify_import: {
|
|
1738
|
+
def: {
|
|
1739
|
+
description:
|
|
1740
|
+
"Read-only preview of an approved Apify actor's already-produced dataset before importing it. mapping is {lead_field: dotted.source.path} where lead_field is one of name, email, phone, domain, organization_name, city, state, country, linkedin_url, title. Normalizes the whole dataset (rows with no email/phone/domain/linkedin/name are skipped), returns the first 25 rows plus record_count (the true total) and a preview_hash. Pass that preview_hash to import_apify_dataset: the import re-verifies it and refuses (409) if the dataset changed. Does NOT run the actor or spend: it reads an existing dataset_id. Returns {dataset_id, actor_id, record_count, preview, preview_truncated, preview_hash}.",
|
|
1741
|
+
inputSchema: obj(
|
|
1742
|
+
{
|
|
1743
|
+
dataset_id: { ...S, description: 'Apify dataset id from a completed actor run.' },
|
|
1744
|
+
actor_id: { ...S, description: 'Approved actor id (owner/name), e.g. apify/google-search-scraper.' },
|
|
1745
|
+
mapping: { ...O, description: '{lead_field: "dotted.source.path"} onto the lead spine.' },
|
|
1746
|
+
},
|
|
1747
|
+
['dataset_id', 'actor_id', 'mapping'],
|
|
1748
|
+
),
|
|
1749
|
+
},
|
|
1750
|
+
run: (a) => api('POST', '/apify/preview', { dataset_id: a.dataset_id, actor_id: a.actor_id, mapping: a.mapping }),
|
|
1751
|
+
},
|
|
1752
|
+
import_apify_dataset: {
|
|
1753
|
+
def: {
|
|
1754
|
+
description:
|
|
1755
|
+
'Import a previewed Apify dataset (≤1000 rows) into a private Scrapeloop Table through the bring-your-own-leads funnel (dedupe + identity matching + Table attach reused; no lead credits). Call preview_apify_import first and pass its preview_hash: the import re-reads the dataset and returns 409 (preview_drift) if it changed since preview. Each row gets a stable external_id (apify:{dataset}:{row}) so re-importing updates rather than duplicates; company lands in business_name and the contact name/title ride through as custom fields. Target list_id, or omit for a new "Apify import" Table. Returns the import funnel result {inserted, updated, deduped, invalid, ...}.',
|
|
1756
|
+
inputSchema: obj(
|
|
1757
|
+
{
|
|
1758
|
+
dataset_id: S,
|
|
1759
|
+
actor_id: S,
|
|
1760
|
+
mapping: O,
|
|
1761
|
+
preview_hash: { ...S, description: 'The preview_hash returned by preview_apify_import.' },
|
|
1762
|
+
list_id: S,
|
|
1763
|
+
list_name: S,
|
|
1764
|
+
},
|
|
1765
|
+
['dataset_id', 'actor_id', 'mapping', 'preview_hash'],
|
|
1766
|
+
),
|
|
1767
|
+
},
|
|
1768
|
+
run: (a) =>
|
|
1769
|
+
api('POST', '/apify/import', {
|
|
1770
|
+
dataset_id: a.dataset_id,
|
|
1771
|
+
actor_id: a.actor_id,
|
|
1772
|
+
mapping: a.mapping,
|
|
1773
|
+
preview_hash: a.preview_hash,
|
|
1774
|
+
...(a.list_id ? { list_id: a.list_id } : {}),
|
|
1775
|
+
...(a.list_name ? { list_name: a.list_name } : {}),
|
|
1776
|
+
}),
|
|
1777
|
+
},
|
|
1314
1778
|
capture_community_intent: {
|
|
1315
1779
|
def: {
|
|
1316
1780
|
description:
|
|
@@ -1466,7 +1930,7 @@ const TOOLS = {
|
|
|
1466
1930
|
preview_import: {
|
|
1467
1931
|
def: {
|
|
1468
1932
|
description:
|
|
1469
|
-
'Dry-run for import_leads: report exactly what an import WOULD do
|
|
1933
|
+
'Dry-run for import_leads: report exactly what an import WOULD do ({would_insert, would_update, deduped, invalid, invalid_reasons, list_exists}) WITHOUT writing anything (no leads, no list, no tags created). Same inputs as import_leads. Call this first for a bring-your-own batch, show the user the plan (e.g. "adds 340 new, updates 12, skips 3 bad rows"), then confirm before import_leads.',
|
|
1470
1934
|
inputSchema: obj(
|
|
1471
1935
|
{
|
|
1472
1936
|
list_id: S,
|
|
@@ -1720,6 +2184,60 @@ const TOOLS = {
|
|
|
1720
2184
|
error: { detail: 'campaign_id and rule_id are required.' },
|
|
1721
2185
|
},
|
|
1722
2186
|
},
|
|
2187
|
+
test_flow_rule: {
|
|
2188
|
+
def: {
|
|
2189
|
+
description:
|
|
2190
|
+
'Test one saved flow rule against one lead. Dry-run is the safe default and reports whether the row matches plus the action that would run. Set execute=true only after confirmation because the action may spend credits or change data.',
|
|
2191
|
+
inputSchema: obj(
|
|
2192
|
+
{
|
|
2193
|
+
list_id: { ...S, description: 'Source Table id. Provide this or campaign_id, never both.' },
|
|
2194
|
+
campaign_id: { ...S, description: 'Source campaign id. Provide this or list_id, never both.' },
|
|
2195
|
+
rule_id: S,
|
|
2196
|
+
lead_id: { ...S, description: 'Optional lead id. The newest source row is used when omitted.' },
|
|
2197
|
+
execute: { ...B, description: 'False for dry-run. True executes the action once.' },
|
|
2198
|
+
},
|
|
2199
|
+
['rule_id'],
|
|
2200
|
+
),
|
|
2201
|
+
},
|
|
2202
|
+
run: (a) => {
|
|
2203
|
+
if (!!a.list_id === !!a.campaign_id) {
|
|
2204
|
+
return {
|
|
2205
|
+
ok: false,
|
|
2206
|
+
status: 400,
|
|
2207
|
+
error: { detail: 'Provide exactly one of list_id or campaign_id.' },
|
|
2208
|
+
};
|
|
2209
|
+
}
|
|
2210
|
+
const source = a.list_id
|
|
2211
|
+
? `/lists/${enc(a.list_id)}`
|
|
2212
|
+
: `/campaigns/${enc(a.campaign_id)}`;
|
|
2213
|
+
return api('POST', `${source}/flow-rules/${enc(a.rule_id)}/test`, {
|
|
2214
|
+
...(a.lead_id ? { lead_id: a.lead_id } : {}),
|
|
2215
|
+
execute: !!a.execute,
|
|
2216
|
+
});
|
|
2217
|
+
},
|
|
2218
|
+
},
|
|
2219
|
+
get_flow_rule_stats: {
|
|
2220
|
+
def: {
|
|
2221
|
+
description:
|
|
2222
|
+
'Get the frozen 24-hour and 7-day fire, failure, last-run, and paused-state counters for every saved rule on one Table or campaign.',
|
|
2223
|
+
inputSchema: obj({
|
|
2224
|
+
list_id: { ...S, description: 'Source Table id. Provide this or campaign_id, never both.' },
|
|
2225
|
+
campaign_id: { ...S, description: 'Source campaign id. Provide this or list_id, never both.' },
|
|
2226
|
+
}),
|
|
2227
|
+
},
|
|
2228
|
+
run: (a) => {
|
|
2229
|
+
if (!!a.list_id === !!a.campaign_id) {
|
|
2230
|
+
return {
|
|
2231
|
+
ok: false,
|
|
2232
|
+
status: 400,
|
|
2233
|
+
error: { detail: 'Provide exactly one of list_id or campaign_id.' },
|
|
2234
|
+
};
|
|
2235
|
+
}
|
|
2236
|
+
return a.list_id
|
|
2237
|
+
? api('GET', `/lists/${enc(a.list_id)}/flow-rules/stats`)
|
|
2238
|
+
: api('GET', `/campaigns/${enc(a.campaign_id)}/flow-rules/stats`);
|
|
2239
|
+
},
|
|
2240
|
+
},
|
|
1723
2241
|
preview_send_to_table: {
|
|
1724
2242
|
def: {
|
|
1725
2243
|
description:
|
|
@@ -1834,7 +2352,7 @@ const TOOLS = {
|
|
|
1834
2352
|
list_table_rows: {
|
|
1835
2353
|
def: {
|
|
1836
2354
|
description:
|
|
1837
|
-
"Read a table's rows the way the grid sees them
|
|
2355
|
+
"Read a table's rows the way the grid sees them: lead identity fields (name/email/phone/domain/city/state/country/status) plus every column's cell (status + value). Server-side: sort is a JSON-array string like [{\"key\":\"lead.name\",\"dir\":\"asc\"},{\"key\":\"col.company_size\",\"dir\":\"desc\"}] (keys are lead.<field> or col.<column_key>, ≤3 levels); q is a full-text search across identity fields + visible column cells; filter is a FilterExpr object (same grammar the Tables filter UI uses), e.g. {\"and\":[{\"field\":\"lead.email_status\",\"op\":\"eq\",\"value\":\"valid\"},{\"field\":\"col.company_size\",\"op\":\"gte\",\"value\":50}]}, and adds total_filtered alongside total; pass the previous response's next_cursor back as cursor to page (null next_cursor = last page). Distinct from get_list_rows (the plain import read-back). limit ≤ 500.",
|
|
1838
2356
|
inputSchema: obj(
|
|
1839
2357
|
{
|
|
1840
2358
|
list_id: S,
|
|
@@ -1842,7 +2360,7 @@ const TOOLS = {
|
|
|
1842
2360
|
limit: N,
|
|
1843
2361
|
sort: S,
|
|
1844
2362
|
q: S,
|
|
1845
|
-
filter: { ...O, description: 'FilterExpr JSON (object)
|
|
2363
|
+
filter: { ...O, description: 'FilterExpr JSON (object): server-side row filter; adds total_filtered.' },
|
|
1846
2364
|
view_id: { ...S, description: 'A saved custom view uuid or a system key (errored_rows/fully_enriched/data_only) from list_table_views. An ad-hoc filter/sort replaces the view\'s.' },
|
|
1847
2365
|
},
|
|
1848
2366
|
['list_id'],
|
|
@@ -1866,7 +2384,7 @@ const TOOLS = {
|
|
|
1866
2384
|
list_table_views: {
|
|
1867
2385
|
def: {
|
|
1868
2386
|
description:
|
|
1869
|
-
"List a table's views
|
|
2387
|
+
"List a table's views: three always-current SYSTEM views (errored_rows / fully_enriched / data_only) plus the user's saved custom views. Pass a returned id (or a system key) as view_id to list_table_rows to read that view (a saved filter + sorts + column overlay).",
|
|
1870
2388
|
inputSchema: obj({ list_id: S }, ['list_id']),
|
|
1871
2389
|
},
|
|
1872
2390
|
run: (a) =>
|
|
@@ -1922,7 +2440,7 @@ const TOOLS = {
|
|
|
1922
2440
|
update_table_view: {
|
|
1923
2441
|
def: {
|
|
1924
2442
|
description:
|
|
1925
|
-
'Update a saved table view
|
|
2443
|
+
'Update a saved table view: any field (name/description/filter_expression/sorts/column_state/row_window); a field set to null clears it. System views (errored_rows/fully_enriched/data_only) are read-only.',
|
|
1926
2444
|
inputSchema: obj(
|
|
1927
2445
|
{ list_id: S, view_id: S, name: S, description: S, filter_expression: O, sorts: ARR(O), column_state: O, row_window: O },
|
|
1928
2446
|
['list_id', 'view_id'],
|
|
@@ -1939,7 +2457,7 @@ const TOOLS = {
|
|
|
1939
2457
|
delete_table_view: {
|
|
1940
2458
|
def: {
|
|
1941
2459
|
description:
|
|
1942
|
-
'Delete a saved table view
|
|
2460
|
+
'Delete a saved table view. Rows in the table are unaffected. Confirm with the user first. System views cannot be deleted.',
|
|
1943
2461
|
inputSchema: obj({ list_id: S, view_id: S }, ['list_id', 'view_id']),
|
|
1944
2462
|
},
|
|
1945
2463
|
run: (a) =>
|
|
@@ -1950,7 +2468,7 @@ const TOOLS = {
|
|
|
1950
2468
|
estimate_table_column: {
|
|
1951
2469
|
def: {
|
|
1952
2470
|
description:
|
|
1953
|
-
"FREE cost preview for running one enrichment column, optionally scoped to a view or selection (view_id + selection{mode:'query', filter:<FilterExpr>, exclude_ids} or {mode:'explicit', ids}). cell_filter picks which cells: 'empty_or_stale' (default in the UI
|
|
2471
|
+
"FREE cost preview for running one enrichment column, optionally scoped to a view or selection (view_id + selection{mode:'query', filter:<FilterExpr>, exclude_ids} or {mode:'explicit', ids}). cell_filter picks which cells: 'empty_or_stale' (default in the UI; skips results still fresh under the current column config), 'failed', or 'all'; n_rows/start_row window the run (1-based). Returns lead_count + estimated_cost_usd + scope_resolved. ALWAYS show this to the user before run_table_column; it spends nothing.",
|
|
1954
2472
|
inputSchema: obj(
|
|
1955
2473
|
{ list_id: S, column_id: S, view_id: S, selection: O, cell_filter: S, n_rows: N, start_row: N, only_failed: B },
|
|
1956
2474
|
['list_id', 'column_id'],
|
|
@@ -1964,7 +2482,7 @@ const TOOLS = {
|
|
|
1964
2482
|
run_table_column: {
|
|
1965
2483
|
def: {
|
|
1966
2484
|
description:
|
|
1967
|
-
"SPENDS vendor credits
|
|
2485
|
+
"SPENDS vendor credits: enrich one column across the scoped rows (view_id + selection + cell_filter + n_rows/start_row, same shape as estimate_table_column). cell_filter='empty_or_stale' re-runs only rows that are empty or out of date; fresh results are never re-billed. Run estimate_table_column first and confirm with the user.",
|
|
1968
2486
|
inputSchema: obj(
|
|
1969
2487
|
{ list_id: S, column_id: S, view_id: S, selection: O, cell_filter: S, n_rows: N, start_row: N, only_failed: B },
|
|
1970
2488
|
['list_id', 'column_id'],
|
|
@@ -1978,7 +2496,7 @@ const TOOLS = {
|
|
|
1978
2496
|
run_scope_summary: {
|
|
1979
2497
|
def: {
|
|
1980
2498
|
description:
|
|
1981
|
-
'FREE counts + estimates for a column run scope
|
|
2499
|
+
'FREE counts + estimates for a column run scope: total, empty_or_stale, stale, failed rows + est_all/est_empty_or_stale/est_first_10 USD + budget. Use to decide what to run (which cell_filter) before estimate/run_table_column. Read-only.',
|
|
1982
2500
|
inputSchema: obj({ list_id: S, column_id: S, view_id: S, selection: O }, ['list_id', 'column_id']),
|
|
1983
2501
|
},
|
|
1984
2502
|
run: (a) =>
|
|
@@ -1992,7 +2510,7 @@ const TOOLS = {
|
|
|
1992
2510
|
estimate_table_run_all: {
|
|
1993
2511
|
def: {
|
|
1994
2512
|
description:
|
|
1995
|
-
'FREE per-column preview + total for running EVERY enrichment column in dependency order, optionally scoped (view_id + selection). Show before run_table_all
|
|
2513
|
+
'FREE per-column preview + total for running EVERY enrichment column in dependency order, optionally scoped (view_id + selection). Show before run_table_all; spends nothing.',
|
|
1996
2514
|
inputSchema: obj(
|
|
1997
2515
|
{ list_id: S, column_ids: ARR(S), view_id: S, selection: O, only_empty: B },
|
|
1998
2516
|
['list_id'],
|
|
@@ -2011,7 +2529,7 @@ const TOOLS = {
|
|
|
2011
2529
|
run_table_all: {
|
|
2012
2530
|
def: {
|
|
2013
2531
|
description:
|
|
2014
|
-
'SPENDS vendor credits
|
|
2532
|
+
'SPENDS vendor credits: run every enrichment column in dependency order over the scoped rows (only empty cells run by default; cached results are free). Run estimate_table_run_all first and confirm with the user.',
|
|
2015
2533
|
inputSchema: obj(
|
|
2016
2534
|
{ list_id: S, column_ids: ARR(S), view_id: S, selection: O, only_empty: B },
|
|
2017
2535
|
['list_id'],
|
|
@@ -2041,7 +2559,7 @@ const TOOLS = {
|
|
|
2041
2559
|
remove_table_rows: {
|
|
2042
2560
|
def: {
|
|
2043
2561
|
description:
|
|
2044
|
-
'Remove rows from a Scrapeloop table (list) by lead id
|
|
2562
|
+
'Remove rows from a Scrapeloop table (list) by lead id. The leads stay in the workspace pool; only the table membership and that table\'s enrichment cells are removed. Get ids from list_table_rows. To delete the whole table use delete_list.',
|
|
2045
2563
|
inputSchema: obj({ list_id: S, lead_ids: ARR(S) }, ['list_id', 'lead_ids']),
|
|
2046
2564
|
},
|
|
2047
2565
|
run: (a) =>
|
|
@@ -2130,7 +2648,7 @@ const TOOLS = {
|
|
|
2130
2648
|
delete_list: {
|
|
2131
2649
|
def: {
|
|
2132
2650
|
description:
|
|
2133
|
-
'Delete a list (the list + its membership; the underlying leads stay in the workspace). Works for any Scrapeloop list. Irreversible
|
|
2651
|
+
'Delete a list (the list + its membership; the underlying leads stay in the workspace). Works for any Scrapeloop list. Irreversible: confirm with the user first.',
|
|
2134
2652
|
inputSchema: obj({ list_id: S }, ['list_id']),
|
|
2135
2653
|
},
|
|
2136
2654
|
run: (a) =>
|
|
@@ -2141,7 +2659,7 @@ const TOOLS = {
|
|
|
2141
2659
|
delete_campaign: {
|
|
2142
2660
|
def: {
|
|
2143
2661
|
description:
|
|
2144
|
-
'Delete a campaign (the Scrapeloop campaign + its ledger; the Instantly campaign itself is not deleted). Irreversible
|
|
2662
|
+
'Delete a campaign (the Scrapeloop campaign + its ledger; the Instantly campaign itself is not deleted). Irreversible. If you only want to stop sending, pause_campaign instead. Confirm with the user first.',
|
|
2145
2663
|
inputSchema: obj({ campaign_id: S }, ['campaign_id']),
|
|
2146
2664
|
},
|
|
2147
2665
|
run: (a) =>
|
|
@@ -2154,7 +2672,7 @@ const TOOLS = {
|
|
|
2154
2672
|
list_replies: {
|
|
2155
2673
|
def: {
|
|
2156
2674
|
description:
|
|
2157
|
-
'List inbound campaign replies, newest first. Filters: sentiment (interested|not_interested|unsubscribed), handled (true = already actioned, false = needs attention), lead_id, campaign_id, status (pending|auto_classified|human_required|human_reviewed
|
|
2675
|
+
'List inbound campaign replies, newest first. Filters: sentiment (interested|not_interested|unsubscribed), handled (true = already actioned, false = needs attention), lead_id, campaign_id, status (pending|auto_classified|human_required|human_reviewed; human_required is the review queue).',
|
|
2158
2676
|
inputSchema: obj({ sentiment: S, handled: B, lead_id: S, campaign_id: S, status: S, limit: N, offset: N }),
|
|
2159
2677
|
},
|
|
2160
2678
|
run: (a) =>
|
|
@@ -2174,7 +2692,7 @@ const TOOLS = {
|
|
|
2174
2692
|
update_reply: {
|
|
2175
2693
|
def: {
|
|
2176
2694
|
description:
|
|
2177
|
-
"Update a reply: mark it handled/unhandled and/or set its sentiment. Setting sentiment is a human classification
|
|
2695
|
+
"Update a reply: mark it handled/unhandled and/or set its sentiment. Setting sentiment is a human classification: it propagates to the lead (reply_sentiment, replied_at, status transition, the unsubscribe guard for 'unsubscribed', and the workspace's auto-suppress feed).",
|
|
2178
2696
|
inputSchema: obj(
|
|
2179
2697
|
{
|
|
2180
2698
|
reply_id: S,
|
|
@@ -2285,7 +2803,7 @@ const TOOLS = {
|
|
|
2285
2803
|
{
|
|
2286
2804
|
slug: S,
|
|
2287
2805
|
status: { ...S, enum: ['planned', 'in_progress', 'deployed'] },
|
|
2288
|
-
pr_url: { ...S, description: 'PR link
|
|
2806
|
+
pr_url: { ...S, description: 'PR link: set it when marking deployed' },
|
|
2289
2807
|
},
|
|
2290
2808
|
['slug', 'status'],
|
|
2291
2809
|
),
|
|
@@ -2335,7 +2853,7 @@ const TOOLS = {
|
|
|
2335
2853
|
roadmap_update_plan: {
|
|
2336
2854
|
def: {
|
|
2337
2855
|
description:
|
|
2338
|
-
"Update a roadmap card's fields
|
|
2856
|
+
"Update a roadmap card's fields, most importantly plan_md (the execution plan). Also: title, summary, kind, track, area, effort, depends_on, files_hint, position, pr_url.",
|
|
2339
2857
|
inputSchema: obj(
|
|
2340
2858
|
{
|
|
2341
2859
|
slug: S,
|
|
@@ -2361,25 +2879,70 @@ const TOOLS = {
|
|
|
2361
2879
|
);
|
|
2362
2880
|
return Object.keys(body).length
|
|
2363
2881
|
? api('PATCH', `/roadmap/${enc(slug)}`, body)
|
|
2364
|
-
: { ok: false, status: 400, error: { detail: 'nothing to update
|
|
2882
|
+
: { ok: false, status: 400, error: { detail: 'nothing to update: pass at least one field' } };
|
|
2365
2883
|
},
|
|
2366
2884
|
},
|
|
2367
2885
|
};
|
|
2368
2886
|
|
|
2887
|
+
TOOLS.preview_pipeline.def.inputSchema = TOOLS.create_pipeline.def.inputSchema;
|
|
2888
|
+
|
|
2889
|
+
const extractHttpMethods = (run) => [
|
|
2890
|
+
...run.toString().matchAll(/api\(\s*['"](GET|POST|PUT|PATCH|DELETE)['"]/g),
|
|
2891
|
+
].map((match) => match[1]);
|
|
2892
|
+
|
|
2893
|
+
export const TOOL_HTTP_OPERATIONS = Object.freeze(
|
|
2894
|
+
Object.fromEntries(
|
|
2895
|
+
Object.entries(TOOLS).map(([name, tool]) => [
|
|
2896
|
+
name,
|
|
2897
|
+
Object.freeze([...new Set(extractHttpMethods(tool.run))]),
|
|
2898
|
+
]),
|
|
2899
|
+
),
|
|
2900
|
+
);
|
|
2901
|
+
|
|
2902
|
+
for (const [name, tool] of Object.entries(TOOLS)) {
|
|
2903
|
+
const originalRun = tool.run;
|
|
2904
|
+
const policy = MUTATION_POLICIES[name];
|
|
2905
|
+
const methods = TOOL_HTTP_OPERATIONS[name];
|
|
2906
|
+
const retryCopy = policy
|
|
2907
|
+
? RETRY_CONTRACT_COPY[policy.retry]
|
|
2908
|
+
: methods.every((method) => method === 'GET')
|
|
2909
|
+
? RETRY_CONTRACT_COPY.bounded
|
|
2910
|
+
: 'Policy missing. This operation is blocked until its retry contract is reviewed.';
|
|
2911
|
+
tool.def.description = `${tool.def.description} Retry contract: ${retryCopy}`;
|
|
2912
|
+
if (policy?.retry === 'idempotency_key') {
|
|
2913
|
+
const schema = tool.def.inputSchema;
|
|
2914
|
+
tool.def.inputSchema = {
|
|
2915
|
+
...schema,
|
|
2916
|
+
properties: {
|
|
2917
|
+
...schema.properties,
|
|
2918
|
+
idempotency_key: {
|
|
2919
|
+
type: 'string',
|
|
2920
|
+
description: 'Stable UUID for this exact mutation. Reuse it after a timeout or lost response.',
|
|
2921
|
+
},
|
|
2922
|
+
},
|
|
2923
|
+
required: [...new Set([...(schema.required || []), 'idempotency_key'])],
|
|
2924
|
+
};
|
|
2925
|
+
}
|
|
2926
|
+
tool.run = (args = {}) => toolCallContext.run(
|
|
2927
|
+
{ toolName: name, policy, args },
|
|
2928
|
+
() => originalRun(args),
|
|
2929
|
+
);
|
|
2930
|
+
}
|
|
2931
|
+
|
|
2369
2932
|
async function serve() {
|
|
2370
|
-
// Start cleanly even without a key
|
|
2933
|
+
// Start cleanly even without a key: never crash or hang. The first tools/call
|
|
2371
2934
|
// returns a clear, structured auth error (api() handles the missing key), and we
|
|
2372
2935
|
// log exactly one warning to stderr here. The key itself is never logged.
|
|
2373
2936
|
if (!API_KEY) {
|
|
2374
2937
|
warnStartupOnce(
|
|
2375
2938
|
'missing_api_key',
|
|
2376
|
-
'scrapeloop-mcp: SCRAPELOOP_API_KEY is not set
|
|
2939
|
+
'scrapeloop-mcp: SCRAPELOOP_API_KEY is not set, so tools will return an auth error until it is. ' +
|
|
2377
2940
|
'Generate a key in Scrapeloop → Settings → API access.'
|
|
2378
2941
|
);
|
|
2379
2942
|
}
|
|
2380
2943
|
|
|
2381
2944
|
const server = new Server(
|
|
2382
|
-
{ name: 'scrapeloop-mcp', version: '0.
|
|
2945
|
+
{ name: 'scrapeloop-mcp', version: '0.8.0' },
|
|
2383
2946
|
{ capabilities: { tools: {} } }
|
|
2384
2947
|
);
|
|
2385
2948
|
|
|
@@ -2420,9 +2983,10 @@ async function serve() {
|
|
|
2420
2983
|
// Exported for the manifest/tools drift test. Tests set SCRAPELOOP_MCP_NO_SERVE
|
|
2421
2984
|
// before importing so the stdio server (and the API-key requirement) stay dormant;
|
|
2422
2985
|
// the CLI leaves it unset and serves. (An entrypoint-URL check is unreliable when
|
|
2423
|
-
// the install path contains spaces
|
|
2424
|
-
// process.argv[1] does not
|
|
2986
|
+
// the install path contains spaces: import.meta.url percent-encodes them but
|
|
2987
|
+
// process.argv[1] does not, so an explicit opt-out flag is used instead.)
|
|
2425
2988
|
export {
|
|
2989
|
+
MUTATION_POLICIES,
|
|
2426
2990
|
TOOLS,
|
|
2427
2991
|
classifyApiKeyPreflight,
|
|
2428
2992
|
formatApiKeyPreflightWarning,
|