@extraktor/cli 0.0.0-stage → 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/run.js ADDED
@@ -0,0 +1,504 @@
1
+ import { writeFile } from "node:fs/promises";
2
+ import path from "node:path";
3
+ import { parseArgs } from "node:util";
4
+ import { detectAgent } from "./agent.js";
5
+ import { deleteSavedApiKey, resolveApiKey, saveApiKey } from "./auth.js";
6
+ import { callToolCached } from "./cache.js";
7
+ import { CliError, EXIT } from "./errors.js";
8
+ import { EXTRACT_HELP, LOGIN_HELP, LOGOUT_HELP, MAIN_HELP, MCP_HELP, SEARCH_HELP, } from "./help.js";
9
+ import { closest, optionError, usageError } from "./usage.js";
10
+ import { VERSION } from "./version.js";
11
+ const DEFAULT_URL = "https://extraktor.app";
12
+ const UNSAFE_FILE_CHARACTERS = /[^a-z0-9.-]/giu;
13
+ const LOOKS_LIKE_URL = /^(?:https?:\/\/)?[\w-]+(?:\.[\w-]+)+(?:[/?#]\S*)?$/iu;
14
+ const WHITESPACE = /\s/u;
15
+ /** At most 5 pages in one command, so that the output stays easy to read. */
16
+ const MAX_URLS = 5;
17
+ /**
18
+ * The characters that one command prints. Agents see about 30,000 characters
19
+ * of a command's output (Claude Code cuts the rest), so each output fits
20
+ * completely. Several pages share it.
21
+ */
22
+ const MAX_OUTPUT_CHARACTERS = 24_000;
23
+ const SOME_PAGES_FAILED = "SOME_PAGES_FAILED";
24
+ const SOME_AGENTS_FAILED = "SOME_AGENTS_FAILED";
25
+ const EXTRACT_FLAGS = {
26
+ find: { type: "string", multiple: true },
27
+ offset: { type: "string" },
28
+ excerpts: { type: "boolean" },
29
+ summary: { type: "boolean" },
30
+ focus: { type: "string" },
31
+ screenshot: { type: "boolean" },
32
+ "screenshot-file": { type: "string" },
33
+ save: { type: "string" },
34
+ contacts: { type: "boolean" },
35
+ messaging: { type: "boolean" },
36
+ seo: { type: "boolean" },
37
+ keyword: { type: "string" },
38
+ deep: { type: "boolean" },
39
+ design: { type: "boolean" },
40
+ "design-file": { type: "string" },
41
+ vision: { type: "boolean" },
42
+ "agent-access": { type: "boolean" },
43
+ json: { type: "boolean" },
44
+ "no-cache": { type: "boolean" },
45
+ help: { type: "boolean", short: "h" },
46
+ };
47
+ const SEARCH_FLAGS = {
48
+ serp: { type: "boolean" },
49
+ json: { type: "boolean" },
50
+ "no-cache": { type: "boolean" },
51
+ help: { type: "boolean", short: "h" },
52
+ };
53
+ const LOGIN_FLAGS = {
54
+ "no-browser": { type: "boolean" },
55
+ "with-key": { type: "boolean" },
56
+ help: { type: "boolean", short: "h" },
57
+ };
58
+ const MCP_FLAGS = {
59
+ agent: { type: "string", multiple: true },
60
+ json: { type: "boolean" },
61
+ help: { type: "boolean", short: "h" },
62
+ };
63
+ const HELP_ONLY = {
64
+ help: { type: "boolean", short: "h" },
65
+ };
66
+ /** parseArgs with usage errors that tell the agent what to change. */
67
+ const parse = (args, options, command) => {
68
+ try {
69
+ return parseArgs({ args, options, allowPositionals: true, strict: true });
70
+ }
71
+ catch (error) {
72
+ throw optionError(error instanceof Error ? error.message : String(error), options, command);
73
+ }
74
+ };
75
+ const baseUrlOf = (env) => {
76
+ const value = env.EXTRAKTOR_URL || DEFAULT_URL;
77
+ const url = URL.parse(value);
78
+ if (!url) {
79
+ throw new CliError(`EXTRAKTOR_URL is not a correct URL: ${value}`, EXIT.usage);
80
+ }
81
+ return url.origin;
82
+ };
83
+ const requireKey = async (baseUrl, env) => {
84
+ const apiKey = await resolveApiKey(baseUrl, env);
85
+ if (!apiKey) {
86
+ throw new CliError("You are not signed in.", EXIT.access, 'Run "extraktor login", or set EXTRAKTOR_API_KEY to an API key from https://extraktor.app/developers.', "SIGN_IN_REQUIRED");
87
+ }
88
+ return apiKey;
89
+ };
90
+ const screenshotName = (url) => {
91
+ const host = URL.parse(url.includes("://") ? url : `https://${url}`)?.hostname;
92
+ return `screenshot-${host?.replaceAll(UNSAFE_FILE_CHARACTERS, "_") || "page"}-${Date.now()}.png`;
93
+ };
94
+ /** Saves Markdown in a file, with a line break at the end. */
95
+ const saveMarkdown = async (markdown, file, io) => {
96
+ const absolute = path.resolve(io.cwd, file);
97
+ await writeFile(absolute, markdown.endsWith("\n") ? markdown : `${markdown}\n`);
98
+ return absolute;
99
+ };
100
+ /**
101
+ * Saves the complete page text for --save, and DESIGN.md for --design-file
102
+ * when the design output completed.
103
+ */
104
+ const saveFiles = async (result, values, io) => {
105
+ if (values.save === undefined && values["design-file"] === undefined) {
106
+ return { savedFile: undefined, designFile: undefined };
107
+ }
108
+ // zod loaded while the request was on the network.
109
+ // oxlint-disable-next-line react-doctor/server-sequential-independent-await
110
+ const { parsePage } = await import("./format.js");
111
+ const { design, markdown } = parsePage(result);
112
+ const designFile = values["design-file"];
113
+ return {
114
+ savedFile: values.save === undefined
115
+ ? undefined
116
+ : await saveMarkdown(markdown, values.save, io),
117
+ designFile: designFile !== undefined && design?.status === "success"
118
+ ? await saveMarkdown(design.markdown, designFile, io)
119
+ : undefined,
120
+ };
121
+ };
122
+ const saveScreenshot = async (result, file, io) => {
123
+ const image = result.content.find((item) => item.type === "image");
124
+ if (!image?.data) {
125
+ io.stderr("Warning: the result has no image data. Use the server copy of the screenshot.\n");
126
+ return;
127
+ }
128
+ const absolute = path.resolve(io.cwd, file);
129
+ await writeFile(absolute, Buffer.from(image.data, "base64"));
130
+ return absolute;
131
+ };
132
+ const parseOffset = (value) => {
133
+ if (value === undefined) {
134
+ return;
135
+ }
136
+ const offset = Number(value);
137
+ if (!Number.isInteger(offset) || offset < 0) {
138
+ throw usageError("--offset must be a whole number, 0 or more.", "extract");
139
+ }
140
+ return offset;
141
+ };
142
+ /** Option combinations that do not work. */
143
+ const checkExtractOptions = (values) => {
144
+ if (values.focus !== undefined && !values.excerpts && !values.summary) {
145
+ throw usageError('Use --focus with --excerpts or --summary. To get only the parts of the page with some text, use --find "text".', "extract");
146
+ }
147
+ if (values.find?.some((find) => !find.trim())) {
148
+ throw usageError('Give --find a text, for example --find "price".', "extract");
149
+ }
150
+ if (values.find !== undefined && values.offset !== undefined) {
151
+ throw usageError("Use --find or --offset, not both. --find searches the complete page.", "extract");
152
+ }
153
+ if (values.save !== undefined &&
154
+ (values.find !== undefined || values.offset !== undefined)) {
155
+ throw usageError("--save saves the complete page text. Do not use it with --find or --offset. Use grep on the file.", "extract");
156
+ }
157
+ };
158
+ /** The page text that the command prints. The CLI cuts it from the complete text. */
159
+ const textRequest = (values) => ({
160
+ finds: [...new Set(values.find?.map((find) => find.trim()))],
161
+ offset: parseOffset(values.offset),
162
+ });
163
+ /** Turns the command options into the extract tool input. */
164
+ const extractArgs = (url, values) => {
165
+ const focus = values.focus?.trim() || undefined;
166
+ const keyword = values.keyword?.trim() || undefined;
167
+ const enabled = true;
168
+ const wants = {
169
+ screenshot: values.screenshot || values["screenshot-file"] !== undefined,
170
+ seo: values.seo || values.keyword !== undefined || values.deep,
171
+ design: values.design || values.vision || values["design-file"] !== undefined,
172
+ };
173
+ return {
174
+ url,
175
+ excerpts: values.excerpts ? { enabled, query: focus } : undefined,
176
+ summary: values.summary ? { enabled, query: focus } : undefined,
177
+ screenshot: wants.screenshot ? { enabled } : undefined,
178
+ contacts: values.contacts ? { enabled } : undefined,
179
+ messaging: values.messaging ? { enabled } : undefined,
180
+ seo: wants.seo
181
+ ? { enabled, keyword, deep: values.deep || undefined }
182
+ : undefined,
183
+ design: wants.design
184
+ ? { enabled, vision: values.vision || undefined }
185
+ : undefined,
186
+ agentAccess: values["agent-access"] ? { enabled } : undefined,
187
+ };
188
+ };
189
+ /** A URL has a host name with a dot and no spaces. Catches a query given to extract. */
190
+ const isUrlLike = (value) => !WHITESPACE.test(value) &&
191
+ (URL.parse(value.includes("://") ? value : `https://${value}`)?.hostname ??
192
+ "").includes(".");
193
+ const checkUrls = (urls, values) => {
194
+ if (urls.length === 0) {
195
+ throw usageError("Give one or more URLs.", "extract");
196
+ }
197
+ const notUrl = urls.find((url) => !isUrlLike(url));
198
+ if (notUrl) {
199
+ throw usageError(`"${notUrl}" is not a URL. To find pages about it, run: extraktor search "${notUrl.replaceAll('"', "")}"`, "extract");
200
+ }
201
+ if (urls.length > MAX_URLS) {
202
+ throw usageError(`Give ${MAX_URLS} URLs or fewer in one command. Run a second command for the other pages.`, "extract");
203
+ }
204
+ if (urls.length > 1 && values.offset !== undefined) {
205
+ throw usageError("--offset is for one page. Give one URL with --offset.", "extract");
206
+ }
207
+ if (urls.length > 1 && values["screenshot-file"] !== undefined) {
208
+ throw usageError("--screenshot-file is for one page. Use --screenshot to save one PNG for each page.", "extract");
209
+ }
210
+ for (const option of ["save", "design-file"]) {
211
+ if (urls.length > 1 && values[option] !== undefined) {
212
+ throw usageError(`--${option} is for one page. Run one command for each page.`, "extract");
213
+ }
214
+ }
215
+ };
216
+ const readPage = async (url, values, connection, io) => {
217
+ const toolArgs = extractArgs(url, values);
218
+ const result = await callToolCached({ ...connection, name: "extract", args: toolArgs }, io.env, !values["no-cache"]);
219
+ const screenshotFile = toolArgs.screenshot
220
+ ? await saveScreenshot(result, values["screenshot-file"] ?? screenshotName(url), io)
221
+ : undefined;
222
+ return {
223
+ url,
224
+ result,
225
+ screenshotFile,
226
+ ...(await saveFiles(result, values, io)),
227
+ };
228
+ };
229
+ const connect = async (io) => {
230
+ const baseUrl = baseUrlOf(io.env);
231
+ return {
232
+ baseUrl,
233
+ apiKey: await requireKey(baseUrl, io.env),
234
+ agent: detectAgent(io.env),
235
+ };
236
+ };
237
+ const extract = async (args, io) => {
238
+ const { values, positionals: urls } = parse(args, EXTRACT_FLAGS, "extract");
239
+ if (values.help) {
240
+ io.stdout(EXTRACT_HELP);
241
+ return;
242
+ }
243
+ // Check every option before the first request.
244
+ checkUrls(urls, values);
245
+ checkExtractOptions(values);
246
+ parseOffset(values.offset);
247
+ const connection = await connect(io);
248
+ const reads = await Promise.allSettled(urls.map((url) => readPage(url, values, connection, io)));
249
+ // Import after the requests on purpose: an earlier import would load zod
250
+ // before the requests leave. callTool loaded zod during the requests.
251
+ // oxlint-disable-next-line react-doctor/server-sequential-independent-await
252
+ const { formatExtract, formatPages, jsonPage, resultWarnings } = await import("./format.js");
253
+ const request = textRequest(values);
254
+ const maxCharacters = Math.floor(MAX_OUTPUT_CHARACTERS / urls.length);
255
+ const [first] = reads;
256
+ if (urls.length === 1 && first) {
257
+ if (first.status === "rejected") {
258
+ throw first.reason;
259
+ }
260
+ const { url, result, screenshotFile, designFile, savedFile } = first.value;
261
+ const pageRequest = { ...request, savedFile };
262
+ if (values.json) {
263
+ for (const warning of resultWarnings(result)) {
264
+ io.stderr(`Warning (${warning.code}): ${warning.message}\n`);
265
+ }
266
+ io.stdout(`${JSON.stringify({
267
+ ...jsonPage(result, pageRequest, maxCharacters),
268
+ ...(screenshotFile && { screenshotFile }),
269
+ ...(designFile && { designFile }),
270
+ }, null, 2)}\n`);
271
+ return;
272
+ }
273
+ io.stdout(formatExtract(result, {
274
+ ...pageRequest,
275
+ url,
276
+ baseUrl: connection.baseUrl,
277
+ screenshotFile,
278
+ designFile,
279
+ maxCharacters,
280
+ }));
281
+ return;
282
+ }
283
+ const pages = reads.map((read, index) => {
284
+ const url = urls[index] ?? "";
285
+ if (read.status === "fulfilled") {
286
+ return read.value;
287
+ }
288
+ if (!(read.reason instanceof CliError)) {
289
+ throw read.reason;
290
+ }
291
+ return { url, error: read.reason };
292
+ });
293
+ io.stdout(values.json
294
+ ? `${JSON.stringify(pages.map((page) => {
295
+ if ("error" in page) {
296
+ return { url: page.url, ...page.error.toJSON() };
297
+ }
298
+ return {
299
+ url: page.url,
300
+ ...jsonPage(page.result, request, maxCharacters),
301
+ ...(page.screenshotFile && {
302
+ screenshotFile: page.screenshotFile,
303
+ }),
304
+ };
305
+ }), null, 2)}\n`
306
+ : formatPages(pages, {
307
+ ...request,
308
+ baseUrl: connection.baseUrl,
309
+ maxCharacters,
310
+ }));
311
+ const failed = pages.find((page) => "error" in page);
312
+ if (failed && "error" in failed) {
313
+ // The output above has every page. The exit code tells that one failed.
314
+ const count = pages.filter((page) => "error" in page).length;
315
+ throw new CliError(count === pages.length
316
+ ? `All ${count} pages failed. The output has the error of each page.`
317
+ : `${count} of ${pages.length} pages failed. The output has the other pages.`, failed.error.exitCode, undefined, SOME_PAGES_FAILED);
318
+ }
319
+ };
320
+ const search = async (args, io) => {
321
+ const { values, positionals } = parse(args, SEARCH_FLAGS, "search");
322
+ if (values.help) {
323
+ io.stdout(SEARCH_HELP);
324
+ return;
325
+ }
326
+ // Agents sometimes forget quotes. Join the words into one query.
327
+ const query = positionals.join(" ").trim();
328
+ if (!query) {
329
+ throw usageError("Give a search query.", "search");
330
+ }
331
+ const connection = await connect(io);
332
+ const result = await callToolCached({
333
+ ...connection,
334
+ name: "search",
335
+ args: { query, serp: values.serp || undefined },
336
+ }, io.env, !values["no-cache"]);
337
+ // After the request on purpose. See extract.
338
+ // oxlint-disable-next-line react-doctor/server-sequential-independent-await
339
+ const { formatSearch } = await import("./format.js");
340
+ io.stdout(values.json
341
+ ? `${JSON.stringify(result.structuredContent ?? {}, null, 2)}\n`
342
+ : formatSearch(result, query, values.serp === true));
343
+ };
344
+ const login = async (args, io) => {
345
+ const { values, positionals } = parse(args, LOGIN_FLAGS, "login");
346
+ if (values.help) {
347
+ io.stdout(LOGIN_HELP);
348
+ return;
349
+ }
350
+ if (positionals.length > 0) {
351
+ throw usageError("login does not accept arguments.", "login");
352
+ }
353
+ const baseUrl = baseUrlOf(io.env);
354
+ let key;
355
+ if (values["with-key"]) {
356
+ const input = await io.readStdin();
357
+ key = input.trim();
358
+ }
359
+ else {
360
+ // node:http takes about 9 ms to load. Only login needs it.
361
+ const { loginWithBrowser } = await import("./login.js");
362
+ key = await loginWithBrowser({
363
+ baseUrl,
364
+ browser: !values["no-browser"],
365
+ log: (line) => io.stderr(`${line}\n`),
366
+ });
367
+ }
368
+ const file = await saveApiKey(baseUrl, key, io.env);
369
+ io.stderr(`Signed in to ${baseUrl}. The key is saved in ${file}.\n`);
370
+ if (io.env.EXTRAKTOR_API_KEY) {
371
+ io.stderr("Note: EXTRAKTOR_API_KEY is set. The CLI uses it and not the saved key.\n");
372
+ }
373
+ };
374
+ const logout = async (args, io) => {
375
+ const { values, positionals } = parse(args, HELP_ONLY, "logout");
376
+ if (values.help) {
377
+ io.stdout(LOGOUT_HELP);
378
+ return;
379
+ }
380
+ if (positionals.length > 0) {
381
+ throw usageError("logout does not accept arguments.", "logout");
382
+ }
383
+ const baseUrl = baseUrlOf(io.env);
384
+ const deleted = await deleteSavedApiKey(baseUrl, io.env);
385
+ io.stderr(deleted
386
+ ? `Deleted the saved key for ${baseUrl}. The key continues to work until you delete it on ${baseUrl}/developers.\n`
387
+ : `No key is saved for ${baseUrl}.\n`);
388
+ };
389
+ const MCP_ACTIONS = new Set(["add", "remove"]);
390
+ const isMcpAction = (value) => MCP_ACTIONS.has(value);
391
+ const mcp = async (args, io) => {
392
+ const [action = "", ...rest] = args;
393
+ if (action === "" ||
394
+ action === "help" ||
395
+ action === "--help" ||
396
+ action === "-h") {
397
+ io.stdout(MCP_HELP);
398
+ return;
399
+ }
400
+ if (!isMcpAction(action)) {
401
+ throw usageError(`"${action}" is not an mcp command. Use "extraktor mcp add" or "extraktor mcp remove".`, "mcp");
402
+ }
403
+ const { values, positionals } = parse(rest, MCP_FLAGS, `mcp ${action}`);
404
+ if (values.help) {
405
+ io.stdout(MCP_HELP);
406
+ return;
407
+ }
408
+ if (positionals.length > 0) {
409
+ throw usageError(`mcp ${action} does not accept arguments.`, "mcp");
410
+ }
411
+ const { AGENT_IDS, changeAgents, defaultExec, formatResults, isAgentId } = await import("./install.js");
412
+ const ids = new Set();
413
+ for (const name of values.agent ?? AGENT_IDS) {
414
+ if (!isAgentId(name)) {
415
+ const match = closest(name, [...AGENT_IDS]);
416
+ throw usageError(`"${name}" is not an agent.${match ? ` Did you mean "--agent ${match}"?` : ""} Agents: ${AGENT_IDS.join(", ")}.`, "mcp");
417
+ }
418
+ ids.add(name);
419
+ }
420
+ const url = `${baseUrlOf(io.env)}/mcp`;
421
+ const results = await changeAgents(action, ids, {
422
+ env: io.env,
423
+ exec: io.exec ?? defaultExec,
424
+ platform: io.platform ?? process.platform,
425
+ url,
426
+ });
427
+ const found = results.filter(({ status }) => status !== "not-found");
428
+ if (found.length === 0 && action === "add") {
429
+ throw new CliError("No coding agent was found on this computer.", EXIT.failure, `Install Claude Code, Codex, Cursor, VS Code, Gemini CLI or Windsurf, then run this command again. Or add ${url} to your MCP client by hand.`, "NO_AGENT_FOUND");
430
+ }
431
+ io.stdout(values.json
432
+ ? `${JSON.stringify({ action, url, agents: results }, null, 2)}\n`
433
+ : formatResults(action, results, url));
434
+ const failed = results.filter(({ status }) => status === "failed").length;
435
+ if (failed > 0) {
436
+ throw new CliError(`${failed} of ${found.length} agents failed. The output tells why.`, EXIT.failure, undefined, SOME_AGENTS_FAILED);
437
+ }
438
+ };
439
+ const COMMANDS = { extract, search, login, logout, mcp };
440
+ /** Names that agents guess for mcp add. They run it, with no error. */
441
+ const MCP_ADD_ALIASES = new Set(["install", "setup"]);
442
+ /** Names that agents guess for extract. They run extract, with no error. */
443
+ const EXTRACT_ALIASES = new Set(["read", "fetch", "get", "scrape", "open"]);
444
+ const isCommand = (name) => Object.hasOwn(COMMANDS, name);
445
+ const unknownCommand = (name) => {
446
+ const match = closest(name, Object.keys(COMMANDS));
447
+ return new CliError(`"${name}" is not a command.${match ? ` Did you mean "extraktor ${match}"?` : ""}`, EXIT.usage, 'Commands: extract <url>..., search "<query>", login, logout, mcp add, mcp remove. To read a page, run: extraktor extract <url>');
448
+ };
449
+ const reportError = (error, json, io) => {
450
+ if (json) {
451
+ // The JSON array already has an error entry for each failed page.
452
+ if (error.code === SOME_PAGES_FAILED || error.code === SOME_AGENTS_FAILED) {
453
+ return;
454
+ }
455
+ io.stdout(`${JSON.stringify(error.toJSON(), null, 2)}\n`);
456
+ return;
457
+ }
458
+ io.stderr(`Error: ${error.message}\n`);
459
+ if (error.guidance) {
460
+ io.stderr(`${error.guidance}\n`);
461
+ }
462
+ };
463
+ /** Runs the CLI and returns the exit code. */
464
+ export const run = async (argv, io) => {
465
+ const [command, ...args] = argv;
466
+ if (!command) {
467
+ io.stdout(MAIN_HELP);
468
+ return EXIT.usage;
469
+ }
470
+ if (command === "--help" || command === "-h" || command === "help") {
471
+ io.stdout(MAIN_HELP);
472
+ return EXIT.ok;
473
+ }
474
+ if (command === "--version" || command === "-v") {
475
+ io.stdout(`${VERSION}\n`);
476
+ return EXIT.ok;
477
+ }
478
+ try {
479
+ if (isCommand(command)) {
480
+ await COMMANDS[command](args, io);
481
+ }
482
+ else if (MCP_ADD_ALIASES.has(command)) {
483
+ await mcp(["add", ...args], io);
484
+ }
485
+ else if (EXTRACT_ALIASES.has(command)) {
486
+ await extract(args, io);
487
+ }
488
+ else if (LOOKS_LIKE_URL.test(command)) {
489
+ // Agents often run "extraktor <url>". Read the page, as extract does.
490
+ await extract(argv, io);
491
+ }
492
+ else {
493
+ throw unknownCommand(command);
494
+ }
495
+ return EXIT.ok;
496
+ }
497
+ catch (error) {
498
+ if (!(error instanceof CliError)) {
499
+ throw error;
500
+ }
501
+ reportError(error, argv.includes("--json"), io);
502
+ return error.exitCode;
503
+ }
504
+ };