@extraktor/cli 0.0.0-stage → 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +188 -2
- package/dist/agent.js +26 -0
- package/dist/auth.js +67 -0
- package/dist/cache.js +113 -0
- package/dist/cli.js +11 -0
- package/dist/errors.js +44 -0
- package/dist/format.js +589 -0
- package/dist/help.js +209 -0
- package/dist/install.js +348 -0
- package/dist/login.js +133 -0
- package/dist/mcp.js +135 -0
- package/dist/page-text.js +386 -0
- package/dist/run.js +504 -0
- package/dist/schemas.js +233 -0
- package/dist/usage.js +131 -0
- package/dist/version.js +2 -0
- package/package.json +46 -4
package/dist/schemas.js
ADDED
|
@@ -0,0 +1,233 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Every zod schema of the CLI. The CLI loads this module while the request is
|
|
3
|
+
* on the network, so zod adds no wait before the request starts.
|
|
4
|
+
*/
|
|
5
|
+
import { z } from "zod";
|
|
6
|
+
import { API_KEY_FORMAT } from "./auth.js";
|
|
7
|
+
export const failureSchema = z.object({
|
|
8
|
+
code: z.string(),
|
|
9
|
+
message: z.string(),
|
|
10
|
+
retryable: z.boolean().optional(),
|
|
11
|
+
guidance: z.string().optional(),
|
|
12
|
+
retryAfterMs: z.number().optional(),
|
|
13
|
+
feature: z.string().optional(),
|
|
14
|
+
});
|
|
15
|
+
export const toolResultSchema = z.object({
|
|
16
|
+
content: z.array(z.looseObject({
|
|
17
|
+
type: z.string(),
|
|
18
|
+
text: z.string().optional(),
|
|
19
|
+
data: z.string().optional(),
|
|
20
|
+
mimeType: z.string().optional(),
|
|
21
|
+
})),
|
|
22
|
+
structuredContent: z.record(z.string(), z.json()).optional(),
|
|
23
|
+
isError: z.boolean().optional(),
|
|
24
|
+
});
|
|
25
|
+
export const jsonRpcSchema = z.object({
|
|
26
|
+
id: z.union([z.number(), z.string(), z.null()]).optional(),
|
|
27
|
+
result: toolResultSchema.optional(),
|
|
28
|
+
error: z.object({ code: z.number(), message: z.string() }).optional(),
|
|
29
|
+
});
|
|
30
|
+
export const httpErrorSchema = z.object({ error_description: z.string() });
|
|
31
|
+
export const parseJson = (text) => {
|
|
32
|
+
try {
|
|
33
|
+
// SAFETY: JSON.parse returns only JSON values. The caller parses the shape.
|
|
34
|
+
return JSON.parse(text);
|
|
35
|
+
}
|
|
36
|
+
catch {
|
|
37
|
+
return null;
|
|
38
|
+
}
|
|
39
|
+
};
|
|
40
|
+
/** Reads the tool failure JSON from a text content item, if it has one. */
|
|
41
|
+
export const parseFailure = (text) => text ? (failureSchema.safeParse(parseJson(text)).data ?? null) : null;
|
|
42
|
+
export const credentialsSchema = z.record(z.string(), z.object({ apiKey: z.string().regex(API_KEY_FORMAT), savedAt: z.string() }));
|
|
43
|
+
// The schemas below mirror the agent schemas of the Extraktor MCP tools. They
|
|
44
|
+
// keep only the fields that the Markdown output shows. --json prints the
|
|
45
|
+
// complete result.
|
|
46
|
+
export const outputFailure = z.object({
|
|
47
|
+
status: z.enum(["empty", "error"]),
|
|
48
|
+
message: z.string(),
|
|
49
|
+
});
|
|
50
|
+
export const markdownOutput = z.union([
|
|
51
|
+
z.object({ status: z.literal("success"), markdown: z.string() }),
|
|
52
|
+
outputFailure,
|
|
53
|
+
]);
|
|
54
|
+
export const excerptsOutput = z.union([
|
|
55
|
+
z.object({
|
|
56
|
+
status: z.literal("success"),
|
|
57
|
+
markdown: z.string(),
|
|
58
|
+
passageLinks: z.array(z.string().nullable()),
|
|
59
|
+
}),
|
|
60
|
+
outputFailure,
|
|
61
|
+
]);
|
|
62
|
+
export const contactsOutput = z.union([
|
|
63
|
+
z.object({
|
|
64
|
+
status: z.literal("success"),
|
|
65
|
+
emails: z.array(z.object({ address: z.string(), label: z.string().nullable() })),
|
|
66
|
+
phones: z.array(z.object({
|
|
67
|
+
number: z.string().nullable(),
|
|
68
|
+
text: z.string(),
|
|
69
|
+
label: z.string().nullable(),
|
|
70
|
+
})),
|
|
71
|
+
// Older servers do not send these fields.
|
|
72
|
+
addresses: z
|
|
73
|
+
.array(z.object({
|
|
74
|
+
text: z.string(),
|
|
75
|
+
country: z.string().nullable(),
|
|
76
|
+
label: z.string().nullable(),
|
|
77
|
+
}))
|
|
78
|
+
.default([]),
|
|
79
|
+
profiles: z
|
|
80
|
+
.array(z.object({ network: z.string(), url: z.string() }))
|
|
81
|
+
.default([]),
|
|
82
|
+
organization: z
|
|
83
|
+
.object({
|
|
84
|
+
name: z.string().nullable(),
|
|
85
|
+
legalName: z.string().nullable(),
|
|
86
|
+
registrationId: z.string().nullable(),
|
|
87
|
+
vatId: z.string().nullable(),
|
|
88
|
+
})
|
|
89
|
+
.nullable()
|
|
90
|
+
.default(null),
|
|
91
|
+
contactPages: z.array(z.string()).optional(),
|
|
92
|
+
message: z.string().optional(),
|
|
93
|
+
}),
|
|
94
|
+
z.object({
|
|
95
|
+
status: z.enum(["empty", "error"]),
|
|
96
|
+
message: z.string(),
|
|
97
|
+
contactPages: z.array(z.string()).optional(),
|
|
98
|
+
}),
|
|
99
|
+
]);
|
|
100
|
+
const messagingCta = z
|
|
101
|
+
.object({ text: z.string(), url: z.string().nullable() })
|
|
102
|
+
.nullable();
|
|
103
|
+
export const messagingOutput = z.union([
|
|
104
|
+
z.object({
|
|
105
|
+
status: z.literal("success"),
|
|
106
|
+
headline: z.string().nullable(),
|
|
107
|
+
subheadline: z.string().nullable(),
|
|
108
|
+
primaryCta: messagingCta,
|
|
109
|
+
secondaryCta: messagingCta,
|
|
110
|
+
benefits: z.array(z.string()),
|
|
111
|
+
differentiators: z.array(z.string()),
|
|
112
|
+
proof: z.object({
|
|
113
|
+
metrics: z.array(z.string()),
|
|
114
|
+
testimonials: z.array(z.object({
|
|
115
|
+
quote: z.string(),
|
|
116
|
+
person: z.string().nullable(),
|
|
117
|
+
role: z.string().nullable(),
|
|
118
|
+
company: z.string().nullable(),
|
|
119
|
+
})),
|
|
120
|
+
customers: z.array(z.string()),
|
|
121
|
+
awards: z.array(z.string()),
|
|
122
|
+
certifications: z.array(z.string()),
|
|
123
|
+
reviewBadges: z.array(z.string()),
|
|
124
|
+
}),
|
|
125
|
+
positioning: z.object({
|
|
126
|
+
category: z.string(),
|
|
127
|
+
audience: z.string(),
|
|
128
|
+
valueProposition: z.string(),
|
|
129
|
+
painPoints: z.array(z.string()),
|
|
130
|
+
tone: z.array(z.string()),
|
|
131
|
+
}),
|
|
132
|
+
}),
|
|
133
|
+
outputFailure,
|
|
134
|
+
]);
|
|
135
|
+
/** Reports without a fixed shape. The CLI prints them as JSON. */
|
|
136
|
+
export const reportOutput = z.looseObject({
|
|
137
|
+
status: z.enum(["success", "empty", "error"]),
|
|
138
|
+
message: z.string().optional(),
|
|
139
|
+
/** The report as Markdown, as the site shows it. */
|
|
140
|
+
markdown: z.string().optional(),
|
|
141
|
+
});
|
|
142
|
+
export const screenshotOutput = z.union([
|
|
143
|
+
z.object({
|
|
144
|
+
status: z.literal("success"),
|
|
145
|
+
url: z.string(),
|
|
146
|
+
bytes: z.number(),
|
|
147
|
+
}),
|
|
148
|
+
z.object({ status: z.literal("error"), message: z.string() }),
|
|
149
|
+
]);
|
|
150
|
+
const foundSection = z.object({
|
|
151
|
+
level: z.number(),
|
|
152
|
+
heading: z.string(),
|
|
153
|
+
offset: z.number(),
|
|
154
|
+
});
|
|
155
|
+
/** Any result with page text. --json keeps every other field. */
|
|
156
|
+
export const pageTextSchema = z.looseObject({ markdown: z.string() });
|
|
157
|
+
export const extractSchema = z.object({
|
|
158
|
+
finalUrl: z.string(),
|
|
159
|
+
status: z.number().nullable(),
|
|
160
|
+
markdown: z.string(),
|
|
161
|
+
markdownRange: z
|
|
162
|
+
.object({
|
|
163
|
+
start: z.number(),
|
|
164
|
+
end: z.number(),
|
|
165
|
+
totalCharacters: z.number(),
|
|
166
|
+
nextOffset: z.number().nullable(),
|
|
167
|
+
})
|
|
168
|
+
.optional(),
|
|
169
|
+
outline: z
|
|
170
|
+
.array(z.object({ level: z.number(), heading: z.string(), offset: z.number() }))
|
|
171
|
+
.optional(),
|
|
172
|
+
found: z
|
|
173
|
+
.object({
|
|
174
|
+
query: z.string(),
|
|
175
|
+
matchedBy: z.enum(["phrase", "words"]).nullable(),
|
|
176
|
+
sections: z.array(foundSection),
|
|
177
|
+
omittedSections: z.array(foundSection),
|
|
178
|
+
message: z.string(),
|
|
179
|
+
})
|
|
180
|
+
.optional(),
|
|
181
|
+
metadata: z.object({
|
|
182
|
+
title: z.string().nullable(),
|
|
183
|
+
description: z.string().nullable(),
|
|
184
|
+
byline: z.string().nullable(),
|
|
185
|
+
publishedTime: z.string().nullable(),
|
|
186
|
+
modifiedTime: z.string().nullable(),
|
|
187
|
+
}),
|
|
188
|
+
stats: z.object({ words: z.number() }),
|
|
189
|
+
structuredData: z.array(z.json()).optional(),
|
|
190
|
+
summary: markdownOutput.optional(),
|
|
191
|
+
excerpts: excerptsOutput.optional(),
|
|
192
|
+
contacts: contactsOutput.optional(),
|
|
193
|
+
messaging: messagingOutput.optional(),
|
|
194
|
+
seo: reportOutput.optional(),
|
|
195
|
+
agentAccess: reportOutput.optional(),
|
|
196
|
+
design: markdownOutput.optional(),
|
|
197
|
+
screenshot: screenshotOutput.optional(),
|
|
198
|
+
});
|
|
199
|
+
const jsonValue = z.json();
|
|
200
|
+
const serpEntry = z.record(z.string(), jsonValue);
|
|
201
|
+
const scalarText = z.union([z.string(), z.number()]).transform(String);
|
|
202
|
+
/** A string or a number of a SERP block as text; null for other values. */
|
|
203
|
+
export const serpText = (value) => {
|
|
204
|
+
const parsed = scalarText.safeParse(value);
|
|
205
|
+
return parsed.success ? parsed.data : null;
|
|
206
|
+
};
|
|
207
|
+
export const serpBlockSchema = z.object({
|
|
208
|
+
feature: z.string(),
|
|
209
|
+
y: z.number().optional(),
|
|
210
|
+
column: z.string().optional(),
|
|
211
|
+
ranks: z.array(z.number()).optional(),
|
|
212
|
+
count: z.number().optional(),
|
|
213
|
+
deferred: z.boolean().optional(),
|
|
214
|
+
title: z.string().optional(),
|
|
215
|
+
type: z.string().optional(),
|
|
216
|
+
source: jsonValue.optional(),
|
|
217
|
+
link: z.string().optional(),
|
|
218
|
+
website: z.string().optional(),
|
|
219
|
+
text: z.string().optional(),
|
|
220
|
+
description: z.string().optional(),
|
|
221
|
+
list: z.array(z.string()).optional(),
|
|
222
|
+
sources: z.array(serpEntry).optional(),
|
|
223
|
+
items: z.array(serpEntry).optional(),
|
|
224
|
+
});
|
|
225
|
+
export const searchSchema = z.object({
|
|
226
|
+
organic_results: z.array(z.looseObject({
|
|
227
|
+
title: z.string().optional(),
|
|
228
|
+
link: z.string().optional(),
|
|
229
|
+
snippet: z.string().optional(),
|
|
230
|
+
date: z.string().optional(),
|
|
231
|
+
})),
|
|
232
|
+
serp: z.array(serpBlockSchema).default([]),
|
|
233
|
+
});
|
package/dist/usage.js
ADDED
|
@@ -0,0 +1,131 @@
|
|
|
1
|
+
import { CliError, EXIT } from "./errors.js";
|
|
2
|
+
const UNKNOWN_OPTION = /^Unknown option '(?<option>[^']+)'/u;
|
|
3
|
+
const MISSING_VALUE = /^Option '(?<option>-[^' ]+)[^']*' argument missing/u;
|
|
4
|
+
/** A typo in a long name can have 2 wrong letters, in a short name only 1. */
|
|
5
|
+
const LONG_NAME_LENGTH = 8;
|
|
6
|
+
const MIN_PREFIX_LENGTH = 3;
|
|
7
|
+
/** Names that agents guess, mapped to the option that does the job. */
|
|
8
|
+
const SYNONYMS = new Map(Object.entries({
|
|
9
|
+
filter: "find",
|
|
10
|
+
grep: "find",
|
|
11
|
+
search: "find",
|
|
12
|
+
query: "find",
|
|
13
|
+
match: "find",
|
|
14
|
+
section: "find",
|
|
15
|
+
format: "json",
|
|
16
|
+
output: "save",
|
|
17
|
+
out: "save",
|
|
18
|
+
o: "save",
|
|
19
|
+
file: "save",
|
|
20
|
+
path: "save",
|
|
21
|
+
write: "save",
|
|
22
|
+
download: "save",
|
|
23
|
+
full: "save",
|
|
24
|
+
all: "save",
|
|
25
|
+
page: "offset",
|
|
26
|
+
start: "offset",
|
|
27
|
+
skip: "offset",
|
|
28
|
+
question: "focus",
|
|
29
|
+
topic: "focus",
|
|
30
|
+
prompt: "focus",
|
|
31
|
+
quote: "excerpts",
|
|
32
|
+
quotes: "excerpts",
|
|
33
|
+
summarize: "summary",
|
|
34
|
+
email: "contacts",
|
|
35
|
+
emails: "contacts",
|
|
36
|
+
phone: "contacts",
|
|
37
|
+
phones: "contacts",
|
|
38
|
+
address: "contacts",
|
|
39
|
+
addresses: "contacts",
|
|
40
|
+
social: "contacts",
|
|
41
|
+
socials: "contacts",
|
|
42
|
+
profiles: "contacts",
|
|
43
|
+
imprint: "contacts",
|
|
44
|
+
impressum: "contacts",
|
|
45
|
+
company: "contacts",
|
|
46
|
+
positioning: "messaging",
|
|
47
|
+
marketing: "messaging",
|
|
48
|
+
"value-proposition": "messaging",
|
|
49
|
+
fresh: "no-cache",
|
|
50
|
+
refresh: "no-cache",
|
|
51
|
+
force: "no-cache",
|
|
52
|
+
png: "screenshot",
|
|
53
|
+
image: "screenshot",
|
|
54
|
+
"full-page": "screenshot",
|
|
55
|
+
questions: "serp",
|
|
56
|
+
paa: "serp",
|
|
57
|
+
related: "serp",
|
|
58
|
+
features: "serp",
|
|
59
|
+
"serp-features": "serp",
|
|
60
|
+
}));
|
|
61
|
+
const distance = (a, b) => {
|
|
62
|
+
let previous = Array.from({ length: b.length + 1 }, (_, index) => index);
|
|
63
|
+
for (const [i, charA] of [...a].entries()) {
|
|
64
|
+
const current = [i + 1];
|
|
65
|
+
for (const [j, charB] of [...b].entries()) {
|
|
66
|
+
current.push(Math.min((previous[j + 1] ?? 0) + 1, (current[j] ?? 0) + 1, (previous[j] ?? 0) + (charA === charB ? 0 : 1)));
|
|
67
|
+
}
|
|
68
|
+
previous = current;
|
|
69
|
+
}
|
|
70
|
+
return previous[b.length] ?? Number.POSITIVE_INFINITY;
|
|
71
|
+
};
|
|
72
|
+
/** The closest name: a known synonym, a prefix, or a small typo. */
|
|
73
|
+
export const closest = (name, names) => {
|
|
74
|
+
const synonym = SYNONYMS.get(name.toLowerCase());
|
|
75
|
+
if (synonym && names.includes(synonym)) {
|
|
76
|
+
return synonym;
|
|
77
|
+
}
|
|
78
|
+
if (name.length >= MIN_PREFIX_LENGTH) {
|
|
79
|
+
const prefixed = names.filter((candidate) => candidate.startsWith(name));
|
|
80
|
+
if (prefixed.length === 1) {
|
|
81
|
+
return prefixed[0] ?? null;
|
|
82
|
+
}
|
|
83
|
+
}
|
|
84
|
+
let best = null;
|
|
85
|
+
let bestDistance = name.length >= LONG_NAME_LENGTH ? 3 : 2;
|
|
86
|
+
for (const candidate of names) {
|
|
87
|
+
const value = distance(name.toLowerCase(), candidate);
|
|
88
|
+
if (value < bestDistance) {
|
|
89
|
+
best = candidate;
|
|
90
|
+
bestDistance = value;
|
|
91
|
+
}
|
|
92
|
+
}
|
|
93
|
+
return best;
|
|
94
|
+
};
|
|
95
|
+
const VALUE_NAMES = new Map([
|
|
96
|
+
["offset", "n"],
|
|
97
|
+
["screenshot-file", "path"],
|
|
98
|
+
["design-file", "path"],
|
|
99
|
+
["save", "file"],
|
|
100
|
+
]);
|
|
101
|
+
/** Every option of a command in one line, for example "--find <text>, --json". */
|
|
102
|
+
export const optionList = (flags) => {
|
|
103
|
+
const options = [];
|
|
104
|
+
for (const [name, { type }] of Object.entries(flags)) {
|
|
105
|
+
if (name !== "help") {
|
|
106
|
+
options.push(type === "string"
|
|
107
|
+
? `--${name} <${VALUE_NAMES.get(name) ?? "text"}>`
|
|
108
|
+
: `--${name}`);
|
|
109
|
+
}
|
|
110
|
+
}
|
|
111
|
+
return options.join(", ");
|
|
112
|
+
};
|
|
113
|
+
export const usageError = (message, command) => new CliError(message, EXIT.usage, `Run "extraktor ${command ? `${command} ` : ""}--help" to see the correct usage.`);
|
|
114
|
+
/**
|
|
115
|
+
* Turns a parseArgs error into a usage error that names the fix and lists
|
|
116
|
+
* the options of the command.
|
|
117
|
+
*/
|
|
118
|
+
export const optionError = (message, flags, command) => {
|
|
119
|
+
const options = `Options for ${command}: ${optionList(flags)}.`;
|
|
120
|
+
const unknown = UNKNOWN_OPTION.exec(message)?.groups?.option;
|
|
121
|
+
if (unknown) {
|
|
122
|
+
const name = unknown.replace(/^-+/u, "").split("=")[0] ?? "";
|
|
123
|
+
const match = closest(name, Object.keys(flags));
|
|
124
|
+
return new CliError(`Unknown option ${unknown}.${match ? ` Did you mean --${match}?` : ""}`, EXIT.usage, options);
|
|
125
|
+
}
|
|
126
|
+
const missing = MISSING_VALUE.exec(message)?.groups?.option;
|
|
127
|
+
if (missing) {
|
|
128
|
+
return new CliError(`${missing} needs a value, for example ${missing} "text".`, EXIT.usage, options);
|
|
129
|
+
}
|
|
130
|
+
return new CliError(message.split("\n")[0] ?? message, EXIT.usage, options);
|
|
131
|
+
};
|
package/dist/version.js
ADDED
package/package.json
CHANGED
|
@@ -1,6 +1,48 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@extraktor/cli",
|
|
3
|
-
"version": "0.
|
|
4
|
-
"
|
|
5
|
-
"
|
|
6
|
-
|
|
3
|
+
"version": "0.1.0",
|
|
4
|
+
"description": "Read live web pages as Markdown and search the web from the terminal. Built for AI agents.",
|
|
5
|
+
"keywords": [
|
|
6
|
+
"agent",
|
|
7
|
+
"cli",
|
|
8
|
+
"extract",
|
|
9
|
+
"markdown",
|
|
10
|
+
"mcp",
|
|
11
|
+
"scrape",
|
|
12
|
+
"search",
|
|
13
|
+
"web"
|
|
14
|
+
],
|
|
15
|
+
"homepage": "https://extraktor.app",
|
|
16
|
+
"repository": {
|
|
17
|
+
"type": "git",
|
|
18
|
+
"url": "git+https://github.com/DennisKo/extraktor-cli.git"
|
|
19
|
+
},
|
|
20
|
+
"bugs": {
|
|
21
|
+
"url": "https://github.com/DennisKo/extraktor-cli/issues"
|
|
22
|
+
},
|
|
23
|
+
"license": "MIT",
|
|
24
|
+
"bin": {
|
|
25
|
+
"extraktor": "dist/cli.js"
|
|
26
|
+
},
|
|
27
|
+
"files": [
|
|
28
|
+
"dist",
|
|
29
|
+
"LICENSE"
|
|
30
|
+
],
|
|
31
|
+
"type": "module",
|
|
32
|
+
"scripts": {
|
|
33
|
+
"build": "tsc -p .",
|
|
34
|
+
"typecheck": "tsc -p tsconfig.test.json",
|
|
35
|
+
"test": "node --test test/*.test.ts",
|
|
36
|
+
"prepack": "npm run build"
|
|
37
|
+
},
|
|
38
|
+
"dependencies": {
|
|
39
|
+
"zod": "^4.6.5"
|
|
40
|
+
},
|
|
41
|
+
"devDependencies": {
|
|
42
|
+
"@types/node": "^22.10.2",
|
|
43
|
+
"typescript": "^6.0.2"
|
|
44
|
+
},
|
|
45
|
+
"engines": {
|
|
46
|
+
"node": ">=22"
|
|
47
|
+
}
|
|
48
|
+
}
|