@extraktor/cli 0.0.0-stage → 0.1.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +58 -2
- package/dist/agent.js +26 -0
- package/dist/auth.js +67 -0
- package/dist/cache.js +113 -0
- package/dist/cli.js +11 -0
- package/dist/errors.js +44 -0
- package/dist/format.js +589 -0
- package/dist/help.js +209 -0
- package/dist/install.js +348 -0
- package/dist/login.js +133 -0
- package/dist/mcp.js +135 -0
- package/dist/page-text.js +386 -0
- package/dist/run.js +504 -0
- package/dist/schemas.js +233 -0
- package/dist/usage.js +131 -0
- package/dist/version.js +2 -0
- package/package.json +46 -4
package/dist/format.js
ADDED
|
@@ -0,0 +1,589 @@
|
|
|
1
|
+
import { CliError, EXIT } from "./errors.js";
|
|
2
|
+
import { findPageText, fitPageText, pageOutline } from "./page-text.js";
|
|
3
|
+
import { extractSchema, pageTextSchema, parseFailure, searchSchema, serpText, } from "./schemas.js";
|
|
4
|
+
const SAFE_SHELL_WORD = /^[\w./:@%+=,-]+$/u;
|
|
5
|
+
/** Quotes a value for a POSIX shell command when it needs quotes. */
|
|
6
|
+
const quoteArg = (value) => SAFE_SHELL_WORD.test(value) ? value : `'${value.replaceAll("'", "'\\''")}'`;
|
|
7
|
+
const parseResult = (schema, result) => {
|
|
8
|
+
const parsed = schema.safeParse(result.structuredContent);
|
|
9
|
+
if (!parsed.success) {
|
|
10
|
+
throw new CliError("The server result has an unexpected form.", EXIT.failure, "Update the CLI: npm install --global @extraktor/cli@latest. Or use --json.");
|
|
11
|
+
}
|
|
12
|
+
return parsed.data;
|
|
13
|
+
};
|
|
14
|
+
/** Warnings that the server adds after the result, for example a capture limit. */
|
|
15
|
+
export const resultWarnings = (result) => {
|
|
16
|
+
const warnings = [];
|
|
17
|
+
for (const item of result.content.slice(1)) {
|
|
18
|
+
const failure = item.type === "text" ? parseFailure(item.text) : null;
|
|
19
|
+
// A failed output shows its own message in its section.
|
|
20
|
+
if (failure && failure.code !== "OPTIONAL_FEATURE_FAILED") {
|
|
21
|
+
warnings.push(failure);
|
|
22
|
+
}
|
|
23
|
+
}
|
|
24
|
+
return warnings;
|
|
25
|
+
};
|
|
26
|
+
const failedOutput = ({ message }) => `Not available: ${message || "the output did not complete."}`;
|
|
27
|
+
const renderMarkdown = (output) => output.status === "success" ? output.markdown.trim() : failedOutput(output);
|
|
28
|
+
const renderExcerpts = (output) => {
|
|
29
|
+
if (output.status !== "success") {
|
|
30
|
+
return failedOutput(output);
|
|
31
|
+
}
|
|
32
|
+
const links = [];
|
|
33
|
+
for (const [index, link] of output.passageLinks.entries()) {
|
|
34
|
+
if (link) {
|
|
35
|
+
links.push(`${index + 1}. ${link}`);
|
|
36
|
+
}
|
|
37
|
+
}
|
|
38
|
+
return links.length > 0
|
|
39
|
+
? `${output.markdown.trim()}\n\nPassage links:\n${links.join("\n")}`
|
|
40
|
+
: output.markdown.trim();
|
|
41
|
+
};
|
|
42
|
+
const labelled = (value, label) => label ? `${value} (${label})` : value;
|
|
43
|
+
const NETWORK_NAMES = new Map([
|
|
44
|
+
["bluesky", "Bluesky"],
|
|
45
|
+
["facebook", "Facebook"],
|
|
46
|
+
["github", "GitHub"],
|
|
47
|
+
["instagram", "Instagram"],
|
|
48
|
+
["linkedin", "LinkedIn"],
|
|
49
|
+
["mastodon", "Mastodon"],
|
|
50
|
+
["tiktok", "TikTok"],
|
|
51
|
+
["x", "X"],
|
|
52
|
+
["xing", "XING"],
|
|
53
|
+
["youtube", "YouTube"],
|
|
54
|
+
]);
|
|
55
|
+
const ORGANIZATION_FIELDS = [
|
|
56
|
+
["name", "Name"],
|
|
57
|
+
["legalName", "Legal name"],
|
|
58
|
+
["registrationId", "Registration"],
|
|
59
|
+
["vatId", "VAT ID"],
|
|
60
|
+
];
|
|
61
|
+
/** The message, the contact pages and the command for the first one. */
|
|
62
|
+
const contactPageLines = (output) => {
|
|
63
|
+
const pages = output.contactPages ?? [];
|
|
64
|
+
return [
|
|
65
|
+
output.message ?? "",
|
|
66
|
+
...pages.map((page) => `- ${page}`),
|
|
67
|
+
pages[0] ? `Next: extraktor extract ${quoteArg(pages[0])} --contacts` : "",
|
|
68
|
+
].filter(Boolean);
|
|
69
|
+
};
|
|
70
|
+
const listSection = (title, items) => items.length > 0 ? `${title}:\n${items.join("\n")}` : `${title}: none`;
|
|
71
|
+
const renderContacts = (output) => {
|
|
72
|
+
if (output.status !== "success") {
|
|
73
|
+
const lines = contactPageLines(output);
|
|
74
|
+
return output.status === "empty" && lines.length > 0
|
|
75
|
+
? lines.join("\n")
|
|
76
|
+
: failedOutput(output);
|
|
77
|
+
}
|
|
78
|
+
const emails = output.emails.map(({ address, label }) => `- ${labelled(address, label)}`);
|
|
79
|
+
const phones = output.phones.map(({ number, text, label }) => {
|
|
80
|
+
const value = number && number !== text ? `${text} [${number}]` : text;
|
|
81
|
+
return `- ${labelled(value, label)}`;
|
|
82
|
+
});
|
|
83
|
+
const addresses = output.addresses.map(({ country, label, text }) => {
|
|
84
|
+
const value = country && !text.endsWith(country) ? `${text} [${country}]` : text;
|
|
85
|
+
return `- ${labelled(value, label)}`;
|
|
86
|
+
});
|
|
87
|
+
const profiles = output.profiles.map(({ network, url }) => `- ${NETWORK_NAMES.get(network) ?? network}: ${url}`);
|
|
88
|
+
const { organization } = output;
|
|
89
|
+
const company = organization
|
|
90
|
+
? ORGANIZATION_FIELDS.flatMap(([field, title]) => organization[field] ? [`- ${title}: ${organization[field]}`] : [])
|
|
91
|
+
: [];
|
|
92
|
+
return [
|
|
93
|
+
listSection("Emails", emails),
|
|
94
|
+
listSection("Phones", phones),
|
|
95
|
+
listSection("Addresses", addresses),
|
|
96
|
+
listSection("Profiles", profiles),
|
|
97
|
+
listSection("Organization", company),
|
|
98
|
+
...(output.contactPages ? [contactPageLines(output).join("\n")] : []),
|
|
99
|
+
].join("\n\n");
|
|
100
|
+
};
|
|
101
|
+
const NOT_ON_PAGE = "not on the page";
|
|
102
|
+
const renderCta = (cta) => {
|
|
103
|
+
if (!cta) {
|
|
104
|
+
return NOT_ON_PAGE;
|
|
105
|
+
}
|
|
106
|
+
return cta.url ? `${cta.text} (${cta.url})` : cta.text;
|
|
107
|
+
};
|
|
108
|
+
const bulletList = (title, items) => items.length > 0
|
|
109
|
+
? `${title}:\n${items.map((item) => `- ${item}`).join("\n")}`
|
|
110
|
+
: `${title}: ${NOT_ON_PAGE}`;
|
|
111
|
+
/**
|
|
112
|
+
* What the offer is and who it is for first, then the site's own words. Every
|
|
113
|
+
* page prints the same lines, so pages compare line by line.
|
|
114
|
+
*/
|
|
115
|
+
const renderMessaging = (output) => {
|
|
116
|
+
if (output.status !== "success") {
|
|
117
|
+
return failedOutput(output);
|
|
118
|
+
}
|
|
119
|
+
const { positioning, proof } = output;
|
|
120
|
+
const proofLines = [
|
|
121
|
+
["Metrics", proof.metrics],
|
|
122
|
+
["Customers", proof.customers],
|
|
123
|
+
["Awards", proof.awards],
|
|
124
|
+
["Certifications", proof.certifications],
|
|
125
|
+
["Review badges", proof.reviewBadges],
|
|
126
|
+
].map(([label, items]) => `- ${label}: ${items.length > 0 ? items.join("; ") : NOT_ON_PAGE}`);
|
|
127
|
+
const testimonials = proof.testimonials.map(({ company, person, quote, role }) => {
|
|
128
|
+
const source = [person, role, company].filter(Boolean).join(", ");
|
|
129
|
+
return `> ${quote}${source ? `\n> (${source})` : ""}`;
|
|
130
|
+
});
|
|
131
|
+
return [
|
|
132
|
+
[
|
|
133
|
+
"Positioning (Extraktor's words, not site text):",
|
|
134
|
+
`- Category: ${positioning.category || "none"}`,
|
|
135
|
+
`- Audience: ${positioning.audience || "none"}`,
|
|
136
|
+
`- Value proposition: ${positioning.valueProposition || "none"}`,
|
|
137
|
+
`- Pain points: ${positioning.painPoints.join("; ") || "none"}`,
|
|
138
|
+
`- Tone: ${positioning.tone.join(", ") || "none"}`,
|
|
139
|
+
].join("\n"),
|
|
140
|
+
"Site text, as the page writes it:",
|
|
141
|
+
`Headline: ${output.headline ?? NOT_ON_PAGE}`,
|
|
142
|
+
`Subheadline: ${output.subheadline ?? NOT_ON_PAGE}`,
|
|
143
|
+
`Primary CTA: ${renderCta(output.primaryCta)}`,
|
|
144
|
+
`Secondary CTA: ${renderCta(output.secondaryCta)}`,
|
|
145
|
+
bulletList("Benefits", output.benefits),
|
|
146
|
+
bulletList("Differentiators", output.differentiators),
|
|
147
|
+
`Proof:\n${proofLines.join("\n")}`,
|
|
148
|
+
testimonials.length > 0
|
|
149
|
+
? `Testimonials:\n${testimonials.join("\n\n")}`
|
|
150
|
+
: `Testimonials: ${NOT_ON_PAGE}`,
|
|
151
|
+
].join("\n\n");
|
|
152
|
+
};
|
|
153
|
+
const HEADING_MARKS = /^(?<marks>#{1,5}) /gmu;
|
|
154
|
+
const REPORT_TITLE = /^# .*\n+(?:Page: .*\n+)?/u;
|
|
155
|
+
/**
|
|
156
|
+
* A report as the server writes it in Markdown, one heading level lower, so
|
|
157
|
+
* that it fits under its section. The section heading and the header give
|
|
158
|
+
* its title and the page URL.
|
|
159
|
+
*/
|
|
160
|
+
const renderReport = (output) => {
|
|
161
|
+
if (output.status === "error") {
|
|
162
|
+
return failedOutput(output);
|
|
163
|
+
}
|
|
164
|
+
if (output.markdown) {
|
|
165
|
+
return output.markdown
|
|
166
|
+
.replace(REPORT_TITLE, "")
|
|
167
|
+
.replaceAll(HEADING_MARKS, "#$<marks> ")
|
|
168
|
+
.trim();
|
|
169
|
+
}
|
|
170
|
+
return `\`\`\`json\n${JSON.stringify(output, null, 2)}\n\`\`\``;
|
|
171
|
+
};
|
|
172
|
+
const outputSections = (page) => [
|
|
173
|
+
["Summary (AI)", page.summary && renderMarkdown(page.summary)],
|
|
174
|
+
["Excerpts", page.excerpts && renderExcerpts(page.excerpts)],
|
|
175
|
+
["Contacts", page.contacts && renderContacts(page.contacts)],
|
|
176
|
+
["Messaging", page.messaging && renderMessaging(page.messaging)],
|
|
177
|
+
["SEO", page.seo && renderReport(page.seo)],
|
|
178
|
+
["Agent access", page.agentAccess && renderReport(page.agentAccess)],
|
|
179
|
+
];
|
|
180
|
+
const viewText = (markdown, { finds, offset, savedFile }, maxCharacters) => {
|
|
181
|
+
if (savedFile) {
|
|
182
|
+
return { kind: "saved", file: savedFile, outline: pageOutline(markdown) };
|
|
183
|
+
}
|
|
184
|
+
if (finds.length > 0) {
|
|
185
|
+
const each = Math.floor(maxCharacters / finds.length);
|
|
186
|
+
return {
|
|
187
|
+
kind: "found",
|
|
188
|
+
results: finds.map((find) => findPageText({ markdown }, find, each)),
|
|
189
|
+
};
|
|
190
|
+
}
|
|
191
|
+
return {
|
|
192
|
+
kind: "part",
|
|
193
|
+
part: fitPageText({ markdown }, offset, maxCharacters),
|
|
194
|
+
};
|
|
195
|
+
};
|
|
196
|
+
/** The --json result: the server result with the page text of the request. */
|
|
197
|
+
export const jsonPage = (result, request, maxCharacters) => {
|
|
198
|
+
const parsed = pageTextSchema.safeParse(result.structuredContent);
|
|
199
|
+
if (!parsed.success) {
|
|
200
|
+
return result.structuredContent ?? {};
|
|
201
|
+
}
|
|
202
|
+
const { markdown, ...page } = parsed.data;
|
|
203
|
+
const view = viewText(markdown, request, maxCharacters);
|
|
204
|
+
if (view.kind === "saved") {
|
|
205
|
+
return { ...page, savedFile: view.file, outline: view.outline };
|
|
206
|
+
}
|
|
207
|
+
if (view.kind === "part") {
|
|
208
|
+
return { ...page, ...view.part };
|
|
209
|
+
}
|
|
210
|
+
const [first] = view.results;
|
|
211
|
+
return view.results.length === 1 && first
|
|
212
|
+
? { ...page, ...first }
|
|
213
|
+
: { ...page, finds: view.results };
|
|
214
|
+
};
|
|
215
|
+
// Room for the header and the warnings.
|
|
216
|
+
const HEADER_CHARACTERS = 800;
|
|
217
|
+
const MIN_TEXT_CHARACTERS = 2000;
|
|
218
|
+
const findLine = ({ query, matchedBy, sections, omittedSections, }) => {
|
|
219
|
+
if (!matchedBy) {
|
|
220
|
+
return `Find "${query}": searched the complete page. The page text does not contain it.`;
|
|
221
|
+
}
|
|
222
|
+
const count = sections.length + omittedSections.length;
|
|
223
|
+
return `Find "${query}": searched the complete page. ${count} ${count === 1 ? "section matches" : "sections match"}${matchedBy === "words" ? " all the words (not the exact phrase)" : ""}.`;
|
|
224
|
+
};
|
|
225
|
+
const viewLines = (view) => {
|
|
226
|
+
if (view.kind === "found") {
|
|
227
|
+
return view.results.map(({ found }) => findLine(found));
|
|
228
|
+
}
|
|
229
|
+
const range = view.kind === "part" ? view.part.markdownRange : undefined;
|
|
230
|
+
if (!range) {
|
|
231
|
+
return [];
|
|
232
|
+
}
|
|
233
|
+
// Say it at the top too: agents often read only the first lines.
|
|
234
|
+
return [
|
|
235
|
+
`Part of the page: characters ${range.start} to ${range.end} of ${range.totalCharacters}.${range.nextOffset === null ? "" : ` The next part starts at --offset ${range.nextOffset}.`} The outline with the offset of each heading is at the end.`,
|
|
236
|
+
];
|
|
237
|
+
};
|
|
238
|
+
const header = ({ finalUrl, metadata, stats, status }, view) => {
|
|
239
|
+
const facts = [
|
|
240
|
+
status === null ? null : `HTTP ${status}`,
|
|
241
|
+
`${stats.words.toLocaleString("en-US")} words`,
|
|
242
|
+
].filter(Boolean);
|
|
243
|
+
const lines = [
|
|
244
|
+
`# ${metadata.title || finalUrl}`,
|
|
245
|
+
"",
|
|
246
|
+
`Source: ${finalUrl} (${facts.join(", ")})`,
|
|
247
|
+
...viewLines(view),
|
|
248
|
+
];
|
|
249
|
+
for (const [label, value] of [
|
|
250
|
+
["Description", metadata.description],
|
|
251
|
+
["Author", metadata.byline],
|
|
252
|
+
["Published", metadata.publishedTime],
|
|
253
|
+
["Modified", metadata.modifiedTime],
|
|
254
|
+
]) {
|
|
255
|
+
if (value) {
|
|
256
|
+
lines.push(`${label}: ${value}`);
|
|
257
|
+
}
|
|
258
|
+
}
|
|
259
|
+
return lines.join("\n");
|
|
260
|
+
};
|
|
261
|
+
const outlineLine = ({ level, heading, offset }) => `- ${level > 0 ? `${"#".repeat(level)} ` : ""}${heading} (offset ${offset})`;
|
|
262
|
+
const partNote = ({ markdownRange: range, outline = [] }, url) => {
|
|
263
|
+
if (!range) {
|
|
264
|
+
return null;
|
|
265
|
+
}
|
|
266
|
+
const lines = [
|
|
267
|
+
`This is part of the page: characters ${range.start} to ${range.end} of ${range.totalCharacters}.`,
|
|
268
|
+
];
|
|
269
|
+
if (range.nextOffset !== null) {
|
|
270
|
+
lines.push(`For the next part, run: extraktor extract ${quoteArg(url)} --offset ${range.nextOffset}`, "Other parts and --find use the saved result: no new request and no credit. To summarize the complete page, use --summary. To save it to a file, use --save <file>.");
|
|
271
|
+
}
|
|
272
|
+
if (outline.length > 0) {
|
|
273
|
+
lines.push("", "Page outline. To read a section, use its offset with --offset:", ...outline.map((entry) => outlineLine(entry)));
|
|
274
|
+
}
|
|
275
|
+
return lines.join("\n");
|
|
276
|
+
};
|
|
277
|
+
const foundSections = ({ found, markdown }) => found.matchedBy && markdown.trim()
|
|
278
|
+
? `## Matches for "${found.query}"\n\n${markdown.trim()}`
|
|
279
|
+
: null;
|
|
280
|
+
const foundNote = (results, url) => {
|
|
281
|
+
const read = `extraktor extract ${quoteArg(url)} --offset <offset>`;
|
|
282
|
+
const lines = [];
|
|
283
|
+
for (const { found, outline = [] } of results) {
|
|
284
|
+
if (lines.length > 0) {
|
|
285
|
+
lines.push("");
|
|
286
|
+
}
|
|
287
|
+
if (!found.matchedBy) {
|
|
288
|
+
lines.push(`No match for "${found.query}". Try a shorter or different text, or read a section: ${read}`);
|
|
289
|
+
if (outline.length > 0) {
|
|
290
|
+
lines.push("", "Page outline:", ...outline.map((entry) => outlineLine(entry)));
|
|
291
|
+
}
|
|
292
|
+
continue;
|
|
293
|
+
}
|
|
294
|
+
lines.push(`Sections that match "${found.query}". Large sections show only the matched paragraphs, lines and table rows. To read all of a section, run: ${read}`, ...found.sections.map((entry) => outlineLine(entry)));
|
|
295
|
+
if (found.omittedSections.length > 0) {
|
|
296
|
+
lines.push("More matched sections that did not fit. Read them with --offset:", ...found.omittedSections.map((entry) => outlineLine(entry)));
|
|
297
|
+
}
|
|
298
|
+
}
|
|
299
|
+
return lines.join("\n");
|
|
300
|
+
};
|
|
301
|
+
const textSections = (page, view, url) => {
|
|
302
|
+
if (view.kind === "saved") {
|
|
303
|
+
return {
|
|
304
|
+
text: [
|
|
305
|
+
`## Page text\n\nSaved the complete page text (${page.markdown.length} characters) as Markdown to ${view.file}.`,
|
|
306
|
+
],
|
|
307
|
+
note: view.outline.length > 0
|
|
308
|
+
? [
|
|
309
|
+
"Page outline:",
|
|
310
|
+
...view.outline.map((entry) => outlineLine(entry)),
|
|
311
|
+
].join("\n")
|
|
312
|
+
: null,
|
|
313
|
+
};
|
|
314
|
+
}
|
|
315
|
+
if (view.kind === "found") {
|
|
316
|
+
const text = [];
|
|
317
|
+
for (const result of view.results) {
|
|
318
|
+
const section = foundSections(result);
|
|
319
|
+
if (section) {
|
|
320
|
+
text.push(section);
|
|
321
|
+
}
|
|
322
|
+
}
|
|
323
|
+
return { text, note: foundNote(view.results, url) };
|
|
324
|
+
}
|
|
325
|
+
return {
|
|
326
|
+
text: [`## Page text\n\n${view.part.markdown.trim()}`],
|
|
327
|
+
note: partNote(view.part, url),
|
|
328
|
+
};
|
|
329
|
+
};
|
|
330
|
+
const screenshotSection = (screenshot, { baseUrl, screenshotFile }) => {
|
|
331
|
+
if (screenshot.status !== "success") {
|
|
332
|
+
return failedOutput(screenshot);
|
|
333
|
+
}
|
|
334
|
+
return [
|
|
335
|
+
screenshotFile
|
|
336
|
+
? `Saved to ${screenshotFile} (${screenshot.bytes} bytes).`
|
|
337
|
+
: "The CLI did not save the image.",
|
|
338
|
+
`Copy on the server (needs sign-in): ${new URL(screenshot.url, baseUrl).href}`,
|
|
339
|
+
].join("\n");
|
|
340
|
+
};
|
|
341
|
+
const MAX_FACT_CHARACTERS = 160;
|
|
342
|
+
const MAX_LIST_ITEMS = 5;
|
|
343
|
+
const MAX_NESTED_OBJECTS = 3;
|
|
344
|
+
const MAX_FACT_DEPTH = 2;
|
|
345
|
+
const MAX_FACTS_PER_ITEM = 30;
|
|
346
|
+
const SKIPPED_KEYS = new Set(["@context", "@id"]);
|
|
347
|
+
const WHITESPACE_RUN = /\s+/gu;
|
|
348
|
+
const scalarText = (value) => {
|
|
349
|
+
const text = String(value).replaceAll(WHITESPACE_RUN, " ").trim();
|
|
350
|
+
return text.length > MAX_FACT_CHARACTERS
|
|
351
|
+
? `${text.slice(0, MAX_FACT_CHARACTERS)}…`
|
|
352
|
+
: text;
|
|
353
|
+
};
|
|
354
|
+
const isScalar = (value) => !(value instanceof Object);
|
|
355
|
+
/**
|
|
356
|
+
* schema.org data as short "path: value" lines. Long lists, for example
|
|
357
|
+
* reviews or recipe steps, show only their count. --json has all of it.
|
|
358
|
+
*/
|
|
359
|
+
const factLines = (value, path, depth) => {
|
|
360
|
+
if (value === null || value === "") {
|
|
361
|
+
return [];
|
|
362
|
+
}
|
|
363
|
+
if (isScalar(value)) {
|
|
364
|
+
return [`- ${path}: ${scalarText(value)}`];
|
|
365
|
+
}
|
|
366
|
+
if (Array.isArray(value)) {
|
|
367
|
+
if (value.every((item) => isScalar(item))) {
|
|
368
|
+
const items = value
|
|
369
|
+
.filter((item) => item !== null)
|
|
370
|
+
.map((item) => scalarText(item));
|
|
371
|
+
const more = items.length - MAX_LIST_ITEMS;
|
|
372
|
+
return items.length === 0
|
|
373
|
+
? []
|
|
374
|
+
: [
|
|
375
|
+
`- ${path}: ${items.slice(0, MAX_LIST_ITEMS).join("; ")}${more > 0 ? ` (and ${more} more)` : ""}`,
|
|
376
|
+
];
|
|
377
|
+
}
|
|
378
|
+
if (value.length > MAX_NESTED_OBJECTS || depth >= MAX_FACT_DEPTH) {
|
|
379
|
+
return [`- ${path}: ${value.length} items`];
|
|
380
|
+
}
|
|
381
|
+
return value.flatMap((item) => factLines(item, path, depth + 1));
|
|
382
|
+
}
|
|
383
|
+
if (depth >= MAX_FACT_DEPTH) {
|
|
384
|
+
return [];
|
|
385
|
+
}
|
|
386
|
+
// Plain values first. A nested @type only names the kind of object.
|
|
387
|
+
const entries = Object.entries(value)
|
|
388
|
+
.filter(([key]) => !SKIPPED_KEYS.has(key) && (depth === 0 || key !== "@type"))
|
|
389
|
+
.toSorted(([, a], [, b]) => Number(!isScalar(a)) - Number(!isScalar(b)));
|
|
390
|
+
return entries.flatMap(([key, item]) => factLines(item, path ? `${path}.${key}` : key, depth + 1));
|
|
391
|
+
};
|
|
392
|
+
const structuredDataSection = ({ structuredData }) => {
|
|
393
|
+
if (!structuredData?.length) {
|
|
394
|
+
return null;
|
|
395
|
+
}
|
|
396
|
+
const items = structuredData.map((item) => {
|
|
397
|
+
const lines = factLines(item, "", 0);
|
|
398
|
+
const more = lines.length - MAX_FACTS_PER_ITEM;
|
|
399
|
+
return [
|
|
400
|
+
...lines.slice(0, MAX_FACTS_PER_ITEM),
|
|
401
|
+
...(more > 0 ? [`- (${more} more facts in --json)`] : []),
|
|
402
|
+
].join("\n");
|
|
403
|
+
});
|
|
404
|
+
return `## Structured data (schema.org)\n\nThe site gives these facts to search engines. They can be different from the visible page.\n\n${items.join("\n\n")}`;
|
|
405
|
+
};
|
|
406
|
+
const designSection = (design, designFile) => {
|
|
407
|
+
const body = renderMarkdown(design);
|
|
408
|
+
return designFile && design.status === "success"
|
|
409
|
+
? `Saved DESIGN.md to ${designFile}.\n\n${body}`
|
|
410
|
+
: body;
|
|
411
|
+
};
|
|
412
|
+
/** The requested outputs, each in its section. */
|
|
413
|
+
const outputBlocks = (page, options) => {
|
|
414
|
+
const blocks = [];
|
|
415
|
+
for (const [heading, body] of outputSections(page)) {
|
|
416
|
+
if (body) {
|
|
417
|
+
blocks.push(`## ${heading}\n\n${body}`);
|
|
418
|
+
}
|
|
419
|
+
}
|
|
420
|
+
if (page.design) {
|
|
421
|
+
blocks.push(`## Design (DESIGN.md)\n\n${designSection(page.design, options.designFile)}`);
|
|
422
|
+
}
|
|
423
|
+
if (page.screenshot) {
|
|
424
|
+
blocks.push(`## Screenshot\n\n${screenshotSection(page.screenshot, options)}`);
|
|
425
|
+
}
|
|
426
|
+
return blocks;
|
|
427
|
+
};
|
|
428
|
+
const totalLength = (blocks) => blocks.reduce((total, block) => total + (block?.length ?? 0), 0);
|
|
429
|
+
/** Reads the extract result. Throws when the server sent a different form. */
|
|
430
|
+
export const parsePage = (result) => parseResult(extractSchema, result);
|
|
431
|
+
/** Formats an extract result as Markdown. */
|
|
432
|
+
export const formatExtract = (result, options) => {
|
|
433
|
+
const page = parseResult(extractSchema, result);
|
|
434
|
+
const outputs = outputBlocks(page, options);
|
|
435
|
+
const data = structuredDataSection(page);
|
|
436
|
+
// The outputs and facts come first. The page text gets the rest.
|
|
437
|
+
const available = options.maxCharacters - HEADER_CHARACTERS - totalLength([data, ...outputs]);
|
|
438
|
+
const viewWith = (characters) => viewText(page.markdown, options, Math.max(MIN_TEXT_CHARACTERS, characters));
|
|
439
|
+
let view = viewWith(available);
|
|
440
|
+
// The note after the page text lists offsets. Make room for it.
|
|
441
|
+
const { note } = textSections(page, view, options.url);
|
|
442
|
+
if (note) {
|
|
443
|
+
view = viewWith(available - note.length);
|
|
444
|
+
}
|
|
445
|
+
const sections = [header(page, view)];
|
|
446
|
+
const warnings = resultWarnings(result);
|
|
447
|
+
if (warnings.length > 0) {
|
|
448
|
+
sections.push(warnings
|
|
449
|
+
.map(({ code, message, guidance }) => `> Warning (${code}): ${message} ${guidance ?? ""}`.trim())
|
|
450
|
+
.join("\n"));
|
|
451
|
+
}
|
|
452
|
+
sections.push(...outputs);
|
|
453
|
+
if (data) {
|
|
454
|
+
sections.push(data);
|
|
455
|
+
}
|
|
456
|
+
const text = textSections(page, view, options.url);
|
|
457
|
+
sections.push(...text.text);
|
|
458
|
+
if (text.note) {
|
|
459
|
+
sections.push(`---\n\n${text.note}`);
|
|
460
|
+
}
|
|
461
|
+
return `${sections.join("\n\n")}\n`;
|
|
462
|
+
};
|
|
463
|
+
/** Formats several pages, each after a line that names it. */
|
|
464
|
+
export const formatPages = (pages, options) => pages
|
|
465
|
+
.map((page, index) => {
|
|
466
|
+
const title = `===== Page ${index + 1} of ${pages.length}: ${page.url} =====`;
|
|
467
|
+
if ("error" in page) {
|
|
468
|
+
const { message, guidance } = page.error;
|
|
469
|
+
return `${title}\n\nError: ${message}${guidance ? `\n${guidance}` : ""}\n`;
|
|
470
|
+
}
|
|
471
|
+
return `${title}\n\n${formatExtract(page.result, {
|
|
472
|
+
...options,
|
|
473
|
+
url: page.url,
|
|
474
|
+
screenshotFile: page.screenshotFile,
|
|
475
|
+
designFile: page.designFile,
|
|
476
|
+
})}`;
|
|
477
|
+
})
|
|
478
|
+
.join("\n");
|
|
479
|
+
const ORGANIC = "organic_results";
|
|
480
|
+
/** Fields of a SERP item that name it, or link to it; the others are details. */
|
|
481
|
+
const ITEM_NAMES = ["title", "question", "query", "name"];
|
|
482
|
+
const PIXELS_ABOVE_FOLD = 600;
|
|
483
|
+
/** A detail of a SERP item as text, for example "4.5 stars". */
|
|
484
|
+
const serpDetail = (field, value) => {
|
|
485
|
+
if (Array.isArray(value)) {
|
|
486
|
+
const parts = value.flatMap((part) => serpText(part) ?? []);
|
|
487
|
+
return parts.length > 0 ? parts.join(", ") : null;
|
|
488
|
+
}
|
|
489
|
+
const text = serpText(value);
|
|
490
|
+
if (text && field === "rating") {
|
|
491
|
+
return `${text} stars`;
|
|
492
|
+
}
|
|
493
|
+
return text && field === "reviews" ? `(${text} reviews)` : text;
|
|
494
|
+
};
|
|
495
|
+
/** One SERP item on one line, then its link: name · details. */
|
|
496
|
+
const serpItem = (item, indent) => {
|
|
497
|
+
const name = ITEM_NAMES.map((field) => serpText(item[field])).find(Boolean);
|
|
498
|
+
const details = Object.entries(item).flatMap(([field, value]) => ITEM_NAMES.includes(field) || field === "link"
|
|
499
|
+
? []
|
|
500
|
+
: (serpDetail(field, value) ?? []));
|
|
501
|
+
const link = serpText(item.link);
|
|
502
|
+
return [
|
|
503
|
+
`${indent}- ${[name ?? "(no name)", ...details].join(" · ")}`,
|
|
504
|
+
link && `${indent} ${link}`,
|
|
505
|
+
]
|
|
506
|
+
.filter(Boolean)
|
|
507
|
+
.join("\n");
|
|
508
|
+
};
|
|
509
|
+
/** One line for a block of the results page: its name, size and place. */
|
|
510
|
+
const serpLine = (block, index) => {
|
|
511
|
+
const { feature, y, column, ranks, count, deferred } = block;
|
|
512
|
+
const name = ranks
|
|
513
|
+
? `${feature} #${ranks[0]}${ranks.length > 1 ? `–#${ranks.at(-1)}` : ""}`
|
|
514
|
+
: `${feature}${count === undefined ? "" : ` (${count})`}`;
|
|
515
|
+
const place = [
|
|
516
|
+
y === undefined ? null : `y ${y} px`,
|
|
517
|
+
column ? `${column} column` : null,
|
|
518
|
+
deferred ? "text not available" : null,
|
|
519
|
+
].filter(Boolean);
|
|
520
|
+
return `${index + 1}. ${name}${place.length > 0 ? `, ${place.join(", ")}` : ""}`;
|
|
521
|
+
};
|
|
522
|
+
/** The content of a block, indented under its line. */
|
|
523
|
+
const serpContent = (block) => {
|
|
524
|
+
const indent = " ";
|
|
525
|
+
const lines = [];
|
|
526
|
+
if (block.deferred) {
|
|
527
|
+
lines.push(`${indent}Google loads this AI Overview separately. Its text is not available, and a new search does not give it.`);
|
|
528
|
+
}
|
|
529
|
+
const head = [block.title, block.type, serpText(block.source)]
|
|
530
|
+
.filter(Boolean)
|
|
531
|
+
.join(" · ");
|
|
532
|
+
const text = block.text ?? block.description;
|
|
533
|
+
lines.push(...[head, block.link, block.website].flatMap((line) => line ? [`${indent}${line}`] : []), ...(text?.split("\n").map((line) => `${indent}${line}`) ?? []), ...(block.list?.map((item) => `${indent}- ${item}`) ?? []));
|
|
534
|
+
if (block.sources?.length) {
|
|
535
|
+
lines.push(`${indent}Sources:`, ...block.sources.map((item) => serpItem(item, indent)));
|
|
536
|
+
}
|
|
537
|
+
lines.push(...(block.items ?? []).map((item) => serpItem(item, indent)));
|
|
538
|
+
return lines;
|
|
539
|
+
};
|
|
540
|
+
const serpSection = (blocks, withContent) => {
|
|
541
|
+
if (!blocks.some(({ feature }) => feature !== ORGANIC)) {
|
|
542
|
+
return null;
|
|
543
|
+
}
|
|
544
|
+
const lines = blocks.flatMap((block, index) => [
|
|
545
|
+
serpLine(block, index),
|
|
546
|
+
...(withContent && block.feature !== ORGANIC ? serpContent(block) : []),
|
|
547
|
+
]);
|
|
548
|
+
return `## Google results page (SERP), top first\n\n${lines.join("\n")}\n\ny is the distance from the top of a desktop page 1200 px wide. A block with y less than ${PIXELS_ABOVE_FOLD} shows without scrolling.`;
|
|
549
|
+
};
|
|
550
|
+
/** Formats a search result as Markdown. */
|
|
551
|
+
export const formatSearch = (result, query, withContent = false) => {
|
|
552
|
+
const data = parseResult(searchSchema, result);
|
|
553
|
+
const sections = [`# Search: ${query}`];
|
|
554
|
+
const serp = serpSection(data.serp, withContent);
|
|
555
|
+
// --serp asks for the SERP features, so they come first.
|
|
556
|
+
if (serp && withContent) {
|
|
557
|
+
sections.push(serp);
|
|
558
|
+
}
|
|
559
|
+
const results = data.organic_results.map(({ title, link, snippet, date }, index) => [
|
|
560
|
+
`${index + 1}. ${title || "(no title)"}`,
|
|
561
|
+
link && ` ${link}`,
|
|
562
|
+
date && ` Date: ${date}`,
|
|
563
|
+
snippet && ` ${snippet}`,
|
|
564
|
+
]
|
|
565
|
+
.filter(Boolean)
|
|
566
|
+
.join("\n"));
|
|
567
|
+
sections.push(`## Results\n\n${results.length > 0 ? results.join("\n\n") : "No results."}`);
|
|
568
|
+
if (serp && !withContent) {
|
|
569
|
+
sections.push(serp);
|
|
570
|
+
}
|
|
571
|
+
const next = ["To read a result, run: extraktor extract <link>"];
|
|
572
|
+
const snippetLink = data.serp.find(({ feature }) => feature === "featured_snippet")?.link;
|
|
573
|
+
if (snippetLink) {
|
|
574
|
+
next.push(`The featured snippet quotes ${snippetLink}. To read the complete answer, run: extraktor extract ${quoteArg(snippetLink)}`);
|
|
575
|
+
}
|
|
576
|
+
if (withContent) {
|
|
577
|
+
const links = data.organic_results
|
|
578
|
+
.slice(0, 3)
|
|
579
|
+
.flatMap(({ link }) => (link ? [quoteArg(link)] : []));
|
|
580
|
+
if (links.length > 1) {
|
|
581
|
+
next.push(`To compare the content of the top results, run: extraktor extract ${links.join(" ")}`);
|
|
582
|
+
}
|
|
583
|
+
}
|
|
584
|
+
else if (serp) {
|
|
585
|
+
next.push("Only if the task needs the content of the SERP features, run the search again with --serp.");
|
|
586
|
+
}
|
|
587
|
+
sections.push(next.join("\n"));
|
|
588
|
+
return `${sections.join("\n\n")}\n`;
|
|
589
|
+
};
|