@extraktor/cli 0.0.0-stage → 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/mcp.js ADDED
@@ -0,0 +1,135 @@
1
+ /**
2
+ * The smallest MCP client that the CLI needs: one stateless `tools/call`
3
+ * request to the Extraktor server. The server needs no `initialize` handshake,
4
+ * so each command costs one round trip.
5
+ */
6
+ import { subscribe, unsubscribe } from "node:diagnostics_channel";
7
+ import { CliError, EXIT } from "./errors.js";
8
+ import { VERSION } from "./version.js";
9
+ export const PROTOCOL_VERSION = "2026-07-28";
10
+ const REQUEST_TIMEOUT_MS = 180_000;
11
+ // Node's fetch (undici) publishes this event when the request body is sent.
12
+ const BODY_SENT = "undici:request:bodySent";
13
+ /**
14
+ * Loads the zod schemas (about 20 ms of CPU) after fetch has sent the request.
15
+ * zod then loads while the server works. If it loaded before the request or
16
+ * between the headers and the body, the request would leave later.
17
+ */
18
+ const loadSchemasAfterSend = () => {
19
+ let loading;
20
+ const onSent = () => {
21
+ unsubscribe(BODY_SENT, onSent);
22
+ loading ??= import("./schemas.js");
23
+ };
24
+ subscribe(BODY_SENT, onSent);
25
+ return {
26
+ load: () => {
27
+ onSent();
28
+ return loading ?? import("./schemas.js");
29
+ },
30
+ cancel: () => unsubscribe(BODY_SENT, onSent),
31
+ };
32
+ };
33
+ const unexpectedResponse = () => new CliError("The server response has an unexpected form.", EXIT.failure, "Update the CLI: npm install --global @extraktor/cli@latest", "UNEXPECTED_RESPONSE");
34
+ const ACCESS_CODES = new Set(["BILLING_REQUIRED"]);
35
+ const failureError = (failure) => new CliError(`${failure.message} (${failure.code})`, ACCESS_CODES.has(failure.code) ? EXIT.access : EXIT.failure, failure.guidance, failure.code);
36
+ /** Reads the JSON-RPC response from a JSON body or from server-sent events. */
37
+ const readResponse = async (response, id, { jsonRpcSchema, parseJson }) => {
38
+ const text = await response.text();
39
+ const messages = response.headers
40
+ .get("content-type")
41
+ ?.includes("text/event-stream")
42
+ ? text
43
+ .split("\n")
44
+ .filter((line) => line.startsWith("data:"))
45
+ .map((line) => line.slice("data:".length))
46
+ : [text];
47
+ for (const message of messages.toReversed()) {
48
+ const parsed = jsonRpcSchema.safeParse(parseJson(message));
49
+ if (parsed.success && (parsed.data.id === id || messages.length === 1)) {
50
+ return parsed.data;
51
+ }
52
+ }
53
+ throw unexpectedResponse();
54
+ };
55
+ const SIGN_IN_GUIDANCE = 'Run "extraktor login", or set EXTRAKTOR_API_KEY to an API key from the developers page.';
56
+ const httpError = async (response, { httpErrorSchema, parseJson }) => {
57
+ const description = httpErrorSchema.safeParse(parseJson(await response.text())).data
58
+ ?.error_description ?? null;
59
+ if (response.status === 401 || response.status === 403) {
60
+ return new CliError(description ?? "Sign-in is necessary.", EXIT.access, SIGN_IN_GUIDANCE, "SIGN_IN_REQUIRED");
61
+ }
62
+ return new CliError(`The server returned HTTP ${response.status}.`, EXIT.failure, response.status >= 500
63
+ ? "Try again one time. If it fails again, report the failure."
64
+ : undefined, `HTTP_${response.status}`);
65
+ };
66
+ /** Calls one Extraktor MCP tool. Throws a CliError when the call fails. */
67
+ export const callTool = async ({ baseUrl, apiKey, name, args, agent, }) => {
68
+ const id = 1;
69
+ const schemasLoader = loadSchemasAfterSend();
70
+ let response;
71
+ try {
72
+ const headers = new Headers({
73
+ Accept: "application/json, text/event-stream",
74
+ Authorization: `Bearer ${apiKey}`,
75
+ "Content-Type": "application/json",
76
+ "MCP-Protocol-Version": PROTOCOL_VERSION,
77
+ "Mcp-Method": "tools/call",
78
+ "Mcp-Name": name,
79
+ "User-Agent": `extraktor-cli/${VERSION}`,
80
+ });
81
+ if (agent) {
82
+ headers.set("X-Extraktor-Agent", agent);
83
+ }
84
+ if (name === "extract") {
85
+ // The CLI makes the parts and finds text itself, from one result.
86
+ headers.set("X-Extraktor-Text", "complete");
87
+ }
88
+ response = await fetch(new URL("/mcp", baseUrl), {
89
+ method: "POST",
90
+ headers,
91
+ body: JSON.stringify({
92
+ jsonrpc: "2.0",
93
+ id,
94
+ method: "tools/call",
95
+ params: {
96
+ name,
97
+ arguments: args,
98
+ _meta: {
99
+ "io.modelcontextprotocol/protocolVersion": PROTOCOL_VERSION,
100
+ "io.modelcontextprotocol/clientCapabilities": {},
101
+ },
102
+ },
103
+ }),
104
+ signal: AbortSignal.timeout(REQUEST_TIMEOUT_MS),
105
+ });
106
+ }
107
+ catch (error) {
108
+ schemasLoader.cancel();
109
+ const timedOut = error instanceof Error && error.name === "TimeoutError";
110
+ throw new CliError(timedOut
111
+ ? "The server did not reply in time."
112
+ : `Extraktor at ${baseUrl} is not available.`, EXIT.failure, timedOut
113
+ ? "Try again one time."
114
+ : "Make sure that the computer is online. Check EXTRAKTOR_URL if you set it.", timedOut ? "TIMEOUT" : "SERVER_UNAVAILABLE");
115
+ }
116
+ const schemas = await schemasLoader.load();
117
+ if (!response.ok) {
118
+ throw await httpError(response, schemas);
119
+ }
120
+ const message = await readResponse(response, id, schemas);
121
+ if (message.error) {
122
+ throw new CliError(message.error.message, EXIT.failure);
123
+ }
124
+ const { result } = message;
125
+ if (!result) {
126
+ throw unexpectedResponse();
127
+ }
128
+ if (result.isError) {
129
+ const failure = schemas.parseFailure(result.content[0]?.text);
130
+ throw failure
131
+ ? failureError(failure)
132
+ : new CliError("The tool call did not complete.", EXIT.failure);
133
+ }
134
+ return result;
135
+ };
@@ -0,0 +1,386 @@
1
+ /**
2
+ * Page text in parts, and the sections that match a find. The MCP server uses
3
+ * it for each result. The CLI uses it on the complete page text that it keeps,
4
+ * so that it reads other parts and finds text without a new request.
5
+ */
6
+ /** Readers can stop at this heading: after it come the site's menus. */
7
+ export const SITE_NAVIGATION_HEADING = "Site navigation";
8
+ /** The heading as the Markdown shows it. */
9
+ export const SITE_NAVIGATION_LINE = `## ${SITE_NAVIGATION_HEADING}`;
10
+ // Codex 0.149 gives the model at most about 48,000 bytes of serialized
11
+ // structured content and silently removes the middle of a larger result.
12
+ // Agents then repeat the same read, which returns the same cut. Keep each
13
+ // result below that limit, with room for serialization differences.
14
+ export const MAX_RESULT_BYTES = 44_000;
15
+ const MAX_OUTLINE_ENTRIES = 40;
16
+ const MAX_HEADING_CHARACTERS = 120;
17
+ const HEADING = /^(?<marks>#{1,6})\s+(?<text>.+?)\s*#*\s*$/u;
18
+ const FENCE = /^\s*(?:```|~~~)/u;
19
+ const jsonBytes = (value) => new TextEncoder().encode(JSON.stringify(value)).length;
20
+ const readOutline = (markdown) => {
21
+ const entries = [];
22
+ let offset = 0;
23
+ let inFence = false;
24
+ for (const line of markdown.split("\n")) {
25
+ if (FENCE.test(line)) {
26
+ inFence = !inFence;
27
+ }
28
+ else if (!inFence) {
29
+ const heading = HEADING.exec(line)?.groups;
30
+ if (heading) {
31
+ entries.push({
32
+ level: heading.marks.length,
33
+ heading: heading.text.slice(0, MAX_HEADING_CHARACTERS),
34
+ offset,
35
+ });
36
+ }
37
+ }
38
+ offset += line.length + 1;
39
+ }
40
+ return entries;
41
+ };
42
+ /**
43
+ * Keep the broadest heading levels that fit; a long reference can have
44
+ * hundreds. When even the broadest level has too many, keep evenly spaced
45
+ * entries of it, so the outline covers the whole page, for example every
46
+ * twelfth "Page N" heading of a long PDF.
47
+ */
48
+ const limitOutline = (entries) => {
49
+ for (let level = 6; level >= 1; level -= 1) {
50
+ const kept = entries.filter((entry) => entry.level <= level);
51
+ if (kept.length === 0) {
52
+ break;
53
+ }
54
+ if (kept.length <= MAX_OUTLINE_ENTRIES) {
55
+ return kept;
56
+ }
57
+ }
58
+ if (entries.length === 0) {
59
+ return [];
60
+ }
61
+ const broadest = Math.min(...entries.map((entry) => entry.level));
62
+ const top = entries.filter((entry) => entry.level === broadest);
63
+ const step = top.length / MAX_OUTLINE_ENTRIES;
64
+ return Array.from({ length: MAX_OUTLINE_ENTRIES }, (_, index) => top[Math.floor(index * step)]).filter((entry) => entry !== undefined);
65
+ };
66
+ /** The headings of the page with their offsets, at most 40. */
67
+ export const pageOutline = (markdown) => limitOutline(readOutline(markdown));
68
+ /** Ends on a line break when possible and never splits a surrogate pair. */
69
+ const sliceEnd = (markdown, start, budget) => {
70
+ let used = 0;
71
+ let end = start;
72
+ while (end < markdown.length) {
73
+ const lineBreak = markdown.indexOf("\n", end);
74
+ const lineEnd = lineBreak === -1 ? markdown.length : lineBreak + 1;
75
+ const cost = jsonBytes(markdown.slice(end, lineEnd)) - 2;
76
+ if (used + cost > budget) {
77
+ break;
78
+ }
79
+ used += cost;
80
+ end = lineEnd;
81
+ }
82
+ if (end > start || end === markdown.length) {
83
+ return end;
84
+ }
85
+ // A single line is larger than the budget: cut it by characters.
86
+ while (end < markdown.length) {
87
+ const width = (markdown.codePointAt(end) ?? 0) > 0xff_ff ? 2 : 1;
88
+ const cost = jsonBytes(markdown.slice(end, end + width)) - 2;
89
+ if (used + cost > budget) {
90
+ break;
91
+ }
92
+ used += cost;
93
+ end += width;
94
+ }
95
+ return end;
96
+ };
97
+ const rangeMessage = (start, end, total, skippedPreface) => [
98
+ `This is one part of the page text: Markdown characters ${start}–${end} of ${total}.`,
99
+ skippedPreface
100
+ ? "This part starts at the first main heading. The text before it is usually site navigation. To get that text, send offset 0."
101
+ : "",
102
+ "Use this part if it is sufficient for the task.",
103
+ "To read a different part, call extract with the same url and an offset from nextOffset or outline.",
104
+ "The same request returns the same part again.",
105
+ ]
106
+ .filter(Boolean)
107
+ .join(" ");
108
+ // A matched section up to this size comes back complete, for context.
109
+ const MAX_WHOLE_SECTION_CHARACTERS = 2500;
110
+ // A larger matched block comes back as its matched lines.
111
+ const MAX_WHOLE_BLOCK_CHARACTERS = 1200;
112
+ const MIN_FIND_WORD_LENGTH = 2;
113
+ const BLANK_LINE = /\n[ \t]*\n/u;
114
+ const TABLE_ROW = /^\s*\|/u;
115
+ const FIND_WORD = /[\p{L}\p{N}]+/gu;
116
+ /** Splits the page at each heading. Text before the first heading has level 0. */
117
+ const readSections = (markdown) => {
118
+ const sections = [];
119
+ let current = { level: 0, heading: "", offset: 0, text: "" };
120
+ let offset = 0;
121
+ let inFence = false;
122
+ for (const line of markdown.split("\n")) {
123
+ if (FENCE.test(line)) {
124
+ inFence = !inFence;
125
+ }
126
+ const heading = inFence ? undefined : HEADING.exec(line)?.groups;
127
+ if (heading) {
128
+ sections.push(current);
129
+ current = {
130
+ level: heading.marks.length,
131
+ heading: heading.text.slice(0, MAX_HEADING_CHARACTERS),
132
+ offset,
133
+ text: "",
134
+ };
135
+ }
136
+ current.text += `${line}\n`;
137
+ offset += line.length + 1;
138
+ }
139
+ sections.push(current);
140
+ return sections.filter((section) => section.text.trim());
141
+ };
142
+ /** Splits section text into blocks at blank lines. A code fence stays one block. */
143
+ const readBlocks = (text) => {
144
+ const blocks = [];
145
+ let inFence = false;
146
+ let block = "";
147
+ for (const part of text.split(BLANK_LINE)) {
148
+ block = block ? `${block}\n\n${part}` : part;
149
+ for (const line of part.split("\n")) {
150
+ if (FENCE.test(line)) {
151
+ inFence = !inFence;
152
+ }
153
+ }
154
+ if (!inFence) {
155
+ blocks.push(block);
156
+ block = "";
157
+ }
158
+ }
159
+ if (block) {
160
+ blocks.push(block);
161
+ }
162
+ return blocks;
163
+ };
164
+ const phraseMatcher = (query) => (text) => text.toLowerCase().includes(query.toLowerCase());
165
+ const wordsMatcher = (query) => {
166
+ const words = [...new Set(query.toLowerCase().match(FIND_WORD))].filter((word) => word.length >= MIN_FIND_WORD_LENGTH);
167
+ if (words.length < 2) {
168
+ return null;
169
+ }
170
+ return (text) => {
171
+ const lower = text.toLowerCase();
172
+ return words.every((word) => lower.includes(word));
173
+ };
174
+ };
175
+ /** The matched lines of a large block, with one line of context, or the table header and matched rows. */
176
+ const trimBlock = (block, matches) => {
177
+ const lines = block.split("\n");
178
+ if (lines.every((line) => TABLE_ROW.test(line))) {
179
+ const header = lines.slice(0, 2);
180
+ const rows = lines.slice(2).filter((line) => matches(line));
181
+ return [...header, ...rows].join("\n");
182
+ }
183
+ const keep = new Set();
184
+ for (const [index, line] of lines.entries()) {
185
+ if (matches(line)) {
186
+ keep
187
+ .add(index - 1)
188
+ .add(index)
189
+ .add(index + 1);
190
+ }
191
+ }
192
+ if (keep.size === 0) {
193
+ // The words match across lines: the block is the smallest unit.
194
+ return block.slice(0, MAX_WHOLE_BLOCK_CHARACTERS);
195
+ }
196
+ const kept = [];
197
+ let last = -1;
198
+ for (const [index, line] of lines.entries()) {
199
+ if (keep.has(index)) {
200
+ if (last !== -1 && index > last + 1) {
201
+ kept.push("…");
202
+ }
203
+ kept.push(line);
204
+ last = index;
205
+ }
206
+ }
207
+ return kept.join("\n");
208
+ };
209
+ /** The text of a matched section: all of it when it is short, else its matched blocks. */
210
+ const matchedText = (section, matches) => {
211
+ if (section.text.length <= MAX_WHOLE_SECTION_CHARACTERS) {
212
+ return section.text.trimEnd();
213
+ }
214
+ const [first = "", ...rest] = readBlocks(section.text);
215
+ // Keep the heading line of the section for context.
216
+ const headingLine = section.level > 0 ? first.split("\n")[0] : null;
217
+ const parts = [];
218
+ for (const block of [first, ...rest]) {
219
+ if (matches(block)) {
220
+ parts.push(block.length <= MAX_WHOLE_BLOCK_CHARACTERS
221
+ ? block
222
+ : trimBlock(block, matches));
223
+ }
224
+ }
225
+ if (headingLine && !parts[0]?.startsWith(headingLine)) {
226
+ parts.unshift(headingLine);
227
+ }
228
+ return parts.join("\n\n…\n\n");
229
+ };
230
+ const toFound = ({ level, heading, offset }) => ({
231
+ level,
232
+ heading: heading || "(start of page)",
233
+ offset,
234
+ });
235
+ const findMessage = (query, matchedBy, shown, omitted) => {
236
+ if (!matchedBy) {
237
+ return `The page text does not contain "${query}". Use a shorter or different phrase, or read a section with offset from the outline. Do not send the same find again.`;
238
+ }
239
+ const sections = shown === 1 ? "This section" : `These ${shown} sections`;
240
+ return [
241
+ matchedBy === "phrase"
242
+ ? `${sections} of the page text contain "${query}".`
243
+ : `The exact phrase "${query}" is not in the page text. ${sections} have all of its words.`,
244
+ "Large sections show only their matched paragraphs, lines or table rows.",
245
+ omitted > 0
246
+ ? `${omitted} more matched sections did not fit. Read them with offset from omittedSections.`
247
+ : "",
248
+ "To read all of a section, send its offset.",
249
+ ]
250
+ .filter(Boolean)
251
+ .join(" ");
252
+ };
253
+ /**
254
+ * Returns only the sections of the complete page text that contain the query,
255
+ * so an agent gets a fact from a large page in one call. Matching is
256
+ * deterministic: the exact phrase (any case), else all words of the query in
257
+ * one block. With `completeWhenNoMatch`, a page that fits in `maxBytes` comes
258
+ * back complete when nothing matches, so the agent needs no second call to
259
+ * check that the page does not state the fact.
260
+ */
261
+ export const findPageText = (result, query, maxBytes = MAX_RESULT_BYTES, { completeWhenNoMatch = false } = {}) => {
262
+ const sections = readSections(result.markdown);
263
+ const phrase = phraseMatcher(query);
264
+ const words = wordsMatcher(query);
265
+ let matches = phrase;
266
+ let matchedBy = "phrase";
267
+ let hits = sections.filter((section) => phrase(section.text));
268
+ if (hits.length === 0 && words) {
269
+ matches = words;
270
+ matchedBy = "words";
271
+ hits = sections.filter((section) => readBlocks(section.text).some((block) => words(block)));
272
+ }
273
+ if (hits.length === 0) {
274
+ matches = null;
275
+ matchedBy = null;
276
+ }
277
+ // The site menus come after the page content. Show their matches last.
278
+ const navigation = sections.find((section) => section.text.split("\n")[0] === SITE_NAVIGATION_LINE)?.offset;
279
+ if (navigation !== undefined) {
280
+ hits = [
281
+ ...hits.filter((section) => section.offset < navigation),
282
+ ...hits.filter((section) => section.offset >= navigation),
283
+ ];
284
+ }
285
+ const none = [];
286
+ const base = {
287
+ ...result,
288
+ markdown: "",
289
+ found: {
290
+ query,
291
+ matchedBy,
292
+ sections: none,
293
+ omittedSections: none,
294
+ message: findMessage(query, matchedBy, 0, 0),
295
+ },
296
+ };
297
+ if (!matches) {
298
+ const complete = {
299
+ ...result,
300
+ found: {
301
+ ...base.found,
302
+ message: `The page text does not contain "${query}" or all of its words. markdown has the complete page text. Use it to answer. Do not call extract again for this page.`,
303
+ },
304
+ };
305
+ return completeWhenNoMatch && jsonBytes(complete) <= maxBytes
306
+ ? complete
307
+ : { ...base, outline: limitOutline(readOutline(result.markdown)) };
308
+ }
309
+ // Each section can list as an omitted entry. Keep room for all of them.
310
+ let budget = maxBytes -
311
+ jsonBytes({
312
+ ...base,
313
+ found: { ...base.found, omittedSections: hits.map(toFound) },
314
+ });
315
+ const parts = [];
316
+ const shown = [];
317
+ const omitted = [];
318
+ for (const section of hits) {
319
+ let text = matchedText(section, matches);
320
+ let cost = jsonBytes(`${text}\n\n---\n\n`) - 2;
321
+ if (cost > budget && shown.length === 0) {
322
+ // Never return no text when there is a match: cut the first one to fit.
323
+ text = text.slice(0, sliceEnd(text, 0, budget - 16));
324
+ cost = jsonBytes(`${text}\n\n---\n\n`) - 2;
325
+ }
326
+ if (cost <= budget) {
327
+ parts.push(text);
328
+ shown.push(toFound(section));
329
+ budget -= cost;
330
+ }
331
+ else {
332
+ omitted.push(toFound(section));
333
+ }
334
+ }
335
+ return {
336
+ ...base,
337
+ markdown: parts.join("\n\n---\n\n"),
338
+ found: {
339
+ query,
340
+ matchedBy,
341
+ sections: shown,
342
+ omittedSections: omitted,
343
+ message: findMessage(query, matchedBy, shown.length, omitted.length),
344
+ },
345
+ };
346
+ };
347
+ /**
348
+ * Returns the complete result when it fits. Otherwise returns one part of the
349
+ * Markdown with its character range and the page outline, so another part can
350
+ * be requested explicitly instead of repeating the same read.
351
+ */
352
+ export const fitPageText = (result, offset, maxBytes = MAX_RESULT_BYTES) => {
353
+ if (offset === undefined && jsonBytes(result) <= maxBytes) {
354
+ return { ...result };
355
+ }
356
+ const { markdown } = result;
357
+ const total = markdown.length;
358
+ const outline = readOutline(markdown);
359
+ const firstMainHeading = outline.find((entry) => entry.level === 1)?.offset;
360
+ const skippedPreface = offset === undefined && (firstMainHeading ?? 0) > 0;
361
+ const start = Math.min(offset ?? (skippedPreface ? (firstMainHeading ?? 0) : 0), total);
362
+ const emptyPart = {
363
+ ...result,
364
+ markdown: "",
365
+ markdownRange: {
366
+ start: total,
367
+ end: total,
368
+ totalCharacters: total,
369
+ nextOffset: total,
370
+ message: rangeMessage(total, total, total, skippedPreface),
371
+ },
372
+ outline: limitOutline(outline),
373
+ };
374
+ const end = sliceEnd(markdown, start, Math.max(0, maxBytes - jsonBytes(emptyPart)));
375
+ return {
376
+ ...emptyPart,
377
+ markdown: markdown.slice(start, end),
378
+ markdownRange: {
379
+ start,
380
+ end,
381
+ totalCharacters: total,
382
+ nextOffset: end < total ? end : null,
383
+ message: rangeMessage(start, end, total, skippedPreface),
384
+ },
385
+ };
386
+ };