@octocrawl/sdk 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +17 -0
- package/dist/index.cjs +1278 -0
- package/dist/index.js +1240 -0
- package/dist/types/cjs/client.d.ts +362 -0
- package/dist/types/cjs/contracts/access.d.ts +166 -0
- package/dist/types/cjs/contracts/actions.d.ts +191 -0
- package/dist/types/cjs/contracts/api.d.ts +891 -0
- package/dist/types/cjs/contracts/benchmark.d.ts +116 -0
- package/dist/types/cjs/contracts/checkpoint.d.ts +164 -0
- package/dist/types/cjs/contracts/compliance.d.ts +412 -0
- package/dist/types/cjs/contracts/crawl.d.ts +302 -0
- package/dist/types/cjs/contracts/delivery.d.ts +136 -0
- package/dist/types/cjs/contracts/evidenceRecord.d.ts +192 -0
- package/dist/types/cjs/contracts/execution.d.ts +197 -0
- package/dist/types/cjs/contracts/extractor.d.ts +379 -0
- package/dist/types/cjs/contracts/file.d.ts +117 -0
- package/dist/types/cjs/contracts/firecrawl.d.ts +258 -0
- package/dist/types/cjs/contracts/groundTruth.d.ts +77 -0
- package/dist/types/cjs/contracts/identityBundle.d.ts +70 -0
- package/dist/types/cjs/contracts/index.d.ts +30 -0
- package/dist/types/cjs/contracts/map.d.ts +180 -0
- package/dist/types/cjs/contracts/monitor.d.ts +217 -0
- package/dist/types/cjs/contracts/monitorConfig.d.ts +9 -0
- package/dist/types/cjs/contracts/policy.d.ts +93 -0
- package/dist/types/cjs/contracts/proxy.d.ts +52 -0
- package/dist/types/cjs/contracts/recipe.d.ts +74 -0
- package/dist/types/cjs/contracts/regexSafety.d.ts +31 -0
- package/dist/types/cjs/contracts/result.d.ts +503 -0
- package/dist/types/cjs/contracts/session.d.ts +51 -0
- package/dist/types/cjs/contracts/ssrf.d.ts +16 -0
- package/dist/types/cjs/contracts/status.d.ts +27 -0
- package/dist/types/cjs/contracts/structured.d.ts +185 -0
- package/dist/types/cjs/contracts/tableMarkdown.d.ts +58 -0
- package/dist/types/cjs/contracts/tokens.d.ts +47 -0
- package/dist/types/cjs/index.d.ts +9 -0
- package/dist/types/cjs/package.json +1 -0
- package/dist/types/cjs/version.d.ts +8 -0
- package/dist/types/cjs/watcher.d.ts +151 -0
- package/dist/types/esm/client.d.ts +362 -0
- package/dist/types/esm/contracts/access.d.ts +166 -0
- package/dist/types/esm/contracts/actions.d.ts +191 -0
- package/dist/types/esm/contracts/api.d.ts +891 -0
- package/dist/types/esm/contracts/benchmark.d.ts +116 -0
- package/dist/types/esm/contracts/checkpoint.d.ts +164 -0
- package/dist/types/esm/contracts/compliance.d.ts +412 -0
- package/dist/types/esm/contracts/crawl.d.ts +302 -0
- package/dist/types/esm/contracts/delivery.d.ts +136 -0
- package/dist/types/esm/contracts/evidenceRecord.d.ts +192 -0
- package/dist/types/esm/contracts/execution.d.ts +197 -0
- package/dist/types/esm/contracts/extractor.d.ts +379 -0
- package/dist/types/esm/contracts/file.d.ts +117 -0
- package/dist/types/esm/contracts/firecrawl.d.ts +258 -0
- package/dist/types/esm/contracts/groundTruth.d.ts +77 -0
- package/dist/types/esm/contracts/identityBundle.d.ts +70 -0
- package/dist/types/esm/contracts/index.d.ts +30 -0
- package/dist/types/esm/contracts/map.d.ts +180 -0
- package/dist/types/esm/contracts/monitor.d.ts +217 -0
- package/dist/types/esm/contracts/monitorConfig.d.ts +9 -0
- package/dist/types/esm/contracts/policy.d.ts +93 -0
- package/dist/types/esm/contracts/proxy.d.ts +52 -0
- package/dist/types/esm/contracts/recipe.d.ts +74 -0
- package/dist/types/esm/contracts/regexSafety.d.ts +31 -0
- package/dist/types/esm/contracts/result.d.ts +503 -0
- package/dist/types/esm/contracts/session.d.ts +51 -0
- package/dist/types/esm/contracts/ssrf.d.ts +16 -0
- package/dist/types/esm/contracts/status.d.ts +27 -0
- package/dist/types/esm/contracts/structured.d.ts +185 -0
- package/dist/types/esm/contracts/tableMarkdown.d.ts +58 -0
- package/dist/types/esm/contracts/tokens.d.ts +47 -0
- package/dist/types/esm/index.d.ts +9 -0
- package/dist/types/esm/version.d.ts +8 -0
- package/dist/types/esm/watcher.d.ts +151 -0
- package/package.json +40 -0
|
@@ -0,0 +1,185 @@
|
|
|
1
|
+
import type { AdapterDescriptor, ExtractedEntity, ProductFactSource } from './extractor.js';
|
|
2
|
+
export type JsonPrimitive = string | number | boolean | null;
|
|
3
|
+
export type JsonValue = JsonPrimitive | JsonValue[] | {
|
|
4
|
+
[key: string]: JsonValue;
|
|
5
|
+
};
|
|
6
|
+
/**
|
|
7
|
+
* The JSON Schema subset accepted by scrape (`readSchema` in api.ts checks
|
|
8
|
+
* it): the structure extraction maps, the assertions it checks on the result
|
|
9
|
+
* and annotations it ignores. `default` is never filled in and `format` is
|
|
10
|
+
* not checked.
|
|
11
|
+
*/
|
|
12
|
+
export interface JsonSchema {
|
|
13
|
+
/** Root only: draft-07, 2019-09 or 2020-12. */
|
|
14
|
+
$schema?: string;
|
|
15
|
+
/** Root only. */
|
|
16
|
+
$id?: string;
|
|
17
|
+
$ref?: string;
|
|
18
|
+
type?: 'object' | 'array' | 'string' | 'number' | 'integer' | 'boolean' | 'null' | readonly ('object' | 'array' | 'string' | 'number' | 'integer' | 'boolean' | 'null')[];
|
|
19
|
+
properties?: Readonly<Record<string, JsonSchema>>;
|
|
20
|
+
required?: readonly string[];
|
|
21
|
+
items?: JsonSchema;
|
|
22
|
+
enum?: readonly JsonValue[];
|
|
23
|
+
const?: JsonValue;
|
|
24
|
+
/** A schema and `{ type: 'null' }`, or primitive types only. */
|
|
25
|
+
anyOf?: readonly JsonSchema[];
|
|
26
|
+
oneOf?: readonly JsonSchema[];
|
|
27
|
+
additionalProperties?: boolean | JsonSchema;
|
|
28
|
+
$defs?: Readonly<Record<string, JsonSchema>>;
|
|
29
|
+
definitions?: Readonly<Record<string, JsonSchema>>;
|
|
30
|
+
minimum?: number;
|
|
31
|
+
maximum?: number;
|
|
32
|
+
exclusiveMinimum?: number;
|
|
33
|
+
exclusiveMaximum?: number;
|
|
34
|
+
multipleOf?: number;
|
|
35
|
+
minLength?: number;
|
|
36
|
+
maxLength?: number;
|
|
37
|
+
pattern?: string;
|
|
38
|
+
minItems?: number;
|
|
39
|
+
maxItems?: number;
|
|
40
|
+
uniqueItems?: boolean;
|
|
41
|
+
title?: string;
|
|
42
|
+
description?: string;
|
|
43
|
+
$comment?: string;
|
|
44
|
+
default?: JsonValue;
|
|
45
|
+
examples?: readonly JsonValue[];
|
|
46
|
+
deprecated?: boolean;
|
|
47
|
+
readOnly?: boolean;
|
|
48
|
+
writeOnly?: boolean;
|
|
49
|
+
format?: string;
|
|
50
|
+
}
|
|
51
|
+
export interface JsonFormatRequest {
|
|
52
|
+
type: 'json';
|
|
53
|
+
schema: JsonSchema;
|
|
54
|
+
prompt?: string;
|
|
55
|
+
/** Page content leaves the process only when this is explicitly true. */
|
|
56
|
+
modelFallback?: boolean;
|
|
57
|
+
}
|
|
58
|
+
/** One CSS selector and the HTML attribute to read from every element it matches (the `attributes` format). */
|
|
59
|
+
export interface AttributeSelector {
|
|
60
|
+
/** A non-empty selector of at most 200 characters, within what W2L matches (`invalidSelector`), like `includeTags`. */
|
|
61
|
+
selector: string;
|
|
62
|
+
/** An HTML attribute name: `^[A-Za-z_][A-Za-z0-9_:.-]*$`, at most 100 characters. */
|
|
63
|
+
attribute: string;
|
|
64
|
+
}
|
|
65
|
+
/** Firecrawl's `{ type: 'attributes', selectors }`: 1 to 50 selectors, at most one such entry per request. */
|
|
66
|
+
export interface AttributesFormatRequest {
|
|
67
|
+
type: 'attributes';
|
|
68
|
+
selectors: readonly AttributeSelector[];
|
|
69
|
+
}
|
|
70
|
+
/** One column of a `list` format: what to read from each record. */
|
|
71
|
+
export interface ListField {
|
|
72
|
+
/** The column's name: 1 to 64 characters, unique within the format. */
|
|
73
|
+
name: string;
|
|
74
|
+
/** A selector matched within the record (the first match, in document order); omitted, the record element itself. */
|
|
75
|
+
selector?: string;
|
|
76
|
+
/** An HTML attribute to read instead of the text (`href`, `src`, `data-sku`, ...); a link or source is made absolute. */
|
|
77
|
+
attribute?: string;
|
|
78
|
+
}
|
|
79
|
+
/**
|
|
80
|
+
* W2L's `{ type: 'list', itemSelector, fields }`: every element `itemSelector`
|
|
81
|
+
* matches is a record (one nested in another matched one is not a record of
|
|
82
|
+
* its own), and each field is read from it. At most one such entry per request.
|
|
83
|
+
* Without `itemSelector` W2L finds the page's list itself, and without
|
|
84
|
+
* `fields` the fields its items hold (`fields` needs an `itemSelector`):
|
|
85
|
+
* what it chose is the result's `itemSelector` and `detected.fields`.
|
|
86
|
+
*/
|
|
87
|
+
export interface ListFormatRequest {
|
|
88
|
+
type: 'list';
|
|
89
|
+
itemSelector?: string;
|
|
90
|
+
/** 1 to 50 fields. */
|
|
91
|
+
fields?: readonly ListField[];
|
|
92
|
+
}
|
|
93
|
+
/** A list format with its items and fields named: as asked, or as W2L found them on the page. */
|
|
94
|
+
export interface ListSpec {
|
|
95
|
+
type: 'list';
|
|
96
|
+
itemSelector: string;
|
|
97
|
+
fields: readonly ListField[];
|
|
98
|
+
}
|
|
99
|
+
/** The window a screenshot is taken at, in CSS pixels: integers within the declared screen (`readFormats` checks 320..1920 by 240..1080, the desktop identity's 1920x1080 screen). */
|
|
100
|
+
export interface ScreenshotViewport {
|
|
101
|
+
width: number;
|
|
102
|
+
height: number;
|
|
103
|
+
}
|
|
104
|
+
/**
|
|
105
|
+
* What a screenshot entry asks of the browser lane. Every field is optional:
|
|
106
|
+
* the string `screenshot` alone is a PNG of the declared viewport (1280x800
|
|
107
|
+
* for the desktop identity). `fullPage` captures the document's whole height
|
|
108
|
+
* at the viewport's width, without scrolling first; `quality` (1 to 100)
|
|
109
|
+
* gives a JPEG at that quality instead of a PNG; `viewport` lays the page out
|
|
110
|
+
* in that window, a size within the declared screen and not a change of the
|
|
111
|
+
* identity (User-Agent, client hints, locale, time zone, screen and scale
|
|
112
|
+
* factor stay as declared).
|
|
113
|
+
*/
|
|
114
|
+
export interface ScreenshotOptions {
|
|
115
|
+
fullPage?: boolean;
|
|
116
|
+
quality?: number;
|
|
117
|
+
viewport?: ScreenshotViewport;
|
|
118
|
+
}
|
|
119
|
+
/** Firecrawl's `{ type: 'screenshot', fullPage, quality, viewport }`: at most one screenshot entry per request; the strings `screenshot` and `screenshot@fullPage` (Firecrawl v1's full-page spelling) are its shorthands. */
|
|
120
|
+
export interface ScreenshotFormatRequest extends ScreenshotOptions {
|
|
121
|
+
type: 'screenshot';
|
|
122
|
+
}
|
|
123
|
+
/**
|
|
124
|
+
* String json returns the canonical envelope; an object maps into a caller schema.
|
|
125
|
+
* `html` is the cleaned HTML the Markdown is written from (the main content,
|
|
126
|
+
* the whole page when onlyMainContent is false, or an includeTags selection);
|
|
127
|
+
* `rawHtml` is the page as the lane received it. `images` is every image URL
|
|
128
|
+
* of the whole document, an attributes entry reads named attributes off the
|
|
129
|
+
* elements its selectors match, and `screenshot` (or a screenshot entry)
|
|
130
|
+
* captures the rendered page as an image on the browser lane.
|
|
131
|
+
*/
|
|
132
|
+
export type ScrapeFormat = 'markdown' | 'links' | 'json' | 'html' | 'rawHtml' | 'images' | 'tables' | 'screenshot' | JsonFormatRequest | AttributesFormatRequest | ScreenshotFormatRequest | ListFormatRequest;
|
|
133
|
+
export interface StructuredFieldEvidence {
|
|
134
|
+
path: string;
|
|
135
|
+
/**
|
|
136
|
+
* `fetch`: the value is the fetch's own (`finalUrl`, `requestedUrl`), not read from the page.
|
|
137
|
+
* `pdf`: a `Label: value` line of a PDF's text; its evidencePath is `page N "label"`.
|
|
138
|
+
*/
|
|
139
|
+
source: ProductFactSource | 'hydration' | 'fetch' | 'pdf';
|
|
140
|
+
evidencePath?: string;
|
|
141
|
+
/** For a number read from the page's text: that text, whitespace collapsed (`1.299,00 €`), so the reading can be checked. */
|
|
142
|
+
text?: string;
|
|
143
|
+
}
|
|
144
|
+
export type StructuredIssueCode = 'adapter_unavailable' | 'subject_unverified' | 'field_unavailable'
|
|
145
|
+
/** Page labels matching the field state different values, so none was chosen. */
|
|
146
|
+
| 'field_ambiguous' | 'missing_required' | 'schema_invalid' | 'model_unavailable' | 'model_provider_error' | 'model_output_invalid' | 'model_timeout'
|
|
147
|
+
/** The page status is neither `success` nor `partial`, so no fields were read from it. */
|
|
148
|
+
| 'page_unsuccessful'
|
|
149
|
+
/** The page is `partial` (a timeout ended the scrape): fields come from the content fetched so far, never a complete result. */
|
|
150
|
+
| 'page_partial';
|
|
151
|
+
export interface StructuredExtractionIssue {
|
|
152
|
+
code: StructuredIssueCode;
|
|
153
|
+
message: string;
|
|
154
|
+
path?: string;
|
|
155
|
+
}
|
|
156
|
+
export interface StructuredModelUsage {
|
|
157
|
+
model: string;
|
|
158
|
+
attempts: number;
|
|
159
|
+
inputTokens: number | null;
|
|
160
|
+
outputTokens: number | null;
|
|
161
|
+
/** Unknown provider pricing stays unknown. */
|
|
162
|
+
externalCostUsd: number | null;
|
|
163
|
+
/**
|
|
164
|
+
* True when the request used strict structured outputs with a strict-safe
|
|
165
|
+
* schema W2L derived from the caller's; false when it sent the caller's
|
|
166
|
+
* schema without strict mode, which cannot express it (`strictReason`).
|
|
167
|
+
*/
|
|
168
|
+
strict?: boolean;
|
|
169
|
+
strictReason?: string;
|
|
170
|
+
}
|
|
171
|
+
export interface CanonicalStructuredData {
|
|
172
|
+
adapter: AdapterDescriptor;
|
|
173
|
+
pageType: string;
|
|
174
|
+
entities: readonly ExtractedEntity[];
|
|
175
|
+
}
|
|
176
|
+
export interface StructuredExtractionResult {
|
|
177
|
+
status: 'complete' | 'incomplete' | 'invalid';
|
|
178
|
+
data: JsonValue | CanonicalStructuredData | null;
|
|
179
|
+
/** Present for a caller supplied schema; omitted for canonical entity JSON. */
|
|
180
|
+
schemaSha256?: string;
|
|
181
|
+
evidence: readonly StructuredFieldEvidence[];
|
|
182
|
+
issues: readonly StructuredExtractionIssue[];
|
|
183
|
+
modelUsage?: StructuredModelUsage | null;
|
|
184
|
+
}
|
|
185
|
+
//# sourceMappingURL=structured.d.ts.map
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Structural table assertions for benchmark ground truth.
|
|
3
|
+
*
|
|
4
|
+
* mustContain/mustNotContain are substring checks: an extraction that drops a
|
|
5
|
+
* column, shifts a row, or flattens the table into prose still contains every
|
|
6
|
+
* required string. These helpers parse GFM tables out of extracted markdown and
|
|
7
|
+
* score column/row geometry against an ExpectedTable annotation.
|
|
8
|
+
*/
|
|
9
|
+
/** A parsed GFM table. Rows and columns are logical (span spacer cells counted). */
|
|
10
|
+
export interface GfmTable {
|
|
11
|
+
columns: number;
|
|
12
|
+
rows: number;
|
|
13
|
+
/** Trimmed cell text at a logical position; undefined outside bounds. */
|
|
14
|
+
cellAt: (row: number, col: number) => string | undefined;
|
|
15
|
+
}
|
|
16
|
+
/**
|
|
17
|
+
* Structural assertion for one table in the extracted markdown.
|
|
18
|
+
*
|
|
19
|
+
* Cell keys are zero-indexed "row,col". A position covered by a colspan/rowspan
|
|
20
|
+
* spacer cell is addressable but always empty, so annotations should address
|
|
21
|
+
* span origins, not continuations. Cell text is compared exactly against the
|
|
22
|
+
* trimmed markdown cell, so fixture cells must be unique strings.
|
|
23
|
+
*/
|
|
24
|
+
export interface ExpectedTable {
|
|
25
|
+
/** Logical column count after colspans are expanded. */
|
|
26
|
+
columns: number;
|
|
27
|
+
/** Logical row count after rowspans are expanded (header row included). */
|
|
28
|
+
rows: number;
|
|
29
|
+
/** Exact cell text by logical position. Use unique cell strings. */
|
|
30
|
+
cells?: Readonly<Record<string, string>>;
|
|
31
|
+
/** Cell texts that must land in the same table row. */
|
|
32
|
+
sameRow?: readonly string[];
|
|
33
|
+
/** Cell texts that must land in the same table column. */
|
|
34
|
+
sameColumn?: readonly string[];
|
|
35
|
+
/**
|
|
36
|
+
* Require the content to be a real GFM table. Set false for fixtures that
|
|
37
|
+
* accept any representation; presence is then enforced by mustContain alone.
|
|
38
|
+
* Defaults to true.
|
|
39
|
+
*/
|
|
40
|
+
requireMarkdown?: boolean;
|
|
41
|
+
}
|
|
42
|
+
export interface ExpectedTableCheck {
|
|
43
|
+
pass: boolean;
|
|
44
|
+
/** Human-readable failures, one per violated constraint. */
|
|
45
|
+
issues: readonly string[];
|
|
46
|
+
/** The parsed table, or null when no GFM table was found. */
|
|
47
|
+
table: GfmTable | null;
|
|
48
|
+
}
|
|
49
|
+
/**
|
|
50
|
+
* Find the first GFM table in markdown: a header row, a delimiter row with the
|
|
51
|
+
* same cell count whose cells are hyphens (optionally colon-aligned), and zero
|
|
52
|
+
* or more body rows of the same cell count. Fenced code blocks are skipped.
|
|
53
|
+
* Returns null when no table is present.
|
|
54
|
+
*/
|
|
55
|
+
export declare function parseGfmTable(markdown: string): GfmTable | null;
|
|
56
|
+
/** Score extracted markdown against an ExpectedTable annotation. */
|
|
57
|
+
export declare function evaluateExpectedTable(markdown: string, spec: ExpectedTable): ExpectedTableCheck;
|
|
58
|
+
//# sourceMappingURL=tableMarkdown.d.ts.map
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Canonical token estimator for benchmark purposes.
|
|
3
|
+
*
|
|
4
|
+
* Deliberately simple and dependency-free so that ground-truth ranges and
|
|
5
|
+
* measured values are produced by the same function. Swapping this out
|
|
6
|
+
* invalidates every stored `expectedMainTokens` range — treat a change here
|
|
7
|
+
* as a suite version bump.
|
|
8
|
+
*
|
|
9
|
+
* Approximates BPE behaviour: ~4 chars/token for prose, with CJK counted
|
|
10
|
+
* closer to 1 char/token.
|
|
11
|
+
*/
|
|
12
|
+
export declare function estimateTokens(text: string): number;
|
|
13
|
+
/**
|
|
14
|
+
* A successful HTTP extraction at or below this many tokens is thin enough
|
|
15
|
+
* that the ladder should offer it to a higher lane. This is NOT a content
|
|
16
|
+
* floor that hides loss — thin content is reported as success with its real
|
|
17
|
+
* token count, and the escalation is an addition, never a rewrite.
|
|
18
|
+
*
|
|
19
|
+
* Calibrated against the live-comparison measurements: producthunt.com's JS
|
|
20
|
+
* shell yields ~100-150 tokens over HTTP while the rendered page yields
|
|
21
|
+
* ~24k. A threshold of 200 separates that shell from a real server-rendered
|
|
22
|
+
* page without ever mistaking an honest short page (a terse buy-box, a 404)
|
|
23
|
+
* for a shell — those stay exactly what they are.
|
|
24
|
+
*/
|
|
25
|
+
export declare const QUALITY_ESCALATION_MAX_TOKENS = 200;
|
|
26
|
+
/**
|
|
27
|
+
* Extraction confidence at or below this value marks the result as
|
|
28
|
+
* low-confidence, the second half of the quality signal. `confidenceOf`
|
|
29
|
+
* caps non-article pages at 0.75 and suspiciously small main regions at 0.4;
|
|
30
|
+
* a value of 0.3 sits below both, so only genuinely uncertain extractions
|
|
31
|
+
* escalate.
|
|
32
|
+
*/
|
|
33
|
+
export declare const QUALITY_ESCALATION_MAX_CONFIDENCE = 0.3;
|
|
34
|
+
/**
|
|
35
|
+
* A rendered answer (the browser or a provider lane's) at or below this many
|
|
36
|
+
* main-content tokens, extracted at or below QUALITY_ESCALATION_MAX_CONFIDENCE,
|
|
37
|
+
* carries the `low_content_yield` warning: it is the run's answer, so the
|
|
38
|
+
* caveat is what tells the reader it holds little. Set from the browser lane's
|
|
39
|
+
* yield on the real-site set (research/parity/runs/2026-10-04-rendered-yield-
|
|
40
|
+
* calibration-b9dd658.md): IMF's datamapper renders 226 tokens of social and
|
|
41
|
+
* navigation links at confidence 0 while its figures are drawn by script; the
|
|
42
|
+
* thinnest real listing at confidence 0 there held 363 (ten quotes), and an
|
|
43
|
+
* honest short page (example.com, 267 tokens) is extracted at confidence 1.
|
|
44
|
+
* Listings are often extracted at confidence 0, so the tokens decide.
|
|
45
|
+
*/
|
|
46
|
+
export declare const RENDERED_LOW_YIELD_MAX_TOKENS = 300;
|
|
47
|
+
//# sourceMappingURL=tokens.d.ts.map
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
export { chunkUrls, MAP_ANSWER_MARGIN_MS, SDK_ORIGIN, W2L, W2LError, WaitTimeoutError } from './client.js';
|
|
2
|
+
export { DEFAULT_WATCH_POLL_INTERVAL_MS, JobWatcher, MIN_WATCH_POLL_INTERVAL_MS, pageCursor, parseSseBlock } from './watcher.js';
|
|
3
|
+
export type { WatchKind, WatchOptions, WatchTransport, WatcherClient, WatcherError, WatcherEvent, WatcherWebSocket, WatcherWebSocketConstructor } from './watcher.js';
|
|
4
|
+
export { SDK_VERSION } from './version.js';
|
|
5
|
+
export type { W2LOptions, RequestOptions, WaitOptions, CreateMonitorRequest, ReviseMonitorRequest, RunMonitorRequest, ScrapeOptions } from './client.js';
|
|
6
|
+
export type { AppendToBatchOptions, ChunkedBatchJob, ChunkedBatchOptions, ChunkedBatchResult } from './client.js';
|
|
7
|
+
export type { BatchCollected, BatchDocuments, CrawlCollected, CrawlDocuments, PageCollection, PagedListOptions, PaginationEnd, PaginationLimits, PaginationStop } from './client.js';
|
|
8
|
+
export type { EvidenceRecord, PageTable, PdfPageMarkdown, PdfParser, ActiveCrawl, ActiveCrawlList, ActiveCrawlOptions, AgentHints, ApiErrorBody, ApiErrorCode, ApiErrorDetails, AttributeExtraction, AttributeSelector, AttributesFormatRequest, FetchWarning, JsonFormatRequest, PageMetadata, ScrapeFormat, ScreenshotEvidence, ScreenshotFormatRequest, ScreenshotOptions, ScreenshotViewport, RateLimitedBody, RequestAttribution, ScrapeMetadata, ScrapeRecord, ScrapeResponseMetadata, BatchAccepted, BatchErrorItem, BatchErrorStatus, BatchErrorsQuery, BatchErrorsResponse, BatchStartRequest, BatchStatusResponse, CompactScrapeResponse, ScrapeResponse, CrawlAccepted, DeliveryAttempt, DeliveryDestination, DeliveryDestinationInput, DeliveryDetail, DeliveryQuery, DeliveryState, WebhookDelivery, WebhookEventEnvelope, JobWebhookEnvelope, JobWebhookStatus, WebhookConfig, WebhookEvent, WebhookOption, WebhookPayload, WebhookPayloadFormat, FirecrawlWebhookPayload, DeliveryDestinationKind, CrawlReport, CrawlError, CrawlPage, CrawlPageList, CrawlPageQuery, CrawlStartRequest, JobStreamEventType, JobStreamFrame, JobStreamReport, SitemapMode, SitemapDiscovery, SitemapFileRecord, CrawlDiscovery, FetchResult, DocumentMonitorConfig, DocumentFields, FieldValue, MonitorFieldRule, MonitorRevision, MonitorRun, MonitorView, MonitorEvent, MonitorSnapshot, ScrapeRequest, } from './contracts/index.js';
|
|
9
|
+
//# sourceMappingURL=index.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{ "type": "commonjs" }
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The SDK's own version, pinned to packages/sdk/package.json by its test. It
|
|
3
|
+
* is what the SDK writes as the `origin` of every scrape, crawl and batch it
|
|
4
|
+
* starts (`js-sdk@<version>`), for W2L's own records; nothing goes to the
|
|
5
|
+
* target.
|
|
6
|
+
*/
|
|
7
|
+
export declare const SDK_VERSION = "0.3.0";
|
|
8
|
+
//# sourceMappingURL=version.d.ts.map
|
|
@@ -0,0 +1,151 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The job watcher: a crawl's or batch's pages as they are recorded, from the
|
|
3
|
+
* API's stream routes (WebSocket, then server-sent events) or, where neither
|
|
4
|
+
* answers, from the status and listing routes polled. Every transport
|
|
5
|
+
* delivers the same compact pages the items routes list; a document is
|
|
6
|
+
* emitted once per step id whatever route, reconnection or transport switch
|
|
7
|
+
* it arrived by. Purely client-side: the server records nothing about a
|
|
8
|
+
* watcher, and `close()` stops watching without touching the job.
|
|
9
|
+
*/
|
|
10
|
+
import { type BatchStatusResponse, type CrawlError, type CrawlPage, type CrawlPageList, type CrawlPageQuery, type CrawlReport, type JobStreamFrame, type JobStreamReport, type TaskStatus } from './contracts/index.js';
|
|
11
|
+
export type WatchKind = 'crawl' | 'batch';
|
|
12
|
+
export type WatchTransport = 'websocket' | 'sse' | 'poll';
|
|
13
|
+
/** What the watcher reports as `error`: the API's code where it has one (`not_found`, `unauthorized`), else the watcher's own (`watcher_timeout`, `transport_unavailable`, `poll_failed`). */
|
|
14
|
+
export interface WatcherError {
|
|
15
|
+
code: string;
|
|
16
|
+
message: string;
|
|
17
|
+
}
|
|
18
|
+
/** The events a watcher dispatches (CustomEvent `detail`) and yields from its async iterator. */
|
|
19
|
+
export type WatcherEvent = {
|
|
20
|
+
type: 'document';
|
|
21
|
+
data: CrawlPage;
|
|
22
|
+
} | {
|
|
23
|
+
type: 'snapshot';
|
|
24
|
+
data: JobStreamReport;
|
|
25
|
+
} | {
|
|
26
|
+
type: 'done';
|
|
27
|
+
data: JobStreamReport;
|
|
28
|
+
} | {
|
|
29
|
+
type: 'error';
|
|
30
|
+
error: WatcherError;
|
|
31
|
+
};
|
|
32
|
+
/** The part of a WebSocket the watcher uses; the platform's WebSocket satisfies it. */
|
|
33
|
+
export interface WatcherWebSocket {
|
|
34
|
+
readonly readyState: number;
|
|
35
|
+
readonly protocol: string;
|
|
36
|
+
close(code?: number, reason?: string): void;
|
|
37
|
+
addEventListener(type: 'open' | 'message' | 'close' | 'error', listener: (event: {
|
|
38
|
+
data?: unknown;
|
|
39
|
+
code?: number;
|
|
40
|
+
reason?: string;
|
|
41
|
+
}) => void): void;
|
|
42
|
+
}
|
|
43
|
+
export type WatcherWebSocketConstructor = new (url: string, protocols?: string | string[]) => WatcherWebSocket;
|
|
44
|
+
export interface WatchOptions {
|
|
45
|
+
/** Which kind of job the id names. Default `crawl`. */
|
|
46
|
+
kind?: WatchKind;
|
|
47
|
+
/**
|
|
48
|
+
* `auto` (default) tries a WebSocket, then the server-sent events route, then
|
|
49
|
+
* polling, switching once per level when a transport is unavailable or ends
|
|
50
|
+
* before `done`; one of the three forces it (an error event when it cannot run).
|
|
51
|
+
*/
|
|
52
|
+
transport?: 'auto' | WatchTransport;
|
|
53
|
+
/** Delay between polls on the poll transport, at least 250. Default 2000. */
|
|
54
|
+
pollIntervalMs?: number;
|
|
55
|
+
/** After this long without `done`, an `error` of code `watcher_timeout` ends the watch; the job keeps running. Default: no limit. */
|
|
56
|
+
timeoutMs?: number;
|
|
57
|
+
/** The step cursor (a document's `cursor` / SSE `id`) to resume after: documents up to it are not delivered again. */
|
|
58
|
+
after?: string;
|
|
59
|
+
/** Aborting it closes the watcher, with no event. */
|
|
60
|
+
signal?: AbortSignal;
|
|
61
|
+
/** The WebSocket constructor to use; omitted takes the platform's (Node 22+, browsers); `null` tries no WebSocket. */
|
|
62
|
+
WebSocket?: WatcherWebSocketConstructor | null;
|
|
63
|
+
}
|
|
64
|
+
/** What the watcher needs from the SDK client: the server, its token and fetch, and the status and listing routes it polls. */
|
|
65
|
+
export interface WatcherClient {
|
|
66
|
+
readonly baseUrl: string;
|
|
67
|
+
readonly token: string | undefined;
|
|
68
|
+
readonly fetch: typeof fetch;
|
|
69
|
+
headers(extra?: Record<string, string>): Record<string, string>;
|
|
70
|
+
getCrawl(id: string, request?: {
|
|
71
|
+
signal?: AbortSignal;
|
|
72
|
+
}): Promise<CrawlReport>;
|
|
73
|
+
getBatch(id: string, request?: {
|
|
74
|
+
signal?: AbortSignal;
|
|
75
|
+
}): Promise<BatchStatusResponse>;
|
|
76
|
+
getCrawlPages(id: string, options?: CrawlPageQuery, request?: {
|
|
77
|
+
signal?: AbortSignal;
|
|
78
|
+
}): Promise<CrawlPageList<CrawlPage>>;
|
|
79
|
+
getCrawlErrors(id: string, options?: CrawlPageQuery, request?: {
|
|
80
|
+
signal?: AbortSignal;
|
|
81
|
+
}): Promise<CrawlPageList<CrawlError>>;
|
|
82
|
+
getBatchItems(id: string, options?: CrawlPageQuery, request?: {
|
|
83
|
+
signal?: AbortSignal;
|
|
84
|
+
}): Promise<CrawlPageList<CrawlPage>>;
|
|
85
|
+
}
|
|
86
|
+
export declare const DEFAULT_WATCH_POLL_INTERVAL_MS = 2000;
|
|
87
|
+
export declare const MIN_WATCH_POLL_INTERVAL_MS = 250;
|
|
88
|
+
/**
|
|
89
|
+
* A page's step cursor as the API issues it (the `(createdAt, id)` position
|
|
90
|
+
* the listing routes take): what polling continues from after the last
|
|
91
|
+
* page of a listing, which carries no `nextCursor`.
|
|
92
|
+
*/
|
|
93
|
+
export declare function pageCursor(page: Pick<CrawlPage, 'createdAt' | 'id'>): string;
|
|
94
|
+
/** One server-sent event block as a stream frame; null for a comment, an event W2L does not send or data that is not JSON. */
|
|
95
|
+
export declare function parseSseBlock(block: string): JobStreamFrame | null;
|
|
96
|
+
/**
|
|
97
|
+
* Watches one job. Events: `document` (CustomEvent<CrawlPage>), `snapshot`
|
|
98
|
+
* (CustomEvent<CrawlReport | BatchStatusResponse>, the report as the stream
|
|
99
|
+
* opens and after each page), `done` (the terminal report) and `error`
|
|
100
|
+
* (CustomEvent<WatcherError>); `for await (const event of watcher)` yields
|
|
101
|
+
* the same events from the start. `data` accumulates every document, each
|
|
102
|
+
* once; `status` is the last status seen; `transport` the one in use.
|
|
103
|
+
*/
|
|
104
|
+
export declare class JobWatcher extends EventTarget {
|
|
105
|
+
readonly jobId: string;
|
|
106
|
+
readonly kind: WatchKind;
|
|
107
|
+
/** Every document delivered so far, each step once, in arrival order. */
|
|
108
|
+
readonly data: CrawlPage[];
|
|
109
|
+
/** The job's status as last reported; null before the first report. */
|
|
110
|
+
status: TaskStatus | null;
|
|
111
|
+
/** The transport delivering events; null before one is open and after the watch ends. */
|
|
112
|
+
transport: WatchTransport | null;
|
|
113
|
+
private readonly client;
|
|
114
|
+
private readonly transportOption;
|
|
115
|
+
private readonly pollIntervalMs;
|
|
116
|
+
private readonly socketConstructor;
|
|
117
|
+
private readonly controller;
|
|
118
|
+
private readonly seen;
|
|
119
|
+
private readonly log;
|
|
120
|
+
private readonly waiters;
|
|
121
|
+
/** The cursor of the last document delivered: where the next transport, or a poll, continues from. */
|
|
122
|
+
private cursor;
|
|
123
|
+
private lastSnapshot;
|
|
124
|
+
private finished;
|
|
125
|
+
private closed;
|
|
126
|
+
private timer;
|
|
127
|
+
constructor(client: WatcherClient, jobId: string, options?: WatchOptions);
|
|
128
|
+
/** Stops watching: no further event, the job untouched. */
|
|
129
|
+
close(): void;
|
|
130
|
+
/** The events from the start, then as they arrive, until done, error or close. */
|
|
131
|
+
[Symbol.asyncIterator](): AsyncIterator<WatcherEvent>;
|
|
132
|
+
private get stopped();
|
|
133
|
+
private run;
|
|
134
|
+
/** The stream route of this job, with the cursor to resume after when one is known. */
|
|
135
|
+
private streamUrl;
|
|
136
|
+
/** The WebSocket route; the token, when there is one, as the `w2l.token.<token>` subprotocol, since the WebSocket API sets no header. */
|
|
137
|
+
private watchSocket;
|
|
138
|
+
/** The server-sent events route: 404 leaves it to polling, 401/403 ends the watch, a stream that ends before done hands over from the last cursor. */
|
|
139
|
+
private watchEvents;
|
|
140
|
+
/** The status and listing routes every pollIntervalMs: the documents since the last cursor, a snapshot, and done once the status is terminal and the last pages are read. */
|
|
141
|
+
private poll;
|
|
142
|
+
private handle;
|
|
143
|
+
private document;
|
|
144
|
+
private snapshot;
|
|
145
|
+
private done;
|
|
146
|
+
private fail;
|
|
147
|
+
private emit;
|
|
148
|
+
private stop;
|
|
149
|
+
private wake;
|
|
150
|
+
}
|
|
151
|
+
//# sourceMappingURL=watcher.d.ts.map
|