@octocrawl/sdk 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +17 -0
- package/dist/index.cjs +1278 -0
- package/dist/index.js +1240 -0
- package/dist/types/cjs/client.d.ts +362 -0
- package/dist/types/cjs/contracts/access.d.ts +166 -0
- package/dist/types/cjs/contracts/actions.d.ts +191 -0
- package/dist/types/cjs/contracts/api.d.ts +891 -0
- package/dist/types/cjs/contracts/benchmark.d.ts +116 -0
- package/dist/types/cjs/contracts/checkpoint.d.ts +164 -0
- package/dist/types/cjs/contracts/compliance.d.ts +412 -0
- package/dist/types/cjs/contracts/crawl.d.ts +302 -0
- package/dist/types/cjs/contracts/delivery.d.ts +136 -0
- package/dist/types/cjs/contracts/evidenceRecord.d.ts +192 -0
- package/dist/types/cjs/contracts/execution.d.ts +197 -0
- package/dist/types/cjs/contracts/extractor.d.ts +379 -0
- package/dist/types/cjs/contracts/file.d.ts +117 -0
- package/dist/types/cjs/contracts/firecrawl.d.ts +258 -0
- package/dist/types/cjs/contracts/groundTruth.d.ts +77 -0
- package/dist/types/cjs/contracts/identityBundle.d.ts +70 -0
- package/dist/types/cjs/contracts/index.d.ts +30 -0
- package/dist/types/cjs/contracts/map.d.ts +180 -0
- package/dist/types/cjs/contracts/monitor.d.ts +217 -0
- package/dist/types/cjs/contracts/monitorConfig.d.ts +9 -0
- package/dist/types/cjs/contracts/policy.d.ts +93 -0
- package/dist/types/cjs/contracts/proxy.d.ts +52 -0
- package/dist/types/cjs/contracts/recipe.d.ts +74 -0
- package/dist/types/cjs/contracts/regexSafety.d.ts +31 -0
- package/dist/types/cjs/contracts/result.d.ts +503 -0
- package/dist/types/cjs/contracts/session.d.ts +51 -0
- package/dist/types/cjs/contracts/ssrf.d.ts +16 -0
- package/dist/types/cjs/contracts/status.d.ts +27 -0
- package/dist/types/cjs/contracts/structured.d.ts +185 -0
- package/dist/types/cjs/contracts/tableMarkdown.d.ts +58 -0
- package/dist/types/cjs/contracts/tokens.d.ts +47 -0
- package/dist/types/cjs/index.d.ts +9 -0
- package/dist/types/cjs/package.json +1 -0
- package/dist/types/cjs/version.d.ts +8 -0
- package/dist/types/cjs/watcher.d.ts +151 -0
- package/dist/types/esm/client.d.ts +362 -0
- package/dist/types/esm/contracts/access.d.ts +166 -0
- package/dist/types/esm/contracts/actions.d.ts +191 -0
- package/dist/types/esm/contracts/api.d.ts +891 -0
- package/dist/types/esm/contracts/benchmark.d.ts +116 -0
- package/dist/types/esm/contracts/checkpoint.d.ts +164 -0
- package/dist/types/esm/contracts/compliance.d.ts +412 -0
- package/dist/types/esm/contracts/crawl.d.ts +302 -0
- package/dist/types/esm/contracts/delivery.d.ts +136 -0
- package/dist/types/esm/contracts/evidenceRecord.d.ts +192 -0
- package/dist/types/esm/contracts/execution.d.ts +197 -0
- package/dist/types/esm/contracts/extractor.d.ts +379 -0
- package/dist/types/esm/contracts/file.d.ts +117 -0
- package/dist/types/esm/contracts/firecrawl.d.ts +258 -0
- package/dist/types/esm/contracts/groundTruth.d.ts +77 -0
- package/dist/types/esm/contracts/identityBundle.d.ts +70 -0
- package/dist/types/esm/contracts/index.d.ts +30 -0
- package/dist/types/esm/contracts/map.d.ts +180 -0
- package/dist/types/esm/contracts/monitor.d.ts +217 -0
- package/dist/types/esm/contracts/monitorConfig.d.ts +9 -0
- package/dist/types/esm/contracts/policy.d.ts +93 -0
- package/dist/types/esm/contracts/proxy.d.ts +52 -0
- package/dist/types/esm/contracts/recipe.d.ts +74 -0
- package/dist/types/esm/contracts/regexSafety.d.ts +31 -0
- package/dist/types/esm/contracts/result.d.ts +503 -0
- package/dist/types/esm/contracts/session.d.ts +51 -0
- package/dist/types/esm/contracts/ssrf.d.ts +16 -0
- package/dist/types/esm/contracts/status.d.ts +27 -0
- package/dist/types/esm/contracts/structured.d.ts +185 -0
- package/dist/types/esm/contracts/tableMarkdown.d.ts +58 -0
- package/dist/types/esm/contracts/tokens.d.ts +47 -0
- package/dist/types/esm/index.d.ts +9 -0
- package/dist/types/esm/version.d.ts +8 -0
- package/dist/types/esm/watcher.d.ts +151 -0
- package/package.json +40 -0
|
@@ -0,0 +1,197 @@
|
|
|
1
|
+
import type { RobotsOverride } from './compliance.js';
|
|
2
|
+
import type { FetchWarning, TraceEvent } from './result.js';
|
|
3
|
+
import type { AttributeSelector, ListFormatRequest, ScreenshotOptions } from './structured.js';
|
|
4
|
+
import type { PageAction } from './actions.js';
|
|
5
|
+
/** In-process cancellation and an absolute UTC deadline. Never serialize signal. */
|
|
6
|
+
export interface ExecutionContext {
|
|
7
|
+
signal?: AbortSignal;
|
|
8
|
+
deadlineAt?: number;
|
|
9
|
+
/** Persist publisher cooldown immediately, before an interruptible inline wait. */
|
|
10
|
+
onRetryAfter?: (url: string, retryAt: number) => void;
|
|
11
|
+
/**
|
|
12
|
+
* Hear of a recorded robots override the moment a lane applies it, before
|
|
13
|
+
* its request goes out: the ladder keeps the override on the run's answer
|
|
14
|
+
* when that lane never returns (a deadline) or another rung's result answers.
|
|
15
|
+
*/
|
|
16
|
+
onRobotsOverride?: (applied: RobotsOverrideApplied) => void;
|
|
17
|
+
}
|
|
18
|
+
/** What a lane reports when it sets a robots.txt rule aside under a recorded override. */
|
|
19
|
+
export interface RobotsOverrideApplied {
|
|
20
|
+
/** The lane's `robots_checked`, `robots_disallowed` and `robots_overridden` trace events, in that order. */
|
|
21
|
+
trace: readonly TraceEvent[];
|
|
22
|
+
/** The `robots_overridden` warning the lane's own result carries. */
|
|
23
|
+
warning: FetchWarning;
|
|
24
|
+
}
|
|
25
|
+
/**
|
|
26
|
+
* What the caller asked a lane to capture from one page. It changes the
|
|
27
|
+
* content a lane emits, never the Evidence (hashes, status rules).
|
|
28
|
+
*/
|
|
29
|
+
export interface FetchOptions {
|
|
30
|
+
/** false: Markdown of the whole page body, header, navigation and footer kept. Default true. */
|
|
31
|
+
onlyMainContent?: boolean;
|
|
32
|
+
/** Milliseconds a browser rung waits after load and stability before capture. Default 0. */
|
|
33
|
+
waitFor?: number;
|
|
34
|
+
/**
|
|
35
|
+
* The caller's own `timeout`, when it set one. The deadline itself is the
|
|
36
|
+
* ExecutionContext's; this says the caller chose it, so a lane waits for a
|
|
37
|
+
* slow server (HTTP headers and body, browser navigation) until that
|
|
38
|
+
* deadline instead of stopping at its default caps.
|
|
39
|
+
*/
|
|
40
|
+
timeout?: number;
|
|
41
|
+
/**
|
|
42
|
+
* Bytes a file (PDF, CSV, XLSX, ZIP, JSON, text) may have, below the
|
|
43
|
+
* operator's cap (`NetworkPolicy.maxFileBytes`); a larger file is failed
|
|
44
|
+
* with `body_too_large` and not saved. Web pages keep `maxBodyBytes`.
|
|
45
|
+
*/
|
|
46
|
+
maxFileBytes?: number;
|
|
47
|
+
/**
|
|
48
|
+
* How a PDF response is read (Firecrawl's `parsers`): absent reads every
|
|
49
|
+
* PDF's text layer with W2L's defaults; `[]` reads none (the file is saved
|
|
50
|
+
* as received, without text); one `pdf` entry sets the options. Other
|
|
51
|
+
* files are unaffected.
|
|
52
|
+
*/
|
|
53
|
+
parsers?: readonly PdfParser[];
|
|
54
|
+
/**
|
|
55
|
+
* A recorded decision to fetch this one URL although its host's robots.txt
|
|
56
|
+
* disallows it. robots.txt is still read and its verdict recorded; the
|
|
57
|
+
* override goes into the trace, the warnings and, in the browser lane, the
|
|
58
|
+
* compliance record. An unreachable robots.txt is not set aside. The HTTP
|
|
59
|
+
* and local browser lanes apply it; the provider lane takes none.
|
|
60
|
+
* Set per URL by the caller (a scrape's `robotsOverride`, a batch's
|
|
61
|
+
* `robotsOverrides` entry), never by a batch or crawl for every page.
|
|
62
|
+
*/
|
|
63
|
+
robotsOverride?: RobotsOverride;
|
|
64
|
+
/**
|
|
65
|
+
* CSS selectors naming the only elements to keep. The content is those
|
|
66
|
+
* elements, in document order, copied from the page before anything is
|
|
67
|
+
* cleaned away, so a named navigation stays; `onlyMainContent` no longer
|
|
68
|
+
* chooses the content. Nothing matching is an empty answer. The page's
|
|
69
|
+
* type, title and metadata are still read from the whole page. The API
|
|
70
|
+
* refuses a selector the extractor does not match (@w2l/extract-tf
|
|
71
|
+
* `invalidSelector`) and a list of more parts than it matches for one
|
|
72
|
+
* list (`MAX_SELECTOR_PARTS`); a lane given either reads it as naming
|
|
73
|
+
* nothing.
|
|
74
|
+
*/
|
|
75
|
+
includeTags?: readonly string[];
|
|
76
|
+
/**
|
|
77
|
+
* CSS selectors removed, with everything inside them, before the content
|
|
78
|
+
* is taken: from the main content, from the whole page
|
|
79
|
+
* (`onlyMainContent: false`) and from an `includeTags` selection alike.
|
|
80
|
+
* The same selectors as `includeTags`.
|
|
81
|
+
*/
|
|
82
|
+
excludeTags?: readonly string[];
|
|
83
|
+
/**
|
|
84
|
+
* Extra request headers, lower-cased names, validated by the API
|
|
85
|
+
* (`readHeaders`): never the User-Agent, a client hint, a credential or a
|
|
86
|
+
* transport header. The HTTP and local browser lanes send them with the
|
|
87
|
+
* requested URL and its same-origin hops and subresources, after the
|
|
88
|
+
* declared identity, and record them (`request_headers_added`). Both lanes
|
|
89
|
+
* fetch a redirect hop to another origin with the identity alone and say so
|
|
90
|
+
* (`custom_headers_withheld`); on the browser lane the headers are added per
|
|
91
|
+
* request through Chromium's request interception, which judges every hop
|
|
92
|
+
* and every file the page loads by its own origin, so a navigation the page
|
|
93
|
+
* makes to another origin gets none either. robots.txt is fetched with the
|
|
94
|
+
* identity alone. Everything here is on the record.
|
|
95
|
+
*/
|
|
96
|
+
headers?: Readonly<Record<string, string>>;
|
|
97
|
+
/**
|
|
98
|
+
* Fetch as the declared mobile Chrome identity (Android User-Agent, mobile
|
|
99
|
+
* client hints, 412x915 viewport, touch) instead of the desktop one. A
|
|
100
|
+
* second declared identity, not a disguise: it passes the same coherence
|
|
101
|
+
* and honesty checks, and robots.txt is evaluated against its User-Agent.
|
|
102
|
+
* Default false. Refused with research mode, which declares a bot.
|
|
103
|
+
*/
|
|
104
|
+
mobile?: boolean;
|
|
105
|
+
/**
|
|
106
|
+
* Local only: load a site whose certificate does not verify (self-signed,
|
|
107
|
+
* expired, wrong name). The lanes relax verification for this one fetch
|
|
108
|
+
* and its robots.txt lookup, say so in the trace
|
|
109
|
+
* (`tls_verification_skipped`) and in a `tls_unverified` warning. Default
|
|
110
|
+
* false: a certificate failure is `failed` / `tls_error`. A hosted engine
|
|
111
|
+
* refuses the option.
|
|
112
|
+
*/
|
|
113
|
+
skipTlsVerification?: boolean;
|
|
114
|
+
/**
|
|
115
|
+
* Abort requests to a bundled list of ad-serving hosts on the local
|
|
116
|
+
* browser lane (`ads_blocked`), and remove ad and cookie-banner elements
|
|
117
|
+
* before extraction on every lane. Default true; false keeps them in the
|
|
118
|
+
* Markdown and `html`. The hosted browser's host allowlist stays in force
|
|
119
|
+
* whatever this says.
|
|
120
|
+
*/
|
|
121
|
+
blockAds?: boolean;
|
|
122
|
+
/**
|
|
123
|
+
* Carry `html` on a contentful result: the cleaned HTML its Markdown was
|
|
124
|
+
* written from. Set from the requested formats (`html`), not by a caller.
|
|
125
|
+
*/
|
|
126
|
+
includeHtml?: boolean;
|
|
127
|
+
/**
|
|
128
|
+
* Carry `rawHtml` on a contentful result: the page as the lane received
|
|
129
|
+
* it. Set from the requested formats (`rawHtml`), not by a caller.
|
|
130
|
+
*/
|
|
131
|
+
includeRawHtml?: boolean;
|
|
132
|
+
/**
|
|
133
|
+
* Carry `images` on a contentful result: every image URL of the whole
|
|
134
|
+
* document as received (the rendered DOM on a browser lane). Set from the
|
|
135
|
+
* requested formats (`images`), not by a caller.
|
|
136
|
+
*/
|
|
137
|
+
includeImages?: boolean;
|
|
138
|
+
/**
|
|
139
|
+
* Carry `tables` on a contentful result: every data table of the content
|
|
140
|
+
* the Markdown was written from, as data and CSV. Set from the requested
|
|
141
|
+
* formats (`tables`), not by a caller.
|
|
142
|
+
*/
|
|
143
|
+
includeTables?: boolean;
|
|
144
|
+
/**
|
|
145
|
+
* Carry `attributes` on a contentful result: for each selector, the named
|
|
146
|
+
* attribute's values on the elements it matches in the document as
|
|
147
|
+
* received. Set from the requested formats (an `attributes` entry), not
|
|
148
|
+
* by a caller; the API has checked the selectors (`invalidSelector`).
|
|
149
|
+
*/
|
|
150
|
+
attributes?: readonly AttributeSelector[];
|
|
151
|
+
/** The `list` format's request, set from the requested formats, not by a caller; the API has checked its selectors. */
|
|
152
|
+
list?: ListFormatRequest;
|
|
153
|
+
/**
|
|
154
|
+
* Whether an `<img>` whose `src` is a `data:` URI is left out of the
|
|
155
|
+
* Markdown, its alt text kept (Firecrawl's `removeBase64Images`). Default
|
|
156
|
+
* true, which every lane always did; false keeps the image as
|
|
157
|
+
* ``, which the token count then counts. `html` and
|
|
158
|
+
* `rawHtml` are never rewritten. A rendering choice, not a fetch fact: no
|
|
159
|
+
* trace event or warning.
|
|
160
|
+
*/
|
|
161
|
+
removeBase64Images?: boolean;
|
|
162
|
+
/**
|
|
163
|
+
* Capture the rendered page as an image (the `screenshot` format): a PNG,
|
|
164
|
+
* or a JPEG at `quality`, of the viewport (the request's `viewport` within
|
|
165
|
+
* the declared screen, else the declared one) or of the document's whole
|
|
166
|
+
* height (`fullPage`, without scrolling first), taken after load,
|
|
167
|
+
* stability and `waitFor` and before the DOM is read, CSS-pixel sized. The
|
|
168
|
+
* local browser lane alone honours it, and the API selects that lane alone
|
|
169
|
+
* for such a request; the http and provider lanes ignore it. Set from the
|
|
170
|
+
* requested formats (a `screenshot` entry), not by a caller.
|
|
171
|
+
*/
|
|
172
|
+
screenshot?: ScreenshotOptions;
|
|
173
|
+
/**
|
|
174
|
+
* Steps the local browser runs on the page after load, stability and
|
|
175
|
+
* `waitFor`, and before the screenshot format and the DOM are read
|
|
176
|
+
* (`actions`). The local browser lanes alone run them; the API selects
|
|
177
|
+
* those lanes alone for such a request.
|
|
178
|
+
*/
|
|
179
|
+
actions?: readonly PageAction[];
|
|
180
|
+
}
|
|
181
|
+
/** The largest `maxPages` a pdf parser entry may ask for. */
|
|
182
|
+
export declare const MAX_PDF_PAGES = 10000;
|
|
183
|
+
/**
|
|
184
|
+
* The `pdf` entry of `parsers`. W2L reads a PDF's text layer and runs no
|
|
185
|
+
* OCR, so `mode` is `fast` or `auto`, both that reader (`ocr` is refused).
|
|
186
|
+
*/
|
|
187
|
+
export interface PdfParser {
|
|
188
|
+
type: 'pdf';
|
|
189
|
+
mode?: 'fast' | 'auto';
|
|
190
|
+
/** Read at most this many pages, from the first (1 to MAX_PDF_PAGES); default 1000. A document cut by it is `success` with a `page_cap` warning: the cut was asked for. */
|
|
191
|
+
maxPages?: number;
|
|
192
|
+
/** Also return each page's Markdown as `pages: [{ pageNumber, markdown }]`. Default false. */
|
|
193
|
+
pages?: boolean;
|
|
194
|
+
/** A `<!-- page N -->` line before each page's text. Default true natively, false on `/fc` as on Firecrawl. */
|
|
195
|
+
pageMarkers?: boolean;
|
|
196
|
+
}
|
|
197
|
+
//# sourceMappingURL=execution.d.ts.map
|
|
@@ -0,0 +1,379 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Extractor contract: the seam every main-content extractor implements
|
|
3
|
+
* (extract-tf, the v0 readability wrapper, and any future tier).
|
|
4
|
+
*/
|
|
5
|
+
/** Page shape the extractor routed to. */
|
|
6
|
+
export type PageType = 'article' | 'listing' | 'collection' | 'product' | 'forum';
|
|
7
|
+
/** Extraction strategy that produced mainHtml, independent of pageType. */
|
|
8
|
+
export type ExtractStrategy = 'article' | 'list' | 'table' | 'product';
|
|
9
|
+
/**
|
|
10
|
+
* Where a product fact came from. The ordering is a strength ordering:
|
|
11
|
+
* `jsonld` and `microdata` are the publisher's own machine-readable claim,
|
|
12
|
+
* `meta` is a tag written for machines, `text` is our reading of rendered
|
|
13
|
+
* prose, `inferred` is our derivation from context (e.g., currency from domain).
|
|
14
|
+
* A price we matched out of visible text is a weaker claim than one
|
|
15
|
+
* the publisher declared, and a consumer is entitled to know which it got.
|
|
16
|
+
*/
|
|
17
|
+
export type ProductFactSource = 'jsonld' | 'microdata' | 'meta' | 'dom' | 'text' | 'inferred' | 'model';
|
|
18
|
+
/** Evidence classes allowed on the normalized cross-site entity surface. */
|
|
19
|
+
export type EntityFieldSource = 'jsonld' | 'microdata' | 'meta' | 'hydration' | 'dom' | 'inferred';
|
|
20
|
+
export type EntityFieldStatus = 'confirmed' | 'unconfirmed';
|
|
21
|
+
export type EntityType = 'product' | 'post' | 'thread' | 'comment' | 'profile' | 'community' | 'video' | 'article';
|
|
22
|
+
export type AdapterStatus = 'generic' | 'beta adapter' | 'verified adapter' | 'unsupported';
|
|
23
|
+
export type EntityValue = string | number | boolean | null | readonly EntityValue[] | {
|
|
24
|
+
readonly [key: string]: EntityValue;
|
|
25
|
+
};
|
|
26
|
+
export interface EntityField<T extends EntityValue = EntityValue> {
|
|
27
|
+
/** Value exactly as observed on the page. */
|
|
28
|
+
raw: T;
|
|
29
|
+
/** Stable value used across adapters. */
|
|
30
|
+
normalized: T;
|
|
31
|
+
source: EntityFieldSource;
|
|
32
|
+
/** CSS selector, JSON Pointer, URL component, or another public-page location. */
|
|
33
|
+
path: string;
|
|
34
|
+
status: EntityFieldStatus;
|
|
35
|
+
}
|
|
36
|
+
export interface ExtractedEntity {
|
|
37
|
+
type: EntityType;
|
|
38
|
+
id: string | null;
|
|
39
|
+
fields: Readonly<Record<string, EntityField>>;
|
|
40
|
+
/** IDs of other entities in this response, e.g. parent/author/community. */
|
|
41
|
+
relationships: Readonly<Record<string, string | readonly string[] | null>>;
|
|
42
|
+
}
|
|
43
|
+
export interface AdapterDescriptor {
|
|
44
|
+
id: string;
|
|
45
|
+
version: string;
|
|
46
|
+
status: AdapterStatus;
|
|
47
|
+
}
|
|
48
|
+
/** Identity and provenance checks performed by a site adapter. */
|
|
49
|
+
export interface AdapterValidation {
|
|
50
|
+
valid: boolean;
|
|
51
|
+
issues: readonly string[];
|
|
52
|
+
}
|
|
53
|
+
/** One product fact plus the evidence class it was drawn from. */
|
|
54
|
+
export interface ProductFact {
|
|
55
|
+
/** The value exactly as the page carried it. Never normalized — a
|
|
56
|
+
* normalized price is a claim we would be making, not one we read. */
|
|
57
|
+
value: string;
|
|
58
|
+
source: ProductFactSource;
|
|
59
|
+
/** JSON Pointer or CSS selector locating the evidence when available. */
|
|
60
|
+
path?: string;
|
|
61
|
+
}
|
|
62
|
+
export interface ProductPrice {
|
|
63
|
+
amount: ProductFact;
|
|
64
|
+
currency: ProductFact | null;
|
|
65
|
+
priceType: 'current' | 'list' | 'unit' | 'subscription' | 'other';
|
|
66
|
+
seller: ProductFact | null;
|
|
67
|
+
}
|
|
68
|
+
export interface ProductVariant {
|
|
69
|
+
name: string;
|
|
70
|
+
value: string;
|
|
71
|
+
selected: boolean;
|
|
72
|
+
source: ProductFactSource;
|
|
73
|
+
path?: string;
|
|
74
|
+
}
|
|
75
|
+
/**
|
|
76
|
+
* Product identity verification result with multi-evidence approach.
|
|
77
|
+
* R1-B: Ensures we never output data for the wrong product.
|
|
78
|
+
*/
|
|
79
|
+
export interface ProductIdentity {
|
|
80
|
+
/** ASIN or SKU requested by the user (from URL). */
|
|
81
|
+
requestedId: string;
|
|
82
|
+
/** ASIN or SKU observed as selected on the page (from DOM multi-evidence). */
|
|
83
|
+
observedSelectedId: string;
|
|
84
|
+
/** Parent ASIN if this is a variant product. */
|
|
85
|
+
parentId: string | null;
|
|
86
|
+
/** Selected variant attributes if applicable. */
|
|
87
|
+
selectedVariants: readonly ProductVariant[];
|
|
88
|
+
/** Why the selected subject could or could not be verified. */
|
|
89
|
+
status: 'matched' | 'mismatched' | 'unverified' | 'conflicting';
|
|
90
|
+
/** True only when the observed selected product is the requested product. */
|
|
91
|
+
identityMatch: boolean;
|
|
92
|
+
/** CSS selectors or paths that contributed to identity determination. */
|
|
93
|
+
identityEvidence: readonly string[];
|
|
94
|
+
}
|
|
95
|
+
/**
|
|
96
|
+
* Quote/price state classification.
|
|
97
|
+
* R1-C: Distinguishes "definitely absent" from "not found yet" from "present".
|
|
98
|
+
*/
|
|
99
|
+
export declare enum QuoteState {
|
|
100
|
+
/** Quote found and extracted successfully. */
|
|
101
|
+
Present = "present",
|
|
102
|
+
/** Evidence that quote does not exist (e.g., "Currently unavailable"). */
|
|
103
|
+
AbsentObserved = "absent_observed",
|
|
104
|
+
/** Not found in current extraction, may exist elsewhere. */
|
|
105
|
+
Unobserved = "unobserved",
|
|
106
|
+
/** Multiple conflicting quotes found. */
|
|
107
|
+
Conflicting = "conflicting"
|
|
108
|
+
}
|
|
109
|
+
/**
|
|
110
|
+
* Facts a product-detail page asserted about the product it is about.
|
|
111
|
+
* Every field is independently nullable: a page may declare a price and no
|
|
112
|
+
* SKU, and inventing the missing one is worse than reporting null.
|
|
113
|
+
*/
|
|
114
|
+
export interface ProductFacts {
|
|
115
|
+
name: ProductFact | null;
|
|
116
|
+
price: ProductFact | null;
|
|
117
|
+
priceCurrency: ProductFact | null;
|
|
118
|
+
sku: ProductFact | null;
|
|
119
|
+
brand: ProductFact | null;
|
|
120
|
+
availability: ProductFact | null;
|
|
121
|
+
/** Rich product facts are additive so older extractors remain valid. */
|
|
122
|
+
kind?: 'physical' | 'subscription' | 'unknown';
|
|
123
|
+
subjectId?: ProductFact | null;
|
|
124
|
+
prices?: readonly ProductPrice[];
|
|
125
|
+
seller?: ProductFact | null;
|
|
126
|
+
deliveryLocation?: ProductFact | null;
|
|
127
|
+
rating?: ProductFact | null;
|
|
128
|
+
reviewCount?: ProductFact | null;
|
|
129
|
+
images?: readonly ProductFact[];
|
|
130
|
+
variants?: readonly ProductVariant[];
|
|
131
|
+
specifications?: Readonly<Record<string, ProductFact>>;
|
|
132
|
+
/** R1-B: Multi-evidence identity verification. */
|
|
133
|
+
identity?: ProductIdentity;
|
|
134
|
+
/** R1-C: Quote state classification. */
|
|
135
|
+
quoteState?: QuoteState;
|
|
136
|
+
}
|
|
137
|
+
/**
|
|
138
|
+
* A value the page states under its own label: a two-cell table row (a `<th>`
|
|
139
|
+
* label and a `<td>` value) or a definition-list pair (one `<dt>`, one
|
|
140
|
+
* `<dd>`) in the main content. Both texts are as the page shows them, with
|
|
141
|
+
* whitespace collapsed; nothing is normalized.
|
|
142
|
+
*/
|
|
143
|
+
export interface LabelledValue {
|
|
144
|
+
label: string;
|
|
145
|
+
value: string;
|
|
146
|
+
/**
|
|
147
|
+
* `table[i] tr[j]` or `dl[i] dt[j]`: zero-based, in document order, the
|
|
148
|
+
* table or list among those in the main content and the row or term in it.
|
|
149
|
+
*/
|
|
150
|
+
path: string;
|
|
151
|
+
}
|
|
152
|
+
/**
|
|
153
|
+
* What the page's own markup declares about the page, read from the whole
|
|
154
|
+
* document before cleaning. Each value is that declaration or null when the
|
|
155
|
+
* page makes none: nothing is inferred from the URL, the content or another
|
|
156
|
+
* tag (no `og:description` for a missing description, no `/favicon.ico` for a
|
|
157
|
+
* missing icon).
|
|
158
|
+
*/
|
|
159
|
+
export interface PageMetadata {
|
|
160
|
+
/**
|
|
161
|
+
* The document's `<title>`, whitespace collapsed as `document.title` does
|
|
162
|
+
* (an SVG `<title>` does not count). Unlike `document.title`, the content
|
|
163
|
+
* title, it never comes from a heading.
|
|
164
|
+
*/
|
|
165
|
+
title: string | null;
|
|
166
|
+
/** `<meta name="description">`. */
|
|
167
|
+
description: string | null;
|
|
168
|
+
/** `<html lang>`; when `<html>` has no lang attribute, `<meta http-equiv="content-language">`. */
|
|
169
|
+
language: string | null;
|
|
170
|
+
/** `<meta name="keywords">` as declared, not split. */
|
|
171
|
+
keywords: string | null;
|
|
172
|
+
/** `<meta name="robots">`. */
|
|
173
|
+
robots: string | null;
|
|
174
|
+
/** The first `<link rel~="icon">` that resolves, against the document base URL, to an http(s) URL. */
|
|
175
|
+
favicon: string | null;
|
|
176
|
+
/** The first `<link rel~="canonical">` that resolves to an http(s) URL. */
|
|
177
|
+
canonicalUrl: string | null;
|
|
178
|
+
/** `<meta property="og:title">` (or `name=`). */
|
|
179
|
+
ogTitle?: string;
|
|
180
|
+
/** `og:description`; an empty `content` is no declaration. */
|
|
181
|
+
ogDescription?: string;
|
|
182
|
+
/** `og:url`, resolved against the document base URL when it parses, else as written. */
|
|
183
|
+
ogUrl?: string;
|
|
184
|
+
/** `og:image`, else `og:image:secure_url`, else `og:image:url`; resolved like `ogUrl`. */
|
|
185
|
+
ogImage?: string;
|
|
186
|
+
/** `og:audio`, resolved like `ogUrl`. */
|
|
187
|
+
ogAudio?: string;
|
|
188
|
+
/** `og:video`, else `og:video:secure_url`, else `og:video:url`; resolved like `ogUrl`. */
|
|
189
|
+
ogVideo?: string;
|
|
190
|
+
/** `og:determiner`. */
|
|
191
|
+
ogDeterminer?: string;
|
|
192
|
+
/** `og:locale`. */
|
|
193
|
+
ogLocale?: string;
|
|
194
|
+
/** Every `og:locale:alternate`, in document order. */
|
|
195
|
+
ogLocaleAlternate?: readonly string[];
|
|
196
|
+
/** `og:site_name`. */
|
|
197
|
+
ogSiteName?: string;
|
|
198
|
+
/** `<meta name="dcterms.created">`. */
|
|
199
|
+
dcTermsCreated?: string;
|
|
200
|
+
/** `<meta name="dc.date.created">`. */
|
|
201
|
+
dcDateCreated?: string;
|
|
202
|
+
/** `<meta name="dc.date">`. */
|
|
203
|
+
dcDate?: string;
|
|
204
|
+
/** `<meta name="dcterms.type">`. */
|
|
205
|
+
dcTermsType?: string;
|
|
206
|
+
/** `<meta name="dc.type">`. */
|
|
207
|
+
dcType?: string;
|
|
208
|
+
/** `<meta name="dcterms.audience">`. */
|
|
209
|
+
dcTermsAudience?: string;
|
|
210
|
+
/** `<meta name="dcterms.subject">`. */
|
|
211
|
+
dcTermsSubject?: string;
|
|
212
|
+
/** `<meta name="dc.subject">`. */
|
|
213
|
+
dcSubject?: string;
|
|
214
|
+
/** `<meta name="dc.description">`. */
|
|
215
|
+
dcDescription?: string;
|
|
216
|
+
/** `<meta name="dcterms.keywords">`. */
|
|
217
|
+
dcTermsKeywords?: string;
|
|
218
|
+
/** `article:published_time`, as the page writes it: no date normalisation. */
|
|
219
|
+
publishedTime?: string;
|
|
220
|
+
/** `article:modified_time`, as written. */
|
|
221
|
+
modifiedTime?: string;
|
|
222
|
+
/** Every `article:tag`, in document order. */
|
|
223
|
+
articleTag?: readonly string[];
|
|
224
|
+
/** `article:section`. */
|
|
225
|
+
articleSection?: string;
|
|
226
|
+
}
|
|
227
|
+
export interface DocumentExtraction {
|
|
228
|
+
title: string | null;
|
|
229
|
+
pageType: PageType;
|
|
230
|
+
strategy: ExtractStrategy;
|
|
231
|
+
confidence: number;
|
|
232
|
+
product: ProductFacts | null;
|
|
233
|
+
adapter: AdapterDescriptor;
|
|
234
|
+
entities: readonly ExtractedEntity[];
|
|
235
|
+
adapterValidation?: AdapterValidation;
|
|
236
|
+
/** Label/value pairs of the main content; JSON extraction matches them to schema keys. */
|
|
237
|
+
labelledValues?: readonly LabelledValue[];
|
|
238
|
+
}
|
|
239
|
+
/** The rule that decided a page's data is most likely rendered client-side (see RenderSignals). */
|
|
240
|
+
export type RenderReason = 'empty_table_with_scripts' | 'empty_app_root' | 'script_shell' | 'js_fallback' | 'hydration_shell' | 'aria_busy';
|
|
241
|
+
/** A client-side rendering marker found in the page as received (see RenderSignals). */
|
|
242
|
+
export type RenderMarker = 'hydration_state' | 'app_root_empty' | 'noscript_notice' | 'js_fallback_marker' | 'aria_busy';
|
|
243
|
+
/**
|
|
244
|
+
* Evidence that a page fills its data in with JavaScript after load, read
|
|
245
|
+
* from the server HTML. A shell with an empty app root, a table with no
|
|
246
|
+
* cells beside kilobytes of script, or an explicit "enable JavaScript"
|
|
247
|
+
* fallback all mean the HTTP capture is not the page a browser shows.
|
|
248
|
+
*/
|
|
249
|
+
export interface RenderSignals {
|
|
250
|
+
/** Visible text characters after boilerplate cleaning. */
|
|
251
|
+
textChars: number;
|
|
252
|
+
/** Characters of inline script in the raw document. */
|
|
253
|
+
scriptChars: number;
|
|
254
|
+
/** `<table>` elements with no data cells. */
|
|
255
|
+
emptyTables: number;
|
|
256
|
+
/** Markers found in the page as received. */
|
|
257
|
+
markers: readonly RenderMarker[];
|
|
258
|
+
/** True when the signals say the data is most likely rendered client-side. */
|
|
259
|
+
clientRendered: boolean;
|
|
260
|
+
/**
|
|
261
|
+
* The rule that decided `clientRendered`, null when false. Every rule pairs
|
|
262
|
+
* a structural gap with script presence; a `noscript` notice alone counts
|
|
263
|
+
* only on a thin page or beside hydration state.
|
|
264
|
+
*/
|
|
265
|
+
reason: RenderReason | null;
|
|
266
|
+
}
|
|
267
|
+
export interface ExtractorOutput {
|
|
268
|
+
/** Page title, or null when none could be found. */
|
|
269
|
+
title: string | null;
|
|
270
|
+
/** Extracted main content as HTML. Markdown conversion happens later in the pipeline. */
|
|
271
|
+
mainHtml: string;
|
|
272
|
+
/**
|
|
273
|
+
* Base URL for the page's relative URLs: the first `<base href>` resolved
|
|
274
|
+
* against `options.url`, else `options.url`. mainHtml is a fragment without
|
|
275
|
+
* the page's `<base>` element, so Markdown conversion takes this instead.
|
|
276
|
+
* Null when no absolute URL is known.
|
|
277
|
+
*/
|
|
278
|
+
baseUrl: string | null;
|
|
279
|
+
/** The page's own declarations (title, description, language, ...), from the whole document. */
|
|
280
|
+
metadata: PageMetadata;
|
|
281
|
+
/** 0..1 self-assessed extraction confidence. */
|
|
282
|
+
confidence: number;
|
|
283
|
+
/**
|
|
284
|
+
* True when this page should be routed to a higher tier (LLM/neural).
|
|
285
|
+
* The escalation target is intentionally unimplemented in v0.
|
|
286
|
+
*/
|
|
287
|
+
escalate: boolean;
|
|
288
|
+
/**
|
|
289
|
+
* True when mainHtml came only from the last resort, a list the page repeats
|
|
290
|
+
* (`selectDetectedList`), after every strategy found nothing. A lane still
|
|
291
|
+
* checks such a page for a wall (a login form, a challenge) as it does one
|
|
292
|
+
* with nothing found, since a list sits beside many walls.
|
|
293
|
+
*/
|
|
294
|
+
lastResort?: boolean;
|
|
295
|
+
/** Page type the router detected. */
|
|
296
|
+
pageType: PageType;
|
|
297
|
+
/**
|
|
298
|
+
* The strategy that produced mainHtml. Independent of pageType: a product
|
|
299
|
+
* page may use the table strategy, a forum thread the article cascade.
|
|
300
|
+
*/
|
|
301
|
+
strategy: ExtractStrategy;
|
|
302
|
+
/**
|
|
303
|
+
* Product facts, present only when pageType is 'product'. Null on every
|
|
304
|
+
* other page type — an article has no price, and an empty ProductFacts
|
|
305
|
+
* object would read as "we looked and found none".
|
|
306
|
+
*/
|
|
307
|
+
product?: ProductFacts | null;
|
|
308
|
+
/** Adapter identity and normalized entities are produced directly from HTML. */
|
|
309
|
+
adapter: AdapterDescriptor;
|
|
310
|
+
entities: readonly ExtractedEntity[];
|
|
311
|
+
adapterValidation?: AdapterValidation;
|
|
312
|
+
/**
|
|
313
|
+
* Tables in the fetched HTML with no rows at all: an empty `<thead>` and
|
|
314
|
+
* `<tbody>` waiting for a script to fill them. The data is not in this HTML.
|
|
315
|
+
*/
|
|
316
|
+
emptyTableShells?: number;
|
|
317
|
+
/**
|
|
318
|
+
* Data the page declares its scripts will fetch once they run
|
|
319
|
+
* (`<link rel="preload" as="fetch">`). Whatever the scripts build from it,
|
|
320
|
+
* a table or a chart, is not in this HTML.
|
|
321
|
+
*/
|
|
322
|
+
fetchPreloads?: number;
|
|
323
|
+
/**
|
|
324
|
+
* Client-side rendering signals read from the page as received: whether
|
|
325
|
+
* its data is most likely filled in by scripts after load, and why. A lane
|
|
326
|
+
* that cannot run scripts reads `clientRendered` as a caveat on its
|
|
327
|
+
* capture; one that rendered the page has no use for it.
|
|
328
|
+
*/
|
|
329
|
+
render?: RenderSignals;
|
|
330
|
+
/** Label/value pairs of the main content (see LabelledValue). */
|
|
331
|
+
labelledValues?: readonly LabelledValue[];
|
|
332
|
+
/** Monotonic extractor stage timings. */
|
|
333
|
+
timings: {
|
|
334
|
+
parseMs: number;
|
|
335
|
+
extractMs: number;
|
|
336
|
+
};
|
|
337
|
+
}
|
|
338
|
+
export interface ExtractorOptions {
|
|
339
|
+
/** Final URL after redirects. Site adapters use it only as an identity signal. */
|
|
340
|
+
url?: string;
|
|
341
|
+
/**
|
|
342
|
+
* Prefer less text but correct extraction (tighten thresholds, require a
|
|
343
|
+
* semantic container). Mirrors trafilatura's favor_precision.
|
|
344
|
+
*/
|
|
345
|
+
favorPrecision?: boolean;
|
|
346
|
+
/** When unsure, prefer more text (loosen thresholds). Mirrors favor_recall. */
|
|
347
|
+
favorRecall?: boolean;
|
|
348
|
+
/**
|
|
349
|
+
* Extra CSS selectors to prune from the tree before extraction. Like
|
|
350
|
+
* `includeSelectors`, limited to the selectors that are matched in time
|
|
351
|
+
* proportional to the page (@w2l/extract-tf `invalidSelector`); any other
|
|
352
|
+
* names nothing.
|
|
353
|
+
*/
|
|
354
|
+
pruneSelectors?: readonly string[];
|
|
355
|
+
/**
|
|
356
|
+
* CSS selectors naming the only elements to keep. mainHtml is then a
|
|
357
|
+
* `<body>` holding those elements in document order, copied from the page
|
|
358
|
+
* before cleaning and without `pruneSelectors`, and its confidence is 1
|
|
359
|
+
* when they hold any text or image: what the caller named is the content.
|
|
360
|
+
* Nothing matching gives an empty mainHtml. The page type, title, metadata
|
|
361
|
+
* and product facts are still read from the whole page, and `escalate`
|
|
362
|
+
* stays the page's own signal (the cascade found no main content): a lane
|
|
363
|
+
* reads it for its block check and its offer to the browser, and does not
|
|
364
|
+
* fail a selection for it.
|
|
365
|
+
*/
|
|
366
|
+
includeSelectors?: readonly string[];
|
|
367
|
+
/**
|
|
368
|
+
* Remove ad containers (id or class tokens such as `ad`, `advertisement`,
|
|
369
|
+
* `sponsored`, `promo`) and cookie-consent banners before the cascade
|
|
370
|
+
* runs. Default true, which is what every extraction did before the
|
|
371
|
+
* switch existed; false keeps them. The structural cleaning (scripts,
|
|
372
|
+
* styles, navigation, forms) is not affected.
|
|
373
|
+
*/
|
|
374
|
+
blockAds?: boolean;
|
|
375
|
+
}
|
|
376
|
+
export interface Extractor {
|
|
377
|
+
extract(html: string, options?: ExtractorOptions): ExtractorOutput;
|
|
378
|
+
}
|
|
379
|
+
//# sourceMappingURL=extractor.d.ts.map
|