@octocrawl/sdk 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (74) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +17 -0
  3. package/dist/index.cjs +1278 -0
  4. package/dist/index.js +1240 -0
  5. package/dist/types/cjs/client.d.ts +362 -0
  6. package/dist/types/cjs/contracts/access.d.ts +166 -0
  7. package/dist/types/cjs/contracts/actions.d.ts +191 -0
  8. package/dist/types/cjs/contracts/api.d.ts +891 -0
  9. package/dist/types/cjs/contracts/benchmark.d.ts +116 -0
  10. package/dist/types/cjs/contracts/checkpoint.d.ts +164 -0
  11. package/dist/types/cjs/contracts/compliance.d.ts +412 -0
  12. package/dist/types/cjs/contracts/crawl.d.ts +302 -0
  13. package/dist/types/cjs/contracts/delivery.d.ts +136 -0
  14. package/dist/types/cjs/contracts/evidenceRecord.d.ts +192 -0
  15. package/dist/types/cjs/contracts/execution.d.ts +197 -0
  16. package/dist/types/cjs/contracts/extractor.d.ts +379 -0
  17. package/dist/types/cjs/contracts/file.d.ts +117 -0
  18. package/dist/types/cjs/contracts/firecrawl.d.ts +258 -0
  19. package/dist/types/cjs/contracts/groundTruth.d.ts +77 -0
  20. package/dist/types/cjs/contracts/identityBundle.d.ts +70 -0
  21. package/dist/types/cjs/contracts/index.d.ts +30 -0
  22. package/dist/types/cjs/contracts/map.d.ts +180 -0
  23. package/dist/types/cjs/contracts/monitor.d.ts +217 -0
  24. package/dist/types/cjs/contracts/monitorConfig.d.ts +9 -0
  25. package/dist/types/cjs/contracts/policy.d.ts +93 -0
  26. package/dist/types/cjs/contracts/proxy.d.ts +52 -0
  27. package/dist/types/cjs/contracts/recipe.d.ts +74 -0
  28. package/dist/types/cjs/contracts/regexSafety.d.ts +31 -0
  29. package/dist/types/cjs/contracts/result.d.ts +503 -0
  30. package/dist/types/cjs/contracts/session.d.ts +51 -0
  31. package/dist/types/cjs/contracts/ssrf.d.ts +16 -0
  32. package/dist/types/cjs/contracts/status.d.ts +27 -0
  33. package/dist/types/cjs/contracts/structured.d.ts +185 -0
  34. package/dist/types/cjs/contracts/tableMarkdown.d.ts +58 -0
  35. package/dist/types/cjs/contracts/tokens.d.ts +47 -0
  36. package/dist/types/cjs/index.d.ts +9 -0
  37. package/dist/types/cjs/package.json +1 -0
  38. package/dist/types/cjs/version.d.ts +8 -0
  39. package/dist/types/cjs/watcher.d.ts +151 -0
  40. package/dist/types/esm/client.d.ts +362 -0
  41. package/dist/types/esm/contracts/access.d.ts +166 -0
  42. package/dist/types/esm/contracts/actions.d.ts +191 -0
  43. package/dist/types/esm/contracts/api.d.ts +891 -0
  44. package/dist/types/esm/contracts/benchmark.d.ts +116 -0
  45. package/dist/types/esm/contracts/checkpoint.d.ts +164 -0
  46. package/dist/types/esm/contracts/compliance.d.ts +412 -0
  47. package/dist/types/esm/contracts/crawl.d.ts +302 -0
  48. package/dist/types/esm/contracts/delivery.d.ts +136 -0
  49. package/dist/types/esm/contracts/evidenceRecord.d.ts +192 -0
  50. package/dist/types/esm/contracts/execution.d.ts +197 -0
  51. package/dist/types/esm/contracts/extractor.d.ts +379 -0
  52. package/dist/types/esm/contracts/file.d.ts +117 -0
  53. package/dist/types/esm/contracts/firecrawl.d.ts +258 -0
  54. package/dist/types/esm/contracts/groundTruth.d.ts +77 -0
  55. package/dist/types/esm/contracts/identityBundle.d.ts +70 -0
  56. package/dist/types/esm/contracts/index.d.ts +30 -0
  57. package/dist/types/esm/contracts/map.d.ts +180 -0
  58. package/dist/types/esm/contracts/monitor.d.ts +217 -0
  59. package/dist/types/esm/contracts/monitorConfig.d.ts +9 -0
  60. package/dist/types/esm/contracts/policy.d.ts +93 -0
  61. package/dist/types/esm/contracts/proxy.d.ts +52 -0
  62. package/dist/types/esm/contracts/recipe.d.ts +74 -0
  63. package/dist/types/esm/contracts/regexSafety.d.ts +31 -0
  64. package/dist/types/esm/contracts/result.d.ts +503 -0
  65. package/dist/types/esm/contracts/session.d.ts +51 -0
  66. package/dist/types/esm/contracts/ssrf.d.ts +16 -0
  67. package/dist/types/esm/contracts/status.d.ts +27 -0
  68. package/dist/types/esm/contracts/structured.d.ts +185 -0
  69. package/dist/types/esm/contracts/tableMarkdown.d.ts +58 -0
  70. package/dist/types/esm/contracts/tokens.d.ts +47 -0
  71. package/dist/types/esm/index.d.ts +9 -0
  72. package/dist/types/esm/version.d.ts +8 -0
  73. package/dist/types/esm/watcher.d.ts +151 -0
  74. package/package.json +40 -0
@@ -0,0 +1,197 @@
1
+ import type { RobotsOverride } from './compliance.js';
2
+ import type { FetchWarning, TraceEvent } from './result.js';
3
+ import type { AttributeSelector, ListFormatRequest, ScreenshotOptions } from './structured.js';
4
+ import type { PageAction } from './actions.js';
5
+ /** In-process cancellation and an absolute UTC deadline. Never serialize signal. */
6
+ export interface ExecutionContext {
7
+ signal?: AbortSignal;
8
+ deadlineAt?: number;
9
+ /** Persist publisher cooldown immediately, before an interruptible inline wait. */
10
+ onRetryAfter?: (url: string, retryAt: number) => void;
11
+ /**
12
+ * Hear of a recorded robots override the moment a lane applies it, before
13
+ * its request goes out: the ladder keeps the override on the run's answer
14
+ * when that lane never returns (a deadline) or another rung's result answers.
15
+ */
16
+ onRobotsOverride?: (applied: RobotsOverrideApplied) => void;
17
+ }
18
+ /** What a lane reports when it sets a robots.txt rule aside under a recorded override. */
19
+ export interface RobotsOverrideApplied {
20
+ /** The lane's `robots_checked`, `robots_disallowed` and `robots_overridden` trace events, in that order. */
21
+ trace: readonly TraceEvent[];
22
+ /** The `robots_overridden` warning the lane's own result carries. */
23
+ warning: FetchWarning;
24
+ }
25
+ /**
26
+ * What the caller asked a lane to capture from one page. It changes the
27
+ * content a lane emits, never the Evidence (hashes, status rules).
28
+ */
29
+ export interface FetchOptions {
30
+ /** false: Markdown of the whole page body, header, navigation and footer kept. Default true. */
31
+ onlyMainContent?: boolean;
32
+ /** Milliseconds a browser rung waits after load and stability before capture. Default 0. */
33
+ waitFor?: number;
34
+ /**
35
+ * The caller's own `timeout`, when it set one. The deadline itself is the
36
+ * ExecutionContext's; this says the caller chose it, so a lane waits for a
37
+ * slow server (HTTP headers and body, browser navigation) until that
38
+ * deadline instead of stopping at its default caps.
39
+ */
40
+ timeout?: number;
41
+ /**
42
+ * Bytes a file (PDF, CSV, XLSX, ZIP, JSON, text) may have, below the
43
+ * operator's cap (`NetworkPolicy.maxFileBytes`); a larger file is failed
44
+ * with `body_too_large` and not saved. Web pages keep `maxBodyBytes`.
45
+ */
46
+ maxFileBytes?: number;
47
+ /**
48
+ * How a PDF response is read (Firecrawl's `parsers`): absent reads every
49
+ * PDF's text layer with W2L's defaults; `[]` reads none (the file is saved
50
+ * as received, without text); one `pdf` entry sets the options. Other
51
+ * files are unaffected.
52
+ */
53
+ parsers?: readonly PdfParser[];
54
+ /**
55
+ * A recorded decision to fetch this one URL although its host's robots.txt
56
+ * disallows it. robots.txt is still read and its verdict recorded; the
57
+ * override goes into the trace, the warnings and, in the browser lane, the
58
+ * compliance record. An unreachable robots.txt is not set aside. The HTTP
59
+ * and local browser lanes apply it; the provider lane takes none.
60
+ * Set per URL by the caller (a scrape's `robotsOverride`, a batch's
61
+ * `robotsOverrides` entry), never by a batch or crawl for every page.
62
+ */
63
+ robotsOverride?: RobotsOverride;
64
+ /**
65
+ * CSS selectors naming the only elements to keep. The content is those
66
+ * elements, in document order, copied from the page before anything is
67
+ * cleaned away, so a named navigation stays; `onlyMainContent` no longer
68
+ * chooses the content. Nothing matching is an empty answer. The page's
69
+ * type, title and metadata are still read from the whole page. The API
70
+ * refuses a selector the extractor does not match (@w2l/extract-tf
71
+ * `invalidSelector`) and a list of more parts than it matches for one
72
+ * list (`MAX_SELECTOR_PARTS`); a lane given either reads it as naming
73
+ * nothing.
74
+ */
75
+ includeTags?: readonly string[];
76
+ /**
77
+ * CSS selectors removed, with everything inside them, before the content
78
+ * is taken: from the main content, from the whole page
79
+ * (`onlyMainContent: false`) and from an `includeTags` selection alike.
80
+ * The same selectors as `includeTags`.
81
+ */
82
+ excludeTags?: readonly string[];
83
+ /**
84
+ * Extra request headers, lower-cased names, validated by the API
85
+ * (`readHeaders`): never the User-Agent, a client hint, a credential or a
86
+ * transport header. The HTTP and local browser lanes send them with the
87
+ * requested URL and its same-origin hops and subresources, after the
88
+ * declared identity, and record them (`request_headers_added`). Both lanes
89
+ * fetch a redirect hop to another origin with the identity alone and say so
90
+ * (`custom_headers_withheld`); on the browser lane the headers are added per
91
+ * request through Chromium's request interception, which judges every hop
92
+ * and every file the page loads by its own origin, so a navigation the page
93
+ * makes to another origin gets none either. robots.txt is fetched with the
94
+ * identity alone. Everything here is on the record.
95
+ */
96
+ headers?: Readonly<Record<string, string>>;
97
+ /**
98
+ * Fetch as the declared mobile Chrome identity (Android User-Agent, mobile
99
+ * client hints, 412x915 viewport, touch) instead of the desktop one. A
100
+ * second declared identity, not a disguise: it passes the same coherence
101
+ * and honesty checks, and robots.txt is evaluated against its User-Agent.
102
+ * Default false. Refused with research mode, which declares a bot.
103
+ */
104
+ mobile?: boolean;
105
+ /**
106
+ * Local only: load a site whose certificate does not verify (self-signed,
107
+ * expired, wrong name). The lanes relax verification for this one fetch
108
+ * and its robots.txt lookup, say so in the trace
109
+ * (`tls_verification_skipped`) and in a `tls_unverified` warning. Default
110
+ * false: a certificate failure is `failed` / `tls_error`. A hosted engine
111
+ * refuses the option.
112
+ */
113
+ skipTlsVerification?: boolean;
114
+ /**
115
+ * Abort requests to a bundled list of ad-serving hosts on the local
116
+ * browser lane (`ads_blocked`), and remove ad and cookie-banner elements
117
+ * before extraction on every lane. Default true; false keeps them in the
118
+ * Markdown and `html`. The hosted browser's host allowlist stays in force
119
+ * whatever this says.
120
+ */
121
+ blockAds?: boolean;
122
+ /**
123
+ * Carry `html` on a contentful result: the cleaned HTML its Markdown was
124
+ * written from. Set from the requested formats (`html`), not by a caller.
125
+ */
126
+ includeHtml?: boolean;
127
+ /**
128
+ * Carry `rawHtml` on a contentful result: the page as the lane received
129
+ * it. Set from the requested formats (`rawHtml`), not by a caller.
130
+ */
131
+ includeRawHtml?: boolean;
132
+ /**
133
+ * Carry `images` on a contentful result: every image URL of the whole
134
+ * document as received (the rendered DOM on a browser lane). Set from the
135
+ * requested formats (`images`), not by a caller.
136
+ */
137
+ includeImages?: boolean;
138
+ /**
139
+ * Carry `tables` on a contentful result: every data table of the content
140
+ * the Markdown was written from, as data and CSV. Set from the requested
141
+ * formats (`tables`), not by a caller.
142
+ */
143
+ includeTables?: boolean;
144
+ /**
145
+ * Carry `attributes` on a contentful result: for each selector, the named
146
+ * attribute's values on the elements it matches in the document as
147
+ * received. Set from the requested formats (an `attributes` entry), not
148
+ * by a caller; the API has checked the selectors (`invalidSelector`).
149
+ */
150
+ attributes?: readonly AttributeSelector[];
151
+ /** The `list` format's request, set from the requested formats, not by a caller; the API has checked its selectors. */
152
+ list?: ListFormatRequest;
153
+ /**
154
+ * Whether an `<img>` whose `src` is a `data:` URI is left out of the
155
+ * Markdown, its alt text kept (Firecrawl's `removeBase64Images`). Default
156
+ * true, which every lane always did; false keeps the image as
157
+ * `![alt](data:…)`, which the token count then counts. `html` and
158
+ * `rawHtml` are never rewritten. A rendering choice, not a fetch fact: no
159
+ * trace event or warning.
160
+ */
161
+ removeBase64Images?: boolean;
162
+ /**
163
+ * Capture the rendered page as an image (the `screenshot` format): a PNG,
164
+ * or a JPEG at `quality`, of the viewport (the request's `viewport` within
165
+ * the declared screen, else the declared one) or of the document's whole
166
+ * height (`fullPage`, without scrolling first), taken after load,
167
+ * stability and `waitFor` and before the DOM is read, CSS-pixel sized. The
168
+ * local browser lane alone honours it, and the API selects that lane alone
169
+ * for such a request; the http and provider lanes ignore it. Set from the
170
+ * requested formats (a `screenshot` entry), not by a caller.
171
+ */
172
+ screenshot?: ScreenshotOptions;
173
+ /**
174
+ * Steps the local browser runs on the page after load, stability and
175
+ * `waitFor`, and before the screenshot format and the DOM are read
176
+ * (`actions`). The local browser lanes alone run them; the API selects
177
+ * those lanes alone for such a request.
178
+ */
179
+ actions?: readonly PageAction[];
180
+ }
181
+ /** The largest `maxPages` a pdf parser entry may ask for. */
182
+ export declare const MAX_PDF_PAGES = 10000;
183
+ /**
184
+ * The `pdf` entry of `parsers`. W2L reads a PDF's text layer and runs no
185
+ * OCR, so `mode` is `fast` or `auto`, both that reader (`ocr` is refused).
186
+ */
187
+ export interface PdfParser {
188
+ type: 'pdf';
189
+ mode?: 'fast' | 'auto';
190
+ /** Read at most this many pages, from the first (1 to MAX_PDF_PAGES); default 1000. A document cut by it is `success` with a `page_cap` warning: the cut was asked for. */
191
+ maxPages?: number;
192
+ /** Also return each page's Markdown as `pages: [{ pageNumber, markdown }]`. Default false. */
193
+ pages?: boolean;
194
+ /** A `<!-- page N -->` line before each page's text. Default true natively, false on `/fc` as on Firecrawl. */
195
+ pageMarkers?: boolean;
196
+ }
197
+ //# sourceMappingURL=execution.d.ts.map
@@ -0,0 +1,379 @@
1
+ /**
2
+ * Extractor contract: the seam every main-content extractor implements
3
+ * (extract-tf, the v0 readability wrapper, and any future tier).
4
+ */
5
+ /** Page shape the extractor routed to. */
6
+ export type PageType = 'article' | 'listing' | 'collection' | 'product' | 'forum';
7
+ /** Extraction strategy that produced mainHtml, independent of pageType. */
8
+ export type ExtractStrategy = 'article' | 'list' | 'table' | 'product';
9
+ /**
10
+ * Where a product fact came from. The ordering is a strength ordering:
11
+ * `jsonld` and `microdata` are the publisher's own machine-readable claim,
12
+ * `meta` is a tag written for machines, `text` is our reading of rendered
13
+ * prose, `inferred` is our derivation from context (e.g., currency from domain).
14
+ * A price we matched out of visible text is a weaker claim than one
15
+ * the publisher declared, and a consumer is entitled to know which it got.
16
+ */
17
+ export type ProductFactSource = 'jsonld' | 'microdata' | 'meta' | 'dom' | 'text' | 'inferred' | 'model';
18
+ /** Evidence classes allowed on the normalized cross-site entity surface. */
19
+ export type EntityFieldSource = 'jsonld' | 'microdata' | 'meta' | 'hydration' | 'dom' | 'inferred';
20
+ export type EntityFieldStatus = 'confirmed' | 'unconfirmed';
21
+ export type EntityType = 'product' | 'post' | 'thread' | 'comment' | 'profile' | 'community' | 'video' | 'article';
22
+ export type AdapterStatus = 'generic' | 'beta adapter' | 'verified adapter' | 'unsupported';
23
+ export type EntityValue = string | number | boolean | null | readonly EntityValue[] | {
24
+ readonly [key: string]: EntityValue;
25
+ };
26
+ export interface EntityField<T extends EntityValue = EntityValue> {
27
+ /** Value exactly as observed on the page. */
28
+ raw: T;
29
+ /** Stable value used across adapters. */
30
+ normalized: T;
31
+ source: EntityFieldSource;
32
+ /** CSS selector, JSON Pointer, URL component, or another public-page location. */
33
+ path: string;
34
+ status: EntityFieldStatus;
35
+ }
36
+ export interface ExtractedEntity {
37
+ type: EntityType;
38
+ id: string | null;
39
+ fields: Readonly<Record<string, EntityField>>;
40
+ /** IDs of other entities in this response, e.g. parent/author/community. */
41
+ relationships: Readonly<Record<string, string | readonly string[] | null>>;
42
+ }
43
+ export interface AdapterDescriptor {
44
+ id: string;
45
+ version: string;
46
+ status: AdapterStatus;
47
+ }
48
+ /** Identity and provenance checks performed by a site adapter. */
49
+ export interface AdapterValidation {
50
+ valid: boolean;
51
+ issues: readonly string[];
52
+ }
53
+ /** One product fact plus the evidence class it was drawn from. */
54
+ export interface ProductFact {
55
+ /** The value exactly as the page carried it. Never normalized — a
56
+ * normalized price is a claim we would be making, not one we read. */
57
+ value: string;
58
+ source: ProductFactSource;
59
+ /** JSON Pointer or CSS selector locating the evidence when available. */
60
+ path?: string;
61
+ }
62
+ export interface ProductPrice {
63
+ amount: ProductFact;
64
+ currency: ProductFact | null;
65
+ priceType: 'current' | 'list' | 'unit' | 'subscription' | 'other';
66
+ seller: ProductFact | null;
67
+ }
68
+ export interface ProductVariant {
69
+ name: string;
70
+ value: string;
71
+ selected: boolean;
72
+ source: ProductFactSource;
73
+ path?: string;
74
+ }
75
+ /**
76
+ * Product identity verification result with multi-evidence approach.
77
+ * R1-B: Ensures we never output data for the wrong product.
78
+ */
79
+ export interface ProductIdentity {
80
+ /** ASIN or SKU requested by the user (from URL). */
81
+ requestedId: string;
82
+ /** ASIN or SKU observed as selected on the page (from DOM multi-evidence). */
83
+ observedSelectedId: string;
84
+ /** Parent ASIN if this is a variant product. */
85
+ parentId: string | null;
86
+ /** Selected variant attributes if applicable. */
87
+ selectedVariants: readonly ProductVariant[];
88
+ /** Why the selected subject could or could not be verified. */
89
+ status: 'matched' | 'mismatched' | 'unverified' | 'conflicting';
90
+ /** True only when the observed selected product is the requested product. */
91
+ identityMatch: boolean;
92
+ /** CSS selectors or paths that contributed to identity determination. */
93
+ identityEvidence: readonly string[];
94
+ }
95
+ /**
96
+ * Quote/price state classification.
97
+ * R1-C: Distinguishes "definitely absent" from "not found yet" from "present".
98
+ */
99
+ export declare enum QuoteState {
100
+ /** Quote found and extracted successfully. */
101
+ Present = "present",
102
+ /** Evidence that quote does not exist (e.g., "Currently unavailable"). */
103
+ AbsentObserved = "absent_observed",
104
+ /** Not found in current extraction, may exist elsewhere. */
105
+ Unobserved = "unobserved",
106
+ /** Multiple conflicting quotes found. */
107
+ Conflicting = "conflicting"
108
+ }
109
+ /**
110
+ * Facts a product-detail page asserted about the product it is about.
111
+ * Every field is independently nullable: a page may declare a price and no
112
+ * SKU, and inventing the missing one is worse than reporting null.
113
+ */
114
+ export interface ProductFacts {
115
+ name: ProductFact | null;
116
+ price: ProductFact | null;
117
+ priceCurrency: ProductFact | null;
118
+ sku: ProductFact | null;
119
+ brand: ProductFact | null;
120
+ availability: ProductFact | null;
121
+ /** Rich product facts are additive so older extractors remain valid. */
122
+ kind?: 'physical' | 'subscription' | 'unknown';
123
+ subjectId?: ProductFact | null;
124
+ prices?: readonly ProductPrice[];
125
+ seller?: ProductFact | null;
126
+ deliveryLocation?: ProductFact | null;
127
+ rating?: ProductFact | null;
128
+ reviewCount?: ProductFact | null;
129
+ images?: readonly ProductFact[];
130
+ variants?: readonly ProductVariant[];
131
+ specifications?: Readonly<Record<string, ProductFact>>;
132
+ /** R1-B: Multi-evidence identity verification. */
133
+ identity?: ProductIdentity;
134
+ /** R1-C: Quote state classification. */
135
+ quoteState?: QuoteState;
136
+ }
137
+ /**
138
+ * A value the page states under its own label: a two-cell table row (a `<th>`
139
+ * label and a `<td>` value) or a definition-list pair (one `<dt>`, one
140
+ * `<dd>`) in the main content. Both texts are as the page shows them, with
141
+ * whitespace collapsed; nothing is normalized.
142
+ */
143
+ export interface LabelledValue {
144
+ label: string;
145
+ value: string;
146
+ /**
147
+ * `table[i] tr[j]` or `dl[i] dt[j]`: zero-based, in document order, the
148
+ * table or list among those in the main content and the row or term in it.
149
+ */
150
+ path: string;
151
+ }
152
+ /**
153
+ * What the page's own markup declares about the page, read from the whole
154
+ * document before cleaning. Each value is that declaration or null when the
155
+ * page makes none: nothing is inferred from the URL, the content or another
156
+ * tag (no `og:description` for a missing description, no `/favicon.ico` for a
157
+ * missing icon).
158
+ */
159
+ export interface PageMetadata {
160
+ /**
161
+ * The document's `<title>`, whitespace collapsed as `document.title` does
162
+ * (an SVG `<title>` does not count). Unlike `document.title`, the content
163
+ * title, it never comes from a heading.
164
+ */
165
+ title: string | null;
166
+ /** `<meta name="description">`. */
167
+ description: string | null;
168
+ /** `<html lang>`; when `<html>` has no lang attribute, `<meta http-equiv="content-language">`. */
169
+ language: string | null;
170
+ /** `<meta name="keywords">` as declared, not split. */
171
+ keywords: string | null;
172
+ /** `<meta name="robots">`. */
173
+ robots: string | null;
174
+ /** The first `<link rel~="icon">` that resolves, against the document base URL, to an http(s) URL. */
175
+ favicon: string | null;
176
+ /** The first `<link rel~="canonical">` that resolves to an http(s) URL. */
177
+ canonicalUrl: string | null;
178
+ /** `<meta property="og:title">` (or `name=`). */
179
+ ogTitle?: string;
180
+ /** `og:description`; an empty `content` is no declaration. */
181
+ ogDescription?: string;
182
+ /** `og:url`, resolved against the document base URL when it parses, else as written. */
183
+ ogUrl?: string;
184
+ /** `og:image`, else `og:image:secure_url`, else `og:image:url`; resolved like `ogUrl`. */
185
+ ogImage?: string;
186
+ /** `og:audio`, resolved like `ogUrl`. */
187
+ ogAudio?: string;
188
+ /** `og:video`, else `og:video:secure_url`, else `og:video:url`; resolved like `ogUrl`. */
189
+ ogVideo?: string;
190
+ /** `og:determiner`. */
191
+ ogDeterminer?: string;
192
+ /** `og:locale`. */
193
+ ogLocale?: string;
194
+ /** Every `og:locale:alternate`, in document order. */
195
+ ogLocaleAlternate?: readonly string[];
196
+ /** `og:site_name`. */
197
+ ogSiteName?: string;
198
+ /** `<meta name="dcterms.created">`. */
199
+ dcTermsCreated?: string;
200
+ /** `<meta name="dc.date.created">`. */
201
+ dcDateCreated?: string;
202
+ /** `<meta name="dc.date">`. */
203
+ dcDate?: string;
204
+ /** `<meta name="dcterms.type">`. */
205
+ dcTermsType?: string;
206
+ /** `<meta name="dc.type">`. */
207
+ dcType?: string;
208
+ /** `<meta name="dcterms.audience">`. */
209
+ dcTermsAudience?: string;
210
+ /** `<meta name="dcterms.subject">`. */
211
+ dcTermsSubject?: string;
212
+ /** `<meta name="dc.subject">`. */
213
+ dcSubject?: string;
214
+ /** `<meta name="dc.description">`. */
215
+ dcDescription?: string;
216
+ /** `<meta name="dcterms.keywords">`. */
217
+ dcTermsKeywords?: string;
218
+ /** `article:published_time`, as the page writes it: no date normalisation. */
219
+ publishedTime?: string;
220
+ /** `article:modified_time`, as written. */
221
+ modifiedTime?: string;
222
+ /** Every `article:tag`, in document order. */
223
+ articleTag?: readonly string[];
224
+ /** `article:section`. */
225
+ articleSection?: string;
226
+ }
227
+ export interface DocumentExtraction {
228
+ title: string | null;
229
+ pageType: PageType;
230
+ strategy: ExtractStrategy;
231
+ confidence: number;
232
+ product: ProductFacts | null;
233
+ adapter: AdapterDescriptor;
234
+ entities: readonly ExtractedEntity[];
235
+ adapterValidation?: AdapterValidation;
236
+ /** Label/value pairs of the main content; JSON extraction matches them to schema keys. */
237
+ labelledValues?: readonly LabelledValue[];
238
+ }
239
+ /** The rule that decided a page's data is most likely rendered client-side (see RenderSignals). */
240
+ export type RenderReason = 'empty_table_with_scripts' | 'empty_app_root' | 'script_shell' | 'js_fallback' | 'hydration_shell' | 'aria_busy';
241
+ /** A client-side rendering marker found in the page as received (see RenderSignals). */
242
+ export type RenderMarker = 'hydration_state' | 'app_root_empty' | 'noscript_notice' | 'js_fallback_marker' | 'aria_busy';
243
+ /**
244
+ * Evidence that a page fills its data in with JavaScript after load, read
245
+ * from the server HTML. A shell with an empty app root, a table with no
246
+ * cells beside kilobytes of script, or an explicit "enable JavaScript"
247
+ * fallback all mean the HTTP capture is not the page a browser shows.
248
+ */
249
+ export interface RenderSignals {
250
+ /** Visible text characters after boilerplate cleaning. */
251
+ textChars: number;
252
+ /** Characters of inline script in the raw document. */
253
+ scriptChars: number;
254
+ /** `<table>` elements with no data cells. */
255
+ emptyTables: number;
256
+ /** Markers found in the page as received. */
257
+ markers: readonly RenderMarker[];
258
+ /** True when the signals say the data is most likely rendered client-side. */
259
+ clientRendered: boolean;
260
+ /**
261
+ * The rule that decided `clientRendered`, null when false. Every rule pairs
262
+ * a structural gap with script presence; a `noscript` notice alone counts
263
+ * only on a thin page or beside hydration state.
264
+ */
265
+ reason: RenderReason | null;
266
+ }
267
+ export interface ExtractorOutput {
268
+ /** Page title, or null when none could be found. */
269
+ title: string | null;
270
+ /** Extracted main content as HTML. Markdown conversion happens later in the pipeline. */
271
+ mainHtml: string;
272
+ /**
273
+ * Base URL for the page's relative URLs: the first `<base href>` resolved
274
+ * against `options.url`, else `options.url`. mainHtml is a fragment without
275
+ * the page's `<base>` element, so Markdown conversion takes this instead.
276
+ * Null when no absolute URL is known.
277
+ */
278
+ baseUrl: string | null;
279
+ /** The page's own declarations (title, description, language, ...), from the whole document. */
280
+ metadata: PageMetadata;
281
+ /** 0..1 self-assessed extraction confidence. */
282
+ confidence: number;
283
+ /**
284
+ * True when this page should be routed to a higher tier (LLM/neural).
285
+ * The escalation target is intentionally unimplemented in v0.
286
+ */
287
+ escalate: boolean;
288
+ /**
289
+ * True when mainHtml came only from the last resort, a list the page repeats
290
+ * (`selectDetectedList`), after every strategy found nothing. A lane still
291
+ * checks such a page for a wall (a login form, a challenge) as it does one
292
+ * with nothing found, since a list sits beside many walls.
293
+ */
294
+ lastResort?: boolean;
295
+ /** Page type the router detected. */
296
+ pageType: PageType;
297
+ /**
298
+ * The strategy that produced mainHtml. Independent of pageType: a product
299
+ * page may use the table strategy, a forum thread the article cascade.
300
+ */
301
+ strategy: ExtractStrategy;
302
+ /**
303
+ * Product facts, present only when pageType is 'product'. Null on every
304
+ * other page type — an article has no price, and an empty ProductFacts
305
+ * object would read as "we looked and found none".
306
+ */
307
+ product?: ProductFacts | null;
308
+ /** Adapter identity and normalized entities are produced directly from HTML. */
309
+ adapter: AdapterDescriptor;
310
+ entities: readonly ExtractedEntity[];
311
+ adapterValidation?: AdapterValidation;
312
+ /**
313
+ * Tables in the fetched HTML with no rows at all: an empty `<thead>` and
314
+ * `<tbody>` waiting for a script to fill them. The data is not in this HTML.
315
+ */
316
+ emptyTableShells?: number;
317
+ /**
318
+ * Data the page declares its scripts will fetch once they run
319
+ * (`<link rel="preload" as="fetch">`). Whatever the scripts build from it,
320
+ * a table or a chart, is not in this HTML.
321
+ */
322
+ fetchPreloads?: number;
323
+ /**
324
+ * Client-side rendering signals read from the page as received: whether
325
+ * its data is most likely filled in by scripts after load, and why. A lane
326
+ * that cannot run scripts reads `clientRendered` as a caveat on its
327
+ * capture; one that rendered the page has no use for it.
328
+ */
329
+ render?: RenderSignals;
330
+ /** Label/value pairs of the main content (see LabelledValue). */
331
+ labelledValues?: readonly LabelledValue[];
332
+ /** Monotonic extractor stage timings. */
333
+ timings: {
334
+ parseMs: number;
335
+ extractMs: number;
336
+ };
337
+ }
338
+ export interface ExtractorOptions {
339
+ /** Final URL after redirects. Site adapters use it only as an identity signal. */
340
+ url?: string;
341
+ /**
342
+ * Prefer less text but correct extraction (tighten thresholds, require a
343
+ * semantic container). Mirrors trafilatura's favor_precision.
344
+ */
345
+ favorPrecision?: boolean;
346
+ /** When unsure, prefer more text (loosen thresholds). Mirrors favor_recall. */
347
+ favorRecall?: boolean;
348
+ /**
349
+ * Extra CSS selectors to prune from the tree before extraction. Like
350
+ * `includeSelectors`, limited to the selectors that are matched in time
351
+ * proportional to the page (@w2l/extract-tf `invalidSelector`); any other
352
+ * names nothing.
353
+ */
354
+ pruneSelectors?: readonly string[];
355
+ /**
356
+ * CSS selectors naming the only elements to keep. mainHtml is then a
357
+ * `<body>` holding those elements in document order, copied from the page
358
+ * before cleaning and without `pruneSelectors`, and its confidence is 1
359
+ * when they hold any text or image: what the caller named is the content.
360
+ * Nothing matching gives an empty mainHtml. The page type, title, metadata
361
+ * and product facts are still read from the whole page, and `escalate`
362
+ * stays the page's own signal (the cascade found no main content): a lane
363
+ * reads it for its block check and its offer to the browser, and does not
364
+ * fail a selection for it.
365
+ */
366
+ includeSelectors?: readonly string[];
367
+ /**
368
+ * Remove ad containers (id or class tokens such as `ad`, `advertisement`,
369
+ * `sponsored`, `promo`) and cookie-consent banners before the cascade
370
+ * runs. Default true, which is what every extraction did before the
371
+ * switch existed; false keeps them. The structural cleaning (scripts,
372
+ * styles, navigation, forms) is not affected.
373
+ */
374
+ blockAds?: boolean;
375
+ }
376
+ export interface Extractor {
377
+ extract(html: string, options?: ExtractorOptions): ExtractorOutput;
378
+ }
379
+ //# sourceMappingURL=extractor.d.ts.map