@octocrawl/sdk 0.3.0 → 0.3.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/dist/index.cjs +15 -9
- package/dist/index.js +15 -9
- package/dist/types/cjs/client.d.ts +1 -1
- package/dist/types/cjs/contracts/actions.d.ts +37 -4
- package/dist/types/cjs/contracts/api.d.ts +66 -10
- package/dist/types/cjs/contracts/checkpoint.d.ts +17 -1
- package/dist/types/cjs/contracts/compliance.d.ts +35 -7
- package/dist/types/cjs/contracts/crawl.d.ts +1 -1
- package/dist/types/cjs/contracts/evidenceRecord.d.ts +132 -4
- package/dist/types/cjs/contracts/execution.d.ts +121 -9
- package/dist/types/cjs/contracts/extractor.d.ts +11 -1
- package/dist/types/cjs/contracts/firecrawl.d.ts +1 -1
- package/dist/types/cjs/contracts/map.d.ts +10 -4
- package/dist/types/cjs/contracts/policy.d.ts +2 -1
- package/dist/types/cjs/contracts/proxy.d.ts +8 -0
- package/dist/types/cjs/contracts/result.d.ts +41 -4
- package/dist/types/cjs/contracts/session.d.ts +2 -0
- package/dist/types/cjs/contracts/status.d.ts +1 -1
- package/dist/types/cjs/version.d.ts +1 -1
- package/dist/types/esm/client.d.ts +1 -1
- package/dist/types/esm/contracts/actions.d.ts +37 -4
- package/dist/types/esm/contracts/api.d.ts +66 -10
- package/dist/types/esm/contracts/checkpoint.d.ts +17 -1
- package/dist/types/esm/contracts/compliance.d.ts +35 -7
- package/dist/types/esm/contracts/crawl.d.ts +1 -1
- package/dist/types/esm/contracts/evidenceRecord.d.ts +132 -4
- package/dist/types/esm/contracts/execution.d.ts +121 -9
- package/dist/types/esm/contracts/extractor.d.ts +11 -1
- package/dist/types/esm/contracts/firecrawl.d.ts +1 -1
- package/dist/types/esm/contracts/map.d.ts +10 -4
- package/dist/types/esm/contracts/policy.d.ts +2 -1
- package/dist/types/esm/contracts/proxy.d.ts +8 -0
- package/dist/types/esm/contracts/result.d.ts +41 -4
- package/dist/types/esm/contracts/session.d.ts +2 -0
- package/dist/types/esm/contracts/status.d.ts +1 -1
- package/dist/types/esm/version.d.ts +1 -1
- package/package.json +19 -2
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import type {
|
|
1
|
+
import type { AppliedRobotsOverride } from './compliance.js';
|
|
2
2
|
import type { FetchWarning, TraceEvent } from './result.js';
|
|
3
3
|
import type { AttributeSelector, ListFormatRequest, ScreenshotOptions } from './structured.js';
|
|
4
4
|
import type { PageAction } from './actions.js';
|
|
@@ -14,6 +14,115 @@ export interface ExecutionContext {
|
|
|
14
14
|
* when that lane never returns (a deadline) or another rung's result answers.
|
|
15
15
|
*/
|
|
16
16
|
onRobotsOverride?: (applied: RobotsOverrideApplied) => void;
|
|
17
|
+
/**
|
|
18
|
+
* The task's cookie session (ADR 0005 `egress_sessions`): the cookies a page's responses set,
|
|
19
|
+
* sent again to their site on the task's later pages, by every local rung. Absent: no cookie is
|
|
20
|
+
* kept or sent, as before.
|
|
21
|
+
*/
|
|
22
|
+
cookieSession?: CookieSession;
|
|
23
|
+
/**
|
|
24
|
+
* Hear of each trace event the moment the HTTP lane records it (robots.txt checked, the response arrived, the
|
|
25
|
+
* page extracted), for a caller that shows progress while the fetch runs. The result's `trace` stays the record;
|
|
26
|
+
* a listener's error never changes the fetch. Absent: nothing is told early.
|
|
27
|
+
*/
|
|
28
|
+
onTrace?: (event: TraceEvent) => void;
|
|
29
|
+
/**
|
|
30
|
+
* Hear of each page a paginate step reads the moment it is read (ROADMAP PA item 3), for a caller that keeps
|
|
31
|
+
* them in the task's checkpoint: a run cut at page N then resumes from them through `listResume`. A listener's
|
|
32
|
+
* error never changes the fetch. Absent: nothing is told.
|
|
33
|
+
*/
|
|
34
|
+
onListPage?: (page: ListPageRead) => void;
|
|
35
|
+
/**
|
|
36
|
+
* The pages a paginate step of this URL read before an earlier run was cut, from the task's checkpoint. The
|
|
37
|
+
* browser lane counts them as read: it passes over them on its way along the site's own Next links (the only way
|
|
38
|
+
* to page N+1 when pages have no address of their own), tells and reads the pages after them, and merges every
|
|
39
|
+
* page once. Absent: the list starts from its first page.
|
|
40
|
+
*/
|
|
41
|
+
listResume?: {
|
|
42
|
+
pages: readonly ListPageRead[];
|
|
43
|
+
};
|
|
44
|
+
/**
|
|
45
|
+
* The run's spend ledger (ROADMAP PA item 4): a rung that costs a third party reserves its price ceiling here before
|
|
46
|
+
* its call and settles after, so a run's concurrent pages, retries and providers share one cap and none overruns it.
|
|
47
|
+
* Absent: nothing is reserved, as before (a run with no third-party rung).
|
|
48
|
+
*/
|
|
49
|
+
spend?: SpendLedger;
|
|
50
|
+
}
|
|
51
|
+
/**
|
|
52
|
+
* A spend ledger: the cap a run (or one page of it) may spend on third parties, what is settled and what is reserved.
|
|
53
|
+
* A paid call reserves its price ceiling first and is not made when the ceiling does not fit.
|
|
54
|
+
*/
|
|
55
|
+
export interface SpendLedger {
|
|
56
|
+
/** Reserve `ceilingUsd` against this ledger's cap and every enclosing one; null when it does not fit. */
|
|
57
|
+
reserve(ceilingUsd: number): SpendReservation | null;
|
|
58
|
+
/** A ledger over the same totals whose own reservations are also capped at `capUsd` (one page's `perRequestUsd`); null: no cap of its own. */
|
|
59
|
+
child(capUsd: number | null): SpendLedger;
|
|
60
|
+
/** Spend settled so far, in US dollars: each call at its reported price, or at its ceiling when none was reported. */
|
|
61
|
+
readonly settledUsd: number;
|
|
62
|
+
/** Ceilings reserved by calls not yet settled. */
|
|
63
|
+
readonly reservedUsd: number;
|
|
64
|
+
/** The cap; null: none. */
|
|
65
|
+
readonly capUsd: number | null;
|
|
66
|
+
}
|
|
67
|
+
/** One reserved call. */
|
|
68
|
+
export interface SpendReservation {
|
|
69
|
+
/** The call was made: settle at the price the provider reported, or at the ceiling when it reported none. Returns what was charged. */
|
|
70
|
+
settle(reportedUsd: number | null): number;
|
|
71
|
+
/** The call was not made: give the reservation back. */
|
|
72
|
+
release(): void;
|
|
73
|
+
}
|
|
74
|
+
/** One page a paginate step read: what `onListPage` tells and `listResume` gives back. */
|
|
75
|
+
export interface ListPageRead {
|
|
76
|
+
/** Index of the paginate step among the request's actions. */
|
|
77
|
+
step: number;
|
|
78
|
+
/** 1-based position among the pages the step read, the resumed ones included. */
|
|
79
|
+
page: number;
|
|
80
|
+
/** The URL the browser showed when the page was read. */
|
|
81
|
+
url: string;
|
|
82
|
+
html: string;
|
|
83
|
+
/** The page's state key as the lane computed it (its URL and its items or words), to know the page again on a resume. */
|
|
84
|
+
state: string;
|
|
85
|
+
/** The hash of the elements `itemSelector` matched, or null without one. */
|
|
86
|
+
items: string | null;
|
|
87
|
+
/** The hash of the links and sources inside those elements alone, which a changed price or date leaves as it was; null without `itemSelector` or when no item has one. */
|
|
88
|
+
itemRefs: string | null;
|
|
89
|
+
/** How many elements `itemSelector` matched on the page; null without one. */
|
|
90
|
+
count: number | null;
|
|
91
|
+
}
|
|
92
|
+
/** A cookie as a browser context takes and gives it (Playwright's shape). */
|
|
93
|
+
export interface ContextCookie {
|
|
94
|
+
name: string;
|
|
95
|
+
value: string;
|
|
96
|
+
domain: string;
|
|
97
|
+
path: string;
|
|
98
|
+
/** Seconds since the epoch; -1 for a session cookie. */
|
|
99
|
+
expires: number;
|
|
100
|
+
httpOnly: boolean;
|
|
101
|
+
secure: boolean;
|
|
102
|
+
sameSite: 'Strict' | 'Lax' | 'None';
|
|
103
|
+
}
|
|
104
|
+
/**
|
|
105
|
+
* One task's cookies, matched to a URL by RFC 6265 (domain, path, secure, expiry). Values never
|
|
106
|
+
* leave it into a record or a trace: a lane reports the session's `id` and counts.
|
|
107
|
+
*/
|
|
108
|
+
export interface CookieSession {
|
|
109
|
+
/** An opaque id for the record, unrelated to any cookie value. */
|
|
110
|
+
readonly id: string;
|
|
111
|
+
/** The `Cookie` header for a request to this URL; empty when none applies. */
|
|
112
|
+
cookieHeader(url: string): Promise<string>;
|
|
113
|
+
/** Keep the `Set-Cookie` lines a response to this URL carried; returns how many were kept. */
|
|
114
|
+
store(url: string, setCookies: readonly string[]): Promise<number>;
|
|
115
|
+
/** The cookies a browser context loading this URL should start with. */
|
|
116
|
+
browserCookies(url: string): Promise<ContextCookie[]>;
|
|
117
|
+
/**
|
|
118
|
+
* Keep what a browser context changed: given the cookies it started with and the ones it holds
|
|
119
|
+
* after the page, store the new and changed ones and delete the ones it dropped, unless another
|
|
120
|
+
* page of the task changed that cookie meanwhile. Unchanged cookies are left as the session has them.
|
|
121
|
+
*/
|
|
122
|
+
storeBrowserChanges(startedWith: readonly ContextCookie[], held: readonly ContextCookie[]): Promise<{
|
|
123
|
+
kept: number;
|
|
124
|
+
removed: number;
|
|
125
|
+
}>;
|
|
17
126
|
}
|
|
18
127
|
/** What a lane reports when it sets a robots.txt rule aside under a recorded override. */
|
|
19
128
|
export interface RobotsOverrideApplied {
|
|
@@ -52,15 +161,18 @@ export interface FetchOptions {
|
|
|
52
161
|
*/
|
|
53
162
|
parsers?: readonly PdfParser[];
|
|
54
163
|
/**
|
|
55
|
-
* A
|
|
56
|
-
* disallows it. robots.txt is still read and its
|
|
57
|
-
*
|
|
58
|
-
*
|
|
59
|
-
* and local browser lanes apply it; the provider lane takes none.
|
|
60
|
-
* Set
|
|
61
|
-
* `robotsOverrides` entry
|
|
164
|
+
* A decision to fetch this one URL although its host's robots.txt
|
|
165
|
+
* disallows it or could not be read. robots.txt is still read and its
|
|
166
|
+
* verdict recorded, Crawl-delay included; the override goes into the
|
|
167
|
+
* trace, the warnings and, in the browser lane, the compliance record. The
|
|
168
|
+
* HTTP and local browser lanes apply it; the provider lane takes none.
|
|
169
|
+
* Set by the engine: a scrape's `robotsOverride` or a batch's
|
|
170
|
+
* `robotsOverrides` entry, else, on a local server, `user_named_url` for
|
|
171
|
+
* every URL a scrape or batch names and `ignore_robots_txt` for a crawl's
|
|
172
|
+
* pages when the crawl asked. A navigation a page's steps make to another
|
|
173
|
+
* URL is checked against robots.txt whatever this says.
|
|
62
174
|
*/
|
|
63
|
-
robotsOverride?:
|
|
175
|
+
robotsOverride?: AppliedRobotsOverride;
|
|
64
176
|
/**
|
|
65
177
|
* CSS selectors naming the only elements to keep. The content is those
|
|
66
178
|
* elements, in document order, copied from the page before anything is
|
|
@@ -237,7 +237,7 @@ export interface DocumentExtraction {
|
|
|
237
237
|
labelledValues?: readonly LabelledValue[];
|
|
238
238
|
}
|
|
239
239
|
/** The rule that decided a page's data is most likely rendered client-side (see RenderSignals). */
|
|
240
|
-
export type RenderReason = 'empty_table_with_scripts' | 'empty_app_root' | 'script_shell' | 'js_fallback' | 'hydration_shell' | 'aria_busy';
|
|
240
|
+
export type RenderReason = 'empty_table_with_scripts' | 'empty_app_root' | 'script_shell' | 'js_fallback' | 'hydration_shell' | 'aria_busy' | 'hydration_list_partial';
|
|
241
241
|
/** A client-side rendering marker found in the page as received (see RenderSignals). */
|
|
242
242
|
export type RenderMarker = 'hydration_state' | 'app_root_empty' | 'noscript_notice' | 'js_fallback_marker' | 'aria_busy';
|
|
243
243
|
/**
|
|
@@ -255,6 +255,16 @@ export interface RenderSignals {
|
|
|
255
255
|
emptyTables: number;
|
|
256
256
|
/** Markers found in the page as received. */
|
|
257
257
|
markers: readonly RenderMarker[];
|
|
258
|
+
/**
|
|
259
|
+
* On a listing page, the hydration data's list of named records that most
|
|
260
|
+
* outnumbers the ones its markup shows: how many it lists, and how many
|
|
261
|
+
* of their names are in the visible text. Present only when such a list
|
|
262
|
+
* decided `hydration_list_partial`.
|
|
263
|
+
*/
|
|
264
|
+
listRecords?: {
|
|
265
|
+
declared: number;
|
|
266
|
+
shown: number;
|
|
267
|
+
};
|
|
258
268
|
/** True when the signals say the data is most likely rendered client-side. */
|
|
259
269
|
clientRendered: boolean;
|
|
260
270
|
/**
|
|
@@ -23,7 +23,7 @@ export declare const FIRECRAWL_SHIM_SNAPSHOT: {
|
|
|
23
23
|
paths: readonly ["/scrape", "/crawl", "/crawl/:id", "/map"];
|
|
24
24
|
notCovered: readonly ["search", "interact", "agent", "monitor", "extract"];
|
|
25
25
|
};
|
|
26
|
-
export declare const FIRECRAWL_SHIM_DIFFS: readonly ["Challenge / block pages are success: false (Firecrawl often returns them as success markdown).", "A page with no main content is success: false (failed: empty_unverified) with the whole page in data.markdown as evidence; with onlyMainContent: false it is success: true.", "A page whose server HTML is a shell for data its scripts fill in is fetched again on the browser rung, and the rendered page is the answer when it holds more; otherwise the HTTP page is returned with client_rendered_suspected and low_content_yield warnings on the native response, whose messages /fc passes through as data.warning (one string, joined with a space), as it does every native warning. Firecrawl renders every page in a browser.", "No fire-engine, proxy pools or JSON extract.", "actions (scrape only; a crawl's scrapeOptions.actions is refused) run on the local browser rung alone, which such a request selects, after load, stability and waitFor and before the formats are read: wait (milliseconds up to 60000, or a selector, waited for up to 60 s within the scrape's timeout), click (all: true clicks every match), write (into the focused element), press, scroll (one screen up or down, of the page or the element a selector names), screenshot, scrape, executeJavascript (a function body; return gives the value) and pdf; at most 50 steps. data.actions holds screenshots and pdfs as data: URIs (Firecrawl returns URLs), scrapes as { url, html } and javascriptReturns as { type, value }. A step that fails stops the steps after it: success is false with data.actions.failed naming the step, its code and message, and data.markdown is the page as it stood. A step that leads the page to a URL robots.txt or the egress policy refuses fails with navigation_refused and that page is not read. A hosted server refuses actions, and the cache is not used with them.", "An omitted maxAge reuses nothing: every page is fetched live unless the request sets maxAge above 0, minAge or lockdown (Firecrawl reuses its own index by default; its Python SDK sends maxAge 4 hours). A reused page is one W2L itself stored, under the same options, on this server (its task root), never a shared index; only a success is stored, and data.metadata says cacheState hit with cachedAt (its fetch time) or miss when one was looked up. lockdown with no stored result is HTTP 404 SCRAPE_LOCKDOWN_CACHE_MISS on scrape, and nothing is fetched; a crawl in lockdown needs sitemap skip, and each page with no stored result is failed with cache_miss. Mode authed neither stores nor reuses. A Firecrawl body never sets useCached, W2L's reuse of a crawl's own pages on resume.", "Omitted limit / maxDepth stay unbounded on a local server; a hosted server takes its crawl limit for an omitted or null limit and refuses a larger one. Firecrawl defaults are 10000 / 10.", "maxDepth counts link hops from the start URL (Firecrawl calls that maxDiscoveryDepth); Firecrawl maxDepth counts URL path depth.", "Crawl start is mapped onto native POST /v1/crawl; the shim itself returns 200 {success,id,url}.", "creditsUsed and expiresAt are null: W2L counts no credits and keeps crawl results until their task directory is deleted.", "Crawl status describes the latest attempt: completed counts its successful pages, total adds its failed, blocked and duplicate pages and, while this API process runs the crawl, the pages in flight and queued (null for a paused crawl); data lists the failed and blocked pages too (with metadata.error) but not the duplicates, whose content is an earlier entry's, up to 100 per response (limit 1 to 1000) with next carrying a W2L cursor; skip is rejected.", "Scrape maps url, formats, onlyMainContent, includeTags, excludeTags, waitFor, timeout, headers, mobile, skipTlsVerification, fastMode, blockAds, removeBase64Images, maxAge, minAge, storeInCache, lockdown, actions, origin and integration; crawl maps url, limit (as maxPages), maxDepth, includePaths, excludePaths, regexOnFullURL, ignoreQueryParameters, deduplicateSimilarURLs, crawlEntireDomain (and its v1 name allowBackwardLinks), allowSubdomains, allowExternalLinks, sitemap (v2; v1 ignoreSitemap true is skip and false include, sitemapOnly true is only), maxConcurrency, origin, integration and the same scrapeOptions (applied to every page). The formats are markdown, links, html, rawHtml, images, screenshot (also screenshot@fullPage, and { type: \"screenshot\", fullPage, quality, viewport }) and an { type: \"attributes\", selectors } entry; other formats and parameters the shim does not map (proxy, location, json, ...) are rejected by name with HTTP 400 and success: false; a refusal of stealth, proxy: stealth or enhanced, or ignoreRobotsTxt names the supported route in agent_hints.", "A crawl follows links inside the start URL's path subtree on its host and www twin by default (crawlEntireDomain false), folds /a and /a/, / and /index.html, www and apex, http and https into one page (deduplicateSimilarURLs true) and reports every collapsed or refused link in the native crawl status (discovery) and each page's trace (links_offered); allowSubdomains takes every host under the start URL's apex (no public-suffix list), allowExternalLinks every host, each page with its own robots.txt read.", "sitemap (default include, as in Firecrawl) reads the sitemaps the start URL's robots.txt names, or /sitemap.xml, with the crawl's own http identity, robots.txt verdict, SSRF checks and proxy, and queues their URLs ahead of the start page's links under the same host, subtree, path and depth rules; only follows no page link; skip reads none. The native crawl status lists every sitemap file read, refused or unreadable in discovery.sitemap; the shim's status carries nothing of it, and sitemap fetches have no signed compliance record. maxConcurrency caps the pages one crawl fetches at once, at most the service's worker count (HTTP 400 above it), and never raises the per-host ceiling.", "screenshot (data.screenshot, a data:image/png;base64 string, or image/jpeg with quality 1 to 100) is captured on the local browser rung alone, which such a request selects (no http attempt; a server without a browser rung refuses the format with HTTP 400): after load, stability and waitFor, before the DOM is read, CSS-pixel sized at the declared 1280x800 viewport (device scale factor 2 is declared, not baked into the image) or at the viewport asked for (integers 320..1920 by 240..1080, within the declared screen; a window size, not a change of identity); fullPage captures the document's whole height at that width without scrolling first, so sections a page loads on scroll may show unloaded. A capture the browser could not make leaves data.screenshot null with a screenshot_unavailable warning while the page stands; a file or a page that was not rendered has null too. Firecrawl captures at its own viewport and may return a URL instead of the image.", "images (data.images) lists every image URL of the whole document as received: img src and srcset candidates, picture sources, lazy data-src/data-srcset/data-lazy-src/data-original, video posters, image_src links, og:image and twitter:image, absolute http(s) with the fragment stripped, each once, in document order, data: URIs left out; includeTags, excludeTags and onlyMainContent do not narrow it. attributes (data.attributes) gives, per selector, the named attribute's values as written, elements without it skipped; a selector W2L does not match is HTTP 400 by name, as for includeTags. Both are absent for a file and for a page that is success: false. removeBase64Images (default true) keeps an image's alt text where Firecrawl writes a (<Base64-Image-Removed>) placeholder; false keeps the data: URI in the Markdown.", "origin (the Firecrawl SDKs' client label) and integration are stored, not echoed: the scrape record (GET /v1/scrapes/:id) and the crawl task carry them, and nothing sent to the target changes.", "data.metadata carries scrapeId (a UUID per call, which GET /v1/scrapes/:id looks up), proxyUsed (operator for the server's environment proxy, user for the caller's own egress, else null), timezone (the browser rung's declared zone, null on the HTTP rung), creditsUsed: null (W2L counts no credits), concurrencyLimited and concurrencyQueueDurationMs (whether and how long the per-origin ceiling held the fetch back), and cacheState and cachedAt when the cache was asked (never a guessed miss).", "A page whose result W2L has advice about (a login wall, a robots.txt rule, a gate, a cut, a script-filled shell) carries data.agent_hints, one sentence each; the native response calls them agentHints. A request refused for an option W2L does not offer carries agent_hints in the error envelope, and a caller over the server's per-minute rate limit gets HTTP 429 { success: false, error, code: rate_limited, agent_hints } with Retry-After.", "headers never override the User-Agent, the client hints, a credential (authorization, cookie) or a transport header: such a header is HTTP 400 naming it, where Firecrawl sends it. The headers go to the requested origin after the declared identity and are on the record (the trace, the browser lane's signed sentHeaders); both rungs withhold them from a redirect hop to another origin and say so (custom_headers_withheld).", "mobile selects a declared Android Chrome identity (User-Agent, client hints, 412x915 viewport, touch) that robots.txt is evaluated against and the record carries; the page is whatever the site serves to it, with no DOM rewriting. It is refused with mode research.", "skipTlsVerification relaxes certificate verification for one local request and its robots.txt lookup, recorded in the trace (tls_verification_skipped) and a tls_unverified warning the native response carries; a hosted W2L refuses it with HTTP 400. Without it a bad certificate is success: false with failed: tls_error. Firecrawl's Python SDK sends true by default; W2L verifies by default.", "fastMode keeps the http rung alone: a page that needs scripts is success: false with failed: empty_unverified, never rendered; waitFor has no effect under it. Firecrawl's fast mode still renders.", "blockAds (default true) aborts requests to a bundled list of about 50 ad-serving hosts on the local browser rung and removes ad and cookie-banner elements before extraction; false keeps them. The list is curated, not EasyList: ads from hosts outside it are not blocked.", "html is the cleaned HTML the markdown is written from: the main content, the whole page without scripts, styles, form controls and embedded media when onlyMainContent is false, or a <body> holding the includeTags elements. rawHtml is the page as the answering rung received it: the response body on the HTTP rung, the rendered DOM on a browser rung. Both are null for a file and for a page that is success: false.", "includeTags keeps only the named elements, in document order, whatever onlyMainContent says; excludeTags removes elements from the main content, the whole page and an includeTags selection. A selector that does not parse, or that uses a sibling combinator, a positional pseudo-class, :has() or another pseudo-class W2L does not match, is rejected with HTTP 400.", "An omitted timeout stays 300000 ms (Firecrawl: 30000). A timeout is answered with HTTP 200: success: true with the content fetched so far (native status partial), or success: false with failed: timeout; Firecrawl answers it with an error.", "waitFor skips the HTTP rung, which cannot run scripts, and starts at the browser rung; the wait counts toward timeout.", "metadata has title, description, language, keywords, robots and favicon only when the page declares them, and the Open Graph (ogTitle, ogDescription, ogUrl, ogImage, ogAudio, ogVideo, ogDeterminer, ogLocale, ogLocaleAlternate, ogSiteName), Dublin Core (dcTermsCreated, dcDateCreated, dcDate, dcTermsType, dcType, dcTermsAudience, dcTermsSubject, dcSubject, dcDescription, dcTermsKeywords) and article (publishedTime, modifiedTime, articleTag, articleSection) tags under Firecrawl's names, each only when the page states it, as written (no date normalisation, no fallback from another tag); twitter:* and other meta tags are not passed through, and a failed or blocked page has none.", "A PDF answers success: true with its text layer as markdown and metadata.numPages (the document's page count). parsers maps Firecrawl's pdf entry (the string or { type: \"pdf\", mode, maxPages, pages, pageMarkers }); mode fast and auto both read the text layer, and mode ocr and the image parser are refused by name. pageMarkers is false unless asked, as on Firecrawl; with true W2L writes a <!-- page N --> line before each page (Firecrawl writes --- and the marker between pages). pages: true adds data.pages, [{ pageNumber, markdown }]. maxPages (1 to 10000) reads the first pages, and a cut it asked for stays success: true. parsers [] or v1 parsePDF false reads no PDF: success: true with markdown null and the file saved as received. A PDF without a text layer is success: false with failed: empty_unverified (no OCR). CSV, JSON and text files give their text as received; XLSX, XLS and ZIP files are success: true with markdown null. A file over W2L_MAX_FILE_BYTES is success: false with failed: body_too_large.", "Map (POST /fc/v1/map) maps url, search, sitemap (v2; v1 ignoreSitemap true is skip and false include, sitemapOnly true is only; both v1 flags true is HTTP 400), includeSubdomains, ignoreQueryParameters, limit (1 to 100000, default 5000), timeout (1000 to 300000 ms for the whole map, default 60000; Firecrawl documents no default), origin and integration onto native POST /v1/map, and answers 200 { success: true, id, links: [url strings], warning?, agent_hints? }, or 200 { success: false, id, error, links: [] } when the map found nothing because a source failed or its deadline passed; useIndex, location, ignoreCache, threatProtection and auditMetadata are refused by name (useIndex with the hint that W2L keeps no URL index). Omitted options take W2L's defaults: includeSubdomains and ignoreQueryParameters are false, where Firecrawl v2 documents true for both. search keeps the URLs in which every word appears in the decoded URL or the title in hand, in discovery order; Firecrawl orders by relevance. A map reads the sitemaps the site declares and one page body (the start URL, http rung only), so a site without a sitemap maps only its start page's links; a title is the start page's own, an anchor's text or a sitemap's <news:title>, never fetched from the target; robots-disallowed URLs are left out and counted on the native response (GET /v1/maps/:id), which also records every sitemap file read.", "A crawl's webhook (a URL string or { url, headers, metadata, events }) is mapped onto the native webhook and its receiver gets Firecrawl's payload shape: { success, type: crawl.started | crawl.page | crawl.completed | crawl.failed, id, data: [page], metadata, error? }, one durable delivery per event with retries, every request carrying x-w2l-event-id, x-w2l-event-version and x-w2l-delivery-id (and the signature pair with secretEnv, a native option). A cancelled crawl is crawl.failed with error \"cancelled\". The native rules apply: https (plain http for a loopback receiver of a local server only), no content-type, host or x-w2l-* header, at most 32 headers and 32 metadata strings; a hosted server takes public https receivers only. GET /v1/deliveries?jobId=<id> on the native API lists the deliveries."];
|
|
26
|
+
export declare const FIRECRAWL_SHIM_DIFFS: readonly ["Challenge / block pages are success: false (Firecrawl often returns them as success markdown).", "A page with no main content is success: false (failed: empty_unverified) with the whole page in data.markdown as evidence; with onlyMainContent: false it is success: true.", "A page whose server HTML is a shell for data its scripts fill in is fetched again on the browser rung, and the rendered page is the answer when it holds more; otherwise the HTTP page is returned with client_rendered_suspected and low_content_yield warnings on the native response, whose messages /fc passes through as data.warning (one string, joined with a space), as it does every native warning. Firecrawl renders every page in a browser.", "No fire-engine, proxy pools or JSON extract.", "actions (scrape only; a crawl's scrapeOptions.actions is refused) run on the local browser rung alone, which such a request selects, after load, stability and waitFor and before the formats are read: wait (milliseconds up to 60000, or a selector, waited for up to 60 s within the scrape's timeout), click (all: true clicks every match), write (into the focused element), press, scroll (one screen up or down, of the page or the element a selector names), screenshot, scrape, executeJavascript (a function body; return gives the value) and pdf; at most 50 steps. data.actions holds screenshots and pdfs as data: URIs (Firecrawl returns URLs), scrapes as { url, html } and javascriptReturns as { type, value }. A step that fails stops the steps after it: success is false with data.actions.failed naming the step, its code and message, and data.markdown is the page as it stood. A step that leads the page to a URL robots.txt or the egress policy refuses fails with navigation_refused and that page is not read. A hosted server refuses actions, and the cache is not used with them.", "An omitted maxAge reuses nothing: every page is fetched live unless the request sets maxAge above 0, minAge or lockdown (Firecrawl reuses its own index by default; its Python SDK sends maxAge 4 hours). A reused page is one Octocrawl itself stored, under the same options, on this server (its task root), never a shared index; only a success is stored, and data.metadata says cacheState hit with cachedAt (its fetch time) or miss when one was looked up. lockdown with no stored result is HTTP 404 SCRAPE_LOCKDOWN_CACHE_MISS on scrape, and nothing is fetched; a crawl in lockdown needs sitemap skip, and each page with no stored result is failed with cache_miss. Mode authed neither stores nor reuses. A Firecrawl body never sets useCached, Octocrawl's reuse of a crawl's own pages on resume.", "Omitted limit / maxDepth stay unbounded on a local server; a hosted server takes its crawl limit for an omitted or null limit and refuses a larger one. Firecrawl defaults are 10000 / 10.", "maxDepth counts link hops from the start URL (Firecrawl calls that maxDiscoveryDepth); Firecrawl maxDepth counts URL path depth.", "Crawl start is mapped onto native POST /v1/crawl; the shim itself returns 200 {success,id,url}.", "creditsUsed and expiresAt are null: Octocrawl counts no credits and keeps crawl results until their task directory is deleted.", "Crawl status describes the latest attempt: completed counts its successful pages, total adds its failed, blocked and duplicate pages and, while this API process runs the crawl, the pages in flight and queued (null for a paused crawl); data lists the failed and blocked pages too (with metadata.error) but not the duplicates, whose content is an earlier entry's, up to 100 per response (limit 1 to 1000) with next carrying an Octocrawl cursor; skip is rejected.", "Scrape maps url, formats, onlyMainContent, includeTags, excludeTags, waitFor, timeout, headers, mobile, skipTlsVerification, fastMode, blockAds, removeBase64Images, maxAge, minAge, storeInCache, lockdown, actions, proxy (basic as access standard; stealth and auto as access enhanced, which a server without an access grant of tier enhanced refuses by name), origin and integration; crawl maps url, limit (as maxPages), maxDepth, includePaths, excludePaths, regexOnFullURL, ignoreQueryParameters, deduplicateSimilarURLs, crawlEntireDomain (and its v1 name allowBackwardLinks), allowSubdomains, allowExternalLinks, sitemap (v2; v1 ignoreSitemap true is skip and false include, sitemapOnly true is only), maxConcurrency, ignoreRobotsTxt (v2; a local server only), origin, integration and the same scrapeOptions (applied to every page). The formats are markdown, links, html, rawHtml, images, screenshot (also screenshot@fullPage, and { type: \"screenshot\", fullPage, quality, viewport }) and an { type: \"attributes\", selectors } entry; other formats and parameters the shim does not map (location, json, ...) are rejected by name with HTTP 400 and success: false; a refusal of stealth, of access enhanced (proxy: stealth or auto) on a server without the grant, or of ignoreRobotsTxt on a scrape or map names the supported route in agent_hints.", "A crawl follows links inside the start URL's path subtree on its host and www twin by default (crawlEntireDomain false), folds /a and /a/, / and /index.html, www and apex, http and https into one page (deduplicateSimilarURLs true) and reports every collapsed or refused link in the native crawl status (discovery) and each page's trace (links_offered); allowSubdomains takes every host under the start URL's apex (no public-suffix list), allowExternalLinks every host, each page with its own robots.txt read.", "sitemap (default include, as in Firecrawl) reads the sitemaps the start URL's robots.txt names, or /sitemap.xml, with the crawl's own http identity, robots.txt verdict, SSRF checks and proxy, and queues their URLs ahead of the start page's links under the same host, subtree, path and depth rules; only follows no page link; skip reads none. The native crawl status lists every sitemap file read, refused or unreadable in discovery.sitemap; the shim's status carries nothing of it, and sitemap fetches have no signed compliance record. maxConcurrency caps the pages one crawl fetches at once, at most the service's worker count (HTTP 400 above it), and never raises the per-host ceiling.", "screenshot (data.screenshot, a data:image/png;base64 string, or image/jpeg with quality 1 to 100) is captured on the local browser rung alone, which such a request selects (no http attempt; a server without a browser rung refuses the format with HTTP 400): after load, stability and waitFor, before the DOM is read, CSS-pixel sized at the declared 1280x800 viewport (device scale factor 2 is declared, not baked into the image) or at the viewport asked for (integers 320..1920 by 240..1080, within the declared screen; a window size, not a change of identity); fullPage captures the document's whole height at that width without scrolling first, so sections a page loads on scroll may show unloaded. A capture the browser could not make leaves data.screenshot null with a screenshot_unavailable warning while the page stands; a file or a page that was not rendered has null too. Firecrawl captures at its own viewport and may return a URL instead of the image.", "images (data.images) lists every image URL of the whole document as received: img src and srcset candidates, picture sources, lazy data-src/data-srcset/data-lazy-src/data-original, video posters, image_src links, og:image and twitter:image, absolute http(s) with the fragment stripped, each once, in document order, data: URIs left out; includeTags, excludeTags and onlyMainContent do not narrow it. attributes (data.attributes) gives, per selector, the named attribute's values as written, elements without it skipped; a selector Octocrawl does not match is HTTP 400 by name, as for includeTags. Both are absent for a file and for a page that is success: false. removeBase64Images (default true) keeps an image's alt text where Firecrawl writes a (<Base64-Image-Removed>) placeholder; false keeps the data: URI in the Markdown.", "origin (the Firecrawl SDKs' client label) and integration are stored, not echoed: the scrape record (GET /v1/scrapes/:id) and the crawl task carry them, and nothing sent to the target changes.", "data.metadata carries scrapeId (a UUID per call, which GET /v1/scrapes/:id looks up), proxyUsed (operator for the server's environment proxy, user for the caller's own egress, else null), timezone (the browser rung's declared zone, null on the HTTP rung), creditsUsed: null (Octocrawl counts no credits), concurrencyLimited and concurrencyQueueDurationMs (whether and how long the per-origin ceiling held the fetch back), and cacheState and cachedAt when the cache was asked (never a guessed miss).", "A page whose result Octocrawl has advice about (a login wall, a robots.txt rule, a gate, a cut, a script-filled shell) carries data.agent_hints, one sentence each; the native response calls them agentHints. A request refused for an option Octocrawl does not offer carries agent_hints in the error envelope, and a caller over the server's per-minute rate limit gets HTTP 429 { success: false, error, code: rate_limited, agent_hints } with Retry-After.", "headers never override the User-Agent, the client hints, a credential (authorization, cookie) or a transport header: such a header is HTTP 400 naming it, where Firecrawl sends it. The headers go to the requested origin after the declared identity and are on the record (the trace, the browser lane's signed sentHeaders); both rungs withhold them from a redirect hop to another origin and say so (custom_headers_withheld).", "mobile selects a declared Android Chrome identity (User-Agent, client hints, 412x915 viewport, touch) that robots.txt is evaluated against and the record carries; the page is whatever the site serves to it, with no DOM rewriting. It is refused with mode research.", "skipTlsVerification relaxes certificate verification for one local request and its robots.txt lookup, recorded in the trace (tls_verification_skipped) and a tls_unverified warning the native response carries; a hosted Octocrawl refuses it with HTTP 400. Without it a bad certificate is success: false with failed: tls_error. Firecrawl's Python SDK sends true by default; Octocrawl verifies by default.", "fastMode keeps the http rung alone: a page that needs scripts is success: false with failed: empty_unverified, never rendered; waitFor has no effect under it. Firecrawl's fast mode still renders.", "blockAds (default true) aborts requests to a bundled list of about 50 ad-serving hosts on the local browser rung and removes ad and cookie-banner elements before extraction; false keeps them. The list is curated, not EasyList: ads from hosts outside it are not blocked.", "html is the cleaned HTML the markdown is written from: the main content, the whole page without scripts, styles, form controls and embedded media when onlyMainContent is false, or a <body> holding the includeTags elements. rawHtml is the page as the answering rung received it: the response body on the HTTP rung, the rendered DOM on a browser rung. Both are null for a file and for a page that is success: false.", "includeTags keeps only the named elements, in document order, whatever onlyMainContent says; excludeTags removes elements from the main content, the whole page and an includeTags selection. A selector that does not parse, or that uses a sibling combinator, a positional pseudo-class, :has() or another pseudo-class Octocrawl does not match, is rejected with HTTP 400.", "An omitted timeout stays 300000 ms (Firecrawl: 30000). A timeout is answered with HTTP 200: success: true with the content fetched so far (native status partial), or success: false with failed: timeout; Firecrawl answers it with an error.", "waitFor skips the HTTP rung, which cannot run scripts, and starts at the browser rung; the wait counts toward timeout.", "metadata has title, description, language, keywords, robots and favicon only when the page declares them, and the Open Graph (ogTitle, ogDescription, ogUrl, ogImage, ogAudio, ogVideo, ogDeterminer, ogLocale, ogLocaleAlternate, ogSiteName), Dublin Core (dcTermsCreated, dcDateCreated, dcDate, dcTermsType, dcType, dcTermsAudience, dcTermsSubject, dcSubject, dcDescription, dcTermsKeywords) and article (publishedTime, modifiedTime, articleTag, articleSection) tags under Firecrawl's names, each only when the page states it, as written (no date normalisation, no fallback from another tag); twitter:* and other meta tags are not passed through, and a failed or blocked page has none.", "A PDF answers success: true with its text layer as markdown and metadata.numPages (the document's page count). parsers maps Firecrawl's pdf entry (the string or { type: \"pdf\", mode, maxPages, pages, pageMarkers }); mode fast and auto both read the text layer, and mode ocr and the image parser are refused by name. pageMarkers is false unless asked, as on Firecrawl; with true Octocrawl writes a <!-- page N --> line before each page (Firecrawl writes --- and the marker between pages). pages: true adds data.pages, [{ pageNumber, markdown }]. maxPages (1 to 10000) reads the first pages, and a cut it asked for stays success: true. parsers [] or v1 parsePDF false reads no PDF: success: true with markdown null and the file saved as received. A PDF without a text layer is success: false with failed: empty_unverified (no OCR). CSV, JSON and text files give their text as received; XLSX, XLS and ZIP files are success: true with markdown null. A file over W2L_MAX_FILE_BYTES is success: false with failed: body_too_large.", "Map (POST /fc/v1/map) maps url, search, sitemap (v2; v1 ignoreSitemap true is skip and false include, sitemapOnly true is only; both v1 flags true is HTTP 400), includeSubdomains, ignoreQueryParameters, limit (1 to 100000, default 5000), timeout (1000 to 300000 ms for the whole map, default 60000; Firecrawl documents no default), origin and integration onto native POST /v1/map, and answers 200 { success: true, id, links: [url strings], warning?, agent_hints? }, or 200 { success: false, id, error, links: [] } when the map found nothing because a source failed or its deadline passed; useIndex, location, ignoreCache, threatProtection and auditMetadata are refused by name (useIndex with the hint that Octocrawl keeps no URL index). Omitted options take Octocrawl's defaults: includeSubdomains and ignoreQueryParameters are false, where Firecrawl v2 documents true for both. search keeps the URLs in which every word appears in the decoded URL or the title in hand, in discovery order; Firecrawl orders by relevance. A map reads the sitemaps the site declares and one page body (the start URL, http rung only), so a site without a sitemap maps only its start page's links; a title is the start page's own, an anchor's text or a sitemap's <news:title>, never fetched from the target; robots-disallowed URLs are left out and counted on the native response (GET /v1/maps/:id), which also records every sitemap file read.", "A crawl's webhook (a URL string or { url, headers, metadata, events }) is mapped onto the native webhook and its receiver gets Firecrawl's payload shape: { success, type: crawl.started | crawl.page | crawl.completed | crawl.failed, id, data: [page], metadata, error? }, one durable delivery per event with retries, every request carrying x-w2l-event-id, x-w2l-event-version and x-w2l-delivery-id (and the signature pair with secretEnv, a native option). A cancelled crawl is crawl.failed with error \"cancelled\". The native rules apply: https (plain http for a loopback receiver of a local server only), no content-type, host or x-w2l-* header, at most 32 headers and 32 metadata strings; a hosted server takes public https receivers only. GET /v1/deliveries?jobId=<id> on the native API lists the deliveries."];
|
|
27
27
|
export interface FirecrawlPage {
|
|
28
28
|
markdown: string | null;
|
|
29
29
|
/** Present when the `html` format was asked for; null when the page has none (a file, a page that did not succeed). */
|
|
@@ -40,11 +40,16 @@ export interface MapLink {
|
|
|
40
40
|
sitemapFile?: string;
|
|
41
41
|
/** The `<lastmod>` the sitemap gave, as written. */
|
|
42
42
|
lastmod?: string;
|
|
43
|
-
/**
|
|
44
|
-
|
|
43
|
+
/**
|
|
44
|
+
* The robots.txt verdict for the URL under the map's declared identity. A
|
|
45
|
+
* URL robots.txt disallows (`disallowed`), or on a host whose robots.txt
|
|
46
|
+
* could not be read (`unreachable`), is a link only in a map started with
|
|
47
|
+
* ignoreRobotsTxt; otherwise it is refused.
|
|
48
|
+
*/
|
|
49
|
+
robots: 'allowed' | 'no_robots' | 'disallowed' | 'unreachable';
|
|
45
50
|
}
|
|
46
51
|
export type MapStatus = 'completed' | 'partial' | 'failed';
|
|
47
|
-
/** What the start page read gave, or why it was not read (`failed/policy_denied` when robots.txt disallows the start URL or could not be read). */
|
|
52
|
+
/** What the start page read gave, or why it was not read (`failed/policy_denied` when robots.txt disallows the start URL or could not be read, unless the map was started with ignoreRobotsTxt). */
|
|
48
53
|
export interface MapStartPage {
|
|
49
54
|
url: string;
|
|
50
55
|
finalUrl: string | null;
|
|
@@ -164,7 +169,8 @@ export interface MapStartPageRead {
|
|
|
164
169
|
/** A robots.txt verdict for one URL under the map's declared identity. */
|
|
165
170
|
export type MapRobotsVerdict = 'allowed' | 'no_robots' | {
|
|
166
171
|
disallowed: true;
|
|
167
|
-
unreachable?: string;
|
|
172
|
+
unreachable?: string; /** A rule robots.txt wrote for Octocrawl itself, which ignoreRobotsTxt does not set aside. */
|
|
173
|
+
octocrawl?: true;
|
|
168
174
|
};
|
|
169
175
|
/** The fetch paths a map uses; the API engine wires the real ones, tests inject fakes. */
|
|
170
176
|
export interface MapSources {
|
|
@@ -63,7 +63,8 @@ export interface NetworkPolicy {
|
|
|
63
63
|
* because the browser lane can route through only one proxy.
|
|
64
64
|
*/
|
|
65
65
|
export interface EgressProxy {
|
|
66
|
-
|
|
66
|
+
/** `environment`: HTTPS_PROXY / HTTP_PROXY. `pool`: one of the operator's W2L_EGRESS_PROXIES (ADR 0005 egress_sessions). */
|
|
67
|
+
source: 'environment' | 'pool';
|
|
67
68
|
/** From HTTPS_PROXY / https_proxy. Null sends https: URLs direct. */
|
|
68
69
|
https: ProxyServer | null;
|
|
69
70
|
/** From HTTP_PROXY / http_proxy. Null sends http: URLs direct. */
|
|
@@ -25,6 +25,14 @@ export declare class ProxyConfigError extends Error {
|
|
|
25
25
|
export declare function environmentProxy(env: Env): EgressProxy | null;
|
|
26
26
|
/** A local-mode policy that routes through the environment's proxy, when one is set. */
|
|
27
27
|
export declare function withEnvironmentProxy(policy: NetworkPolicy, env: Env): NetworkPolicy;
|
|
28
|
+
/**
|
|
29
|
+
* The operator's egress proxies (`W2L_EGRESS_PROXIES`, ADR 0005 `egress_sessions`): http:// or https://
|
|
30
|
+
* URLs, credentials in their userinfo, separated by commas or white space. Each one is a whole egress:
|
|
31
|
+
* every URL's scheme goes through it. Duplicates (the same endpoint) are kept once.
|
|
32
|
+
*/
|
|
33
|
+
export declare function egressProxies(raw: string | undefined): ProxyServer[];
|
|
34
|
+
/** A policy whose every request leaves through this egress proxy; NO_PROXY entries and loopback still go direct. */
|
|
35
|
+
export declare function withPoolProxy(policy: NetworkPolicy, server: ProxyServer): NetworkPolicy;
|
|
28
36
|
/** The operator proxy a URL leaves through under this policy; null means a direct connection. */
|
|
29
37
|
export declare function proxyFor(url: string | URL, policy: Pick<NetworkPolicy, 'origin' | 'egressProxy'>): ProxyServer | null;
|
|
30
38
|
export type NoProxyEntry = {
|
|
@@ -53,11 +53,25 @@ export interface ResourceUsage {
|
|
|
53
53
|
contentTokens: number | null;
|
|
54
54
|
browserMs: number;
|
|
55
55
|
/**
|
|
56
|
-
*
|
|
57
|
-
*
|
|
58
|
-
*
|
|
56
|
+
* What this fetch spent on third-party services it called: a provider's
|
|
57
|
+
* browser minutes or its challenge solving. `0` when it called none (the
|
|
58
|
+
* local lanes, a refusal before any request, a scrape answered from the
|
|
59
|
+
* cache; a cached batch or crawl page keeps the cost of the fetch that
|
|
60
|
+
* stored it). For a run
|
|
61
|
+
* through several rungs it is the whole run's spend. A model call for JSON
|
|
62
|
+
* extraction is reported apart (`modelUsage`) and not counted here. `null` when it called
|
|
63
|
+
* one that did not say what it cost: unknown, never free, and a run with a
|
|
64
|
+
* cost cap stops on it rather than guess (`budgetExceeded: cost_unknown`).
|
|
65
|
+
* The network path is not included: a proxy's own bill is recorded apart
|
|
66
|
+
* once proxy sessions declare a price.
|
|
59
67
|
*/
|
|
60
68
|
externalCostUsd: number | null;
|
|
69
|
+
/**
|
|
70
|
+
* What the run's spend ledger charged for this fetch's paid calls (ROADMAP PA item 4): each call at the price its
|
|
71
|
+
* provider reported, or at its price ceiling when it reported none, so it is never below the cost and stands in for
|
|
72
|
+
* it in the run's cap. Absent when no paid call was made through a ledger; `externalCostUsd` stays the exact cost or null.
|
|
73
|
+
*/
|
|
74
|
+
externalCostChargedUsd?: number;
|
|
61
75
|
/** Stage timings use a monotonic clock. Optional for legacy producers. */
|
|
62
76
|
timings?: ResourceTimings;
|
|
63
77
|
/**
|
|
@@ -152,9 +166,30 @@ export interface TraceEvent {
|
|
|
152
166
|
event: string;
|
|
153
167
|
detail?: Record<string, unknown>;
|
|
154
168
|
}
|
|
169
|
+
/**
|
|
170
|
+
* The paid provider calls (`paid_calls`, ROADMAP PA item 4) of a result that is not the page's answer: a read given up for
|
|
171
|
+
* another egress, or the run whose stopped page the person read in their Chrome. They were paid for, so they stay on the
|
|
172
|
+
* page's record, none of them its answer.
|
|
173
|
+
*/
|
|
174
|
+
export declare function givenUpPaidCalls(trace: readonly TraceEvent[]): TraceEvent[];
|
|
175
|
+
/**
|
|
176
|
+
* A run that threw after paid provider calls carries them on its error, as the `paid_calls` event its answer would have
|
|
177
|
+
* had, so whoever turns the error into the page's result keeps them on its record. The error is returned as it was.
|
|
178
|
+
*/
|
|
179
|
+
export declare function carryPaidCalls(error: unknown, event: TraceEvent): unknown;
|
|
180
|
+
/** The `paid_calls` event a thrown run carried (carryPaidCalls); null when it carried none. */
|
|
181
|
+
export declare function paidCallsOfError(error: unknown): TraceEvent | null;
|
|
155
182
|
export interface LadderAttempt {
|
|
183
|
+
/** The rung's id (http, http_compat, browser_local, authed_session, provider, ...), `(retry)` after a handoff. */
|
|
156
184
|
channel: string;
|
|
157
185
|
result: FetchResult;
|
|
186
|
+
/** 1 for the run's first attempt, then in order; absent on an attempt recorded before it was added. */
|
|
187
|
+
ordinal?: number;
|
|
188
|
+
/** UTC ISO 8601 times the rung was asked and answered; absent when not measured. */
|
|
189
|
+
startedAt?: string;
|
|
190
|
+
endedAt?: string;
|
|
191
|
+
/** The vendor of a provider rung. */
|
|
192
|
+
vendorId?: string;
|
|
158
193
|
}
|
|
159
194
|
export interface LadderExecutionSummary {
|
|
160
195
|
channelsTried: readonly string[];
|
|
@@ -255,7 +290,9 @@ export interface FetchWarning {
|
|
|
255
290
|
* `screenshot_unavailable`: the `screenshot` format was asked for and the
|
|
256
291
|
* browser lane rendered the page but could not capture it
|
|
257
292
|
* (`screenshot_failed` in the trace); `screenshot` is null and the page
|
|
258
|
-
* result stands.
|
|
293
|
+
* result stands. `page_still_loading`: a browser or provider lane read the
|
|
294
|
+
* page as content while it still showed a loading indicator, after waiting
|
|
295
|
+
* for it (`loading_wait` in the trace): its data may not be in the answer.
|
|
259
296
|
*/
|
|
260
297
|
code: string;
|
|
261
298
|
message: string;
|
|
@@ -18,6 +18,8 @@ export interface ManagedSessionRef {
|
|
|
18
18
|
/** Private registry only; never include in public status/logs. */
|
|
19
19
|
cdpEndpoint?: string;
|
|
20
20
|
}
|
|
21
|
+
/** A managed session as the API answers with it: no server path, no CDP endpoint. */
|
|
22
|
+
export type PublicManagedSessionRef = Omit<ManagedSessionRef, 'profileDir' | 'cdpEndpoint'>;
|
|
21
23
|
export interface SessionHandoff {
|
|
22
24
|
handoffId: string;
|
|
23
25
|
reason: string;
|
|
@@ -16,7 +16,7 @@ export type BlockReason = (typeof BLOCK_REASON)[number];
|
|
|
16
16
|
export declare const BUDGET_KIND: readonly ["tokens", "tokens_unknown", "time", "cost", "cost_unknown", "pages", "retries"];
|
|
17
17
|
export type BudgetKind = (typeof BUDGET_KIND)[number];
|
|
18
18
|
/** Execution tiers of the escalation ladder (PHASE1_ENGINEERING_NOTES §2.5). */
|
|
19
|
-
export declare const LANE: readonly ["http", "browser_local", "browser_local_authed", "browser_proxy", "provider"];
|
|
19
|
+
export declare const LANE: readonly ["http", "browser_local", "browser_local_authed", "browser_proxy", "provider", "my_browser"];
|
|
20
20
|
export type Lane = (typeof LANE)[number];
|
|
21
21
|
/**
|
|
22
22
|
* Public canary benchmarks only evaluate these lanes. Authenticated and
|
package/package.json
CHANGED
|
@@ -1,13 +1,30 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@octocrawl/sdk",
|
|
3
|
-
"version": "0.3.
|
|
3
|
+
"version": "0.3.2",
|
|
4
4
|
"description": "TypeScript client for the Octocrawl API: scrape, map, crawl and batch with an Evidence Record on every page.",
|
|
5
5
|
"license": "MIT",
|
|
6
|
+
"keywords": [
|
|
7
|
+
"octocrawl",
|
|
8
|
+
"web-scraping",
|
|
9
|
+
"web-crawler",
|
|
10
|
+
"scraper",
|
|
11
|
+
"markdown",
|
|
12
|
+
"html-to-markdown",
|
|
13
|
+
"llm",
|
|
14
|
+
"rag",
|
|
15
|
+
"ai-agents",
|
|
16
|
+
"evidence",
|
|
17
|
+
"sdk",
|
|
18
|
+
"typescript"
|
|
19
|
+
],
|
|
6
20
|
"repository": {
|
|
7
21
|
"type": "git",
|
|
8
22
|
"url": "git+https://github.com/77777R7/Octocrawl.git"
|
|
9
23
|
},
|
|
10
|
-
"homepage": "https://
|
|
24
|
+
"homepage": "https://octocrawl.dev",
|
|
25
|
+
"bugs": {
|
|
26
|
+
"url": "https://github.com/77777R7/Octocrawl/issues"
|
|
27
|
+
},
|
|
11
28
|
"type": "module",
|
|
12
29
|
"files": [
|
|
13
30
|
"dist",
|