@octocrawl/sdk 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (74) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +17 -0
  3. package/dist/index.cjs +1278 -0
  4. package/dist/index.js +1240 -0
  5. package/dist/types/cjs/client.d.ts +362 -0
  6. package/dist/types/cjs/contracts/access.d.ts +166 -0
  7. package/dist/types/cjs/contracts/actions.d.ts +191 -0
  8. package/dist/types/cjs/contracts/api.d.ts +891 -0
  9. package/dist/types/cjs/contracts/benchmark.d.ts +116 -0
  10. package/dist/types/cjs/contracts/checkpoint.d.ts +164 -0
  11. package/dist/types/cjs/contracts/compliance.d.ts +412 -0
  12. package/dist/types/cjs/contracts/crawl.d.ts +302 -0
  13. package/dist/types/cjs/contracts/delivery.d.ts +136 -0
  14. package/dist/types/cjs/contracts/evidenceRecord.d.ts +192 -0
  15. package/dist/types/cjs/contracts/execution.d.ts +197 -0
  16. package/dist/types/cjs/contracts/extractor.d.ts +379 -0
  17. package/dist/types/cjs/contracts/file.d.ts +117 -0
  18. package/dist/types/cjs/contracts/firecrawl.d.ts +258 -0
  19. package/dist/types/cjs/contracts/groundTruth.d.ts +77 -0
  20. package/dist/types/cjs/contracts/identityBundle.d.ts +70 -0
  21. package/dist/types/cjs/contracts/index.d.ts +30 -0
  22. package/dist/types/cjs/contracts/map.d.ts +180 -0
  23. package/dist/types/cjs/contracts/monitor.d.ts +217 -0
  24. package/dist/types/cjs/contracts/monitorConfig.d.ts +9 -0
  25. package/dist/types/cjs/contracts/policy.d.ts +93 -0
  26. package/dist/types/cjs/contracts/proxy.d.ts +52 -0
  27. package/dist/types/cjs/contracts/recipe.d.ts +74 -0
  28. package/dist/types/cjs/contracts/regexSafety.d.ts +31 -0
  29. package/dist/types/cjs/contracts/result.d.ts +503 -0
  30. package/dist/types/cjs/contracts/session.d.ts +51 -0
  31. package/dist/types/cjs/contracts/ssrf.d.ts +16 -0
  32. package/dist/types/cjs/contracts/status.d.ts +27 -0
  33. package/dist/types/cjs/contracts/structured.d.ts +185 -0
  34. package/dist/types/cjs/contracts/tableMarkdown.d.ts +58 -0
  35. package/dist/types/cjs/contracts/tokens.d.ts +47 -0
  36. package/dist/types/cjs/index.d.ts +9 -0
  37. package/dist/types/cjs/package.json +1 -0
  38. package/dist/types/cjs/version.d.ts +8 -0
  39. package/dist/types/cjs/watcher.d.ts +151 -0
  40. package/dist/types/esm/client.d.ts +362 -0
  41. package/dist/types/esm/contracts/access.d.ts +166 -0
  42. package/dist/types/esm/contracts/actions.d.ts +191 -0
  43. package/dist/types/esm/contracts/api.d.ts +891 -0
  44. package/dist/types/esm/contracts/benchmark.d.ts +116 -0
  45. package/dist/types/esm/contracts/checkpoint.d.ts +164 -0
  46. package/dist/types/esm/contracts/compliance.d.ts +412 -0
  47. package/dist/types/esm/contracts/crawl.d.ts +302 -0
  48. package/dist/types/esm/contracts/delivery.d.ts +136 -0
  49. package/dist/types/esm/contracts/evidenceRecord.d.ts +192 -0
  50. package/dist/types/esm/contracts/execution.d.ts +197 -0
  51. package/dist/types/esm/contracts/extractor.d.ts +379 -0
  52. package/dist/types/esm/contracts/file.d.ts +117 -0
  53. package/dist/types/esm/contracts/firecrawl.d.ts +258 -0
  54. package/dist/types/esm/contracts/groundTruth.d.ts +77 -0
  55. package/dist/types/esm/contracts/identityBundle.d.ts +70 -0
  56. package/dist/types/esm/contracts/index.d.ts +30 -0
  57. package/dist/types/esm/contracts/map.d.ts +180 -0
  58. package/dist/types/esm/contracts/monitor.d.ts +217 -0
  59. package/dist/types/esm/contracts/monitorConfig.d.ts +9 -0
  60. package/dist/types/esm/contracts/policy.d.ts +93 -0
  61. package/dist/types/esm/contracts/proxy.d.ts +52 -0
  62. package/dist/types/esm/contracts/recipe.d.ts +74 -0
  63. package/dist/types/esm/contracts/regexSafety.d.ts +31 -0
  64. package/dist/types/esm/contracts/result.d.ts +503 -0
  65. package/dist/types/esm/contracts/session.d.ts +51 -0
  66. package/dist/types/esm/contracts/ssrf.d.ts +16 -0
  67. package/dist/types/esm/contracts/status.d.ts +27 -0
  68. package/dist/types/esm/contracts/structured.d.ts +185 -0
  69. package/dist/types/esm/contracts/tableMarkdown.d.ts +58 -0
  70. package/dist/types/esm/contracts/tokens.d.ts +47 -0
  71. package/dist/types/esm/index.d.ts +9 -0
  72. package/dist/types/esm/version.d.ts +8 -0
  73. package/dist/types/esm/watcher.d.ts +151 -0
  74. package/package.json +40 -0
@@ -0,0 +1,117 @@
1
+ /**
2
+ * A response W2L took for a file rather than a web page: a PDF, CSV, JSON,
3
+ * plain-text, XLSX, XLS or ZIP file. The bytes are saved as received, with
4
+ * their SHA-256 and size, and never sent to the browser as a page would be.
5
+ * Types and the size cap only; detection lives in @w2l/extract-tf, saving in
6
+ * @w2l/bench.
7
+ */
8
+ import type { NetworkPolicy } from './policy.js';
9
+ export declare const FILE_KINDS: readonly ["pdf", "csv", "json", "text", "xlsx", "xls", "zip"];
10
+ export type FileKind = (typeof FILE_KINDS)[number];
11
+ /** The operator cap on a file's size when `W2L_MAX_FILE_BYTES` is not set: 50 MiB. */
12
+ export declare const DEFAULT_MAX_FILE_BYTES: number;
13
+ /** The largest cap `W2L_MAX_FILE_BYTES` may set: 500 MiB. A file is held in memory while it is hashed and read. */
14
+ export declare const MAX_FILE_BYTES_CEILING: number;
15
+ /** What the PDF declares about itself; null where it declares nothing (pdfToMarkdown's `info`). */
16
+ export interface PdfDocumentInfo {
17
+ title: string | null;
18
+ author: string | null;
19
+ subject: string | null;
20
+ keywords: string | null;
21
+ creator: string | null;
22
+ producer: string | null;
23
+ /** PDF date strings as declared, e.g. `D:20240315120000+01'00'`. */
24
+ creationDate: string | null;
25
+ modificationDate: string | null;
26
+ language: string | null;
27
+ pdfVersion: string | null;
28
+ encrypted: boolean;
29
+ }
30
+ /** The text of a PDF as W2L read it into `markdown`. */
31
+ export interface FilePdfText {
32
+ /** Pages in the file; null when it could not be opened. */
33
+ pageCount: number | null;
34
+ /** Pages read into `markdown`, from the first. */
35
+ pagesRead: number;
36
+ /**
37
+ * Every page read: its 1-based number, the page label the PDF declares
38
+ * (a printed number such as "xii"), and the offsets of its text in
39
+ * `markdown` (`markdown.slice(start, end)`), which follows the line
40
+ * `<!-- page N -->`.
41
+ */
42
+ pages: readonly {
43
+ number: number;
44
+ label: string | null;
45
+ start: number;
46
+ end: number;
47
+ }[];
48
+ info: PdfDocumentInfo | null;
49
+ /**
50
+ * `tables_unverified` on every PDF with text; `no_text_layer` per page
51
+ * without text (no OCR is run); `page_error`, `page_cap`, `time_budget`.
52
+ */
53
+ warnings: readonly {
54
+ code: string;
55
+ message: string;
56
+ page?: number;
57
+ }[];
58
+ /** Why no text could be read at all (`not_pdf`, `encrypted`, `malformed`, `time_budget`); null otherwise. */
59
+ error: {
60
+ code: string;
61
+ message: string;
62
+ } | null;
63
+ }
64
+ export interface FileWarning {
65
+ /**
66
+ * `text_not_decoded`: the bytes are not valid text in their declared or
67
+ * default (UTF-8) encoding, so no text is returned. `pdf_not_parsed`: the
68
+ * request's `parsers` named no `pdf` entry, so a PDF's text was not read.
69
+ */
70
+ code: 'text_not_decoded' | 'pdf_not_parsed';
71
+ message: string;
72
+ }
73
+ export interface FileDescription {
74
+ kind: FileKind;
75
+ /**
76
+ * `content_type`: the response's Content-Type named the kind. `content`:
77
+ * the bytes did, when the Content-Type did not (a `%PDF-` or ZIP header
78
+ * under `application/octet-stream`, or no Content-Type at all).
79
+ */
80
+ detectedBy: 'content_type' | 'content';
81
+ /** The Content-Type header as received; null when the response had none. */
82
+ contentType: string | null;
83
+ /** The Content-Length the server declared; null when it declared none. */
84
+ declaredBytes: number | null;
85
+ /** The size cap that applied: the operator's `W2L_MAX_FILE_BYTES`, or a request's lower `maxFileBytes`. */
86
+ maxBytes: number;
87
+ /** Bytes received, all of them; null when the file was not read (over the cap). */
88
+ bytes: number | null;
89
+ /** SHA-256 (hex) of the bytes received; null when not read. */
90
+ sha256: string | null;
91
+ /**
92
+ * Where the bytes were saved, as received: `<task root>/files/<sha256>.<ext>`.
93
+ * Null when they were not saved (over the cap, or no file store is configured).
94
+ */
95
+ path: string | null;
96
+ /**
97
+ * What `markdown` carries: `pdf_text` (the PDF's text layer, one
98
+ * `<!-- page N -->` marker per page), `text` (a CSV, JSON or plain-text
99
+ * file's text as received), or null (no text: binary kinds, or text that
100
+ * could not be decoded).
101
+ */
102
+ markdownFrom: 'pdf_text' | 'text' | null;
103
+ /** The encoding text was decoded with (the declared charset, else UTF-8); null without text. */
104
+ encoding: string | null;
105
+ warnings: readonly FileWarning[];
106
+ /** PDF files that were read; null otherwise. */
107
+ pdf: FilePdfText | null;
108
+ }
109
+ /**
110
+ * The operator's file cap from `W2L_MAX_FILE_BYTES` (bytes, an integer from 1
111
+ * to MAX_FILE_BYTES_CEILING), else `fallback`. Anything else is an error, so
112
+ * a mistyped cap stops the service at start instead of being ignored.
113
+ */
114
+ export declare function maxFileBytesFromEnv(env: Readonly<Record<string, string | undefined>>, fallback?: number): number;
115
+ /** The cap for one fetch: the policy's (operator) cap, lowered by the request's `maxFileBytes` when it sets one. */
116
+ export declare function fileByteCap(policy: Pick<NetworkPolicy, 'maxFileBytes'>, requested?: number): number;
117
+ //# sourceMappingURL=file.d.ts.map
@@ -0,0 +1,258 @@
1
+ /**
2
+ * Firecrawl v1 scrape/crawl snapshot, frozen 2026-09-18, and its map path (added 2026-10-03).
3
+ *
4
+ * A one-shot migration shim: map the main paths onto the native
5
+ * contract. Not a compatibility layer. A parameter or format the shim cannot
6
+ * honour is rejected by name (HTTP 400, success: false), never ignored.
7
+ */
8
+ import type { AgentHints, CrawlAccepted, MapRequest, ParsedCrawlStartRequest, ScrapeMetadata, ScrapeRequest, ScrapeResponse } from './api.js';
9
+ import type { MapResponse } from './map.js';
10
+ import type { CrawlReport } from './crawl.js';
11
+ import type { JobWebhookEnvelope } from './delivery.js';
12
+ import type { FetchResult } from './result.js';
13
+ import type { StepRecord, StepStatus, TaskStatus } from './checkpoint.js';
14
+ export declare const FIRECRAWL_SHIM_SNAPSHOT: {
15
+ capturedAt: string;
16
+ apiVersion: "v1";
17
+ docs: {
18
+ scrape: string;
19
+ crawl: string;
20
+ crawlStatus: string;
21
+ map: string;
22
+ };
23
+ paths: readonly ["/scrape", "/crawl", "/crawl/:id", "/map"];
24
+ notCovered: readonly ["search", "interact", "agent", "monitor", "extract"];
25
+ };
26
+ export declare const FIRECRAWL_SHIM_DIFFS: readonly ["Challenge / block pages are success: false (Firecrawl often returns them as success markdown).", "A page with no main content is success: false (failed: empty_unverified) with the whole page in data.markdown as evidence; with onlyMainContent: false it is success: true.", "A page whose server HTML is a shell for data its scripts fill in is fetched again on the browser rung, and the rendered page is the answer when it holds more; otherwise the HTTP page is returned with client_rendered_suspected and low_content_yield warnings on the native response, whose messages /fc passes through as data.warning (one string, joined with a space), as it does every native warning. Firecrawl renders every page in a browser.", "No fire-engine, proxy pools or JSON extract.", "actions (scrape only; a crawl's scrapeOptions.actions is refused) run on the local browser rung alone, which such a request selects, after load, stability and waitFor and before the formats are read: wait (milliseconds up to 60000, or a selector, waited for up to 60 s within the scrape's timeout), click (all: true clicks every match), write (into the focused element), press, scroll (one screen up or down, of the page or the element a selector names), screenshot, scrape, executeJavascript (a function body; return gives the value) and pdf; at most 50 steps. data.actions holds screenshots and pdfs as data: URIs (Firecrawl returns URLs), scrapes as { url, html } and javascriptReturns as { type, value }. A step that fails stops the steps after it: success is false with data.actions.failed naming the step, its code and message, and data.markdown is the page as it stood. A step that leads the page to a URL robots.txt or the egress policy refuses fails with navigation_refused and that page is not read. A hosted server refuses actions, and the cache is not used with them.", "An omitted maxAge reuses nothing: every page is fetched live unless the request sets maxAge above 0, minAge or lockdown (Firecrawl reuses its own index by default; its Python SDK sends maxAge 4 hours). A reused page is one W2L itself stored, under the same options, on this server (its task root), never a shared index; only a success is stored, and data.metadata says cacheState hit with cachedAt (its fetch time) or miss when one was looked up. lockdown with no stored result is HTTP 404 SCRAPE_LOCKDOWN_CACHE_MISS on scrape, and nothing is fetched; a crawl in lockdown needs sitemap skip, and each page with no stored result is failed with cache_miss. Mode authed neither stores nor reuses. A Firecrawl body never sets useCached, W2L's reuse of a crawl's own pages on resume.", "Omitted limit / maxDepth stay unbounded on a local server; a hosted server takes its crawl limit for an omitted or null limit and refuses a larger one. Firecrawl defaults are 10000 / 10.", "maxDepth counts link hops from the start URL (Firecrawl calls that maxDiscoveryDepth); Firecrawl maxDepth counts URL path depth.", "Crawl start is mapped onto native POST /v1/crawl; the shim itself returns 200 {success,id,url}.", "creditsUsed and expiresAt are null: W2L counts no credits and keeps crawl results until their task directory is deleted.", "Crawl status describes the latest attempt: completed counts its successful pages, total adds its failed, blocked and duplicate pages and, while this API process runs the crawl, the pages in flight and queued (null for a paused crawl); data lists the failed and blocked pages too (with metadata.error) but not the duplicates, whose content is an earlier entry's, up to 100 per response (limit 1 to 1000) with next carrying a W2L cursor; skip is rejected.", "Scrape maps url, formats, onlyMainContent, includeTags, excludeTags, waitFor, timeout, headers, mobile, skipTlsVerification, fastMode, blockAds, removeBase64Images, maxAge, minAge, storeInCache, lockdown, actions, origin and integration; crawl maps url, limit (as maxPages), maxDepth, includePaths, excludePaths, regexOnFullURL, ignoreQueryParameters, deduplicateSimilarURLs, crawlEntireDomain (and its v1 name allowBackwardLinks), allowSubdomains, allowExternalLinks, sitemap (v2; v1 ignoreSitemap true is skip and false include, sitemapOnly true is only), maxConcurrency, origin, integration and the same scrapeOptions (applied to every page). The formats are markdown, links, html, rawHtml, images, screenshot (also screenshot@fullPage, and { type: \"screenshot\", fullPage, quality, viewport }) and an { type: \"attributes\", selectors } entry; other formats and parameters the shim does not map (proxy, location, json, ...) are rejected by name with HTTP 400 and success: false; a refusal of stealth, proxy: stealth or enhanced, or ignoreRobotsTxt names the supported route in agent_hints.", "A crawl follows links inside the start URL's path subtree on its host and www twin by default (crawlEntireDomain false), folds /a and /a/, / and /index.html, www and apex, http and https into one page (deduplicateSimilarURLs true) and reports every collapsed or refused link in the native crawl status (discovery) and each page's trace (links_offered); allowSubdomains takes every host under the start URL's apex (no public-suffix list), allowExternalLinks every host, each page with its own robots.txt read.", "sitemap (default include, as in Firecrawl) reads the sitemaps the start URL's robots.txt names, or /sitemap.xml, with the crawl's own http identity, robots.txt verdict, SSRF checks and proxy, and queues their URLs ahead of the start page's links under the same host, subtree, path and depth rules; only follows no page link; skip reads none. The native crawl status lists every sitemap file read, refused or unreadable in discovery.sitemap; the shim's status carries nothing of it, and sitemap fetches have no signed compliance record. maxConcurrency caps the pages one crawl fetches at once, at most the service's worker count (HTTP 400 above it), and never raises the per-host ceiling.", "screenshot (data.screenshot, a data:image/png;base64 string, or image/jpeg with quality 1 to 100) is captured on the local browser rung alone, which such a request selects (no http attempt; a server without a browser rung refuses the format with HTTP 400): after load, stability and waitFor, before the DOM is read, CSS-pixel sized at the declared 1280x800 viewport (device scale factor 2 is declared, not baked into the image) or at the viewport asked for (integers 320..1920 by 240..1080, within the declared screen; a window size, not a change of identity); fullPage captures the document's whole height at that width without scrolling first, so sections a page loads on scroll may show unloaded. A capture the browser could not make leaves data.screenshot null with a screenshot_unavailable warning while the page stands; a file or a page that was not rendered has null too. Firecrawl captures at its own viewport and may return a URL instead of the image.", "images (data.images) lists every image URL of the whole document as received: img src and srcset candidates, picture sources, lazy data-src/data-srcset/data-lazy-src/data-original, video posters, image_src links, og:image and twitter:image, absolute http(s) with the fragment stripped, each once, in document order, data: URIs left out; includeTags, excludeTags and onlyMainContent do not narrow it. attributes (data.attributes) gives, per selector, the named attribute's values as written, elements without it skipped; a selector W2L does not match is HTTP 400 by name, as for includeTags. Both are absent for a file and for a page that is success: false. removeBase64Images (default true) keeps an image's alt text where Firecrawl writes a (<Base64-Image-Removed>) placeholder; false keeps the data: URI in the Markdown.", "origin (the Firecrawl SDKs' client label) and integration are stored, not echoed: the scrape record (GET /v1/scrapes/:id) and the crawl task carry them, and nothing sent to the target changes.", "data.metadata carries scrapeId (a UUID per call, which GET /v1/scrapes/:id looks up), proxyUsed (operator for the server's environment proxy, user for the caller's own egress, else null), timezone (the browser rung's declared zone, null on the HTTP rung), creditsUsed: null (W2L counts no credits), concurrencyLimited and concurrencyQueueDurationMs (whether and how long the per-origin ceiling held the fetch back), and cacheState and cachedAt when the cache was asked (never a guessed miss).", "A page whose result W2L has advice about (a login wall, a robots.txt rule, a gate, a cut, a script-filled shell) carries data.agent_hints, one sentence each; the native response calls them agentHints. A request refused for an option W2L does not offer carries agent_hints in the error envelope, and a caller over the server's per-minute rate limit gets HTTP 429 { success: false, error, code: rate_limited, agent_hints } with Retry-After.", "headers never override the User-Agent, the client hints, a credential (authorization, cookie) or a transport header: such a header is HTTP 400 naming it, where Firecrawl sends it. The headers go to the requested origin after the declared identity and are on the record (the trace, the browser lane's signed sentHeaders); both rungs withhold them from a redirect hop to another origin and say so (custom_headers_withheld).", "mobile selects a declared Android Chrome identity (User-Agent, client hints, 412x915 viewport, touch) that robots.txt is evaluated against and the record carries; the page is whatever the site serves to it, with no DOM rewriting. It is refused with mode research.", "skipTlsVerification relaxes certificate verification for one local request and its robots.txt lookup, recorded in the trace (tls_verification_skipped) and a tls_unverified warning the native response carries; a hosted W2L refuses it with HTTP 400. Without it a bad certificate is success: false with failed: tls_error. Firecrawl's Python SDK sends true by default; W2L verifies by default.", "fastMode keeps the http rung alone: a page that needs scripts is success: false with failed: empty_unverified, never rendered; waitFor has no effect under it. Firecrawl's fast mode still renders.", "blockAds (default true) aborts requests to a bundled list of about 50 ad-serving hosts on the local browser rung and removes ad and cookie-banner elements before extraction; false keeps them. The list is curated, not EasyList: ads from hosts outside it are not blocked.", "html is the cleaned HTML the markdown is written from: the main content, the whole page without scripts, styles, form controls and embedded media when onlyMainContent is false, or a <body> holding the includeTags elements. rawHtml is the page as the answering rung received it: the response body on the HTTP rung, the rendered DOM on a browser rung. Both are null for a file and for a page that is success: false.", "includeTags keeps only the named elements, in document order, whatever onlyMainContent says; excludeTags removes elements from the main content, the whole page and an includeTags selection. A selector that does not parse, or that uses a sibling combinator, a positional pseudo-class, :has() or another pseudo-class W2L does not match, is rejected with HTTP 400.", "An omitted timeout stays 300000 ms (Firecrawl: 30000). A timeout is answered with HTTP 200: success: true with the content fetched so far (native status partial), or success: false with failed: timeout; Firecrawl answers it with an error.", "waitFor skips the HTTP rung, which cannot run scripts, and starts at the browser rung; the wait counts toward timeout.", "metadata has title, description, language, keywords, robots and favicon only when the page declares them, and the Open Graph (ogTitle, ogDescription, ogUrl, ogImage, ogAudio, ogVideo, ogDeterminer, ogLocale, ogLocaleAlternate, ogSiteName), Dublin Core (dcTermsCreated, dcDateCreated, dcDate, dcTermsType, dcType, dcTermsAudience, dcTermsSubject, dcSubject, dcDescription, dcTermsKeywords) and article (publishedTime, modifiedTime, articleTag, articleSection) tags under Firecrawl's names, each only when the page states it, as written (no date normalisation, no fallback from another tag); twitter:* and other meta tags are not passed through, and a failed or blocked page has none.", "A PDF answers success: true with its text layer as markdown and metadata.numPages (the document's page count). parsers maps Firecrawl's pdf entry (the string or { type: \"pdf\", mode, maxPages, pages, pageMarkers }); mode fast and auto both read the text layer, and mode ocr and the image parser are refused by name. pageMarkers is false unless asked, as on Firecrawl; with true W2L writes a <!-- page N --> line before each page (Firecrawl writes --- and the marker between pages). pages: true adds data.pages, [{ pageNumber, markdown }]. maxPages (1 to 10000) reads the first pages, and a cut it asked for stays success: true. parsers [] or v1 parsePDF false reads no PDF: success: true with markdown null and the file saved as received. A PDF without a text layer is success: false with failed: empty_unverified (no OCR). CSV, JSON and text files give their text as received; XLSX, XLS and ZIP files are success: true with markdown null. A file over W2L_MAX_FILE_BYTES is success: false with failed: body_too_large.", "Map (POST /fc/v1/map) maps url, search, sitemap (v2; v1 ignoreSitemap true is skip and false include, sitemapOnly true is only; both v1 flags true is HTTP 400), includeSubdomains, ignoreQueryParameters, limit (1 to 100000, default 5000), timeout (1000 to 300000 ms for the whole map, default 60000; Firecrawl documents no default), origin and integration onto native POST /v1/map, and answers 200 { success: true, id, links: [url strings], warning?, agent_hints? }, or 200 { success: false, id, error, links: [] } when the map found nothing because a source failed or its deadline passed; useIndex, location, ignoreCache, threatProtection and auditMetadata are refused by name (useIndex with the hint that W2L keeps no URL index). Omitted options take W2L's defaults: includeSubdomains and ignoreQueryParameters are false, where Firecrawl v2 documents true for both. search keeps the URLs in which every word appears in the decoded URL or the title in hand, in discovery order; Firecrawl orders by relevance. A map reads the sitemaps the site declares and one page body (the start URL, http rung only), so a site without a sitemap maps only its start page's links; a title is the start page's own, an anchor's text or a sitemap's <news:title>, never fetched from the target; robots-disallowed URLs are left out and counted on the native response (GET /v1/maps/:id), which also records every sitemap file read.", "A crawl's webhook (a URL string or { url, headers, metadata, events }) is mapped onto the native webhook and its receiver gets Firecrawl's payload shape: { success, type: crawl.started | crawl.page | crawl.completed | crawl.failed, id, data: [page], metadata, error? }, one durable delivery per event with retries, every request carrying x-w2l-event-id, x-w2l-event-version and x-w2l-delivery-id (and the signature pair with secretEnv, a native option). A cancelled crawl is crawl.failed with error \"cancelled\". The native rules apply: https (plain http for a loopback receiver of a local server only), no content-type, host or x-w2l-* header, at most 32 headers and 32 metadata strings; a hosted server takes public https receivers only. GET /v1/deliveries?jobId=<id> on the native API lists the deliveries."];
27
+ export interface FirecrawlPage {
28
+ markdown: string | null;
29
+ /** Present when the `html` format was asked for; null when the page has none (a file, a page that did not succeed). */
30
+ html?: string | null;
31
+ /** Present when the `rawHtml` format was asked for; null when the page has none. */
32
+ rawHtml?: string | null;
33
+ links?: string[];
34
+ /** A PDF's pages, when the request's pdf parser asked for them (`pages: true`). */
35
+ pages?: {
36
+ pageNumber: number;
37
+ markdown: string;
38
+ }[];
39
+ /** Present when the `images` format was asked for and the page was read as content: every image URL of the whole document. */
40
+ images?: string[];
41
+ /** Present when an `attributes` entry was asked for and the page was read as content: per selector, the attribute's values as written. */
42
+ attributes?: Array<{
43
+ selector: string;
44
+ attribute: string;
45
+ values: string[];
46
+ }>;
47
+ /** Present when a `screenshot` entry was asked for: the capture as a `data:image/png;base64,…` (or `image/jpeg`) string, as Firecrawl v1 clients read it; null when the browser rung rendered no page or could not capture it. */
48
+ screenshot?: string | null;
49
+ /**
50
+ * Present when the request ran `actions`: screenshots and PDFs as data: URIs, the HTML of each scrape step, each
51
+ * script's return, and the step that failed if one did.
52
+ */
53
+ actions?: {
54
+ screenshots: string[];
55
+ scrapes: Array<{
56
+ url: string;
57
+ html: string;
58
+ }>;
59
+ javascriptReturns: Array<{
60
+ type: string;
61
+ value: unknown;
62
+ }>;
63
+ pdfs: string[];
64
+ /** W2L's own list steps (scrollToEnd, loadMore, paginate): what each did and why it stopped; present when the request had one. */
65
+ lists?: Array<{
66
+ index: number;
67
+ type: string;
68
+ stoppedBy: string;
69
+ rounds: number;
70
+ items: number | null;
71
+ itemsRead?: number | null;
72
+ }>;
73
+ failed?: {
74
+ index: number;
75
+ type: string;
76
+ code: string;
77
+ message: string;
78
+ };
79
+ };
80
+ /** The native `warnings` as one string, their messages joined with a space; present when the result has any. */
81
+ warning?: string;
82
+ /** What to change about the request next time, one sentence each (the native `agentHints`); present when W2L has any. */
83
+ agent_hints?: string[];
84
+ /** Page fields appear only when the page declares them (W2L's `metadata`, null values left out). */
85
+ metadata: {
86
+ title?: string;
87
+ description?: string;
88
+ language?: string;
89
+ keywords?: string;
90
+ robots?: string;
91
+ favicon?: string;
92
+ /** Open Graph, Dublin Core and article tags under Firecrawl's names, each only when the page states it (see PageMetadata). */
93
+ ogTitle?: string;
94
+ ogDescription?: string;
95
+ ogUrl?: string;
96
+ ogImage?: string;
97
+ ogAudio?: string;
98
+ ogVideo?: string;
99
+ ogDeterminer?: string;
100
+ ogLocale?: string;
101
+ ogLocaleAlternate?: string[];
102
+ ogSiteName?: string;
103
+ dcTermsCreated?: string;
104
+ dcDateCreated?: string;
105
+ dcDate?: string;
106
+ dcTermsType?: string;
107
+ dcType?: string;
108
+ dcTermsAudience?: string;
109
+ dcTermsSubject?: string;
110
+ dcSubject?: string;
111
+ dcDescription?: string;
112
+ dcTermsKeywords?: string;
113
+ publishedTime?: string;
114
+ modifiedTime?: string;
115
+ /** Every `article:tag`, joined with `, ` as Firecrawl writes it. */
116
+ articleTag?: string;
117
+ articleSection?: string;
118
+ sourceURL: string;
119
+ /** The final URL, after redirects (`evidence.finalUrl`). */
120
+ url: string;
121
+ /** The status of the response that answered `url` (`evidence.httpStatus`); null when none did. */
122
+ statusCode: number | null;
123
+ /** That response's `content-type` header; left out when there was none. */
124
+ contentType?: string;
125
+ /** A PDF's page count, read or not; left out for anything else. */
126
+ numPages?: number;
127
+ /** On a page that did not succeed: W2L's failure, block or budget reason code (`http_error`, `cloudflare_challenge`, ...), or its status when it has none (`empty_verified`). */
128
+ error?: string;
129
+ /** The facts of the scrape call (native `metadata`), on a scrape response; a crawl status page has no call of its own and leaves them out. */
130
+ scrapeId?: string;
131
+ proxyUsed?: ScrapeMetadata['proxyUsed'];
132
+ timezone?: string | null;
133
+ /** Null: W2L counts no credits. */
134
+ creditsUsed?: null;
135
+ concurrencyLimited?: boolean;
136
+ concurrencyQueueDurationMs?: number;
137
+ /** `hit` when a stored result answered (`maxAge`, `minAge`, `lockdown`), `miss` when one was looked up and none fit; absent when the cache was not asked. */
138
+ cacheState?: 'hit' | 'miss';
139
+ /** On a hit, when the reused result was fetched. */
140
+ cachedAt?: string;
141
+ };
142
+ }
143
+ export interface FirecrawlScrapeResponse {
144
+ success: boolean;
145
+ data: FirecrawlPage;
146
+ error?: string;
147
+ }
148
+ /** POST /fc/v1/map: the URLs as strings; a map that found nothing because a source failed or its deadline passed is `success: false`. */
149
+ export type FirecrawlMapResponse = {
150
+ success: true;
151
+ id: string;
152
+ links: string[];
153
+ warning?: string;
154
+ agent_hints?: AgentHints;
155
+ } | {
156
+ success: false;
157
+ id: string;
158
+ error: string;
159
+ links: [];
160
+ agent_hints?: AgentHints;
161
+ };
162
+ export interface FirecrawlCrawlStarted {
163
+ success: true;
164
+ id: string;
165
+ url: string;
166
+ }
167
+ /** The type of one Firecrawl-shaped webhook payload: the job kind and the event (`cancelled` is sent as `failed` with `error: 'cancelled'`). */
168
+ export type FirecrawlWebhookType = `${'crawl' | 'batch_scrape'}.${'started' | 'page' | 'completed' | 'failed'}`;
169
+ /**
170
+ * A job event as Firecrawl's webhook receivers read it: `data` holds the page
171
+ * of a `page` event (as `/fc` crawl status lists it) and is empty otherwise;
172
+ * `metadata` is the request's `webhook.metadata`. The W2L envelope's fields
173
+ * are on the delivery's headers (`x-w2l-event-id`, `x-w2l-event-version`).
174
+ */
175
+ export interface FirecrawlWebhookPayload {
176
+ success: true;
177
+ type: FirecrawlWebhookType;
178
+ id: string;
179
+ data: FirecrawlPage[];
180
+ metadata: Readonly<Record<string, string>>;
181
+ error?: string;
182
+ }
183
+ export type FirecrawlCrawlJobStatus = 'scraping' | 'completed' | 'failed' | 'cancelled';
184
+ export interface FirecrawlCrawlStatus {
185
+ status: FirecrawlCrawlJobStatus;
186
+ /**
187
+ * The latest attempt's pages and errors, plus, while this API process runs
188
+ * the crawl, the pages in flight and those queued within its page limit.
189
+ * Null while the crawl is unfinished and no process here runs it (paused).
190
+ */
191
+ total: number | null;
192
+ /** The latest attempt's pages that succeeded (status success or partial): the data entries without metadata.error. */
193
+ completed: number;
194
+ /** Null: W2L counts no credits, and an unknown count is not zero. */
195
+ creditsUsed: number | null;
196
+ /** Null: crawl results stay until their task directory is deleted. */
197
+ expiresAt: string | null;
198
+ /** The URL of the next page of `data`: there while more pages are stored or the crawl is running, left out after the last. */
199
+ next?: string;
200
+ /** One page of the latest attempt's steps in the order they were recorded, errors included (with `metadata.error`). */
201
+ data: FirecrawlPage[];
202
+ }
203
+ /** What the API knows about a crawl beyond its steps, for its Firecrawl status. */
204
+ export interface FirecrawlCrawlCounts {
205
+ completed: number;
206
+ total: number | null;
207
+ next?: string;
208
+ }
209
+ /**
210
+ * completed and total from how many of the latest attempt's steps have each
211
+ * status: completed is its success and partial pages, total every step it
212
+ * recorded plus, while the crawl is unfinished, `ahead`, the pages it will
213
+ * still record, null when no process here runs it.
214
+ */
215
+ export declare function firecrawlCrawlCounts(status: TaskStatus, steps: Partial<Record<StepStatus, number>>, ahead: number | null): {
216
+ completed: number;
217
+ total: number | null;
218
+ };
219
+ /** Default and largest number of steps one `GET /fc/v1/crawl/:id` returns in `data`. */
220
+ export declare const FIRECRAWL_STATUS_PAGE_SIZE: {
221
+ readonly default: 100;
222
+ readonly max: 1000;
223
+ };
224
+ export declare function parseFirecrawlScrapeRequest(body: unknown): ScrapeRequest;
225
+ export declare function parseFirecrawlCrawlRequest(body: unknown): ParsedCrawlStartRequest;
226
+ /**
227
+ * A Firecrawl map request as the native one. v1 ignoreSitemap true is
228
+ * sitemap skip and false include, sitemapOnly true is only, both true is
229
+ * refused; the v2 `sitemap` wins over the v1 flags, as on the crawl shim.
230
+ * The native parser validates the rest; an omitted option takes W2L's
231
+ * default (includeSubdomains and ignoreQueryParameters false).
232
+ */
233
+ export declare function parseFirecrawlMapRequest(body: unknown): MapRequest;
234
+ /** The native map response as Firecrawl's: the links as URL strings, the warnings' messages joined, the hints as `agent_hints`. */
235
+ export declare function wrapMap(response: MapResponse): FirecrawlMapResponse;
236
+ /**
237
+ * A job event in Firecrawl's webhook shape: the type from the job kind and
238
+ * the event (`cancelled` becomes `failed` with `error: 'cancelled'`), the
239
+ * page of a `page` event as `/fc` crawl status would list it, `metadata` as
240
+ * the request gave it.
241
+ */
242
+ export declare function wrapJobWebhook(envelope: JobWebhookEnvelope, result: FetchResult | null): FirecrawlWebhookPayload;
243
+ /** The native scrape response as Firecrawl's envelope: the page with the call's facts in `data.metadata` and its hints as `data.agent_hints`. */
244
+ export declare function wrapScrape(response: ScrapeResponse): FirecrawlScrapeResponse;
245
+ export declare function wrapCrawlAccepted(native: CrawlAccepted, seedUrl: string): FirecrawlCrawlStarted;
246
+ /**
247
+ * The query of `GET /fc/v1/crawl/:id`: `cursor` (from a `next` URL) and
248
+ * `limit` (1 to 1000, default 100). Anything else, Firecrawl's `skip`
249
+ * included, is rejected by name: pages follow `next`.
250
+ */
251
+ export declare function parseFirecrawlCrawlStatusQuery(query: Record<string, string | undefined>): {
252
+ cursor?: string;
253
+ limit: number;
254
+ };
255
+ export declare function wrapCrawlStatus(report: Pick<CrawlReport, 'status'>, steps: readonly StepRecord[], counts: FirecrawlCrawlCounts): FirecrawlCrawlStatus;
256
+ /** A scrape a cache-only request (`lockdown`) could not answer: nothing was fetched, and Firecrawl answers it with HTTP 404 `SCRAPE_LOCKDOWN_CACHE_MISS`. */
257
+ export declare function isLockdownCacheMiss(response: Pick<FetchResult, 'status' | 'failureReason'>): boolean;
258
+ //# sourceMappingURL=firecrawl.d.ts.map
@@ -0,0 +1,77 @@
1
+ import type { BlockReason, Lane, ResultStatus } from './status.js';
2
+ import type { ExpectedTable } from './tableMarkdown.js';
3
+ /** The five false-success checks. Fixtures evaluate all five; canaries evaluate the evidence-bearing subset. */
4
+ export declare const FALSE_SUCCESS_CHECK: readonly ["missing_required_content", "challenge_text_returned", "content_yield_below_floor", "wrong_page_content", "silent_truncation"];
5
+ export type FalseSuccessCheck = (typeof FALSE_SUCCESS_CHECK)[number];
6
+ /**
7
+ * Checks computable without per-page annotation. Canary runs evaluate exactly these;
8
+ * the annotation-dependent ones are reported as unknown, never as pass.
9
+ */
10
+ export declare const EVIDENCE_ONLY_CHECKS: readonly FalseSuccessCheck[];
11
+ export declare const ANNOTATION_REQUIRED_CHECKS: readonly FalseSuccessCheck[];
12
+ export type CheckOutcome = 'pass' | 'fail' | 'unknown';
13
+ export interface CheckResult {
14
+ check: FalseSuccessCheck;
15
+ outcome: CheckOutcome;
16
+ /** Why it failed, or why it could not be evaluated. */
17
+ detail: string | null;
18
+ }
19
+ export interface CaseBudget {
20
+ maxTokens: number;
21
+ maxWallMs: number;
22
+ maxAttempts: number;
23
+ }
24
+ /**
25
+ * Ground truth for one benchmark case.
26
+ * Fixtures carry the full annotation; canaries may omit the annotation-only fields.
27
+ */
28
+ export interface GroundTruth {
29
+ /** Stable case id, unique across the suite. */
30
+ id: string;
31
+ /** Path relative to the fixture server root, or an absolute https URL for canaries. */
32
+ target: string;
33
+ kind: 'fixture' | 'canary';
34
+ /** What this case is designed to probe. */
35
+ category: string;
36
+ evaluationSet?: 'development' | 'holdout';
37
+ /** Substrings that MUST appear in the extracted markdown. */
38
+ mustContain: readonly string[];
39
+ /** Substrings that MUST NOT appear in delivered content (nav, footer, cookie banner, ads). */
40
+ mustNotContain: readonly string[];
41
+ /**
42
+ * Structural table assertion, evaluated by check `missing_required_content`
43
+ * against the extracted markdown. Omitted for cases without table annotation.
44
+ * Use unique cell strings so cells/anchors locate unambiguously.
45
+ */
46
+ expectedTable?: ExpectedTable | null;
47
+ /** The lane the runtime is expected to settle on. */
48
+ expectedLane: Lane;
49
+ /** True when an empty extraction is the correct answer. */
50
+ emptyIsLegit: boolean;
51
+ /** Inclusive token range for the main content. Null when unannotated (canary). */
52
+ expectedMainTokens: {
53
+ min: number;
54
+ max: number;
55
+ } | null;
56
+ budget: CaseBudget;
57
+ /** Expected terminal status. Lets non-success cases (blocked, failed) be asserted too. */
58
+ expectedStatus: ResultStatus;
59
+ /**
60
+ * Which gate the case is expected to be classified as. Required whenever
61
+ * `expectedStatus` is 'blocked' — an unscored reason field is a reason field
62
+ * that silently drifts. Omitted (undefined) for every other case, which the
63
+ * runner reports as "not graded" rather than as a pass.
64
+ */
65
+ expectedBlockReason?: BlockReason | null;
66
+ notes?: string;
67
+ }
68
+ export interface SuiteMeta {
69
+ name: string;
70
+ version: string;
71
+ /** ISO date the suite content was last curated. */
72
+ curatedAt: string;
73
+ }
74
+ export interface Suite extends SuiteMeta {
75
+ cases: readonly GroundTruth[];
76
+ }
77
+ //# sourceMappingURL=groundTruth.d.ts.map
@@ -0,0 +1,70 @@
1
+ /**
2
+ * Identity bundle: the anti-bot layer we can actually own.
3
+ *
4
+ * FP-Inconsistent (IMC 2025) showed that bots which "escape" DataDome/BotD
5
+ * still leak because they rewrite fingerprint fields that no longer agree
6
+ * with each other. 2606.30119 showed JS stealth often *increases*
7
+ * detectability. So the product rule is: one coherent identity, fail closed
8
+ * on contradiction. This is not disguise — it is refusing to ship a lie.
9
+ */
10
+ import { type BrowserFingerprint, type CrawlMode, type IdentityDevice, type ModeIdentity } from './compliance.js';
11
+ export interface IdentityBundle {
12
+ userAgent: string;
13
+ clientHints: Readonly<Record<string, string>>;
14
+ locale: string;
15
+ timezoneId: string;
16
+ viewport: {
17
+ width: number;
18
+ height: number;
19
+ };
20
+ screen: {
21
+ width: number;
22
+ height: number;
23
+ };
24
+ /** Device pixels per CSS pixel; absent on a bundle built before the mobile identity existed. */
25
+ deviceScaleFactor?: number;
26
+ /** Whether the context reports a mobile device; must agree with `sec-ch-ua-mobile` and the UA. */
27
+ isMobile?: boolean;
28
+ hasTouch?: boolean;
29
+ }
30
+ /** Problems in a bundle. Empty means the identity is internally coherent. */
31
+ export declare function identityBundleIssues(bundle: IdentityBundle): string[];
32
+ /**
33
+ * L4 / vendor wire identity. We never inject our Chrome bundle into a vendor
34
+ * browser — we measure theirs. This check is: does what they actually send
35
+ * contradict the *mode* we claimed, or contradict itself?
36
+ *
37
+ * Missing client hints are allowed (most vendors do not surface them).
38
+ * HeadlessChrome, research-mode Chromium, and UA/hint disagreement are not.
39
+ */
40
+ export declare function vendorIdentityIssues(mode: CrawlMode, userAgent: string, clientHints?: Readonly<Record<string, string>>): string[];
41
+ export declare function assertIdentityBundle(bundle: IdentityBundle): void;
42
+ /**
43
+ * Wire headers implied by a bundle. Asserts first: a contradictory bundle
44
+ * never becomes bytes. HTTP subjects send this object verbatim.
45
+ */
46
+ export declare function headersFromIdentity(bundle: IdentityBundle): Record<string, string>;
47
+ /**
48
+ * Route extras (proxy, cookies, vendor resume) that MUST NOT retune the face.
49
+ * Changing IP ≠ changing identity.
50
+ */
51
+ export interface RouteAccess {
52
+ proxy?: unknown;
53
+ session?: unknown;
54
+ resume?: unknown;
55
+ }
56
+ export type IdentityOverride = Partial<Pick<IdentityBundle, 'userAgent' | 'clientHints' | 'locale' | 'timezoneId' | 'viewport' | 'screen'>>;
57
+ /**
58
+ * The identity a fetch presents, given a mode and optional user route.
59
+ * `access` is accepted so callers cannot "forget" it — it is ignored.
60
+ * An override that would retune timezone/locale/viewport to a proxy geo is
61
+ * refused: that is fingerprint spoofing wearing a routing costume. `device`
62
+ * picks between the two declared browser identities (desktop, mobile): a
63
+ * declared second identity is not an override, and both pass the same checks.
64
+ */
65
+ export declare function identityForRoute(mode: CrawlMode, access?: RouteAccess | null | undefined, chromeMajor?: number, override?: IdentityOverride | null, device?: IdentityDevice): IdentityBundle;
66
+ /** The bundle of an identity with its fingerprint: the device's own (browserFingerprintFor) unless one is given. */
67
+ export declare function identityBundleFrom(identity: ModeIdentity, fingerprint?: Readonly<BrowserFingerprint>): IdentityBundle;
68
+ /** One-line identity for CLI stdout. Never empty for a coherent bundle. */
69
+ export declare function formatIdentitySummary(bundle: IdentityBundle): string;
70
+ //# sourceMappingURL=identityBundle.d.ts.map
@@ -0,0 +1,30 @@
1
+ export * from './status.js';
2
+ export * from './tokens.js';
3
+ export * from './result.js';
4
+ export * from './groundTruth.js';
5
+ export * from './tableMarkdown.js';
6
+ export * from './extractor.js';
7
+ export * from './policy.js';
8
+ export * from './ssrf.js';
9
+ export * from './proxy.js';
10
+ export * from './benchmark.js';
11
+ export * from './compliance.js';
12
+ export * from './identityBundle.js';
13
+ export * from './access.js';
14
+ export * from './checkpoint.js';
15
+ export * from './crawl.js';
16
+ export * from './regexSafety.js';
17
+ export * from './api.js';
18
+ export * from './firecrawl.js';
19
+ export * from './monitor.js';
20
+ export * from './monitorConfig.js';
21
+ export * from './session.js';
22
+ export * from './recipe.js';
23
+ export * from './execution.js';
24
+ export * from './delivery.js';
25
+ export * from './structured.js';
26
+ export * from './evidenceRecord.js';
27
+ export * from './file.js';
28
+ export * from './map.js';
29
+ export * from './actions.js';
30
+ //# sourceMappingURL=index.d.ts.map