context.dev 2.18.0 → 2.20.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (116) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +220 -0
  3. data/README.md +1 -1
  4. data/lib/context_dev/client.rb +0 -4
  5. data/lib/context_dev/models/monitor_retrieve_run_params.rb +26 -0
  6. data/lib/context_dev/models/monitor_retrieve_run_response.rb +232 -0
  7. data/lib/context_dev/models/monitor_rotate_webhook_secret_params.rb +20 -0
  8. data/lib/context_dev/models/monitor_rotate_webhook_secret_response.rb +765 -0
  9. data/lib/context_dev/models/{web_web_scrape_sitemap_params.rb → web_map_urls_params.rb} +14 -14
  10. data/lib/context_dev/models/web_map_urls_response.rb +123 -0
  11. data/lib/context_dev/models/web_scrape_params.rb +1035 -0
  12. data/lib/context_dev/models/web_scrape_response.rb +974 -0
  13. data/lib/context_dev/models/web_screenshot_params.rb +15 -1
  14. data/lib/context_dev/models/web_screenshot_response.rb +3 -3
  15. data/lib/context_dev/models.rb +8 -22
  16. data/lib/context_dev/resources/brand.rb +0 -36
  17. data/lib/context_dev/resources/monitors.rb +47 -0
  18. data/lib/context_dev/resources/web.rb +118 -486
  19. data/lib/context_dev/version.rb +1 -1
  20. data/lib/context_dev.rb +8 -23
  21. data/rbi/context_dev/client.rbi +0 -3
  22. data/rbi/context_dev/models/monitor_retrieve_run_params.rbi +46 -0
  23. data/rbi/context_dev/models/monitor_retrieve_run_response.rbi +452 -0
  24. data/rbi/context_dev/models/monitor_rotate_webhook_secret_params.rbi +38 -0
  25. data/rbi/context_dev/models/monitor_rotate_webhook_secret_response.rbi +1367 -0
  26. data/rbi/context_dev/models/{web_web_scrape_sitemap_params.rbi → web_map_urls_params.rbi} +23 -45
  27. data/rbi/context_dev/models/web_map_urls_response.rbi +226 -0
  28. data/rbi/context_dev/models/web_scrape_params.rbi +2221 -0
  29. data/rbi/context_dev/models/web_scrape_response.rbi +1874 -0
  30. data/rbi/context_dev/models/web_screenshot_params.rbi +23 -0
  31. data/rbi/context_dev/models/web_screenshot_response.rbi +4 -4
  32. data/rbi/context_dev/models.rbi +9 -24
  33. data/rbi/context_dev/resources/brand.rbi +0 -35
  34. data/rbi/context_dev/resources/monitors.rbi +24 -0
  35. data/rbi/context_dev/resources/web.rbi +152 -654
  36. data/sig/context_dev/client.rbs +0 -2
  37. data/sig/context_dev/models/monitor_retrieve_run_params.rbs +28 -0
  38. data/sig/context_dev/models/monitor_retrieve_run_response.rbs +182 -0
  39. data/sig/context_dev/models/monitor_rotate_webhook_secret_params.rbs +23 -0
  40. data/sig/context_dev/models/monitor_rotate_webhook_secret_response.rbs +545 -0
  41. data/sig/context_dev/models/{web_web_scrape_sitemap_params.rbs → web_map_urls_params.rbs} +22 -22
  42. data/sig/context_dev/models/web_map_urls_response.rbs +125 -0
  43. data/sig/context_dev/models/web_scrape_params.rbs +925 -0
  44. data/sig/context_dev/models/web_scrape_response.rbs +782 -0
  45. data/sig/context_dev/models/web_screenshot_params.rbs +7 -0
  46. data/sig/context_dev/models.rbs +8 -22
  47. data/sig/context_dev/resources/brand.rbs +0 -9
  48. data/sig/context_dev/resources/monitors.rbs +11 -0
  49. data/sig/context_dev/resources/web.rbs +33 -128
  50. metadata +26 -71
  51. data/lib/context_dev/models/ai_extract_product_params.rb +0 -125
  52. data/lib/context_dev/models/ai_extract_product_response.rb +0 -402
  53. data/lib/context_dev/models/ai_extract_products_params.rb +0 -233
  54. data/lib/context_dev/models/ai_extract_products_response.rb +0 -370
  55. data/lib/context_dev/models/brand_retrieve_simplified_params.rb +0 -120
  56. data/lib/context_dev/models/brand_retrieve_simplified_response.rb +0 -441
  57. data/lib/context_dev/models/web_extract_fonts_params.rb +0 -114
  58. data/lib/context_dev/models/web_extract_fonts_response.rb +0 -291
  59. data/lib/context_dev/models/web_extract_params.rb +0 -419
  60. data/lib/context_dev/models/web_extract_response.rb +0 -287
  61. data/lib/context_dev/models/web_web_scrape_bytes_params.rb +0 -346
  62. data/lib/context_dev/models/web_web_scrape_bytes_response.rb +0 -135
  63. data/lib/context_dev/models/web_web_scrape_html_params.rb +0 -812
  64. data/lib/context_dev/models/web_web_scrape_html_response.rb +0 -628
  65. data/lib/context_dev/models/web_web_scrape_images_params.rb +0 -373
  66. data/lib/context_dev/models/web_web_scrape_images_response.rb +0 -409
  67. data/lib/context_dev/models/web_web_scrape_md_params.rb +0 -670
  68. data/lib/context_dev/models/web_web_scrape_md_response.rb +0 -567
  69. data/lib/context_dev/models/web_web_scrape_screenshot_params.rb +0 -471
  70. data/lib/context_dev/models/web_web_scrape_screenshot_response.rb +0 -169
  71. data/lib/context_dev/models/web_web_scrape_sitemap_response.rb +0 -144
  72. data/lib/context_dev/resources/ai.rb +0 -71
  73. data/rbi/context_dev/models/ai_extract_product_params.rbi +0 -253
  74. data/rbi/context_dev/models/ai_extract_product_response.rbi +0 -791
  75. data/rbi/context_dev/models/ai_extract_products_params.rbi +0 -490
  76. data/rbi/context_dev/models/ai_extract_products_response.rbi +0 -718
  77. data/rbi/context_dev/models/brand_retrieve_simplified_params.rbi +0 -256
  78. data/rbi/context_dev/models/brand_retrieve_simplified_response.rbi +0 -1040
  79. data/rbi/context_dev/models/web_extract_fonts_params.rbi +0 -223
  80. data/rbi/context_dev/models/web_extract_fonts_response.rbi +0 -563
  81. data/rbi/context_dev/models/web_extract_params.rbi +0 -781
  82. data/rbi/context_dev/models/web_extract_response.rbi +0 -512
  83. data/rbi/context_dev/models/web_web_scrape_bytes_params.rbi +0 -702
  84. data/rbi/context_dev/models/web_web_scrape_bytes_response.rbi +0 -238
  85. data/rbi/context_dev/models/web_web_scrape_html_params.rbi +0 -1690
  86. data/rbi/context_dev/models/web_web_scrape_html_response.rbi +0 -1287
  87. data/rbi/context_dev/models/web_web_scrape_images_params.rbi +0 -728
  88. data/rbi/context_dev/models/web_web_scrape_images_response.rbi +0 -910
  89. data/rbi/context_dev/models/web_web_scrape_md_params.rbi +0 -1052
  90. data/rbi/context_dev/models/web_web_scrape_md_response.rbi +0 -1088
  91. data/rbi/context_dev/models/web_web_scrape_screenshot_params.rbi +0 -1580
  92. data/rbi/context_dev/models/web_web_scrape_screenshot_response.rbi +0 -332
  93. data/rbi/context_dev/models/web_web_scrape_sitemap_response.rbi +0 -258
  94. data/rbi/context_dev/resources/ai.rbi +0 -63
  95. data/sig/context_dev/models/ai_extract_product_params.rbs +0 -106
  96. data/sig/context_dev/models/ai_extract_product_response.rbs +0 -314
  97. data/sig/context_dev/models/ai_extract_products_params.rbs +0 -202
  98. data/sig/context_dev/models/ai_extract_products_response.rbs +0 -293
  99. data/sig/context_dev/models/brand_retrieve_simplified_params.rbs +0 -104
  100. data/sig/context_dev/models/brand_retrieve_simplified_response.rbs +0 -413
  101. data/sig/context_dev/models/web_extract_fonts_params.rbs +0 -93
  102. data/sig/context_dev/models/web_extract_fonts_response.rbs +0 -230
  103. data/sig/context_dev/models/web_extract_params.rbs +0 -324
  104. data/sig/context_dev/models/web_extract_response.rbs +0 -239
  105. data/sig/context_dev/models/web_web_scrape_bytes_params.rbs +0 -531
  106. data/sig/context_dev/models/web_web_scrape_bytes_response.rbs +0 -108
  107. data/sig/context_dev/models/web_web_scrape_html_params.rbs +0 -878
  108. data/sig/context_dev/models/web_web_scrape_html_response.rbs +0 -525
  109. data/sig/context_dev/models/web_web_scrape_images_params.rbs +0 -281
  110. data/sig/context_dev/models/web_web_scrape_images_response.rbs +0 -371
  111. data/sig/context_dev/models/web_web_scrape_md_params.rbs +0 -758
  112. data/sig/context_dev/models/web_web_scrape_md_response.rbs +0 -465
  113. data/sig/context_dev/models/web_web_scrape_screenshot_params.rbs +0 -619
  114. data/sig/context_dev/models/web_web_scrape_screenshot_response.rbs +0 -127
  115. data/sig/context_dev/models/web_web_scrape_sitemap_response.rbs +0 -117
  116. data/sig/context_dev/resources/ai.rbs +0 -21
@@ -49,102 +49,6 @@ module ContextDev
49
49
  )
50
50
  end
51
51
 
52
- # Crawl a website, use the provided JSON Schema and instructions to prioritize
53
- # relevant internal links, and extract structured data from the selected pages.
54
- sig do
55
- params(
56
- schema: T::Hash[Symbol, T.anything],
57
- url: String,
58
- actions:
59
- T::Array[
60
- T.any(
61
- ContextDev::WebExtractParams::Action::Wait::OrHash,
62
- ContextDev::WebExtractParams::Action::Perform::OrHash,
63
- ContextDev::WebExtractParams::Action::Scroll::OrHash
64
- )
65
- ],
66
- fact_check: T::Boolean,
67
- follow_subdomains: T::Boolean,
68
- include_frames: T::Boolean,
69
- instructions: String,
70
- max_age_ms: Integer,
71
- max_depth: Integer,
72
- max_pages: Integer,
73
- pdf: ContextDev::WebExtractParams::Pdf::OrHash,
74
- settle_animations: T::Boolean,
75
- stop_after_ms: Integer,
76
- tags: T::Array[String],
77
- timeout_opts: ContextDev::WebExtractParams::TimeoutOpts::OrHash,
78
- wait_for_ms: Integer,
79
- zdr: ContextDev::WebExtractParams::Zdr::OrSymbol,
80
- request_options: ContextDev::RequestOptions::OrHash
81
- ).returns(ContextDev::Models::WebExtractResponse)
82
- end
83
- def extract(
84
- # JSON Schema for the returned data object. Image fields such as `image_urls` or
85
- # `product_photos` automatically make page image references available to
86
- # extraction, so product data and photos can be returned in one call. TypeScript
87
- # Zod users can pass a JSON Schema generated from a Zod object; Python users can
88
- # pass the equivalent JSON Schema object.
89
- schema:,
90
- # The starting website URL to crawl and extract from. Must include http:// or
91
- # https://.
92
- url:,
93
- # Optional browser actions executed in order on the requested page after it loads,
94
- # before links are discovered or additional pages are crawled. Requires a paid
95
- # plan. When actions are provided and stopAfterMs is omitted, the crawl budget
96
- # defaults to 110000 ms.
97
- actions: nil,
98
- # When true, every returned value must be grounded in facts stated on the page;
99
- # fields that cannot be supported by the page are returned as null/empty. When
100
- # false (default), the model may make reasonable inferences and derivations from
101
- # the page content (e.g. ideal customer, competitor analysis, recommendations)
102
- # while keeping verifiable specifics (names, quotes, URLs, dates, metrics)
103
- # faithful to the source.
104
- fact_check: nil,
105
- # When true, follow links on subdomains of the starting URL's domain.
106
- follow_subdomains: nil,
107
- # When true, iframe contents are included in Markdown before extraction.
108
- include_frames: nil,
109
- # Optional extraction guidance, such as which facts to prioritize or how to
110
- # interpret fields in the schema.
111
- instructions: nil,
112
- # Return cached scrape results if a prior scrape for the same parameters is
113
- # younger than this many milliseconds. Defaults to 7 days (604800000 ms).
114
- max_age_ms: nil,
115
- # Optional maximum link depth from the starting URL (0 = only the starting page).
116
- # If omitted, there is no crawl depth limit.
117
- max_depth: nil,
118
- # Maximum number of pages to analyze for extraction. Hard cap: 50. Defaults to 5.
119
- max_pages: nil,
120
- pdf: nil,
121
- # When true, waits briefly for CSS and transition animations to settle before
122
- # extracting each crawled page. Defaults to false. This adds a bit of latency in
123
- # exchange for more stable output on animated pages.
124
- settle_animations: nil,
125
- # Soft time budget for the crawl in milliseconds. Min: 10000 (10s). Max: 110000
126
- # (110s). Defaults to 80000 (80s), or 110000 (110s) when browser actions are
127
- # provided.
128
- stop_after_ms: nil,
129
- # Optional tags for tracking usage. Up to 20 tags, each 1 to 50 characters.
130
- tags: nil,
131
- # Optional request deadline and behavior on timeout. For GET requests, use
132
- # timeoutOpts[milliseconds]=30000&timeoutOpts[behavior]=fail or a JSON-encoded
133
- # timeoutOpts object.
134
- timeout_opts: nil,
135
- # Optional browser wait time in milliseconds after initial page load for each
136
- # crawled page.
137
- wait_for_ms: nil,
138
- # Set to enabled to bypass shared caches and omit request and response content
139
- # from retained usage logs. Asset uploads are skipped, so hosted image URLs are
140
- # omitted. Requires zero data retention to be enabled for your organization
141
- # (contact support@context.dev), otherwise the request fails with ZDR_NOT_ENABLED.
142
- # Successful ZDR responses include X-Context-ZDR: true.
143
- zdr: nil,
144
- request_options: {}
145
- )
146
- end
147
-
148
52
  # Analyze a company's landing page and web search evidence to return direct
149
53
  # competitors for the same product or market.
150
54
  sig do
@@ -181,43 +85,6 @@ module ContextDev
181
85
  )
182
86
  end
183
87
 
184
- # Scrape font information from a website including font families, usage
185
- # statistics, fallbacks, and element/word counts.
186
- sig do
187
- params(
188
- direct_url: String,
189
- domain: String,
190
- max_age_ms: T.nilable(Integer),
191
- tags: T::Array[String],
192
- timeout_opts: ContextDev::WebExtractFontsParams::TimeoutOpts::OrHash,
193
- request_options: ContextDev::RequestOptions::OrHash
194
- ).returns(ContextDev::Models::WebExtractFontsResponse)
195
- end
196
- def extract_fonts(
197
- # A specific URL to fetch fonts from directly, bypassing domain resolution (e.g.,
198
- # 'https://example.com/design-system'). When provided, fonts are extracted from
199
- # this exact URL. You must provide either 'domain' or 'directUrl', but not both.
200
- direct_url: nil,
201
- # Domain name to extract fonts from (e.g., 'example.com', 'google.com'). The
202
- # domain will be automatically normalized and validated. You must provide either
203
- # 'domain' or 'directUrl', but not both.
204
- domain: nil,
205
- # Maximum age in milliseconds for cached brand data before the API performs a hard
206
- # refresh. Defaults to 3 months (7776000000 ms). Set to 0 to always perform a hard
207
- # refresh. Negative values are clamped to 0; values above 1 year (31536000000 ms)
208
- # are clamped to 1 year.
209
- max_age_ms: nil,
210
- # Comma-separated tags for tracking request usage. Up to 20 tags, each 1-50
211
- # characters.
212
- tags: nil,
213
- # Optional request deadline and behavior on timeout. For GET requests, use
214
- # timeoutOpts[milliseconds]=30000&timeoutOpts[behavior]=fail or a JSON-encoded
215
- # timeoutOpts object.
216
- timeout_opts: nil,
217
- request_options: {}
218
- )
219
- end
220
-
221
88
  # Extract a comprehensive design system from a website including colors,
222
89
  # typography, spacing, shadows, and UI components.
223
90
  sig do
@@ -269,6 +136,149 @@ module ContextDev
269
136
  )
270
137
  end
271
138
 
139
+ # Discovers URLs using the same sitemap crawl, filters, and limits as
140
+ # /web/scrape/sitemap. Each URL includes its available title, description,
141
+ # keywords, and language. URLs without stored enrichment are returned immediately
142
+ # with only the URL and queued for background HTML scraping, so later requests can
143
+ # include their metadata. Responses are never cached as a whole; every request
144
+ # reads the current per-URL enrichment. Zero data retention and credential-bearing
145
+ # discovery requests return URLs without reading or storing shared enrichment or
146
+ # queuing background scrapes. Costs 1 credit, or 2 credits with search.
147
+ sig do
148
+ params(
149
+ domain: String,
150
+ headers: T::Hash[Symbol, String],
151
+ include_subdomains: T::Boolean,
152
+ max_links: Integer,
153
+ search: String,
154
+ sitemap_url: String,
155
+ tags: T::Array[String],
156
+ timeout_opts: ContextDev::WebMapURLsParams::TimeoutOpts::OrHash,
157
+ url_regex: String,
158
+ zdr: ContextDev::WebMapURLsParams::Zdr::OrSymbol,
159
+ request_options: ContextDev::RequestOptions::OrHash
160
+ ).returns(ContextDev::Models::WebMapURLsResponse)
161
+ end
162
+ def map_urls(
163
+ # Domain to build a sitemap for
164
+ domain:,
165
+ # Optional outbound HTTP headers forwarded only to the target URL, sent as
166
+ # deep-object query params such as headers[X-Custom]=value. When provided, caching
167
+ # is bypassed: the result is neither read from nor written to cache.
168
+ headers: nil,
169
+ # When true, discover and include public pages and sitemaps on subdomains of the
170
+ # requested domain. Defaults to false.
171
+ include_subdomains: nil,
172
+ # Maximum number of links to return from the sitemap crawl. Defaults to 10,000.
173
+ # Minimum is 1, maximum is 100,000.
174
+ max_links: nil,
175
+ # Optional search phrase. When provided, the crawled sitemap is filtered to the
176
+ # pages whose URLs are about that phrase, most relevant first, and the request
177
+ # costs 2 credits instead of 1.
178
+ search: nil,
179
+ # Optional explicit sitemap URL. When provided, exactly this sitemap is crawled
180
+ # instead of discovering the domain's sitemaps.
181
+ sitemap_url: nil,
182
+ # Comma-separated tags for tracking request usage. Up to 20 tags, each 1-50
183
+ # characters.
184
+ tags: nil,
185
+ # Optional request deadline and behavior on timeout. For GET requests, use
186
+ # timeoutOpts[milliseconds]=30000&timeoutOpts[behavior]=fail or a JSON-encoded
187
+ # timeoutOpts object.
188
+ timeout_opts: nil,
189
+ # Optional RE2-compatible regex pattern. Only URLs matching this pattern are
190
+ # returned and counted against maxLinks.
191
+ url_regex: nil,
192
+ # Set to enabled to bypass shared caches and omit request and response content
193
+ # from retained usage logs. Asset uploads are skipped, so hosted image URLs are
194
+ # omitted. Requires zero data retention to be enabled for your organization
195
+ # (contact support@context.dev), otherwise the request fails with ZDR_NOT_ENABLED.
196
+ # Successful ZDR responses include X-Context-ZDR: true.
197
+ zdr: nil,
198
+ request_options: {}
199
+ )
200
+ end
201
+
202
+ # Reuse cached outputs independently and capture missing formats in one page
203
+ # visit. Each cache key includes only the settings that affect that output. HTML
204
+ # is shared with Markdown, parsed fields, product data, highlights, and JSON
205
+ # extraction. Cached outputs can come from different visits within maxAgeMs; use 0
206
+ # for a fresh capture. HTML-only requests use the existing fast acquisition path.
207
+ # Highlights return the plain-text passages most relevant to
208
+ # highlightsParams.query. One credit per request, including cache hits and missing
209
+ # pages, or two with browser actions; highlights add 3 credits when passages are
210
+ # returned; JSON extraction adds four credits and runs an LLM over the page
211
+ # Markdown on every request that has text to extract; PDF OCR adds one credit per
212
+ # recovered page on fresh extraction; the product output adds one credit, plus six
213
+ # more when the specialized model is used. Original response bytes and screenshots
214
+ # are limited to 20 MiB each, screenshots to 40 megapixels, and the combined
215
+ # browser capture to 60 MiB.
216
+ sig do
217
+ params(
218
+ formats: ContextDev::WebScrapeParams::Formats::OrHash,
219
+ url: String,
220
+ highlights_params:
221
+ ContextDev::WebScrapeParams::HighlightsParams::OrHash,
222
+ image_params: ContextDev::WebScrapeParams::ImageParams::OrHash,
223
+ json_params: ContextDev::WebScrapeParams::JsonParams::OrHash,
224
+ markdown_params: ContextDev::WebScrapeParams::MarkdownParams::OrHash,
225
+ max_age_ms: Integer,
226
+ parse_params: ContextDev::WebScrapeParams::ParseParams::OrHash,
227
+ product_params: ContextDev::WebScrapeParams::ProductParams::OrHash,
228
+ screenshot_params:
229
+ ContextDev::WebScrapeParams::ScreenshotParams::OrHash,
230
+ shared_params: ContextDev::WebScrapeParams::SharedParams::OrHash,
231
+ tags: T::Array[String],
232
+ timeout_opts: ContextDev::WebScrapeParams::TimeoutOpts::OrHash,
233
+ zdr: ContextDev::WebScrapeParams::Zdr::OrSymbol,
234
+ request_options: ContextDev::RequestOptions::OrHash
235
+ ).returns(ContextDev::Models::WebScrapeResponse)
236
+ end
237
+ def scrape(
238
+ # Outputs to return. Enable at least one; omitted formats are false.
239
+ formats:,
240
+ # The URL to scrape.
241
+ url:,
242
+ # Highlight options. Requires formats.highlights: true.
243
+ highlights_params: nil,
244
+ # Image options. Requires formats.images: true.
245
+ image_params: nil,
246
+ # Required when formats.json is true.
247
+ json_params: nil,
248
+ # Markdown options. Requires formats.markdown: true.
249
+ markdown_params: nil,
250
+ # Maximum age of each cached output. Defaults to 1 day; 0 fetches fresh and
251
+ # updates the requested outputs. Compatible outputs are shared with the individual
252
+ # scrape endpoints. Image results with hosted files refresh after 23 hours; other
253
+ # outputs retain their own freshness.
254
+ max_age_ms: nil,
255
+ # Required when formats.parse is true.
256
+ parse_params: nil,
257
+ # Product options. Requires formats.product: true.
258
+ product_params: nil,
259
+ # Screenshot options. Requires formats.screenshot: true.
260
+ screenshot_params: nil,
261
+ # Shared browser and content settings. Content filters leave screenshots and
262
+ # original bytes unchanged.
263
+ shared_params: nil,
264
+ # Labels for tracking request usage. Not retained when zdr is enabled.
265
+ tags: nil,
266
+ # Total deadline, including navigation, actions, waiting, and all outputs.
267
+ # Defaults to 60000 milliseconds with behavior fail. Use return-partial to capture
268
+ # the current page state and return captured images if image processing cannot
269
+ # finish before the deadline; these responses set isPartial and are not cached.
270
+ # Every requested format must still be available. Fixed waits must fit before a
271
+ # response reserve of up to 5000 milliseconds (at most one quarter of the timeout)
272
+ # when using return-partial.
273
+ timeout_opts: nil,
274
+ # Zero data retention. Bypasses caches and uploads; excludes request/response
275
+ # content and tags from logs. Must be enabled for your organization. Not available
276
+ # with the highlights output.
277
+ zdr: nil,
278
+ request_options: {}
279
+ )
280
+ end
281
+
272
282
  # Capture a screenshot of a website.
273
283
  sig do
274
284
  params(
@@ -280,6 +290,7 @@ module ContextDev
280
290
  full_screenshot:
281
291
  ContextDev::WebScreenshotParams::FullScreenshot::OrSymbol,
282
292
  handle_cookie_popup: T::Boolean,
293
+ headers: T::Hash[Symbol, String],
283
294
  max_age_ms: T.nilable(Integer),
284
295
  page: ContextDev::WebScreenshotParams::Page::OrSymbol,
285
296
  scroll_offset: T.nilable(Integer),
@@ -320,6 +331,14 @@ module ContextDev
320
331
  # dismiss cookie banner before capture. If 'false' or not provided, captures the
321
332
  # page without that step.
322
333
  handle_cookie_popup: nil,
334
+ # Optional outbound HTTP headers, using the same JSON object or deep-object query
335
+ # format as other scrape endpoints (for example headers[Authorization]=Bearer
336
+ # token). Headers are scoped to the target origin during capture. For domain/page
337
+ # requests, discovery receives no custom headers and only pages on the resolved
338
+ # origin are eligible. Non-empty headers bypass screenshot caching and return an
339
+ # in-memory data URL; no screenshot is uploaded. Empty objects behave like omitted
340
+ # headers.
341
+ headers: nil,
323
342
  # Return a cached screenshot if a prior screenshot for the same parameters exists
324
343
  # and is younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
325
344
  # omitted. Max is 30 days (2592000000 ms). Set to 0 to always capture fresh.
@@ -518,527 +537,6 @@ module ContextDev
518
537
  )
519
538
  end
520
539
 
521
- # Downloads a resource and returns its bytes as base64. Supports images, PDFs,
522
- # HTML pages, and any other content type without image conversion, text
523
- # extraction, or character-encoding changes. HTTP compression is decoded before
524
- # base64 encoding. HTML is the original HTTP response; JavaScript is not rendered.
525
- # Follows public redirects and retries failed downloads through ISP and
526
- # residential proxies, with a direct fallback. When country is specified, only a
527
- # residential proxy in that country is used. Supply headers such as Referer for
528
- # images that require a referring page. Downloads are not cached. Maximum decoded
529
- # resource size: 20 MiB (20971520 bytes), before base64 encoding. Successful
530
- # requests cost 1 credit; errors are not billed.
531
- sig do
532
- params(
533
- url: String,
534
- country: ContextDev::WebWebScrapeBytesParams::Country::OrSymbol,
535
- headers: T::Hash[Symbol, String],
536
- tags: T::Array[String],
537
- timeout_opts:
538
- ContextDev::WebWebScrapeBytesParams::TimeoutOpts::OrHash,
539
- zdr: ContextDev::WebWebScrapeBytesParams::Zdr::OrSymbol,
540
- request_options: ContextDev::RequestOptions::OrHash
541
- ).returns(ContextDev::Models::WebWebScrapeBytesResponse)
542
- end
543
- def web_scrape_bytes(
544
- # Full HTTP(S) URL of the resource to download, such as an image, PDF, or page.
545
- url:,
546
- # Fetch the target page through a residential proxy in this country (ISO 3166-1
547
- # alpha-2).
548
- country: nil,
549
- # Optional outbound HTTP headers, such as Referer, Cookie, or Authorization. Send
550
- # as a JSON object or deep-object query params such as
551
- # headers[Referer]=https://example.com/. Host, Content-Length, and hop-by-hop
552
- # transport headers are rejected. Authorization and cookies are removed when a
553
- # redirect changes origin.
554
- headers: nil,
555
- # Comma-separated tags for tracking request usage. Up to 20 tags, each 1-50
556
- # characters.
557
- tags: nil,
558
- # Optional request deadline and behavior on timeout. For GET requests, use
559
- # timeoutOpts[milliseconds]=30000&timeoutOpts[behavior]=fail or a JSON-encoded
560
- # timeoutOpts object.
561
- timeout_opts: nil,
562
- # Set to enabled to bypass shared caches and omit request and response content
563
- # from retained usage logs. Asset uploads are skipped, so hosted image URLs are
564
- # omitted. Requires zero data retention to be enabled for your organization
565
- # (contact support@context.dev), otherwise the request fails with ZDR_NOT_ENABLED.
566
- # Successful ZDR responses include X-Context-ZDR: true.
567
- zdr: nil,
568
- request_options: {}
569
- )
570
- end
571
-
572
- # Scrapes the given URL and returns the HTML content of the page. Optional
573
- # extractRules return deterministic structured data in extracted using CSS
574
- # selectors, attributes, lists, and nested rules, without an LLM or additional
575
- # credits. Rules run on the returned HTML after selector and main-content
576
- # filtering. Send extractRules as a JSON-encoded query parameter. The base request
577
- # costs 1 credit; requests with browser actions cost 2 credits. A request that
578
- # hits its timeoutOpts.milliseconds deadline fails with 408 and is not billed,
579
- # unless timeoutOpts.behavior=return-partial is set — then the page as rendered so
580
- # far is returned with `finalDOMState: "still-loading"` and billed at the base
581
- # cost of 1 credit.
582
- sig do
583
- params(
584
- url: String,
585
- actions:
586
- T.nilable(
587
- T::Array[
588
- T.any(
589
- ContextDev::WebWebScrapeHTMLParams::Action::Wait::OrHash,
590
- ContextDev::WebWebScrapeHTMLParams::Action::Perform::OrHash,
591
- ContextDev::WebWebScrapeHTMLParams::Action::Scroll::OrHash
592
- )
593
- ]
594
- ),
595
- country: ContextDev::WebWebScrapeHTMLParams::Country::OrSymbol,
596
- exclude_selectors: T.nilable(T::Array[String]),
597
- extract_rules:
598
- T::Hash[
599
- Symbol,
600
- T.any(
601
- String,
602
- ContextDev::WebWebScrapeHTMLParams::ExtractRule::UnionMember1::OrHash
603
- )
604
- ],
605
- headers: T::Hash[Symbol, String],
606
- include_frames: T::Boolean,
607
- include_selectors: T.nilable(T::Array[String]),
608
- max_age_ms: T.nilable(Integer),
609
- pdf: ContextDev::WebWebScrapeHTMLParams::Pdf::OrHash,
610
- settle_animations: T::Boolean,
611
- tags: T::Array[String],
612
- timeout_opts: ContextDev::WebWebScrapeHTMLParams::TimeoutOpts::OrHash,
613
- use_main_content_only: T::Boolean,
614
- wait_for_ms: T.nilable(Integer),
615
- zdr: ContextDev::WebWebScrapeHTMLParams::Zdr::OrSymbol,
616
- request_options: ContextDev::RequestOptions::OrHash
617
- ).returns(ContextDev::Models::WebWebScrapeHTMLResponse)
618
- end
619
- def web_scrape_html(
620
- # Full URL to scrape (must include http:// or https:// protocol)
621
- url:,
622
- # Optional browser actions executed in array order after the page loads and before
623
- # content is captured. Requires a paid plan. Send a JSON array in the query
624
- # parameter. Maximum: 5 actions.
625
- actions: nil,
626
- # Fetch the target page through a residential proxy in this country (ISO 3166-1
627
- # alpha-2).
628
- country: nil,
629
- # CSS selectors to remove from the result. Applied after includeSelectors.
630
- # Exclusion takes precedence: an element matching both is removed. Examples:
631
- # "nav", "footer", ".ad-banner", "[aria-hidden=true]".
632
- exclude_selectors: nil,
633
- # Optional CSS extraction rules applied to the returned HTML after selector and
634
- # main-content filtering. Use selector strings ("h1", "a@href") or objects with
635
- # selector, type (item or list), and output (text, html, @attribute, or nested
636
- # rules). Text whitespace is normalized; html includes the matched element;
637
- # attributes are returned as written. Missing items are null and missing lists are
638
- # empty. CSS only; XPath is not supported. Maximum: 100 fields across 5 levels.
639
- # Send a JSON-encoded string in the extractRules query parameter.
640
- extract_rules: nil,
641
- # Optional outbound HTTP headers forwarded only to the target URL, sent as
642
- # deep-object query params such as headers[X-Custom]=value. When provided, caching
643
- # is bypassed: the result is neither read from nor written to cache.
644
- headers: nil,
645
- # When true, iframes are rendered inline into the returned HTML.
646
- include_frames: nil,
647
- # CSS selectors. When provided, only matching subtrees (and their descendants) are
648
- # kept and everything else is dropped. When omitted, the entire document is kept.
649
- # Examples: "article.main", "#content", "[role=main]".
650
- include_selectors: nil,
651
- # Return a cached result if a prior scrape for the same parameters exists and is
652
- # younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
653
- # omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
654
- max_age_ms: nil,
655
- # PDF parsing controls. Use start/end to limit text extraction and embedded-image
656
- # detection/OCR to an inclusive 1-based page range.
657
- pdf: nil,
658
- # When true, waits briefly for CSS and transition animations to settle before
659
- # extracting HTML. Defaults to false. This adds a bit of latency in exchange for
660
- # more stable output on animated pages.
661
- settle_animations: nil,
662
- # Comma-separated tags for tracking request usage. Up to 20 tags, each 1-50
663
- # characters.
664
- tags: nil,
665
- # Optional request deadline and behavior on timeout. For GET requests, use
666
- # timeoutOpts[milliseconds]=30000&timeoutOpts[behavior]=fail or a JSON-encoded
667
- # timeoutOpts object.
668
- timeout_opts: nil,
669
- # When true, return only the page's main content in the HTML response, excluding
670
- # headers, footers, sidebars, and navigation when detectable.
671
- use_main_content_only: nil,
672
- # Optional browser wait time in milliseconds after initial page load. Min: 0. Max:
673
- # 30000 (30 seconds). When combined with timeoutOpts, timeoutOpts.milliseconds
674
- # must be at least waitForMs + 10000 ms; a shorter deadline is rejected with 400
675
- # TIMEOUT_TOO_SHORT_FOR_WAIT.
676
- wait_for_ms: nil,
677
- # Set to enabled to bypass shared caches and omit request and response content
678
- # from retained usage logs. Asset uploads are skipped, so hosted image URLs are
679
- # omitted. Requires zero data retention to be enabled for your organization
680
- # (contact support@context.dev), otherwise the request fails with ZDR_NOT_ENABLED.
681
- # Successful ZDR responses include X-Context-ZDR: true.
682
- zdr: nil,
683
- request_options: {}
684
- )
685
- end
686
-
687
- # Extract image assets from a web page, including standard URLs, inline SVGs, data
688
- # URIs, responsive image sources, metadata, CSS backgrounds, video posters, and
689
- # embeds. The base request costs 1 credit, or 2 credits with browser actions. When
690
- # enrichment is enabled, the entire call costs 5 credits, including requests that
691
- # also use actions.
692
- sig do
693
- params(
694
- url: String,
695
- actions:
696
- T.nilable(
697
- T::Array[
698
- T.any(
699
- ContextDev::WebWebScrapeImagesParams::Action::Wait::OrHash,
700
- ContextDev::WebWebScrapeImagesParams::Action::Perform::OrHash,
701
- ContextDev::WebWebScrapeImagesParams::Action::Scroll::OrHash
702
- )
703
- ]
704
- ),
705
- dedupe: T::Boolean,
706
- enrichment:
707
- T.nilable(ContextDev::WebWebScrapeImagesParams::Enrichment::OrHash),
708
- headers: T::Hash[Symbol, String],
709
- max_age_ms: T.nilable(Integer),
710
- tags: T::Array[String],
711
- timeout_opts:
712
- ContextDev::WebWebScrapeImagesParams::TimeoutOpts::OrHash,
713
- wait_for_ms: T.nilable(Integer),
714
- zdr: ContextDev::WebWebScrapeImagesParams::Zdr::OrSymbol,
715
- request_options: ContextDev::RequestOptions::OrHash
716
- ).returns(ContextDev::Models::WebWebScrapeImagesResponse)
717
- end
718
- def web_scrape_images(
719
- # Page URL to inspect. Must include http:// or https://.
720
- url:,
721
- # Optional browser actions executed in array order after the page loads and before
722
- # content is captured. Requires a paid plan. Send a JSON array in the query
723
- # parameter. Maximum: 5 actions.
724
- actions: nil,
725
- # When true, visually duplicate images are removed: every image is loaded and
726
- # perceptually hashed, and only the highest-resolution copy of each duplicate
727
- # group is kept. Images that cannot be downloaded or hashed are kept. Default:
728
- # false.
729
- dedupe: nil,
730
- # Optional per-image processing, sent as deep-object query params such as
731
- # enrichment[resolution]=true.
732
- enrichment: nil,
733
- # Optional outbound HTTP headers forwarded only to the target URL, sent as
734
- # deep-object query params such as headers[X-Custom]=value. When provided, caching
735
- # is bypassed: the result is neither read from nor written to cache.
736
- headers: nil,
737
- # Reuse a cached result this many milliseconds old or newer. Default: 86400000 (1
738
- # day). Set to 0 to bypass cache. Maximum: 2592000000 (30 days).
739
- max_age_ms: nil,
740
- # Comma-separated tags for tracking request usage. Up to 20 tags, each 1-50
741
- # characters.
742
- tags: nil,
743
- # Optional request deadline and behavior on timeout. For GET requests, use
744
- # timeoutOpts[milliseconds]=30000&timeoutOpts[behavior]=fail or a JSON-encoded
745
- # timeoutOpts object.
746
- timeout_opts: nil,
747
- # Optional browser wait time in milliseconds after initial page load before
748
- # collecting images. Min: 0. Max: 30000 (30 seconds). When combined with
749
- # timeoutOpts, timeoutOpts.milliseconds must be at least waitForMs + 10000 ms; a
750
- # shorter deadline is rejected with 400 TIMEOUT_TOO_SHORT_FOR_WAIT.
751
- wait_for_ms: nil,
752
- # Set to enabled to bypass shared caches and omit request and response content
753
- # from retained usage logs. Asset uploads are skipped, so hosted image URLs are
754
- # omitted. Requires zero data retention to be enabled for your organization
755
- # (contact support@context.dev), otherwise the request fails with ZDR_NOT_ENABLED.
756
- # Successful ZDR responses include X-Context-ZDR: true.
757
- zdr: nil,
758
- request_options: {}
759
- )
760
- end
761
-
762
- # Scrapes the given URL into LLM usable Markdown. Inspect key_metadata on JSON
763
- # responses from a recognized API key; use error_code to distinguish stable
764
- # failure categories.
765
- #
766
- # ### YouTube
767
- #
768
- # YouTube URLs return the video or channel itself rather than the surrounding
769
- # player and navigation chrome. A URL addressing a single video (`/watch`,
770
- # `youtu.be`, `/shorts`, `/embed`, `/live`) returns its title, channel, duration,
771
- # view count, keywords, full description, and the transcript when the video has
772
- # captions that can be retrieved; videos without captions return everything except
773
- # the transcript. A channel URL (`/channel/UC…`, `/@handle`, `/c/…`, `/user/…`)
774
- # returns its name, handle, subscriber count, video count, and full description.
775
- # When `includeImages=true`, video responses also include the thumbnail and
776
- # channel responses include the avatar. Costs the same as any other scrape.
777
- #
778
- # ### Billing & errors
779
- #
780
- # | HTTP status | Billed? | Meaning |
781
- # | ----------- | ----------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
782
- # | 200 | Yes — 1 credit, or 2 credits with actions | Successful scrape, including a zero-length result when includeSelectors matched nothing. A partial result (`finalDOMState: "still-loading"`, only with timeoutOpts.behavior=return-partial) is billed at the base 1 credit with no OCR or actions surcharge |
783
- # | 400 | No | Invalid input, skipped PDF, or the page could not be scraped. error_code WEBSITE_BLOCKED specifically means the site answered with an anti-bot challenge, CAPTCHA wall, or login shell instead of the page (even when the site returned HTTP 200) — retrying later or from another country sometimes succeeds |
784
- # | 401 / 403 | No | Invalid/disabled key, insufficient permissions, or credits exhausted; inspect error_code |
785
- # | 404 | No | Target page returned or fingerprinted as not found |
786
- # | 408 | No | Request timed out. With timeoutOpts.behavior=return-partial this only happens when nothing usable had rendered by the deadline |
787
- # | 413 | No | Target content exceeds the maximum supported size (20 MB) |
788
- # | 415 | No | Unsupported content type |
789
- # | 429 | No | Per-minute rate limit exceeded; honor Retry-After |
790
- # | 500 | No | Internal error |
791
- sig do
792
- params(
793
- url: String,
794
- actions:
795
- T.nilable(
796
- T::Array[
797
- T.any(
798
- ContextDev::WebWebScrapeMdParams::Action::Wait::OrHash,
799
- ContextDev::WebWebScrapeMdParams::Action::Perform::OrHash,
800
- ContextDev::WebWebScrapeMdParams::Action::Scroll::OrHash
801
- )
802
- ]
803
- ),
804
- country: ContextDev::WebWebScrapeMdParams::Country::OrSymbol,
805
- exclude_selectors: T.nilable(T::Array[String]),
806
- headers: T::Hash[Symbol, String],
807
- include_frames: T::Boolean,
808
- include_html: T::Boolean,
809
- include_images: T::Boolean,
810
- include_links: T::Boolean,
811
- include_selectors: T.nilable(T::Array[String]),
812
- max_age_ms: T.nilable(Integer),
813
- pdf: ContextDev::WebWebScrapeMdParams::Pdf::OrHash,
814
- settle_animations: T::Boolean,
815
- shorten_base64_images: T::Boolean,
816
- tags: T::Array[String],
817
- timeout_opts: ContextDev::WebWebScrapeMdParams::TimeoutOpts::OrHash,
818
- use_main_content_only: T::Boolean,
819
- wait_for_ms: T.nilable(Integer),
820
- zdr: ContextDev::WebWebScrapeMdParams::Zdr::OrSymbol,
821
- request_options: ContextDev::RequestOptions::OrHash
822
- ).returns(ContextDev::Models::WebWebScrapeMdResponse)
823
- end
824
- def web_scrape_md(
825
- # Full URL to scrape into LLM usable Markdown (must include http:// or https://
826
- # protocol)
827
- url:,
828
- # Optional browser actions executed in array order after the page loads and before
829
- # content is captured. Requires a paid plan. Send a JSON array in the query
830
- # parameter. Maximum: 5 actions.
831
- actions: nil,
832
- # Fetch the target page through a residential proxy in this country (ISO 3166-1
833
- # alpha-2).
834
- country: nil,
835
- # CSS selectors to remove before conversion to Markdown. Applied after
836
- # includeSelectors. Exclusion takes precedence: an element matching both is
837
- # removed. Examples: "nav", "footer", ".ad-banner", "[aria-hidden=true]".
838
- exclude_selectors: nil,
839
- # Optional outbound HTTP headers forwarded only to the target URL, sent as
840
- # deep-object query params such as headers[X-Custom]=value. When provided, caching
841
- # is bypassed: the result is neither read from nor written to cache.
842
- headers: nil,
843
- # When true, the contents of iframes are rendered to Markdown.
844
- include_frames: nil,
845
- # When true, the response also includes an `html` field with the page HTML the
846
- # Markdown was converted from — the same body the Scrape HTML endpoint returns for
847
- # the equivalent request.
848
- include_html: nil,
849
- # Include image references in Markdown output
850
- include_images: nil,
851
- # Preserve hyperlinks in Markdown output
852
- include_links: nil,
853
- # CSS selectors. When provided, only matching HTML subtrees (and their
854
- # descendants) are kept before conversion to Markdown. When omitted, the entire
855
- # document is kept. Examples: "article.main", "#content", "[role=main]".
856
- include_selectors: nil,
857
- # Return a cached result if a prior scrape for the same parameters exists and is
858
- # younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
859
- # omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
860
- max_age_ms: nil,
861
- # PDF parsing controls. Use start/end to limit text extraction and embedded-image
862
- # detection/OCR to an inclusive 1-based page range.
863
- pdf: nil,
864
- # When true, waits briefly for CSS and transition animations to settle before
865
- # converting to Markdown. Defaults to false. This adds a bit of latency in
866
- # exchange for more stable output on animated pages.
867
- settle_animations: nil,
868
- # Shorten base64-encoded image data in the Markdown output
869
- shorten_base64_images: nil,
870
- # Comma-separated tags for tracking request usage. Up to 20 tags, each 1-50
871
- # characters.
872
- tags: nil,
873
- # Optional request deadline and behavior on timeout. For GET requests, use
874
- # timeoutOpts[milliseconds]=30000&timeoutOpts[behavior]=fail or a JSON-encoded
875
- # timeoutOpts object.
876
- timeout_opts: nil,
877
- # Extract only the main content of the page, excluding headers, footers, sidebars,
878
- # and navigation
879
- use_main_content_only: nil,
880
- # Optional browser wait time in milliseconds after initial page load before
881
- # converting the page to Markdown. Min: 0. Max: 30000 (30 seconds). When combined
882
- # with timeoutOpts, timeoutOpts.milliseconds must be at least waitForMs + 10000
883
- # ms; a shorter deadline is rejected with 400 TIMEOUT_TOO_SHORT_FOR_WAIT.
884
- wait_for_ms: nil,
885
- # Set to enabled to bypass shared caches and omit request and response content
886
- # from retained usage logs. Asset uploads are skipped, so hosted image URLs are
887
- # omitted. Requires zero data retention to be enabled for your organization
888
- # (contact support@context.dev), otherwise the request fails with ZDR_NOT_ENABLED.
889
- # Successful ZDR responses include X-Context-ZDR: true.
890
- zdr: nil,
891
- request_options: {}
892
- )
893
- end
894
-
895
- # Capture the given HTTP or HTTPS URL with configurable viewport, full-page
896
- # capture, wait time, popup handling, theme, scroll offset, cache age, country,
897
- # and request timeout. Defaults to a 1920x1080 viewport, a 3-second wait, and a
898
- # cache age of 1 day. With timeoutOpts.behavior=return-partial, a screenshot of
899
- # the page rendered so far may be returned; inspect finalDOMState to identify an
900
- # incomplete render. Successful requests cost 1 credit; errors are not billed.
901
- sig do
902
- params(
903
- url: String,
904
- clear_popups: T::Boolean,
905
- color_scheme:
906
- ContextDev::WebWebScrapeScreenshotParams::ColorScheme::OrSymbol,
907
- country: ContextDev::WebWebScrapeScreenshotParams::Country::OrSymbol,
908
- full_screenshot:
909
- ContextDev::WebWebScrapeScreenshotParams::FullScreenshot::OrSymbol,
910
- handle_cookie_popup: T::Boolean,
911
- max_age_ms: T.nilable(Integer),
912
- scroll_offset: T.nilable(Integer),
913
- tags: T::Array[String],
914
- timeout_opts:
915
- ContextDev::WebWebScrapeScreenshotParams::TimeoutOpts::OrHash,
916
- viewport: ContextDev::WebWebScrapeScreenshotParams::Viewport::OrHash,
917
- wait_for_ms: T.nilable(Integer),
918
- zdr: ContextDev::WebWebScrapeScreenshotParams::Zdr::OrSymbol,
919
- request_options: ContextDev::RequestOptions::OrHash
920
- ).returns(ContextDev::Models::WebWebScrapeScreenshotResponse)
921
- end
922
- def web_scrape_screenshot(
923
- url:,
924
- # Optional parameter for comprehensive popup cleanup. If 'true', the browser
925
- # dismisses detected cookie/consent UI and clears other detected obstructive
926
- # popups and overlays before capture. If 'false' or not provided, this parameter
927
- # requests no cleanup; handleCookiePopup can still request cookie/consent handling
928
- # independently.
929
- clear_popups: nil,
930
- # Optional parameter to choose the site's visual theme in the screenshot. Use
931
- # 'light' or 'dark' when the site offers both appearances.
932
- color_scheme: nil,
933
- # Fetch the target page through a residential proxy in this country (ISO 3166-1
934
- # alpha-2).
935
- country: nil,
936
- # Optional parameter to determine screenshot type. If 'true', takes a full page
937
- # screenshot capturing all content. If 'false' or not provided, takes a viewport
938
- # screenshot (standard browser view).
939
- full_screenshot: nil,
940
- # Optional parameter to control cookie/consent popup handling. If 'true', we
941
- # dismiss cookie banner before capture. If 'false' or not provided, captures the
942
- # page without that step.
943
- handle_cookie_popup: nil,
944
- # Return a cached screenshot if a prior screenshot for the same parameters exists
945
- # and is younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
946
- # omitted. Max is 30 days (2592000000 ms). Set to 0 to always capture fresh.
947
- max_age_ms: nil,
948
- # Optional vertical scroll offset in pixels for capturing a long page in
949
- # viewport-sized chunks. When provided, the full page is captured once and the
950
- # returned image is the viewport-sized slice that begins at this Y offset (e.g.
951
- # request scrollOffset=0, then 1080, then 2160 to walk a 1920x1080 landing page
952
- # top to bottom). The final slice may be shorter than the viewport height. Takes
953
- # precedence over fullScreenshot. Max: 100000.
954
- scroll_offset: nil,
955
- # Comma-separated tags for tracking request usage. Up to 20 tags, each 1-50
956
- # characters.
957
- tags: nil,
958
- # Optional request deadline and behavior on timeout. For GET requests, use
959
- # timeoutOpts[milliseconds]=30000&timeoutOpts[behavior]=fail or a JSON-encoded
960
- # timeoutOpts object.
961
- timeout_opts: nil,
962
- # Optional browser viewport dimensions for the screenshot. Defaults to 1920x1080.
963
- viewport: nil,
964
- # Optional browser wait time in milliseconds after initial page load before taking
965
- # the screenshot. Min: 0. Max: 30000 (30 seconds). Defaults to 3000 ms when
966
- # omitted. When combined with timeoutOpts, timeoutOpts.milliseconds must be at
967
- # least waitForMs + 10000 ms; a shorter deadline is rejected with 400
968
- # TIMEOUT_TOO_SHORT_FOR_WAIT.
969
- wait_for_ms: nil,
970
- # Set to enabled to bypass shared caches and omit request and response content
971
- # from retained usage logs. Asset uploads are skipped, so hosted image URLs are
972
- # omitted. Requires zero data retention to be enabled for your organization
973
- # (contact support@context.dev), otherwise the request fails with ZDR_NOT_ENABLED.
974
- # Successful ZDR responses include X-Context-ZDR: true.
975
- zdr: nil,
976
- request_options: {}
977
- )
978
- end
979
-
980
- # Crawl an entire website's sitemap and return all discovered page URLs. Set
981
- # `includeSubdomains=true` to also discover public pages and sitemaps on child
982
- # hosts such as `docs.example.com` or `brand.example.com`. Pass `search` to have
983
- # the discovered URLs filtered down to the pages about a phrase (for example
984
- # `pricing and plans` or `api authentication docs`), most relevant first — a
985
- # searched crawl scans the whole sitemap and costs 2 credits instead of 1.
986
- sig do
987
- params(
988
- domain: String,
989
- headers: T::Hash[Symbol, String],
990
- include_subdomains: T::Boolean,
991
- max_links: Integer,
992
- search: String,
993
- sitemap_url: String,
994
- tags: T::Array[String],
995
- timeout_opts:
996
- ContextDev::WebWebScrapeSitemapParams::TimeoutOpts::OrHash,
997
- url_regex: String,
998
- zdr: ContextDev::WebWebScrapeSitemapParams::Zdr::OrSymbol,
999
- request_options: ContextDev::RequestOptions::OrHash
1000
- ).returns(ContextDev::Models::WebWebScrapeSitemapResponse)
1001
- end
1002
- def web_scrape_sitemap(
1003
- # Domain to build a sitemap for
1004
- domain:,
1005
- # Optional outbound HTTP headers forwarded only to the target URL, sent as
1006
- # deep-object query params such as headers[X-Custom]=value. When provided, caching
1007
- # is bypassed: the result is neither read from nor written to cache.
1008
- headers: nil,
1009
- # When true, discover and include public pages and sitemaps on subdomains of the
1010
- # requested domain. Defaults to false.
1011
- include_subdomains: nil,
1012
- # Maximum number of links to return from the sitemap crawl. Defaults to 10,000.
1013
- # Minimum is 1, maximum is 100,000.
1014
- max_links: nil,
1015
- # Optional search phrase. When provided, the crawled sitemap is filtered to the
1016
- # pages whose URLs are about that phrase, most relevant first, and the request
1017
- # costs 2 credits instead of 1.
1018
- search: nil,
1019
- # Optional explicit sitemap URL. When provided, exactly this sitemap is crawled
1020
- # instead of discovering the domain's sitemaps.
1021
- sitemap_url: nil,
1022
- # Comma-separated tags for tracking request usage. Up to 20 tags, each 1-50
1023
- # characters.
1024
- tags: nil,
1025
- # Optional request deadline and behavior on timeout. For GET requests, use
1026
- # timeoutOpts[milliseconds]=30000&timeoutOpts[behavior]=fail or a JSON-encoded
1027
- # timeoutOpts object.
1028
- timeout_opts: nil,
1029
- # Optional RE2-compatible regex pattern. Only URLs matching this pattern are
1030
- # returned and counted against maxLinks.
1031
- url_regex: nil,
1032
- # Set to enabled to bypass shared caches and omit request and response content
1033
- # from retained usage logs. Asset uploads are skipped, so hosted image URLs are
1034
- # omitted. Requires zero data retention to be enabled for your organization
1035
- # (contact support@context.dev), otherwise the request fails with ZDR_NOT_ENABLED.
1036
- # Successful ZDR responses include X-Context-ZDR: true.
1037
- zdr: nil,
1038
- request_options: {}
1039
- )
1040
- end
1041
-
1042
540
  # @api private
1043
541
  sig { params(client: ContextDev::Client).returns(T.attached_class) }
1044
542
  def self.new(client:)