context.dev 2.18.0 → 2.20.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (116) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +220 -0
  3. data/README.md +1 -1
  4. data/lib/context_dev/client.rb +0 -4
  5. data/lib/context_dev/models/monitor_retrieve_run_params.rb +26 -0
  6. data/lib/context_dev/models/monitor_retrieve_run_response.rb +232 -0
  7. data/lib/context_dev/models/monitor_rotate_webhook_secret_params.rb +20 -0
  8. data/lib/context_dev/models/monitor_rotate_webhook_secret_response.rb +765 -0
  9. data/lib/context_dev/models/{web_web_scrape_sitemap_params.rb → web_map_urls_params.rb} +14 -14
  10. data/lib/context_dev/models/web_map_urls_response.rb +123 -0
  11. data/lib/context_dev/models/web_scrape_params.rb +1035 -0
  12. data/lib/context_dev/models/web_scrape_response.rb +974 -0
  13. data/lib/context_dev/models/web_screenshot_params.rb +15 -1
  14. data/lib/context_dev/models/web_screenshot_response.rb +3 -3
  15. data/lib/context_dev/models.rb +8 -22
  16. data/lib/context_dev/resources/brand.rb +0 -36
  17. data/lib/context_dev/resources/monitors.rb +47 -0
  18. data/lib/context_dev/resources/web.rb +118 -486
  19. data/lib/context_dev/version.rb +1 -1
  20. data/lib/context_dev.rb +8 -23
  21. data/rbi/context_dev/client.rbi +0 -3
  22. data/rbi/context_dev/models/monitor_retrieve_run_params.rbi +46 -0
  23. data/rbi/context_dev/models/monitor_retrieve_run_response.rbi +452 -0
  24. data/rbi/context_dev/models/monitor_rotate_webhook_secret_params.rbi +38 -0
  25. data/rbi/context_dev/models/monitor_rotate_webhook_secret_response.rbi +1367 -0
  26. data/rbi/context_dev/models/{web_web_scrape_sitemap_params.rbi → web_map_urls_params.rbi} +23 -45
  27. data/rbi/context_dev/models/web_map_urls_response.rbi +226 -0
  28. data/rbi/context_dev/models/web_scrape_params.rbi +2221 -0
  29. data/rbi/context_dev/models/web_scrape_response.rbi +1874 -0
  30. data/rbi/context_dev/models/web_screenshot_params.rbi +23 -0
  31. data/rbi/context_dev/models/web_screenshot_response.rbi +4 -4
  32. data/rbi/context_dev/models.rbi +9 -24
  33. data/rbi/context_dev/resources/brand.rbi +0 -35
  34. data/rbi/context_dev/resources/monitors.rbi +24 -0
  35. data/rbi/context_dev/resources/web.rbi +152 -654
  36. data/sig/context_dev/client.rbs +0 -2
  37. data/sig/context_dev/models/monitor_retrieve_run_params.rbs +28 -0
  38. data/sig/context_dev/models/monitor_retrieve_run_response.rbs +182 -0
  39. data/sig/context_dev/models/monitor_rotate_webhook_secret_params.rbs +23 -0
  40. data/sig/context_dev/models/monitor_rotate_webhook_secret_response.rbs +545 -0
  41. data/sig/context_dev/models/{web_web_scrape_sitemap_params.rbs → web_map_urls_params.rbs} +22 -22
  42. data/sig/context_dev/models/web_map_urls_response.rbs +125 -0
  43. data/sig/context_dev/models/web_scrape_params.rbs +925 -0
  44. data/sig/context_dev/models/web_scrape_response.rbs +782 -0
  45. data/sig/context_dev/models/web_screenshot_params.rbs +7 -0
  46. data/sig/context_dev/models.rbs +8 -22
  47. data/sig/context_dev/resources/brand.rbs +0 -9
  48. data/sig/context_dev/resources/monitors.rbs +11 -0
  49. data/sig/context_dev/resources/web.rbs +33 -128
  50. metadata +26 -71
  51. data/lib/context_dev/models/ai_extract_product_params.rb +0 -125
  52. data/lib/context_dev/models/ai_extract_product_response.rb +0 -402
  53. data/lib/context_dev/models/ai_extract_products_params.rb +0 -233
  54. data/lib/context_dev/models/ai_extract_products_response.rb +0 -370
  55. data/lib/context_dev/models/brand_retrieve_simplified_params.rb +0 -120
  56. data/lib/context_dev/models/brand_retrieve_simplified_response.rb +0 -441
  57. data/lib/context_dev/models/web_extract_fonts_params.rb +0 -114
  58. data/lib/context_dev/models/web_extract_fonts_response.rb +0 -291
  59. data/lib/context_dev/models/web_extract_params.rb +0 -419
  60. data/lib/context_dev/models/web_extract_response.rb +0 -287
  61. data/lib/context_dev/models/web_web_scrape_bytes_params.rb +0 -346
  62. data/lib/context_dev/models/web_web_scrape_bytes_response.rb +0 -135
  63. data/lib/context_dev/models/web_web_scrape_html_params.rb +0 -812
  64. data/lib/context_dev/models/web_web_scrape_html_response.rb +0 -628
  65. data/lib/context_dev/models/web_web_scrape_images_params.rb +0 -373
  66. data/lib/context_dev/models/web_web_scrape_images_response.rb +0 -409
  67. data/lib/context_dev/models/web_web_scrape_md_params.rb +0 -670
  68. data/lib/context_dev/models/web_web_scrape_md_response.rb +0 -567
  69. data/lib/context_dev/models/web_web_scrape_screenshot_params.rb +0 -471
  70. data/lib/context_dev/models/web_web_scrape_screenshot_response.rb +0 -169
  71. data/lib/context_dev/models/web_web_scrape_sitemap_response.rb +0 -144
  72. data/lib/context_dev/resources/ai.rb +0 -71
  73. data/rbi/context_dev/models/ai_extract_product_params.rbi +0 -253
  74. data/rbi/context_dev/models/ai_extract_product_response.rbi +0 -791
  75. data/rbi/context_dev/models/ai_extract_products_params.rbi +0 -490
  76. data/rbi/context_dev/models/ai_extract_products_response.rbi +0 -718
  77. data/rbi/context_dev/models/brand_retrieve_simplified_params.rbi +0 -256
  78. data/rbi/context_dev/models/brand_retrieve_simplified_response.rbi +0 -1040
  79. data/rbi/context_dev/models/web_extract_fonts_params.rbi +0 -223
  80. data/rbi/context_dev/models/web_extract_fonts_response.rbi +0 -563
  81. data/rbi/context_dev/models/web_extract_params.rbi +0 -781
  82. data/rbi/context_dev/models/web_extract_response.rbi +0 -512
  83. data/rbi/context_dev/models/web_web_scrape_bytes_params.rbi +0 -702
  84. data/rbi/context_dev/models/web_web_scrape_bytes_response.rbi +0 -238
  85. data/rbi/context_dev/models/web_web_scrape_html_params.rbi +0 -1690
  86. data/rbi/context_dev/models/web_web_scrape_html_response.rbi +0 -1287
  87. data/rbi/context_dev/models/web_web_scrape_images_params.rbi +0 -728
  88. data/rbi/context_dev/models/web_web_scrape_images_response.rbi +0 -910
  89. data/rbi/context_dev/models/web_web_scrape_md_params.rbi +0 -1052
  90. data/rbi/context_dev/models/web_web_scrape_md_response.rbi +0 -1088
  91. data/rbi/context_dev/models/web_web_scrape_screenshot_params.rbi +0 -1580
  92. data/rbi/context_dev/models/web_web_scrape_screenshot_response.rbi +0 -332
  93. data/rbi/context_dev/models/web_web_scrape_sitemap_response.rbi +0 -258
  94. data/rbi/context_dev/resources/ai.rbi +0 -63
  95. data/sig/context_dev/models/ai_extract_product_params.rbs +0 -106
  96. data/sig/context_dev/models/ai_extract_product_response.rbs +0 -314
  97. data/sig/context_dev/models/ai_extract_products_params.rbs +0 -202
  98. data/sig/context_dev/models/ai_extract_products_response.rbs +0 -293
  99. data/sig/context_dev/models/brand_retrieve_simplified_params.rbs +0 -104
  100. data/sig/context_dev/models/brand_retrieve_simplified_response.rbs +0 -413
  101. data/sig/context_dev/models/web_extract_fonts_params.rbs +0 -93
  102. data/sig/context_dev/models/web_extract_fonts_response.rbs +0 -230
  103. data/sig/context_dev/models/web_extract_params.rbs +0 -324
  104. data/sig/context_dev/models/web_extract_response.rbs +0 -239
  105. data/sig/context_dev/models/web_web_scrape_bytes_params.rbs +0 -531
  106. data/sig/context_dev/models/web_web_scrape_bytes_response.rbs +0 -108
  107. data/sig/context_dev/models/web_web_scrape_html_params.rbs +0 -878
  108. data/sig/context_dev/models/web_web_scrape_html_response.rbs +0 -525
  109. data/sig/context_dev/models/web_web_scrape_images_params.rbs +0 -281
  110. data/sig/context_dev/models/web_web_scrape_images_response.rbs +0 -371
  111. data/sig/context_dev/models/web_web_scrape_md_params.rbs +0 -758
  112. data/sig/context_dev/models/web_web_scrape_md_response.rbs +0 -465
  113. data/sig/context_dev/models/web_web_scrape_screenshot_params.rbs +0 -619
  114. data/sig/context_dev/models/web_web_scrape_screenshot_response.rbs +0 -127
  115. data/sig/context_dev/models/web_web_scrape_sitemap_response.rbs +0 -117
  116. data/sig/context_dev/resources/ai.rbs +0 -21
@@ -42,64 +42,6 @@ module ContextDev
42
42
  )
43
43
  end
44
44
 
45
- # Some parameter documentations has been truncated, see
46
- # {ContextDev::Models::WebExtractParams} for more details.
47
- #
48
- # Crawl a website, use the provided JSON Schema and instructions to prioritize
49
- # relevant internal links, and extract structured data from the selected pages.
50
- #
51
- # @overload extract(schema:, url:, actions: nil, fact_check: nil, follow_subdomains: nil, include_frames: nil, instructions: nil, max_age_ms: nil, max_depth: nil, max_pages: nil, pdf: nil, settle_animations: nil, stop_after_ms: nil, tags: nil, timeout_opts: nil, wait_for_ms: nil, zdr: nil, request_options: {})
52
- #
53
- # @param schema [Hash{Symbol=>Object}] JSON Schema for the returned data object. Image fields such as `image_urls` or `
54
- #
55
- # @param url [String] The starting website URL to crawl and extract from. Must include http:// or http
56
- #
57
- # @param actions [Array<ContextDev::Models::WebExtractParams::Action::Wait, ContextDev::Models::WebExtractParams::Action::Perform, ContextDev::Models::WebExtractParams::Action::Scroll>] Optional browser actions executed in order on the requested page after it loads,
58
- #
59
- # @param fact_check [Boolean] When true, every returned value must be grounded in facts stated on the page; fi
60
- #
61
- # @param follow_subdomains [Boolean] When true, follow links on subdomains of the starting URL's domain.
62
- #
63
- # @param include_frames [Boolean] When true, iframe contents are included in Markdown before extraction.
64
- #
65
- # @param instructions [String] Optional extraction guidance, such as which facts to prioritize or how to interp
66
- #
67
- # @param max_age_ms [Integer] Return cached scrape results if a prior scrape for the same parameters is younge
68
- #
69
- # @param max_depth [Integer] Optional maximum link depth from the starting URL (0 = only the starting page).
70
- #
71
- # @param max_pages [Integer] Maximum number of pages to analyze for extraction. Hard cap: 50. Defaults to 5.
72
- #
73
- # @param pdf [ContextDev::Models::WebExtractParams::Pdf]
74
- #
75
- # @param settle_animations [Boolean] When true, waits briefly for CSS and transition animations to settle before extr
76
- #
77
- # @param stop_after_ms [Integer] Soft time budget for the crawl in milliseconds. Min: 10000 (10s). Max: 110000 (1
78
- #
79
- # @param tags [Array<String>] Optional tags for tracking usage. Up to 20 tags, each 1 to 50 characters.
80
- #
81
- # @param timeout_opts [ContextDev::Models::WebExtractParams::TimeoutOpts] Optional request deadline and behavior on timeout. For GET requests, use timeout
82
- #
83
- # @param wait_for_ms [Integer] Optional browser wait time in milliseconds after initial page load for each craw
84
- #
85
- # @param zdr [Symbol, ContextDev::Models::WebExtractParams::Zdr] Set to enabled to bypass shared caches and omit request and response content fro
86
- #
87
- # @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
88
- #
89
- # @return [ContextDev::Models::WebExtractResponse]
90
- #
91
- # @see ContextDev::Models::WebExtractParams
92
- def extract(params)
93
- parsed, options = ContextDev::WebExtractParams.dump_request(params)
94
- @client.request(
95
- method: :post,
96
- path: "web/extract",
97
- body: parsed,
98
- model: ContextDev::Models::WebExtractResponse,
99
- options: options
100
- )
101
- end
102
-
103
45
  # Some parameter documentations has been truncated, see
104
46
  # {ContextDev::Models::WebExtractCompetitorsParams} for more details.
105
47
  #
@@ -136,84 +78,166 @@ module ContextDev
136
78
  end
137
79
 
138
80
  # Some parameter documentations has been truncated, see
139
- # {ContextDev::Models::WebExtractFontsParams} for more details.
81
+ # {ContextDev::Models::WebExtractStyleguideParams} for more details.
82
+ #
83
+ # Extract a comprehensive design system from a website including colors,
84
+ # typography, spacing, shadows, and UI components.
140
85
  #
141
- # Scrape font information from a website including font families, usage
142
- # statistics, fallbacks, and element/word counts.
86
+ # @overload extract_styleguide(color_scheme: nil, direct_url: nil, domain: nil, max_age_ms: nil, tags: nil, timeout_opts: nil, zdr: nil, request_options: {})
143
87
  #
144
- # @overload extract_fonts(direct_url: nil, domain: nil, max_age_ms: nil, tags: nil, timeout_opts: nil, request_options: {})
88
+ # @param color_scheme [Symbol, ContextDev::Models::WebExtractStyleguideParams::ColorScheme] Optional browser color scheme to emulate for websites that respond to prefers-co
145
89
  #
146
- # @param direct_url [String] A specific URL to fetch fonts from directly, bypassing domain resolution (e.g.,
90
+ # @param direct_url [String] A specific URL to fetch the styleguide from directly, bypassing domain resolutio
147
91
  #
148
- # @param domain [String] Domain name to extract fonts from (e.g., 'example.com', 'google.com'). The domai
92
+ # @param domain [String] Domain name to extract styleguide from (e.g., 'example.com', 'google.com'). The
149
93
  #
150
94
  # @param max_age_ms [Integer, nil] Maximum age in milliseconds for cached brand data before the API performs a hard
151
95
  #
152
96
  # @param tags [Array<String>] Comma-separated tags for tracking request usage. Up to 20 tags, each 1-50 charac
153
97
  #
154
- # @param timeout_opts [ContextDev::Models::WebExtractFontsParams::TimeoutOpts] Optional request deadline and behavior on timeout. For GET requests, use timeout
98
+ # @param timeout_opts [ContextDev::Models::WebExtractStyleguideParams::TimeoutOpts] Optional request deadline and behavior on timeout. For GET requests, use timeout
99
+ #
100
+ # @param zdr [Symbol, ContextDev::Models::WebExtractStyleguideParams::Zdr] Set to enabled to bypass shared caches and omit request and response content fro
155
101
  #
156
102
  # @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
157
103
  #
158
- # @return [ContextDev::Models::WebExtractFontsResponse]
104
+ # @return [ContextDev::Models::WebExtractStyleguideResponse]
159
105
  #
160
- # @see ContextDev::Models::WebExtractFontsParams
161
- def extract_fonts(params = {})
162
- parsed, options = ContextDev::WebExtractFontsParams.dump_request(params)
106
+ # @see ContextDev::Models::WebExtractStyleguideParams
107
+ def extract_styleguide(params = {})
108
+ parsed, options = ContextDev::WebExtractStyleguideParams.dump_request(params)
163
109
  query = ContextDev::Internal::Util.encode_query_params(parsed)
164
110
  @client.request(
165
111
  method: :get,
166
- path: "web/fonts",
112
+ path: "web/styleguide",
167
113
  query: query.transform_keys(
114
+ color_scheme: "colorScheme",
168
115
  direct_url: "directUrl",
169
116
  max_age_ms: "maxAgeMs",
170
117
  timeout_opts: "timeoutOpts"
171
118
  ),
172
- model: ContextDev::Models::WebExtractFontsResponse,
119
+ model: ContextDev::Models::WebExtractStyleguideResponse,
173
120
  options: options
174
121
  )
175
122
  end
176
123
 
177
124
  # Some parameter documentations has been truncated, see
178
- # {ContextDev::Models::WebExtractStyleguideParams} for more details.
125
+ # {ContextDev::Models::WebMapURLsParams} for more details.
179
126
  #
180
- # Extract a comprehensive design system from a website including colors,
181
- # typography, spacing, shadows, and UI components.
127
+ # Discovers URLs using the same sitemap crawl, filters, and limits as
128
+ # /web/scrape/sitemap. Each URL includes its available title, description,
129
+ # keywords, and language. URLs without stored enrichment are returned immediately
130
+ # with only the URL and queued for background HTML scraping, so later requests can
131
+ # include their metadata. Responses are never cached as a whole; every request
132
+ # reads the current per-URL enrichment. Zero data retention and credential-bearing
133
+ # discovery requests return URLs without reading or storing shared enrichment or
134
+ # queuing background scrapes. Costs 1 credit, or 2 credits with search.
182
135
  #
183
- # @overload extract_styleguide(color_scheme: nil, direct_url: nil, domain: nil, max_age_ms: nil, tags: nil, timeout_opts: nil, zdr: nil, request_options: {})
136
+ # @overload map_urls(domain:, headers: nil, include_subdomains: nil, max_links: nil, search: nil, sitemap_url: nil, tags: nil, timeout_opts: nil, url_regex: nil, zdr: nil, request_options: {})
184
137
  #
185
- # @param color_scheme [Symbol, ContextDev::Models::WebExtractStyleguideParams::ColorScheme] Optional browser color scheme to emulate for websites that respond to prefers-co
138
+ # @param domain [String] Domain to build a sitemap for
186
139
  #
187
- # @param direct_url [String] A specific URL to fetch the styleguide from directly, bypassing domain resolutio
140
+ # @param headers [Hash{Symbol=>String}] Optional outbound HTTP headers forwarded only to the target URL, sent as deep-ob
188
141
  #
189
- # @param domain [String] Domain name to extract styleguide from (e.g., 'example.com', 'google.com'). The
142
+ # @param include_subdomains [Boolean] When true, discover and include public pages and sitemaps on subdomains of the r
190
143
  #
191
- # @param max_age_ms [Integer, nil] Maximum age in milliseconds for cached brand data before the API performs a hard
144
+ # @param max_links [Integer] Maximum number of links to return from the sitemap crawl. Defaults to 10,000. Mi
145
+ #
146
+ # @param search [String] Optional search phrase. When provided, the crawled sitemap is filtered to the pa
147
+ #
148
+ # @param sitemap_url [String] Optional explicit sitemap URL. When provided, exactly this sitemap is crawled in
192
149
  #
193
150
  # @param tags [Array<String>] Comma-separated tags for tracking request usage. Up to 20 tags, each 1-50 charac
194
151
  #
195
- # @param timeout_opts [ContextDev::Models::WebExtractStyleguideParams::TimeoutOpts] Optional request deadline and behavior on timeout. For GET requests, use timeout
152
+ # @param timeout_opts [ContextDev::Models::WebMapURLsParams::TimeoutOpts] Optional request deadline and behavior on timeout. For GET requests, use timeout
196
153
  #
197
- # @param zdr [Symbol, ContextDev::Models::WebExtractStyleguideParams::Zdr] Set to enabled to bypass shared caches and omit request and response content fro
154
+ # @param url_regex [String] Optional RE2-compatible regex pattern. Only URLs matching this pattern are retur
155
+ #
156
+ # @param zdr [Symbol, ContextDev::Models::WebMapURLsParams::Zdr] Set to enabled to bypass shared caches and omit request and response content fro
198
157
  #
199
158
  # @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
200
159
  #
201
- # @return [ContextDev::Models::WebExtractStyleguideResponse]
160
+ # @return [ContextDev::Models::WebMapURLsResponse]
202
161
  #
203
- # @see ContextDev::Models::WebExtractStyleguideParams
204
- def extract_styleguide(params = {})
205
- parsed, options = ContextDev::WebExtractStyleguideParams.dump_request(params)
162
+ # @see ContextDev::Models::WebMapURLsParams
163
+ def map_urls(params)
164
+ parsed, options = ContextDev::WebMapURLsParams.dump_request(params)
206
165
  query = ContextDev::Internal::Util.encode_query_params(parsed)
207
166
  @client.request(
208
167
  method: :get,
209
- path: "web/styleguide",
168
+ path: "web/urls",
210
169
  query: query.transform_keys(
211
- color_scheme: "colorScheme",
212
- direct_url: "directUrl",
213
- max_age_ms: "maxAgeMs",
214
- timeout_opts: "timeoutOpts"
170
+ include_subdomains: "includeSubdomains",
171
+ max_links: "maxLinks",
172
+ sitemap_url: "sitemapUrl",
173
+ timeout_opts: "timeoutOpts",
174
+ url_regex: "urlRegex"
215
175
  ),
216
- model: ContextDev::Models::WebExtractStyleguideResponse,
176
+ model: ContextDev::Models::WebMapURLsResponse,
177
+ options: options
178
+ )
179
+ end
180
+
181
+ # Some parameter documentations has been truncated, see
182
+ # {ContextDev::Models::WebScrapeParams} for more details.
183
+ #
184
+ # Reuse cached outputs independently and capture missing formats in one page
185
+ # visit. Each cache key includes only the settings that affect that output. HTML
186
+ # is shared with Markdown, parsed fields, product data, highlights, and JSON
187
+ # extraction. Cached outputs can come from different visits within maxAgeMs; use 0
188
+ # for a fresh capture. HTML-only requests use the existing fast acquisition path.
189
+ # Highlights return the plain-text passages most relevant to
190
+ # highlightsParams.query. One credit per request, including cache hits and missing
191
+ # pages, or two with browser actions; highlights add 3 credits when passages are
192
+ # returned; JSON extraction adds four credits and runs an LLM over the page
193
+ # Markdown on every request that has text to extract; PDF OCR adds one credit per
194
+ # recovered page on fresh extraction; the product output adds one credit, plus six
195
+ # more when the specialized model is used. Original response bytes and screenshots
196
+ # are limited to 20 MiB each, screenshots to 40 megapixels, and the combined
197
+ # browser capture to 60 MiB.
198
+ #
199
+ # @overload scrape(formats:, url:, highlights_params: nil, image_params: nil, json_params: nil, markdown_params: nil, max_age_ms: nil, parse_params: nil, product_params: nil, screenshot_params: nil, shared_params: nil, tags: nil, timeout_opts: nil, zdr: nil, request_options: {})
200
+ #
201
+ # @param formats [ContextDev::Models::WebScrapeParams::Formats] Outputs to return. Enable at least one; omitted formats are false.
202
+ #
203
+ # @param url [String] The URL to scrape.
204
+ #
205
+ # @param highlights_params [ContextDev::Models::WebScrapeParams::HighlightsParams] Highlight options. Requires formats.highlights: true.
206
+ #
207
+ # @param image_params [ContextDev::Models::WebScrapeParams::ImageParams] Image options. Requires formats.images: true.
208
+ #
209
+ # @param json_params [ContextDev::Models::WebScrapeParams::JsonParams] Required when formats.json is true.
210
+ #
211
+ # @param markdown_params [ContextDev::Models::WebScrapeParams::MarkdownParams] Markdown options. Requires formats.markdown: true.
212
+ #
213
+ # @param max_age_ms [Integer] Maximum age of each cached output. Defaults to 1 day; 0 fetches fresh and update
214
+ #
215
+ # @param parse_params [ContextDev::Models::WebScrapeParams::ParseParams] Required when formats.parse is true.
216
+ #
217
+ # @param product_params [ContextDev::Models::WebScrapeParams::ProductParams] Product options. Requires formats.product: true.
218
+ #
219
+ # @param screenshot_params [ContextDev::Models::WebScrapeParams::ScreenshotParams] Screenshot options. Requires formats.screenshot: true.
220
+ #
221
+ # @param shared_params [ContextDev::Models::WebScrapeParams::SharedParams] Shared browser and content settings. Content filters leave screenshots and origi
222
+ #
223
+ # @param tags [Array<String>] Labels for tracking request usage. Not retained when zdr is enabled.
224
+ #
225
+ # @param timeout_opts [ContextDev::Models::WebScrapeParams::TimeoutOpts] Total deadline, including navigation, actions, waiting, and all outputs. Default
226
+ #
227
+ # @param zdr [Symbol, ContextDev::Models::WebScrapeParams::Zdr] Zero data retention. Bypasses caches and uploads; excludes request/response cont
228
+ #
229
+ # @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
230
+ #
231
+ # @return [ContextDev::Models::WebScrapeResponse]
232
+ #
233
+ # @see ContextDev::Models::WebScrapeParams
234
+ def scrape(params)
235
+ parsed, options = ContextDev::WebScrapeParams.dump_request(params)
236
+ @client.request(
237
+ method: :post,
238
+ path: "web/scrape",
239
+ body: parsed,
240
+ model: ContextDev::Models::WebScrapeResponse,
217
241
  options: options
218
242
  )
219
243
  end
@@ -223,7 +247,7 @@ module ContextDev
223
247
  #
224
248
  # Capture a screenshot of a website.
225
249
  #
226
- # @overload screenshot(clear_popups: nil, color_scheme: nil, country: nil, direct_url: nil, domain: nil, full_screenshot: nil, handle_cookie_popup: nil, max_age_ms: nil, page: nil, scroll_offset: nil, tags: nil, timeout_opts: nil, viewport: nil, wait_for_ms: nil, zdr: nil, request_options: {})
250
+ # @overload screenshot(clear_popups: nil, color_scheme: nil, country: nil, direct_url: nil, domain: nil, full_screenshot: nil, handle_cookie_popup: nil, headers: nil, max_age_ms: nil, page: nil, scroll_offset: nil, tags: nil, timeout_opts: nil, viewport: nil, wait_for_ms: nil, zdr: nil, request_options: {})
227
251
  #
228
252
  # @param clear_popups [Boolean] Optional parameter for comprehensive popup cleanup. If 'true', the browser dismi
229
253
  #
@@ -239,6 +263,8 @@ module ContextDev
239
263
  #
240
264
  # @param handle_cookie_popup [Boolean] Optional parameter to control cookie/consent popup handling. If 'true', we dismi
241
265
  #
266
+ # @param headers [Hash{Symbol=>String}] Optional outbound HTTP headers, using the same JSON object or deep-object query
267
+ #
242
268
  # @param max_age_ms [Integer, nil] Return a cached screenshot if a prior screenshot for the same parameters exists
243
269
  #
244
270
  # @param page [Symbol, ContextDev::Models::WebScreenshotParams::Page] Optional parameter to specify which page type to screenshot. If provided, the sy
@@ -393,400 +419,6 @@ module ContextDev
393
419
  )
394
420
  end
395
421
 
396
- # Some parameter documentations has been truncated, see
397
- # {ContextDev::Models::WebWebScrapeBytesParams} for more details.
398
- #
399
- # Downloads a resource and returns its bytes as base64. Supports images, PDFs,
400
- # HTML pages, and any other content type without image conversion, text
401
- # extraction, or character-encoding changes. HTTP compression is decoded before
402
- # base64 encoding. HTML is the original HTTP response; JavaScript is not rendered.
403
- # Follows public redirects and retries failed downloads through ISP and
404
- # residential proxies, with a direct fallback. When country is specified, only a
405
- # residential proxy in that country is used. Supply headers such as Referer for
406
- # images that require a referring page. Downloads are not cached. Maximum decoded
407
- # resource size: 20 MiB (20971520 bytes), before base64 encoding. Successful
408
- # requests cost 1 credit; errors are not billed.
409
- #
410
- # @overload web_scrape_bytes(url:, country: nil, headers: nil, tags: nil, timeout_opts: nil, zdr: nil, request_options: {})
411
- #
412
- # @param url [String] Full HTTP(S) URL of the resource to download, such as an image, PDF, or page.
413
- #
414
- # @param country [Symbol, ContextDev::Models::WebWebScrapeBytesParams::Country] Fetch the target page through a residential proxy in this country (ISO 3166-1 al
415
- #
416
- # @param headers [Hash{Symbol=>String}] Optional outbound HTTP headers, such as Referer, Cookie, or Authorization. Send
417
- #
418
- # @param tags [Array<String>] Comma-separated tags for tracking request usage. Up to 20 tags, each 1-50 charac
419
- #
420
- # @param timeout_opts [ContextDev::Models::WebWebScrapeBytesParams::TimeoutOpts] Optional request deadline and behavior on timeout. For GET requests, use timeout
421
- #
422
- # @param zdr [Symbol, ContextDev::Models::WebWebScrapeBytesParams::Zdr] Set to enabled to bypass shared caches and omit request and response content fro
423
- #
424
- # @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
425
- #
426
- # @return [ContextDev::Models::WebWebScrapeBytesResponse]
427
- #
428
- # @see ContextDev::Models::WebWebScrapeBytesParams
429
- def web_scrape_bytes(params)
430
- parsed, options = ContextDev::WebWebScrapeBytesParams.dump_request(params)
431
- query = ContextDev::Internal::Util.encode_query_params(parsed)
432
- @client.request(
433
- method: :get,
434
- path: "web/scrape/bytes",
435
- query: query.transform_keys(timeout_opts: "timeoutOpts"),
436
- model: ContextDev::Models::WebWebScrapeBytesResponse,
437
- options: options
438
- )
439
- end
440
-
441
- # Some parameter documentations has been truncated, see
442
- # {ContextDev::Models::WebWebScrapeHTMLParams} for more details.
443
- #
444
- # Scrapes the given URL and returns the HTML content of the page. Optional
445
- # extractRules return deterministic structured data in extracted using CSS
446
- # selectors, attributes, lists, and nested rules, without an LLM or additional
447
- # credits. Rules run on the returned HTML after selector and main-content
448
- # filtering. Send extractRules as a JSON-encoded query parameter. The base request
449
- # costs 1 credit; requests with browser actions cost 2 credits. A request that
450
- # hits its timeoutOpts.milliseconds deadline fails with 408 and is not billed,
451
- # unless timeoutOpts.behavior=return-partial is set — then the page as rendered so
452
- # far is returned with `finalDOMState: "still-loading"` and billed at the base
453
- # cost of 1 credit.
454
- #
455
- # @overload web_scrape_html(url:, actions: nil, country: nil, exclude_selectors: nil, extract_rules: nil, headers: nil, include_frames: nil, include_selectors: nil, max_age_ms: nil, pdf: nil, settle_animations: nil, tags: nil, timeout_opts: nil, use_main_content_only: nil, wait_for_ms: nil, zdr: nil, request_options: {})
456
- #
457
- # @param url [String] Full URL to scrape (must include http:// or https:// protocol)
458
- #
459
- # @param actions [Array<ContextDev::Models::WebWebScrapeHTMLParams::Action::Wait, ContextDev::Models::WebWebScrapeHTMLParams::Action::Perform, ContextDev::Models::WebWebScrapeHTMLParams::Action::Scroll>, nil] Optional browser actions executed in array order after the page loads and before
460
- #
461
- # @param country [Symbol, ContextDev::Models::WebWebScrapeHTMLParams::Country] Fetch the target page through a residential proxy in this country (ISO 3166-1 al
462
- #
463
- # @param exclude_selectors [Array<String>, nil] CSS selectors to remove from the result. Applied after includeSelectors. Exclusi
464
- #
465
- # @param extract_rules [Hash{Symbol=>String, ContextDev::Models::WebWebScrapeHTMLParams::ExtractRule::UnionMember1}] Optional CSS extraction rules applied to the returned HTML after selector and ma
466
- #
467
- # @param headers [Hash{Symbol=>String}] Optional outbound HTTP headers forwarded only to the target URL, sent as deep-ob
468
- #
469
- # @param include_frames [Boolean] When true, iframes are rendered inline into the returned HTML.
470
- #
471
- # @param include_selectors [Array<String>, nil] CSS selectors. When provided, only matching subtrees (and their descendants) are
472
- #
473
- # @param max_age_ms [Integer, nil] Return a cached result if a prior scrape for the same parameters exists and is y
474
- #
475
- # @param pdf [ContextDev::Models::WebWebScrapeHTMLParams::Pdf] PDF parsing controls. Use start/end to limit text extraction and embedded-image
476
- #
477
- # @param settle_animations [Boolean] When true, waits briefly for CSS and transition animations to settle before extr
478
- #
479
- # @param tags [Array<String>] Comma-separated tags for tracking request usage. Up to 20 tags, each 1-50 charac
480
- #
481
- # @param timeout_opts [ContextDev::Models::WebWebScrapeHTMLParams::TimeoutOpts] Optional request deadline and behavior on timeout. For GET requests, use timeout
482
- #
483
- # @param use_main_content_only [Boolean] When true, return only the page's main content in the HTML response, excluding h
484
- #
485
- # @param wait_for_ms [Integer, nil] Optional browser wait time in milliseconds after initial page load. Min: 0. Max:
486
- #
487
- # @param zdr [Symbol, ContextDev::Models::WebWebScrapeHTMLParams::Zdr] Set to enabled to bypass shared caches and omit request and response content fro
488
- #
489
- # @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
490
- #
491
- # @return [ContextDev::Models::WebWebScrapeHTMLResponse]
492
- #
493
- # @see ContextDev::Models::WebWebScrapeHTMLParams
494
- def web_scrape_html(params)
495
- parsed, options = ContextDev::WebWebScrapeHTMLParams.dump_request(params)
496
- query = ContextDev::Internal::Util.encode_query_params(parsed)
497
- @client.request(
498
- method: :get,
499
- path: "web/scrape/html",
500
- query: query.transform_keys(
501
- exclude_selectors: "excludeSelectors",
502
- extract_rules: "extractRules",
503
- include_frames: "includeFrames",
504
- include_selectors: "includeSelectors",
505
- max_age_ms: "maxAgeMs",
506
- settle_animations: "settleAnimations",
507
- timeout_opts: "timeoutOpts",
508
- use_main_content_only: "useMainContentOnly",
509
- wait_for_ms: "waitForMs"
510
- ),
511
- model: ContextDev::Models::WebWebScrapeHTMLResponse,
512
- options: options
513
- )
514
- end
515
-
516
- # Some parameter documentations has been truncated, see
517
- # {ContextDev::Models::WebWebScrapeImagesParams} for more details.
518
- #
519
- # Extract image assets from a web page, including standard URLs, inline SVGs, data
520
- # URIs, responsive image sources, metadata, CSS backgrounds, video posters, and
521
- # embeds. The base request costs 1 credit, or 2 credits with browser actions. When
522
- # enrichment is enabled, the entire call costs 5 credits, including requests that
523
- # also use actions.
524
- #
525
- # @overload web_scrape_images(url:, actions: nil, dedupe: nil, enrichment: nil, headers: nil, max_age_ms: nil, tags: nil, timeout_opts: nil, wait_for_ms: nil, zdr: nil, request_options: {})
526
- #
527
- # @param url [String] Page URL to inspect. Must include http:// or https://.
528
- #
529
- # @param actions [Array<ContextDev::Models::WebWebScrapeImagesParams::Action::Wait, ContextDev::Models::WebWebScrapeImagesParams::Action::Perform, ContextDev::Models::WebWebScrapeImagesParams::Action::Scroll>, nil] Optional browser actions executed in array order after the page loads and before
530
- #
531
- # @param dedupe [Boolean] When true, visually duplicate images are removed: every image is loaded and perc
532
- #
533
- # @param enrichment [ContextDev::Models::WebWebScrapeImagesParams::Enrichment, nil] Optional per-image processing, sent as deep-object query params such as enrichme
534
- #
535
- # @param headers [Hash{Symbol=>String}] Optional outbound HTTP headers forwarded only to the target URL, sent as deep-ob
536
- #
537
- # @param max_age_ms [Integer, nil] Reuse a cached result this many milliseconds old or newer. Default: 86400000 (1
538
- #
539
- # @param tags [Array<String>] Comma-separated tags for tracking request usage. Up to 20 tags, each 1-50 charac
540
- #
541
- # @param timeout_opts [ContextDev::Models::WebWebScrapeImagesParams::TimeoutOpts] Optional request deadline and behavior on timeout. For GET requests, use timeout
542
- #
543
- # @param wait_for_ms [Integer, nil] Optional browser wait time in milliseconds after initial page load before collec
544
- #
545
- # @param zdr [Symbol, ContextDev::Models::WebWebScrapeImagesParams::Zdr] Set to enabled to bypass shared caches and omit request and response content fro
546
- #
547
- # @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
548
- #
549
- # @return [ContextDev::Models::WebWebScrapeImagesResponse]
550
- #
551
- # @see ContextDev::Models::WebWebScrapeImagesParams
552
- def web_scrape_images(params)
553
- parsed, options = ContextDev::WebWebScrapeImagesParams.dump_request(params)
554
- query = ContextDev::Internal::Util.encode_query_params(parsed)
555
- @client.request(
556
- method: :get,
557
- path: "web/scrape/images",
558
- query: query.transform_keys(
559
- max_age_ms: "maxAgeMs",
560
- timeout_opts: "timeoutOpts",
561
- wait_for_ms: "waitForMs"
562
- ),
563
- model: ContextDev::Models::WebWebScrapeImagesResponse,
564
- options: options
565
- )
566
- end
567
-
568
- # Some parameter documentations has been truncated, see
569
- # {ContextDev::Models::WebWebScrapeMdParams} for more details.
570
- #
571
- # Scrapes the given URL into LLM usable Markdown. Inspect key_metadata on JSON
572
- # responses from a recognized API key; use error_code to distinguish stable
573
- # failure categories.
574
- #
575
- # ### YouTube
576
- #
577
- # YouTube URLs return the video or channel itself rather than the surrounding
578
- # player and navigation chrome. A URL addressing a single video (`/watch`,
579
- # `youtu.be`, `/shorts`, `/embed`, `/live`) returns its title, channel, duration,
580
- # view count, keywords, full description, and the transcript when the video has
581
- # captions that can be retrieved; videos without captions return everything except
582
- # the transcript. A channel URL (`/channel/UC…`, `/@handle`, `/c/…`, `/user/…`)
583
- # returns its name, handle, subscriber count, video count, and full description.
584
- # When `includeImages=true`, video responses also include the thumbnail and
585
- # channel responses include the avatar. Costs the same as any other scrape.
586
- #
587
- # ### Billing & errors
588
- #
589
- # | HTTP status | Billed? | Meaning |
590
- # | ----------- | ----------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
591
- # | 200 | Yes — 1 credit, or 2 credits with actions | Successful scrape, including a zero-length result when includeSelectors matched nothing. A partial result (`finalDOMState: "still-loading"`, only with timeoutOpts.behavior=return-partial) is billed at the base 1 credit with no OCR or actions surcharge |
592
- # | 400 | No | Invalid input, skipped PDF, or the page could not be scraped. error_code WEBSITE_BLOCKED specifically means the site answered with an anti-bot challenge, CAPTCHA wall, or login shell instead of the page (even when the site returned HTTP 200) — retrying later or from another country sometimes succeeds |
593
- # | 401 / 403 | No | Invalid/disabled key, insufficient permissions, or credits exhausted; inspect error_code |
594
- # | 404 | No | Target page returned or fingerprinted as not found |
595
- # | 408 | No | Request timed out. With timeoutOpts.behavior=return-partial this only happens when nothing usable had rendered by the deadline |
596
- # | 413 | No | Target content exceeds the maximum supported size (20 MB) |
597
- # | 415 | No | Unsupported content type |
598
- # | 429 | No | Per-minute rate limit exceeded; honor Retry-After |
599
- # | 500 | No | Internal error |
600
- #
601
- # @overload web_scrape_md(url:, actions: nil, country: nil, exclude_selectors: nil, headers: nil, include_frames: nil, include_html: nil, include_images: nil, include_links: nil, include_selectors: nil, max_age_ms: nil, pdf: nil, settle_animations: nil, shorten_base64_images: nil, tags: nil, timeout_opts: nil, use_main_content_only: nil, wait_for_ms: nil, zdr: nil, request_options: {})
602
- #
603
- # @param url [String] Full URL to scrape into LLM usable Markdown (must include http:// or https:// pr
604
- #
605
- # @param actions [Array<ContextDev::Models::WebWebScrapeMdParams::Action::Wait, ContextDev::Models::WebWebScrapeMdParams::Action::Perform, ContextDev::Models::WebWebScrapeMdParams::Action::Scroll>, nil] Optional browser actions executed in array order after the page loads and before
606
- #
607
- # @param country [Symbol, ContextDev::Models::WebWebScrapeMdParams::Country] Fetch the target page through a residential proxy in this country (ISO 3166-1 al
608
- #
609
- # @param exclude_selectors [Array<String>, nil] CSS selectors to remove before conversion to Markdown. Applied after includeSele
610
- #
611
- # @param headers [Hash{Symbol=>String}] Optional outbound HTTP headers forwarded only to the target URL, sent as deep-ob
612
- #
613
- # @param include_frames [Boolean] When true, the contents of iframes are rendered to Markdown.
614
- #
615
- # @param include_html [Boolean] When true, the response also includes an `html` field with the page HTML the Mar
616
- #
617
- # @param include_images [Boolean] Include image references in Markdown output
618
- #
619
- # @param include_links [Boolean] Preserve hyperlinks in Markdown output
620
- #
621
- # @param include_selectors [Array<String>, nil] CSS selectors. When provided, only matching HTML subtrees (and their descendants
622
- #
623
- # @param max_age_ms [Integer, nil] Return a cached result if a prior scrape for the same parameters exists and is y
624
- #
625
- # @param pdf [ContextDev::Models::WebWebScrapeMdParams::Pdf] PDF parsing controls. Use start/end to limit text extraction and embedded-image
626
- #
627
- # @param settle_animations [Boolean] When true, waits briefly for CSS and transition animations to settle before conv
628
- #
629
- # @param shorten_base64_images [Boolean] Shorten base64-encoded image data in the Markdown output
630
- #
631
- # @param tags [Array<String>] Comma-separated tags for tracking request usage. Up to 20 tags, each 1-50 charac
632
- #
633
- # @param timeout_opts [ContextDev::Models::WebWebScrapeMdParams::TimeoutOpts] Optional request deadline and behavior on timeout. For GET requests, use timeout
634
- #
635
- # @param use_main_content_only [Boolean] Extract only the main content of the page, excluding headers, footers, sidebars,
636
- #
637
- # @param wait_for_ms [Integer, nil] Optional browser wait time in milliseconds after initial page load before conver
638
- #
639
- # @param zdr [Symbol, ContextDev::Models::WebWebScrapeMdParams::Zdr] Set to enabled to bypass shared caches and omit request and response content fro
640
- #
641
- # @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
642
- #
643
- # @return [ContextDev::Models::WebWebScrapeMdResponse]
644
- #
645
- # @see ContextDev::Models::WebWebScrapeMdParams
646
- def web_scrape_md(params)
647
- parsed, options = ContextDev::WebWebScrapeMdParams.dump_request(params)
648
- query = ContextDev::Internal::Util.encode_query_params(parsed)
649
- @client.request(
650
- method: :get,
651
- path: "web/scrape/markdown",
652
- query: query.transform_keys(
653
- exclude_selectors: "excludeSelectors",
654
- include_frames: "includeFrames",
655
- include_html: "includeHTML",
656
- include_images: "includeImages",
657
- include_links: "includeLinks",
658
- include_selectors: "includeSelectors",
659
- max_age_ms: "maxAgeMs",
660
- settle_animations: "settleAnimations",
661
- shorten_base64_images: "shortenBase64Images",
662
- timeout_opts: "timeoutOpts",
663
- use_main_content_only: "useMainContentOnly",
664
- wait_for_ms: "waitForMs"
665
- ),
666
- model: ContextDev::Models::WebWebScrapeMdResponse,
667
- options: options
668
- )
669
- end
670
-
671
- # Some parameter documentations has been truncated, see
672
- # {ContextDev::Models::WebWebScrapeScreenshotParams} for more details.
673
- #
674
- # Capture the given HTTP or HTTPS URL with configurable viewport, full-page
675
- # capture, wait time, popup handling, theme, scroll offset, cache age, country,
676
- # and request timeout. Defaults to a 1920x1080 viewport, a 3-second wait, and a
677
- # cache age of 1 day. With timeoutOpts.behavior=return-partial, a screenshot of
678
- # the page rendered so far may be returned; inspect finalDOMState to identify an
679
- # incomplete render. Successful requests cost 1 credit; errors are not billed.
680
- #
681
- # @overload web_scrape_screenshot(url:, clear_popups: nil, color_scheme: nil, country: nil, full_screenshot: nil, handle_cookie_popup: nil, max_age_ms: nil, scroll_offset: nil, tags: nil, timeout_opts: nil, viewport: nil, wait_for_ms: nil, zdr: nil, request_options: {})
682
- #
683
- # @param url [String]
684
- #
685
- # @param clear_popups [Boolean] Optional parameter for comprehensive popup cleanup. If 'true', the browser dismi
686
- #
687
- # @param color_scheme [Symbol, ContextDev::Models::WebWebScrapeScreenshotParams::ColorScheme] Optional parameter to choose the site's visual theme in the screenshot. Use 'lig
688
- #
689
- # @param country [Symbol, ContextDev::Models::WebWebScrapeScreenshotParams::Country] Fetch the target page through a residential proxy in this country (ISO 3166-1 al
690
- #
691
- # @param full_screenshot [Symbol, ContextDev::Models::WebWebScrapeScreenshotParams::FullScreenshot] Optional parameter to determine screenshot type. If 'true', takes a full page sc
692
- #
693
- # @param handle_cookie_popup [Boolean] Optional parameter to control cookie/consent popup handling. If 'true', we dismi
694
- #
695
- # @param max_age_ms [Integer, nil] Return a cached screenshot if a prior screenshot for the same parameters exists
696
- #
697
- # @param scroll_offset [Integer, nil] Optional vertical scroll offset in pixels for capturing a long page in viewport-
698
- #
699
- # @param tags [Array<String>] Comma-separated tags for tracking request usage. Up to 20 tags, each 1-50 charac
700
- #
701
- # @param timeout_opts [ContextDev::Models::WebWebScrapeScreenshotParams::TimeoutOpts] Optional request deadline and behavior on timeout. For GET requests, use timeout
702
- #
703
- # @param viewport [ContextDev::Models::WebWebScrapeScreenshotParams::Viewport] Optional browser viewport dimensions for the screenshot. Defaults to 1920x1080.
704
- #
705
- # @param wait_for_ms [Integer, nil] Optional browser wait time in milliseconds after initial page load before taking
706
- #
707
- # @param zdr [Symbol, ContextDev::Models::WebWebScrapeScreenshotParams::Zdr] Set to enabled to bypass shared caches and omit request and response content fro
708
- #
709
- # @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
710
- #
711
- # @return [ContextDev::Models::WebWebScrapeScreenshotResponse]
712
- #
713
- # @see ContextDev::Models::WebWebScrapeScreenshotParams
714
- def web_scrape_screenshot(params)
715
- parsed, options = ContextDev::WebWebScrapeScreenshotParams.dump_request(params)
716
- query = ContextDev::Internal::Util.encode_query_params(parsed)
717
- @client.request(
718
- method: :get,
719
- path: "web/scrape/screenshot",
720
- query: query.transform_keys(
721
- clear_popups: "clearPopups",
722
- color_scheme: "colorScheme",
723
- full_screenshot: "fullScreenshot",
724
- handle_cookie_popup: "handleCookiePopup",
725
- max_age_ms: "maxAgeMs",
726
- scroll_offset: "scrollOffset",
727
- timeout_opts: "timeoutOpts",
728
- wait_for_ms: "waitForMs"
729
- ),
730
- model: ContextDev::Models::WebWebScrapeScreenshotResponse,
731
- options: options
732
- )
733
- end
734
-
735
- # Some parameter documentations has been truncated, see
736
- # {ContextDev::Models::WebWebScrapeSitemapParams} for more details.
737
- #
738
- # Crawl an entire website's sitemap and return all discovered page URLs. Set
739
- # `includeSubdomains=true` to also discover public pages and sitemaps on child
740
- # hosts such as `docs.example.com` or `brand.example.com`. Pass `search` to have
741
- # the discovered URLs filtered down to the pages about a phrase (for example
742
- # `pricing and plans` or `api authentication docs`), most relevant first — a
743
- # searched crawl scans the whole sitemap and costs 2 credits instead of 1.
744
- #
745
- # @overload web_scrape_sitemap(domain:, headers: nil, include_subdomains: nil, max_links: nil, search: nil, sitemap_url: nil, tags: nil, timeout_opts: nil, url_regex: nil, zdr: nil, request_options: {})
746
- #
747
- # @param domain [String] Domain to build a sitemap for
748
- #
749
- # @param headers [Hash{Symbol=>String}] Optional outbound HTTP headers forwarded only to the target URL, sent as deep-ob
750
- #
751
- # @param include_subdomains [Boolean] When true, discover and include public pages and sitemaps on subdomains of the r
752
- #
753
- # @param max_links [Integer] Maximum number of links to return from the sitemap crawl. Defaults to 10,000. Mi
754
- #
755
- # @param search [String] Optional search phrase. When provided, the crawled sitemap is filtered to the pa
756
- #
757
- # @param sitemap_url [String] Optional explicit sitemap URL. When provided, exactly this sitemap is crawled in
758
- #
759
- # @param tags [Array<String>] Comma-separated tags for tracking request usage. Up to 20 tags, each 1-50 charac
760
- #
761
- # @param timeout_opts [ContextDev::Models::WebWebScrapeSitemapParams::TimeoutOpts] Optional request deadline and behavior on timeout. For GET requests, use timeout
762
- #
763
- # @param url_regex [String] Optional RE2-compatible regex pattern. Only URLs matching this pattern are retur
764
- #
765
- # @param zdr [Symbol, ContextDev::Models::WebWebScrapeSitemapParams::Zdr] Set to enabled to bypass shared caches and omit request and response content fro
766
- #
767
- # @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
768
- #
769
- # @return [ContextDev::Models::WebWebScrapeSitemapResponse]
770
- #
771
- # @see ContextDev::Models::WebWebScrapeSitemapParams
772
- def web_scrape_sitemap(params)
773
- parsed, options = ContextDev::WebWebScrapeSitemapParams.dump_request(params)
774
- query = ContextDev::Internal::Util.encode_query_params(parsed)
775
- @client.request(
776
- method: :get,
777
- path: "web/scrape/sitemap",
778
- query: query.transform_keys(
779
- include_subdomains: "includeSubdomains",
780
- max_links: "maxLinks",
781
- sitemap_url: "sitemapUrl",
782
- timeout_opts: "timeoutOpts",
783
- url_regex: "urlRegex"
784
- ),
785
- model: ContextDev::Models::WebWebScrapeSitemapResponse,
786
- options: options
787
- )
788
- end
789
-
790
422
  # @api private
791
423
  #
792
424
  # @param client [ContextDev::Client]