context.dev 2.18.0 → 2.19.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +18 -0
- data/README.md +1 -1
- data/lib/context_dev/client.rb +0 -4
- data/lib/context_dev/models/monitor_retrieve_run_params.rb +26 -0
- data/lib/context_dev/models/monitor_retrieve_run_response.rb +232 -0
- data/lib/context_dev/models/monitor_rotate_webhook_secret_params.rb +20 -0
- data/lib/context_dev/models/monitor_rotate_webhook_secret_response.rb +765 -0
- data/lib/context_dev/models/{web_web_scrape_sitemap_params.rb → web_map_urls_params.rb} +14 -14
- data/lib/context_dev/models/web_map_urls_response.rb +123 -0
- data/lib/context_dev/models/web_scrape_params.rb +906 -0
- data/lib/context_dev/models/web_scrape_response.rb +659 -0
- data/lib/context_dev/models/web_screenshot_params.rb +15 -1
- data/lib/context_dev/models/web_screenshot_response.rb +3 -3
- data/lib/context_dev/models.rb +8 -22
- data/lib/context_dev/resources/brand.rb +0 -36
- data/lib/context_dev/resources/monitors.rb +47 -0
- data/lib/context_dev/resources/web.rb +106 -486
- data/lib/context_dev/version.rb +1 -1
- data/lib/context_dev.rb +8 -23
- data/rbi/context_dev/client.rbi +0 -3
- data/rbi/context_dev/models/monitor_retrieve_run_params.rbi +46 -0
- data/rbi/context_dev/models/monitor_retrieve_run_response.rbi +452 -0
- data/rbi/context_dev/models/monitor_rotate_webhook_secret_params.rbi +38 -0
- data/rbi/context_dev/models/monitor_rotate_webhook_secret_response.rbi +1367 -0
- data/rbi/context_dev/models/{web_web_scrape_sitemap_params.rbi → web_map_urls_params.rbi} +23 -45
- data/rbi/context_dev/models/web_map_urls_response.rbi +226 -0
- data/rbi/context_dev/models/web_scrape_params.rbi +2000 -0
- data/rbi/context_dev/models/{web_web_scrape_html_response.rbi → web_scrape_response.rbi} +539 -510
- data/rbi/context_dev/models/web_screenshot_params.rbi +23 -0
- data/rbi/context_dev/models/web_screenshot_response.rbi +4 -4
- data/rbi/context_dev/models.rbi +9 -24
- data/rbi/context_dev/resources/brand.rbi +0 -35
- data/rbi/context_dev/resources/monitors.rbi +24 -0
- data/rbi/context_dev/resources/web.rbi +135 -654
- data/sig/context_dev/client.rbs +0 -2
- data/sig/context_dev/models/monitor_retrieve_run_params.rbs +28 -0
- data/sig/context_dev/models/monitor_retrieve_run_response.rbs +182 -0
- data/sig/context_dev/models/monitor_rotate_webhook_secret_params.rbs +23 -0
- data/sig/context_dev/models/monitor_rotate_webhook_secret_response.rbs +545 -0
- data/sig/context_dev/models/{web_web_scrape_sitemap_params.rbs → web_map_urls_params.rbs} +22 -22
- data/sig/context_dev/models/web_map_urls_response.rbs +125 -0
- data/sig/context_dev/models/web_scrape_params.rbs +834 -0
- data/sig/context_dev/models/web_scrape_response.rbs +552 -0
- data/sig/context_dev/models/web_screenshot_params.rbs +7 -0
- data/sig/context_dev/models.rbs +8 -22
- data/sig/context_dev/resources/brand.rbs +0 -9
- data/sig/context_dev/resources/monitors.rbs +11 -0
- data/sig/context_dev/resources/web.rbs +30 -128
- metadata +26 -71
- data/lib/context_dev/models/ai_extract_product_params.rb +0 -125
- data/lib/context_dev/models/ai_extract_product_response.rb +0 -402
- data/lib/context_dev/models/ai_extract_products_params.rb +0 -233
- data/lib/context_dev/models/ai_extract_products_response.rb +0 -370
- data/lib/context_dev/models/brand_retrieve_simplified_params.rb +0 -120
- data/lib/context_dev/models/brand_retrieve_simplified_response.rb +0 -441
- data/lib/context_dev/models/web_extract_fonts_params.rb +0 -114
- data/lib/context_dev/models/web_extract_fonts_response.rb +0 -291
- data/lib/context_dev/models/web_extract_params.rb +0 -419
- data/lib/context_dev/models/web_extract_response.rb +0 -287
- data/lib/context_dev/models/web_web_scrape_bytes_params.rb +0 -346
- data/lib/context_dev/models/web_web_scrape_bytes_response.rb +0 -135
- data/lib/context_dev/models/web_web_scrape_html_params.rb +0 -812
- data/lib/context_dev/models/web_web_scrape_html_response.rb +0 -628
- data/lib/context_dev/models/web_web_scrape_images_params.rb +0 -373
- data/lib/context_dev/models/web_web_scrape_images_response.rb +0 -409
- data/lib/context_dev/models/web_web_scrape_md_params.rb +0 -670
- data/lib/context_dev/models/web_web_scrape_md_response.rb +0 -567
- data/lib/context_dev/models/web_web_scrape_screenshot_params.rb +0 -471
- data/lib/context_dev/models/web_web_scrape_screenshot_response.rb +0 -169
- data/lib/context_dev/models/web_web_scrape_sitemap_response.rb +0 -144
- data/lib/context_dev/resources/ai.rb +0 -71
- data/rbi/context_dev/models/ai_extract_product_params.rbi +0 -253
- data/rbi/context_dev/models/ai_extract_product_response.rbi +0 -791
- data/rbi/context_dev/models/ai_extract_products_params.rbi +0 -490
- data/rbi/context_dev/models/ai_extract_products_response.rbi +0 -718
- data/rbi/context_dev/models/brand_retrieve_simplified_params.rbi +0 -256
- data/rbi/context_dev/models/brand_retrieve_simplified_response.rbi +0 -1040
- data/rbi/context_dev/models/web_extract_fonts_params.rbi +0 -223
- data/rbi/context_dev/models/web_extract_fonts_response.rbi +0 -563
- data/rbi/context_dev/models/web_extract_params.rbi +0 -781
- data/rbi/context_dev/models/web_extract_response.rbi +0 -512
- data/rbi/context_dev/models/web_web_scrape_bytes_params.rbi +0 -702
- data/rbi/context_dev/models/web_web_scrape_bytes_response.rbi +0 -238
- data/rbi/context_dev/models/web_web_scrape_html_params.rbi +0 -1690
- data/rbi/context_dev/models/web_web_scrape_images_params.rbi +0 -728
- data/rbi/context_dev/models/web_web_scrape_images_response.rbi +0 -910
- data/rbi/context_dev/models/web_web_scrape_md_params.rbi +0 -1052
- data/rbi/context_dev/models/web_web_scrape_md_response.rbi +0 -1088
- data/rbi/context_dev/models/web_web_scrape_screenshot_params.rbi +0 -1580
- data/rbi/context_dev/models/web_web_scrape_screenshot_response.rbi +0 -332
- data/rbi/context_dev/models/web_web_scrape_sitemap_response.rbi +0 -258
- data/rbi/context_dev/resources/ai.rbi +0 -63
- data/sig/context_dev/models/ai_extract_product_params.rbs +0 -106
- data/sig/context_dev/models/ai_extract_product_response.rbs +0 -314
- data/sig/context_dev/models/ai_extract_products_params.rbs +0 -202
- data/sig/context_dev/models/ai_extract_products_response.rbs +0 -293
- data/sig/context_dev/models/brand_retrieve_simplified_params.rbs +0 -104
- data/sig/context_dev/models/brand_retrieve_simplified_response.rbs +0 -413
- data/sig/context_dev/models/web_extract_fonts_params.rbs +0 -93
- data/sig/context_dev/models/web_extract_fonts_response.rbs +0 -230
- data/sig/context_dev/models/web_extract_params.rbs +0 -324
- data/sig/context_dev/models/web_extract_response.rbs +0 -239
- data/sig/context_dev/models/web_web_scrape_bytes_params.rbs +0 -531
- data/sig/context_dev/models/web_web_scrape_bytes_response.rbs +0 -108
- data/sig/context_dev/models/web_web_scrape_html_params.rbs +0 -878
- data/sig/context_dev/models/web_web_scrape_html_response.rbs +0 -525
- data/sig/context_dev/models/web_web_scrape_images_params.rbs +0 -281
- data/sig/context_dev/models/web_web_scrape_images_response.rbs +0 -371
- data/sig/context_dev/models/web_web_scrape_md_params.rbs +0 -758
- data/sig/context_dev/models/web_web_scrape_md_response.rbs +0 -465
- data/sig/context_dev/models/web_web_scrape_screenshot_params.rbs +0 -619
- data/sig/context_dev/models/web_web_scrape_screenshot_response.rbs +0 -127
- data/sig/context_dev/models/web_web_scrape_sitemap_response.rbs +0 -117
- data/sig/context_dev/resources/ai.rbs +0 -21
|
@@ -42,64 +42,6 @@ module ContextDev
|
|
|
42
42
|
)
|
|
43
43
|
end
|
|
44
44
|
|
|
45
|
-
# Some parameter documentations has been truncated, see
|
|
46
|
-
# {ContextDev::Models::WebExtractParams} for more details.
|
|
47
|
-
#
|
|
48
|
-
# Crawl a website, use the provided JSON Schema and instructions to prioritize
|
|
49
|
-
# relevant internal links, and extract structured data from the selected pages.
|
|
50
|
-
#
|
|
51
|
-
# @overload extract(schema:, url:, actions: nil, fact_check: nil, follow_subdomains: nil, include_frames: nil, instructions: nil, max_age_ms: nil, max_depth: nil, max_pages: nil, pdf: nil, settle_animations: nil, stop_after_ms: nil, tags: nil, timeout_opts: nil, wait_for_ms: nil, zdr: nil, request_options: {})
|
|
52
|
-
#
|
|
53
|
-
# @param schema [Hash{Symbol=>Object}] JSON Schema for the returned data object. Image fields such as `image_urls` or `
|
|
54
|
-
#
|
|
55
|
-
# @param url [String] The starting website URL to crawl and extract from. Must include http:// or http
|
|
56
|
-
#
|
|
57
|
-
# @param actions [Array<ContextDev::Models::WebExtractParams::Action::Wait, ContextDev::Models::WebExtractParams::Action::Perform, ContextDev::Models::WebExtractParams::Action::Scroll>] Optional browser actions executed in order on the requested page after it loads,
|
|
58
|
-
#
|
|
59
|
-
# @param fact_check [Boolean] When true, every returned value must be grounded in facts stated on the page; fi
|
|
60
|
-
#
|
|
61
|
-
# @param follow_subdomains [Boolean] When true, follow links on subdomains of the starting URL's domain.
|
|
62
|
-
#
|
|
63
|
-
# @param include_frames [Boolean] When true, iframe contents are included in Markdown before extraction.
|
|
64
|
-
#
|
|
65
|
-
# @param instructions [String] Optional extraction guidance, such as which facts to prioritize or how to interp
|
|
66
|
-
#
|
|
67
|
-
# @param max_age_ms [Integer] Return cached scrape results if a prior scrape for the same parameters is younge
|
|
68
|
-
#
|
|
69
|
-
# @param max_depth [Integer] Optional maximum link depth from the starting URL (0 = only the starting page).
|
|
70
|
-
#
|
|
71
|
-
# @param max_pages [Integer] Maximum number of pages to analyze for extraction. Hard cap: 50. Defaults to 5.
|
|
72
|
-
#
|
|
73
|
-
# @param pdf [ContextDev::Models::WebExtractParams::Pdf]
|
|
74
|
-
#
|
|
75
|
-
# @param settle_animations [Boolean] When true, waits briefly for CSS and transition animations to settle before extr
|
|
76
|
-
#
|
|
77
|
-
# @param stop_after_ms [Integer] Soft time budget for the crawl in milliseconds. Min: 10000 (10s). Max: 110000 (1
|
|
78
|
-
#
|
|
79
|
-
# @param tags [Array<String>] Optional tags for tracking usage. Up to 20 tags, each 1 to 50 characters.
|
|
80
|
-
#
|
|
81
|
-
# @param timeout_opts [ContextDev::Models::WebExtractParams::TimeoutOpts] Optional request deadline and behavior on timeout. For GET requests, use timeout
|
|
82
|
-
#
|
|
83
|
-
# @param wait_for_ms [Integer] Optional browser wait time in milliseconds after initial page load for each craw
|
|
84
|
-
#
|
|
85
|
-
# @param zdr [Symbol, ContextDev::Models::WebExtractParams::Zdr] Set to enabled to bypass shared caches and omit request and response content fro
|
|
86
|
-
#
|
|
87
|
-
# @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
|
|
88
|
-
#
|
|
89
|
-
# @return [ContextDev::Models::WebExtractResponse]
|
|
90
|
-
#
|
|
91
|
-
# @see ContextDev::Models::WebExtractParams
|
|
92
|
-
def extract(params)
|
|
93
|
-
parsed, options = ContextDev::WebExtractParams.dump_request(params)
|
|
94
|
-
@client.request(
|
|
95
|
-
method: :post,
|
|
96
|
-
path: "web/extract",
|
|
97
|
-
body: parsed,
|
|
98
|
-
model: ContextDev::Models::WebExtractResponse,
|
|
99
|
-
options: options
|
|
100
|
-
)
|
|
101
|
-
end
|
|
102
|
-
|
|
103
45
|
# Some parameter documentations has been truncated, see
|
|
104
46
|
# {ContextDev::Models::WebExtractCompetitorsParams} for more details.
|
|
105
47
|
#
|
|
@@ -136,84 +78,154 @@ module ContextDev
|
|
|
136
78
|
end
|
|
137
79
|
|
|
138
80
|
# Some parameter documentations has been truncated, see
|
|
139
|
-
# {ContextDev::Models::
|
|
81
|
+
# {ContextDev::Models::WebExtractStyleguideParams} for more details.
|
|
82
|
+
#
|
|
83
|
+
# Extract a comprehensive design system from a website including colors,
|
|
84
|
+
# typography, spacing, shadows, and UI components.
|
|
140
85
|
#
|
|
141
|
-
#
|
|
142
|
-
# statistics, fallbacks, and element/word counts.
|
|
86
|
+
# @overload extract_styleguide(color_scheme: nil, direct_url: nil, domain: nil, max_age_ms: nil, tags: nil, timeout_opts: nil, zdr: nil, request_options: {})
|
|
143
87
|
#
|
|
144
|
-
# @
|
|
88
|
+
# @param color_scheme [Symbol, ContextDev::Models::WebExtractStyleguideParams::ColorScheme] Optional browser color scheme to emulate for websites that respond to prefers-co
|
|
145
89
|
#
|
|
146
|
-
# @param direct_url [String] A specific URL to fetch
|
|
90
|
+
# @param direct_url [String] A specific URL to fetch the styleguide from directly, bypassing domain resolutio
|
|
147
91
|
#
|
|
148
|
-
# @param domain [String] Domain name to extract
|
|
92
|
+
# @param domain [String] Domain name to extract styleguide from (e.g., 'example.com', 'google.com'). The
|
|
149
93
|
#
|
|
150
94
|
# @param max_age_ms [Integer, nil] Maximum age in milliseconds for cached brand data before the API performs a hard
|
|
151
95
|
#
|
|
152
96
|
# @param tags [Array<String>] Comma-separated tags for tracking request usage. Up to 20 tags, each 1-50 charac
|
|
153
97
|
#
|
|
154
|
-
# @param timeout_opts [ContextDev::Models::
|
|
98
|
+
# @param timeout_opts [ContextDev::Models::WebExtractStyleguideParams::TimeoutOpts] Optional request deadline and behavior on timeout. For GET requests, use timeout
|
|
99
|
+
#
|
|
100
|
+
# @param zdr [Symbol, ContextDev::Models::WebExtractStyleguideParams::Zdr] Set to enabled to bypass shared caches and omit request and response content fro
|
|
155
101
|
#
|
|
156
102
|
# @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
|
|
157
103
|
#
|
|
158
|
-
# @return [ContextDev::Models::
|
|
104
|
+
# @return [ContextDev::Models::WebExtractStyleguideResponse]
|
|
159
105
|
#
|
|
160
|
-
# @see ContextDev::Models::
|
|
161
|
-
def
|
|
162
|
-
parsed, options = ContextDev::
|
|
106
|
+
# @see ContextDev::Models::WebExtractStyleguideParams
|
|
107
|
+
def extract_styleguide(params = {})
|
|
108
|
+
parsed, options = ContextDev::WebExtractStyleguideParams.dump_request(params)
|
|
163
109
|
query = ContextDev::Internal::Util.encode_query_params(parsed)
|
|
164
110
|
@client.request(
|
|
165
111
|
method: :get,
|
|
166
|
-
path: "web/
|
|
112
|
+
path: "web/styleguide",
|
|
167
113
|
query: query.transform_keys(
|
|
114
|
+
color_scheme: "colorScheme",
|
|
168
115
|
direct_url: "directUrl",
|
|
169
116
|
max_age_ms: "maxAgeMs",
|
|
170
117
|
timeout_opts: "timeoutOpts"
|
|
171
118
|
),
|
|
172
|
-
model: ContextDev::Models::
|
|
119
|
+
model: ContextDev::Models::WebExtractStyleguideResponse,
|
|
173
120
|
options: options
|
|
174
121
|
)
|
|
175
122
|
end
|
|
176
123
|
|
|
177
124
|
# Some parameter documentations has been truncated, see
|
|
178
|
-
# {ContextDev::Models::
|
|
125
|
+
# {ContextDev::Models::WebMapURLsParams} for more details.
|
|
179
126
|
#
|
|
180
|
-
#
|
|
181
|
-
#
|
|
127
|
+
# Discovers URLs using the same sitemap crawl, filters, and limits as
|
|
128
|
+
# /web/scrape/sitemap. Each URL includes its available title, description,
|
|
129
|
+
# keywords, and language. URLs without stored enrichment are returned immediately
|
|
130
|
+
# with only the URL and queued for background HTML scraping, so later requests can
|
|
131
|
+
# include their metadata. Responses are never cached as a whole; every request
|
|
132
|
+
# reads the current per-URL enrichment. Zero data retention and credential-bearing
|
|
133
|
+
# discovery requests return URLs without reading or storing shared enrichment or
|
|
134
|
+
# queuing background scrapes. Costs 1 credit, or 2 credits with search.
|
|
182
135
|
#
|
|
183
|
-
# @overload
|
|
136
|
+
# @overload map_urls(domain:, headers: nil, include_subdomains: nil, max_links: nil, search: nil, sitemap_url: nil, tags: nil, timeout_opts: nil, url_regex: nil, zdr: nil, request_options: {})
|
|
184
137
|
#
|
|
185
|
-
# @param
|
|
138
|
+
# @param domain [String] Domain to build a sitemap for
|
|
186
139
|
#
|
|
187
|
-
# @param
|
|
140
|
+
# @param headers [Hash{Symbol=>String}] Optional outbound HTTP headers forwarded only to the target URL, sent as deep-ob
|
|
188
141
|
#
|
|
189
|
-
# @param
|
|
142
|
+
# @param include_subdomains [Boolean] When true, discover and include public pages and sitemaps on subdomains of the r
|
|
190
143
|
#
|
|
191
|
-
# @param
|
|
144
|
+
# @param max_links [Integer] Maximum number of links to return from the sitemap crawl. Defaults to 10,000. Mi
|
|
145
|
+
#
|
|
146
|
+
# @param search [String] Optional search phrase. When provided, the crawled sitemap is filtered to the pa
|
|
147
|
+
#
|
|
148
|
+
# @param sitemap_url [String] Optional explicit sitemap URL. When provided, exactly this sitemap is crawled in
|
|
192
149
|
#
|
|
193
150
|
# @param tags [Array<String>] Comma-separated tags for tracking request usage. Up to 20 tags, each 1-50 charac
|
|
194
151
|
#
|
|
195
|
-
# @param timeout_opts [ContextDev::Models::
|
|
152
|
+
# @param timeout_opts [ContextDev::Models::WebMapURLsParams::TimeoutOpts] Optional request deadline and behavior on timeout. For GET requests, use timeout
|
|
196
153
|
#
|
|
197
|
-
# @param
|
|
154
|
+
# @param url_regex [String] Optional RE2-compatible regex pattern. Only URLs matching this pattern are retur
|
|
155
|
+
#
|
|
156
|
+
# @param zdr [Symbol, ContextDev::Models::WebMapURLsParams::Zdr] Set to enabled to bypass shared caches and omit request and response content fro
|
|
198
157
|
#
|
|
199
158
|
# @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
|
|
200
159
|
#
|
|
201
|
-
# @return [ContextDev::Models::
|
|
160
|
+
# @return [ContextDev::Models::WebMapURLsResponse]
|
|
202
161
|
#
|
|
203
|
-
# @see ContextDev::Models::
|
|
204
|
-
def
|
|
205
|
-
parsed, options = ContextDev::
|
|
162
|
+
# @see ContextDev::Models::WebMapURLsParams
|
|
163
|
+
def map_urls(params)
|
|
164
|
+
parsed, options = ContextDev::WebMapURLsParams.dump_request(params)
|
|
206
165
|
query = ContextDev::Internal::Util.encode_query_params(parsed)
|
|
207
166
|
@client.request(
|
|
208
167
|
method: :get,
|
|
209
|
-
path: "web/
|
|
168
|
+
path: "web/urls",
|
|
210
169
|
query: query.transform_keys(
|
|
211
|
-
|
|
212
|
-
|
|
213
|
-
|
|
214
|
-
timeout_opts: "timeoutOpts"
|
|
170
|
+
include_subdomains: "includeSubdomains",
|
|
171
|
+
max_links: "maxLinks",
|
|
172
|
+
sitemap_url: "sitemapUrl",
|
|
173
|
+
timeout_opts: "timeoutOpts",
|
|
174
|
+
url_regex: "urlRegex"
|
|
215
175
|
),
|
|
216
|
-
model: ContextDev::Models::
|
|
176
|
+
model: ContextDev::Models::WebMapURLsResponse,
|
|
177
|
+
options: options
|
|
178
|
+
)
|
|
179
|
+
end
|
|
180
|
+
|
|
181
|
+
# Some parameter documentations has been truncated, see
|
|
182
|
+
# {ContextDev::Models::WebScrapeParams} for more details.
|
|
183
|
+
#
|
|
184
|
+
# Reuse cached outputs independently and capture missing formats in one page
|
|
185
|
+
# visit. Each cache key includes only the settings that affect that output. HTML
|
|
186
|
+
# is shared with Markdown and parsed fields. Cached outputs can come from
|
|
187
|
+
# different visits within maxAgeMs; use 0 for a fresh capture. HTML-only requests
|
|
188
|
+
# use the existing fast acquisition path. One credit per request, including cache
|
|
189
|
+
# hits, or two with browser actions; PDF OCR adds one credit per recovered page on
|
|
190
|
+
# fresh extraction. Original response bytes and screenshots are limited to 20 MiB
|
|
191
|
+
# each, screenshots to 40 megapixels, and the combined browser capture to 60 MiB.
|
|
192
|
+
#
|
|
193
|
+
# @overload scrape(formats:, url:, image_params: nil, markdown_params: nil, max_age_ms: nil, parse_params: nil, screenshot_params: nil, shared_params: nil, tags: nil, timeout_opts: nil, zdr: nil, request_options: {})
|
|
194
|
+
#
|
|
195
|
+
# @param formats [ContextDev::Models::WebScrapeParams::Formats] Outputs to return. Enable at least one; omitted formats are false.
|
|
196
|
+
#
|
|
197
|
+
# @param url [String] The URL to scrape.
|
|
198
|
+
#
|
|
199
|
+
# @param image_params [ContextDev::Models::WebScrapeParams::ImageParams] Image options. Requires formats.images: true.
|
|
200
|
+
#
|
|
201
|
+
# @param markdown_params [ContextDev::Models::WebScrapeParams::MarkdownParams] Markdown options. Requires formats.markdown: true.
|
|
202
|
+
#
|
|
203
|
+
# @param max_age_ms [Integer] Maximum age of each cached output. Defaults to 1 day; 0 fetches fresh and update
|
|
204
|
+
#
|
|
205
|
+
# @param parse_params [ContextDev::Models::WebScrapeParams::ParseParams] Required when formats.parse is true.
|
|
206
|
+
#
|
|
207
|
+
# @param screenshot_params [ContextDev::Models::WebScrapeParams::ScreenshotParams] Screenshot options. Requires formats.screenshot: true.
|
|
208
|
+
#
|
|
209
|
+
# @param shared_params [ContextDev::Models::WebScrapeParams::SharedParams] Shared browser and content settings. Content filters leave screenshots and origi
|
|
210
|
+
#
|
|
211
|
+
# @param tags [Array<String>] Labels for tracking request usage. Not retained when zdr is enabled.
|
|
212
|
+
#
|
|
213
|
+
# @param timeout_opts [ContextDev::Models::WebScrapeParams::TimeoutOpts] Total deadline, including navigation, actions, waiting, and all outputs. Default
|
|
214
|
+
#
|
|
215
|
+
# @param zdr [Symbol, ContextDev::Models::WebScrapeParams::Zdr] Zero data retention. Bypasses caches and uploads; excludes request/response cont
|
|
216
|
+
#
|
|
217
|
+
# @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
|
|
218
|
+
#
|
|
219
|
+
# @return [ContextDev::Models::WebScrapeResponse]
|
|
220
|
+
#
|
|
221
|
+
# @see ContextDev::Models::WebScrapeParams
|
|
222
|
+
def scrape(params)
|
|
223
|
+
parsed, options = ContextDev::WebScrapeParams.dump_request(params)
|
|
224
|
+
@client.request(
|
|
225
|
+
method: :post,
|
|
226
|
+
path: "web/scrape",
|
|
227
|
+
body: parsed,
|
|
228
|
+
model: ContextDev::Models::WebScrapeResponse,
|
|
217
229
|
options: options
|
|
218
230
|
)
|
|
219
231
|
end
|
|
@@ -223,7 +235,7 @@ module ContextDev
|
|
|
223
235
|
#
|
|
224
236
|
# Capture a screenshot of a website.
|
|
225
237
|
#
|
|
226
|
-
# @overload screenshot(clear_popups: nil, color_scheme: nil, country: nil, direct_url: nil, domain: nil, full_screenshot: nil, handle_cookie_popup: nil, max_age_ms: nil, page: nil, scroll_offset: nil, tags: nil, timeout_opts: nil, viewport: nil, wait_for_ms: nil, zdr: nil, request_options: {})
|
|
238
|
+
# @overload screenshot(clear_popups: nil, color_scheme: nil, country: nil, direct_url: nil, domain: nil, full_screenshot: nil, handle_cookie_popup: nil, headers: nil, max_age_ms: nil, page: nil, scroll_offset: nil, tags: nil, timeout_opts: nil, viewport: nil, wait_for_ms: nil, zdr: nil, request_options: {})
|
|
227
239
|
#
|
|
228
240
|
# @param clear_popups [Boolean] Optional parameter for comprehensive popup cleanup. If 'true', the browser dismi
|
|
229
241
|
#
|
|
@@ -239,6 +251,8 @@ module ContextDev
|
|
|
239
251
|
#
|
|
240
252
|
# @param handle_cookie_popup [Boolean] Optional parameter to control cookie/consent popup handling. If 'true', we dismi
|
|
241
253
|
#
|
|
254
|
+
# @param headers [Hash{Symbol=>String}] Optional outbound HTTP headers, using the same JSON object or deep-object query
|
|
255
|
+
#
|
|
242
256
|
# @param max_age_ms [Integer, nil] Return a cached screenshot if a prior screenshot for the same parameters exists
|
|
243
257
|
#
|
|
244
258
|
# @param page [Symbol, ContextDev::Models::WebScreenshotParams::Page] Optional parameter to specify which page type to screenshot. If provided, the sy
|
|
@@ -393,400 +407,6 @@ module ContextDev
|
|
|
393
407
|
)
|
|
394
408
|
end
|
|
395
409
|
|
|
396
|
-
# Some parameter documentations has been truncated, see
|
|
397
|
-
# {ContextDev::Models::WebWebScrapeBytesParams} for more details.
|
|
398
|
-
#
|
|
399
|
-
# Downloads a resource and returns its bytes as base64. Supports images, PDFs,
|
|
400
|
-
# HTML pages, and any other content type without image conversion, text
|
|
401
|
-
# extraction, or character-encoding changes. HTTP compression is decoded before
|
|
402
|
-
# base64 encoding. HTML is the original HTTP response; JavaScript is not rendered.
|
|
403
|
-
# Follows public redirects and retries failed downloads through ISP and
|
|
404
|
-
# residential proxies, with a direct fallback. When country is specified, only a
|
|
405
|
-
# residential proxy in that country is used. Supply headers such as Referer for
|
|
406
|
-
# images that require a referring page. Downloads are not cached. Maximum decoded
|
|
407
|
-
# resource size: 20 MiB (20971520 bytes), before base64 encoding. Successful
|
|
408
|
-
# requests cost 1 credit; errors are not billed.
|
|
409
|
-
#
|
|
410
|
-
# @overload web_scrape_bytes(url:, country: nil, headers: nil, tags: nil, timeout_opts: nil, zdr: nil, request_options: {})
|
|
411
|
-
#
|
|
412
|
-
# @param url [String] Full HTTP(S) URL of the resource to download, such as an image, PDF, or page.
|
|
413
|
-
#
|
|
414
|
-
# @param country [Symbol, ContextDev::Models::WebWebScrapeBytesParams::Country] Fetch the target page through a residential proxy in this country (ISO 3166-1 al
|
|
415
|
-
#
|
|
416
|
-
# @param headers [Hash{Symbol=>String}] Optional outbound HTTP headers, such as Referer, Cookie, or Authorization. Send
|
|
417
|
-
#
|
|
418
|
-
# @param tags [Array<String>] Comma-separated tags for tracking request usage. Up to 20 tags, each 1-50 charac
|
|
419
|
-
#
|
|
420
|
-
# @param timeout_opts [ContextDev::Models::WebWebScrapeBytesParams::TimeoutOpts] Optional request deadline and behavior on timeout. For GET requests, use timeout
|
|
421
|
-
#
|
|
422
|
-
# @param zdr [Symbol, ContextDev::Models::WebWebScrapeBytesParams::Zdr] Set to enabled to bypass shared caches and omit request and response content fro
|
|
423
|
-
#
|
|
424
|
-
# @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
|
|
425
|
-
#
|
|
426
|
-
# @return [ContextDev::Models::WebWebScrapeBytesResponse]
|
|
427
|
-
#
|
|
428
|
-
# @see ContextDev::Models::WebWebScrapeBytesParams
|
|
429
|
-
def web_scrape_bytes(params)
|
|
430
|
-
parsed, options = ContextDev::WebWebScrapeBytesParams.dump_request(params)
|
|
431
|
-
query = ContextDev::Internal::Util.encode_query_params(parsed)
|
|
432
|
-
@client.request(
|
|
433
|
-
method: :get,
|
|
434
|
-
path: "web/scrape/bytes",
|
|
435
|
-
query: query.transform_keys(timeout_opts: "timeoutOpts"),
|
|
436
|
-
model: ContextDev::Models::WebWebScrapeBytesResponse,
|
|
437
|
-
options: options
|
|
438
|
-
)
|
|
439
|
-
end
|
|
440
|
-
|
|
441
|
-
# Some parameter documentations has been truncated, see
|
|
442
|
-
# {ContextDev::Models::WebWebScrapeHTMLParams} for more details.
|
|
443
|
-
#
|
|
444
|
-
# Scrapes the given URL and returns the HTML content of the page. Optional
|
|
445
|
-
# extractRules return deterministic structured data in extracted using CSS
|
|
446
|
-
# selectors, attributes, lists, and nested rules, without an LLM or additional
|
|
447
|
-
# credits. Rules run on the returned HTML after selector and main-content
|
|
448
|
-
# filtering. Send extractRules as a JSON-encoded query parameter. The base request
|
|
449
|
-
# costs 1 credit; requests with browser actions cost 2 credits. A request that
|
|
450
|
-
# hits its timeoutOpts.milliseconds deadline fails with 408 and is not billed,
|
|
451
|
-
# unless timeoutOpts.behavior=return-partial is set — then the page as rendered so
|
|
452
|
-
# far is returned with `finalDOMState: "still-loading"` and billed at the base
|
|
453
|
-
# cost of 1 credit.
|
|
454
|
-
#
|
|
455
|
-
# @overload web_scrape_html(url:, actions: nil, country: nil, exclude_selectors: nil, extract_rules: nil, headers: nil, include_frames: nil, include_selectors: nil, max_age_ms: nil, pdf: nil, settle_animations: nil, tags: nil, timeout_opts: nil, use_main_content_only: nil, wait_for_ms: nil, zdr: nil, request_options: {})
|
|
456
|
-
#
|
|
457
|
-
# @param url [String] Full URL to scrape (must include http:// or https:// protocol)
|
|
458
|
-
#
|
|
459
|
-
# @param actions [Array<ContextDev::Models::WebWebScrapeHTMLParams::Action::Wait, ContextDev::Models::WebWebScrapeHTMLParams::Action::Perform, ContextDev::Models::WebWebScrapeHTMLParams::Action::Scroll>, nil] Optional browser actions executed in array order after the page loads and before
|
|
460
|
-
#
|
|
461
|
-
# @param country [Symbol, ContextDev::Models::WebWebScrapeHTMLParams::Country] Fetch the target page through a residential proxy in this country (ISO 3166-1 al
|
|
462
|
-
#
|
|
463
|
-
# @param exclude_selectors [Array<String>, nil] CSS selectors to remove from the result. Applied after includeSelectors. Exclusi
|
|
464
|
-
#
|
|
465
|
-
# @param extract_rules [Hash{Symbol=>String, ContextDev::Models::WebWebScrapeHTMLParams::ExtractRule::UnionMember1}] Optional CSS extraction rules applied to the returned HTML after selector and ma
|
|
466
|
-
#
|
|
467
|
-
# @param headers [Hash{Symbol=>String}] Optional outbound HTTP headers forwarded only to the target URL, sent as deep-ob
|
|
468
|
-
#
|
|
469
|
-
# @param include_frames [Boolean] When true, iframes are rendered inline into the returned HTML.
|
|
470
|
-
#
|
|
471
|
-
# @param include_selectors [Array<String>, nil] CSS selectors. When provided, only matching subtrees (and their descendants) are
|
|
472
|
-
#
|
|
473
|
-
# @param max_age_ms [Integer, nil] Return a cached result if a prior scrape for the same parameters exists and is y
|
|
474
|
-
#
|
|
475
|
-
# @param pdf [ContextDev::Models::WebWebScrapeHTMLParams::Pdf] PDF parsing controls. Use start/end to limit text extraction and embedded-image
|
|
476
|
-
#
|
|
477
|
-
# @param settle_animations [Boolean] When true, waits briefly for CSS and transition animations to settle before extr
|
|
478
|
-
#
|
|
479
|
-
# @param tags [Array<String>] Comma-separated tags for tracking request usage. Up to 20 tags, each 1-50 charac
|
|
480
|
-
#
|
|
481
|
-
# @param timeout_opts [ContextDev::Models::WebWebScrapeHTMLParams::TimeoutOpts] Optional request deadline and behavior on timeout. For GET requests, use timeout
|
|
482
|
-
#
|
|
483
|
-
# @param use_main_content_only [Boolean] When true, return only the page's main content in the HTML response, excluding h
|
|
484
|
-
#
|
|
485
|
-
# @param wait_for_ms [Integer, nil] Optional browser wait time in milliseconds after initial page load. Min: 0. Max:
|
|
486
|
-
#
|
|
487
|
-
# @param zdr [Symbol, ContextDev::Models::WebWebScrapeHTMLParams::Zdr] Set to enabled to bypass shared caches and omit request and response content fro
|
|
488
|
-
#
|
|
489
|
-
# @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
|
|
490
|
-
#
|
|
491
|
-
# @return [ContextDev::Models::WebWebScrapeHTMLResponse]
|
|
492
|
-
#
|
|
493
|
-
# @see ContextDev::Models::WebWebScrapeHTMLParams
|
|
494
|
-
def web_scrape_html(params)
|
|
495
|
-
parsed, options = ContextDev::WebWebScrapeHTMLParams.dump_request(params)
|
|
496
|
-
query = ContextDev::Internal::Util.encode_query_params(parsed)
|
|
497
|
-
@client.request(
|
|
498
|
-
method: :get,
|
|
499
|
-
path: "web/scrape/html",
|
|
500
|
-
query: query.transform_keys(
|
|
501
|
-
exclude_selectors: "excludeSelectors",
|
|
502
|
-
extract_rules: "extractRules",
|
|
503
|
-
include_frames: "includeFrames",
|
|
504
|
-
include_selectors: "includeSelectors",
|
|
505
|
-
max_age_ms: "maxAgeMs",
|
|
506
|
-
settle_animations: "settleAnimations",
|
|
507
|
-
timeout_opts: "timeoutOpts",
|
|
508
|
-
use_main_content_only: "useMainContentOnly",
|
|
509
|
-
wait_for_ms: "waitForMs"
|
|
510
|
-
),
|
|
511
|
-
model: ContextDev::Models::WebWebScrapeHTMLResponse,
|
|
512
|
-
options: options
|
|
513
|
-
)
|
|
514
|
-
end
|
|
515
|
-
|
|
516
|
-
# Some parameter documentations has been truncated, see
|
|
517
|
-
# {ContextDev::Models::WebWebScrapeImagesParams} for more details.
|
|
518
|
-
#
|
|
519
|
-
# Extract image assets from a web page, including standard URLs, inline SVGs, data
|
|
520
|
-
# URIs, responsive image sources, metadata, CSS backgrounds, video posters, and
|
|
521
|
-
# embeds. The base request costs 1 credit, or 2 credits with browser actions. When
|
|
522
|
-
# enrichment is enabled, the entire call costs 5 credits, including requests that
|
|
523
|
-
# also use actions.
|
|
524
|
-
#
|
|
525
|
-
# @overload web_scrape_images(url:, actions: nil, dedupe: nil, enrichment: nil, headers: nil, max_age_ms: nil, tags: nil, timeout_opts: nil, wait_for_ms: nil, zdr: nil, request_options: {})
|
|
526
|
-
#
|
|
527
|
-
# @param url [String] Page URL to inspect. Must include http:// or https://.
|
|
528
|
-
#
|
|
529
|
-
# @param actions [Array<ContextDev::Models::WebWebScrapeImagesParams::Action::Wait, ContextDev::Models::WebWebScrapeImagesParams::Action::Perform, ContextDev::Models::WebWebScrapeImagesParams::Action::Scroll>, nil] Optional browser actions executed in array order after the page loads and before
|
|
530
|
-
#
|
|
531
|
-
# @param dedupe [Boolean] When true, visually duplicate images are removed: every image is loaded and perc
|
|
532
|
-
#
|
|
533
|
-
# @param enrichment [ContextDev::Models::WebWebScrapeImagesParams::Enrichment, nil] Optional per-image processing, sent as deep-object query params such as enrichme
|
|
534
|
-
#
|
|
535
|
-
# @param headers [Hash{Symbol=>String}] Optional outbound HTTP headers forwarded only to the target URL, sent as deep-ob
|
|
536
|
-
#
|
|
537
|
-
# @param max_age_ms [Integer, nil] Reuse a cached result this many milliseconds old or newer. Default: 86400000 (1
|
|
538
|
-
#
|
|
539
|
-
# @param tags [Array<String>] Comma-separated tags for tracking request usage. Up to 20 tags, each 1-50 charac
|
|
540
|
-
#
|
|
541
|
-
# @param timeout_opts [ContextDev::Models::WebWebScrapeImagesParams::TimeoutOpts] Optional request deadline and behavior on timeout. For GET requests, use timeout
|
|
542
|
-
#
|
|
543
|
-
# @param wait_for_ms [Integer, nil] Optional browser wait time in milliseconds after initial page load before collec
|
|
544
|
-
#
|
|
545
|
-
# @param zdr [Symbol, ContextDev::Models::WebWebScrapeImagesParams::Zdr] Set to enabled to bypass shared caches and omit request and response content fro
|
|
546
|
-
#
|
|
547
|
-
# @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
|
|
548
|
-
#
|
|
549
|
-
# @return [ContextDev::Models::WebWebScrapeImagesResponse]
|
|
550
|
-
#
|
|
551
|
-
# @see ContextDev::Models::WebWebScrapeImagesParams
|
|
552
|
-
def web_scrape_images(params)
|
|
553
|
-
parsed, options = ContextDev::WebWebScrapeImagesParams.dump_request(params)
|
|
554
|
-
query = ContextDev::Internal::Util.encode_query_params(parsed)
|
|
555
|
-
@client.request(
|
|
556
|
-
method: :get,
|
|
557
|
-
path: "web/scrape/images",
|
|
558
|
-
query: query.transform_keys(
|
|
559
|
-
max_age_ms: "maxAgeMs",
|
|
560
|
-
timeout_opts: "timeoutOpts",
|
|
561
|
-
wait_for_ms: "waitForMs"
|
|
562
|
-
),
|
|
563
|
-
model: ContextDev::Models::WebWebScrapeImagesResponse,
|
|
564
|
-
options: options
|
|
565
|
-
)
|
|
566
|
-
end
|
|
567
|
-
|
|
568
|
-
# Some parameter documentations has been truncated, see
|
|
569
|
-
# {ContextDev::Models::WebWebScrapeMdParams} for more details.
|
|
570
|
-
#
|
|
571
|
-
# Scrapes the given URL into LLM usable Markdown. Inspect key_metadata on JSON
|
|
572
|
-
# responses from a recognized API key; use error_code to distinguish stable
|
|
573
|
-
# failure categories.
|
|
574
|
-
#
|
|
575
|
-
# ### YouTube
|
|
576
|
-
#
|
|
577
|
-
# YouTube URLs return the video or channel itself rather than the surrounding
|
|
578
|
-
# player and navigation chrome. A URL addressing a single video (`/watch`,
|
|
579
|
-
# `youtu.be`, `/shorts`, `/embed`, `/live`) returns its title, channel, duration,
|
|
580
|
-
# view count, keywords, full description, and the transcript when the video has
|
|
581
|
-
# captions that can be retrieved; videos without captions return everything except
|
|
582
|
-
# the transcript. A channel URL (`/channel/UC…`, `/@handle`, `/c/…`, `/user/…`)
|
|
583
|
-
# returns its name, handle, subscriber count, video count, and full description.
|
|
584
|
-
# When `includeImages=true`, video responses also include the thumbnail and
|
|
585
|
-
# channel responses include the avatar. Costs the same as any other scrape.
|
|
586
|
-
#
|
|
587
|
-
# ### Billing & errors
|
|
588
|
-
#
|
|
589
|
-
# | HTTP status | Billed? | Meaning |
|
|
590
|
-
# | ----------- | ----------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
|
591
|
-
# | 200 | Yes — 1 credit, or 2 credits with actions | Successful scrape, including a zero-length result when includeSelectors matched nothing. A partial result (`finalDOMState: "still-loading"`, only with timeoutOpts.behavior=return-partial) is billed at the base 1 credit with no OCR or actions surcharge |
|
|
592
|
-
# | 400 | No | Invalid input, skipped PDF, or the page could not be scraped. error_code WEBSITE_BLOCKED specifically means the site answered with an anti-bot challenge, CAPTCHA wall, or login shell instead of the page (even when the site returned HTTP 200) — retrying later or from another country sometimes succeeds |
|
|
593
|
-
# | 401 / 403 | No | Invalid/disabled key, insufficient permissions, or credits exhausted; inspect error_code |
|
|
594
|
-
# | 404 | No | Target page returned or fingerprinted as not found |
|
|
595
|
-
# | 408 | No | Request timed out. With timeoutOpts.behavior=return-partial this only happens when nothing usable had rendered by the deadline |
|
|
596
|
-
# | 413 | No | Target content exceeds the maximum supported size (20 MB) |
|
|
597
|
-
# | 415 | No | Unsupported content type |
|
|
598
|
-
# | 429 | No | Per-minute rate limit exceeded; honor Retry-After |
|
|
599
|
-
# | 500 | No | Internal error |
|
|
600
|
-
#
|
|
601
|
-
# @overload web_scrape_md(url:, actions: nil, country: nil, exclude_selectors: nil, headers: nil, include_frames: nil, include_html: nil, include_images: nil, include_links: nil, include_selectors: nil, max_age_ms: nil, pdf: nil, settle_animations: nil, shorten_base64_images: nil, tags: nil, timeout_opts: nil, use_main_content_only: nil, wait_for_ms: nil, zdr: nil, request_options: {})
|
|
602
|
-
#
|
|
603
|
-
# @param url [String] Full URL to scrape into LLM usable Markdown (must include http:// or https:// pr
|
|
604
|
-
#
|
|
605
|
-
# @param actions [Array<ContextDev::Models::WebWebScrapeMdParams::Action::Wait, ContextDev::Models::WebWebScrapeMdParams::Action::Perform, ContextDev::Models::WebWebScrapeMdParams::Action::Scroll>, nil] Optional browser actions executed in array order after the page loads and before
|
|
606
|
-
#
|
|
607
|
-
# @param country [Symbol, ContextDev::Models::WebWebScrapeMdParams::Country] Fetch the target page through a residential proxy in this country (ISO 3166-1 al
|
|
608
|
-
#
|
|
609
|
-
# @param exclude_selectors [Array<String>, nil] CSS selectors to remove before conversion to Markdown. Applied after includeSele
|
|
610
|
-
#
|
|
611
|
-
# @param headers [Hash{Symbol=>String}] Optional outbound HTTP headers forwarded only to the target URL, sent as deep-ob
|
|
612
|
-
#
|
|
613
|
-
# @param include_frames [Boolean] When true, the contents of iframes are rendered to Markdown.
|
|
614
|
-
#
|
|
615
|
-
# @param include_html [Boolean] When true, the response also includes an `html` field with the page HTML the Mar
|
|
616
|
-
#
|
|
617
|
-
# @param include_images [Boolean] Include image references in Markdown output
|
|
618
|
-
#
|
|
619
|
-
# @param include_links [Boolean] Preserve hyperlinks in Markdown output
|
|
620
|
-
#
|
|
621
|
-
# @param include_selectors [Array<String>, nil] CSS selectors. When provided, only matching HTML subtrees (and their descendants
|
|
622
|
-
#
|
|
623
|
-
# @param max_age_ms [Integer, nil] Return a cached result if a prior scrape for the same parameters exists and is y
|
|
624
|
-
#
|
|
625
|
-
# @param pdf [ContextDev::Models::WebWebScrapeMdParams::Pdf] PDF parsing controls. Use start/end to limit text extraction and embedded-image
|
|
626
|
-
#
|
|
627
|
-
# @param settle_animations [Boolean] When true, waits briefly for CSS and transition animations to settle before conv
|
|
628
|
-
#
|
|
629
|
-
# @param shorten_base64_images [Boolean] Shorten base64-encoded image data in the Markdown output
|
|
630
|
-
#
|
|
631
|
-
# @param tags [Array<String>] Comma-separated tags for tracking request usage. Up to 20 tags, each 1-50 charac
|
|
632
|
-
#
|
|
633
|
-
# @param timeout_opts [ContextDev::Models::WebWebScrapeMdParams::TimeoutOpts] Optional request deadline and behavior on timeout. For GET requests, use timeout
|
|
634
|
-
#
|
|
635
|
-
# @param use_main_content_only [Boolean] Extract only the main content of the page, excluding headers, footers, sidebars,
|
|
636
|
-
#
|
|
637
|
-
# @param wait_for_ms [Integer, nil] Optional browser wait time in milliseconds after initial page load before conver
|
|
638
|
-
#
|
|
639
|
-
# @param zdr [Symbol, ContextDev::Models::WebWebScrapeMdParams::Zdr] Set to enabled to bypass shared caches and omit request and response content fro
|
|
640
|
-
#
|
|
641
|
-
# @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
|
|
642
|
-
#
|
|
643
|
-
# @return [ContextDev::Models::WebWebScrapeMdResponse]
|
|
644
|
-
#
|
|
645
|
-
# @see ContextDev::Models::WebWebScrapeMdParams
|
|
646
|
-
def web_scrape_md(params)
|
|
647
|
-
parsed, options = ContextDev::WebWebScrapeMdParams.dump_request(params)
|
|
648
|
-
query = ContextDev::Internal::Util.encode_query_params(parsed)
|
|
649
|
-
@client.request(
|
|
650
|
-
method: :get,
|
|
651
|
-
path: "web/scrape/markdown",
|
|
652
|
-
query: query.transform_keys(
|
|
653
|
-
exclude_selectors: "excludeSelectors",
|
|
654
|
-
include_frames: "includeFrames",
|
|
655
|
-
include_html: "includeHTML",
|
|
656
|
-
include_images: "includeImages",
|
|
657
|
-
include_links: "includeLinks",
|
|
658
|
-
include_selectors: "includeSelectors",
|
|
659
|
-
max_age_ms: "maxAgeMs",
|
|
660
|
-
settle_animations: "settleAnimations",
|
|
661
|
-
shorten_base64_images: "shortenBase64Images",
|
|
662
|
-
timeout_opts: "timeoutOpts",
|
|
663
|
-
use_main_content_only: "useMainContentOnly",
|
|
664
|
-
wait_for_ms: "waitForMs"
|
|
665
|
-
),
|
|
666
|
-
model: ContextDev::Models::WebWebScrapeMdResponse,
|
|
667
|
-
options: options
|
|
668
|
-
)
|
|
669
|
-
end
|
|
670
|
-
|
|
671
|
-
# Some parameter documentations has been truncated, see
|
|
672
|
-
# {ContextDev::Models::WebWebScrapeScreenshotParams} for more details.
|
|
673
|
-
#
|
|
674
|
-
# Capture the given HTTP or HTTPS URL with configurable viewport, full-page
|
|
675
|
-
# capture, wait time, popup handling, theme, scroll offset, cache age, country,
|
|
676
|
-
# and request timeout. Defaults to a 1920x1080 viewport, a 3-second wait, and a
|
|
677
|
-
# cache age of 1 day. With timeoutOpts.behavior=return-partial, a screenshot of
|
|
678
|
-
# the page rendered so far may be returned; inspect finalDOMState to identify an
|
|
679
|
-
# incomplete render. Successful requests cost 1 credit; errors are not billed.
|
|
680
|
-
#
|
|
681
|
-
# @overload web_scrape_screenshot(url:, clear_popups: nil, color_scheme: nil, country: nil, full_screenshot: nil, handle_cookie_popup: nil, max_age_ms: nil, scroll_offset: nil, tags: nil, timeout_opts: nil, viewport: nil, wait_for_ms: nil, zdr: nil, request_options: {})
|
|
682
|
-
#
|
|
683
|
-
# @param url [String]
|
|
684
|
-
#
|
|
685
|
-
# @param clear_popups [Boolean] Optional parameter for comprehensive popup cleanup. If 'true', the browser dismi
|
|
686
|
-
#
|
|
687
|
-
# @param color_scheme [Symbol, ContextDev::Models::WebWebScrapeScreenshotParams::ColorScheme] Optional parameter to choose the site's visual theme in the screenshot. Use 'lig
|
|
688
|
-
#
|
|
689
|
-
# @param country [Symbol, ContextDev::Models::WebWebScrapeScreenshotParams::Country] Fetch the target page through a residential proxy in this country (ISO 3166-1 al
|
|
690
|
-
#
|
|
691
|
-
# @param full_screenshot [Symbol, ContextDev::Models::WebWebScrapeScreenshotParams::FullScreenshot] Optional parameter to determine screenshot type. If 'true', takes a full page sc
|
|
692
|
-
#
|
|
693
|
-
# @param handle_cookie_popup [Boolean] Optional parameter to control cookie/consent popup handling. If 'true', we dismi
|
|
694
|
-
#
|
|
695
|
-
# @param max_age_ms [Integer, nil] Return a cached screenshot if a prior screenshot for the same parameters exists
|
|
696
|
-
#
|
|
697
|
-
# @param scroll_offset [Integer, nil] Optional vertical scroll offset in pixels for capturing a long page in viewport-
|
|
698
|
-
#
|
|
699
|
-
# @param tags [Array<String>] Comma-separated tags for tracking request usage. Up to 20 tags, each 1-50 charac
|
|
700
|
-
#
|
|
701
|
-
# @param timeout_opts [ContextDev::Models::WebWebScrapeScreenshotParams::TimeoutOpts] Optional request deadline and behavior on timeout. For GET requests, use timeout
|
|
702
|
-
#
|
|
703
|
-
# @param viewport [ContextDev::Models::WebWebScrapeScreenshotParams::Viewport] Optional browser viewport dimensions for the screenshot. Defaults to 1920x1080.
|
|
704
|
-
#
|
|
705
|
-
# @param wait_for_ms [Integer, nil] Optional browser wait time in milliseconds after initial page load before taking
|
|
706
|
-
#
|
|
707
|
-
# @param zdr [Symbol, ContextDev::Models::WebWebScrapeScreenshotParams::Zdr] Set to enabled to bypass shared caches and omit request and response content fro
|
|
708
|
-
#
|
|
709
|
-
# @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
|
|
710
|
-
#
|
|
711
|
-
# @return [ContextDev::Models::WebWebScrapeScreenshotResponse]
|
|
712
|
-
#
|
|
713
|
-
# @see ContextDev::Models::WebWebScrapeScreenshotParams
|
|
714
|
-
def web_scrape_screenshot(params)
|
|
715
|
-
parsed, options = ContextDev::WebWebScrapeScreenshotParams.dump_request(params)
|
|
716
|
-
query = ContextDev::Internal::Util.encode_query_params(parsed)
|
|
717
|
-
@client.request(
|
|
718
|
-
method: :get,
|
|
719
|
-
path: "web/scrape/screenshot",
|
|
720
|
-
query: query.transform_keys(
|
|
721
|
-
clear_popups: "clearPopups",
|
|
722
|
-
color_scheme: "colorScheme",
|
|
723
|
-
full_screenshot: "fullScreenshot",
|
|
724
|
-
handle_cookie_popup: "handleCookiePopup",
|
|
725
|
-
max_age_ms: "maxAgeMs",
|
|
726
|
-
scroll_offset: "scrollOffset",
|
|
727
|
-
timeout_opts: "timeoutOpts",
|
|
728
|
-
wait_for_ms: "waitForMs"
|
|
729
|
-
),
|
|
730
|
-
model: ContextDev::Models::WebWebScrapeScreenshotResponse,
|
|
731
|
-
options: options
|
|
732
|
-
)
|
|
733
|
-
end
|
|
734
|
-
|
|
735
|
-
# Some parameter documentations has been truncated, see
|
|
736
|
-
# {ContextDev::Models::WebWebScrapeSitemapParams} for more details.
|
|
737
|
-
#
|
|
738
|
-
# Crawl an entire website's sitemap and return all discovered page URLs. Set
|
|
739
|
-
# `includeSubdomains=true` to also discover public pages and sitemaps on child
|
|
740
|
-
# hosts such as `docs.example.com` or `brand.example.com`. Pass `search` to have
|
|
741
|
-
# the discovered URLs filtered down to the pages about a phrase (for example
|
|
742
|
-
# `pricing and plans` or `api authentication docs`), most relevant first — a
|
|
743
|
-
# searched crawl scans the whole sitemap and costs 2 credits instead of 1.
|
|
744
|
-
#
|
|
745
|
-
# @overload web_scrape_sitemap(domain:, headers: nil, include_subdomains: nil, max_links: nil, search: nil, sitemap_url: nil, tags: nil, timeout_opts: nil, url_regex: nil, zdr: nil, request_options: {})
|
|
746
|
-
#
|
|
747
|
-
# @param domain [String] Domain to build a sitemap for
|
|
748
|
-
#
|
|
749
|
-
# @param headers [Hash{Symbol=>String}] Optional outbound HTTP headers forwarded only to the target URL, sent as deep-ob
|
|
750
|
-
#
|
|
751
|
-
# @param include_subdomains [Boolean] When true, discover and include public pages and sitemaps on subdomains of the r
|
|
752
|
-
#
|
|
753
|
-
# @param max_links [Integer] Maximum number of links to return from the sitemap crawl. Defaults to 10,000. Mi
|
|
754
|
-
#
|
|
755
|
-
# @param search [String] Optional search phrase. When provided, the crawled sitemap is filtered to the pa
|
|
756
|
-
#
|
|
757
|
-
# @param sitemap_url [String] Optional explicit sitemap URL. When provided, exactly this sitemap is crawled in
|
|
758
|
-
#
|
|
759
|
-
# @param tags [Array<String>] Comma-separated tags for tracking request usage. Up to 20 tags, each 1-50 charac
|
|
760
|
-
#
|
|
761
|
-
# @param timeout_opts [ContextDev::Models::WebWebScrapeSitemapParams::TimeoutOpts] Optional request deadline and behavior on timeout. For GET requests, use timeout
|
|
762
|
-
#
|
|
763
|
-
# @param url_regex [String] Optional RE2-compatible regex pattern. Only URLs matching this pattern are retur
|
|
764
|
-
#
|
|
765
|
-
# @param zdr [Symbol, ContextDev::Models::WebWebScrapeSitemapParams::Zdr] Set to enabled to bypass shared caches and omit request and response content fro
|
|
766
|
-
#
|
|
767
|
-
# @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
|
|
768
|
-
#
|
|
769
|
-
# @return [ContextDev::Models::WebWebScrapeSitemapResponse]
|
|
770
|
-
#
|
|
771
|
-
# @see ContextDev::Models::WebWebScrapeSitemapParams
|
|
772
|
-
def web_scrape_sitemap(params)
|
|
773
|
-
parsed, options = ContextDev::WebWebScrapeSitemapParams.dump_request(params)
|
|
774
|
-
query = ContextDev::Internal::Util.encode_query_params(parsed)
|
|
775
|
-
@client.request(
|
|
776
|
-
method: :get,
|
|
777
|
-
path: "web/scrape/sitemap",
|
|
778
|
-
query: query.transform_keys(
|
|
779
|
-
include_subdomains: "includeSubdomains",
|
|
780
|
-
max_links: "maxLinks",
|
|
781
|
-
sitemap_url: "sitemapUrl",
|
|
782
|
-
timeout_opts: "timeoutOpts",
|
|
783
|
-
url_regex: "urlRegex"
|
|
784
|
-
),
|
|
785
|
-
model: ContextDev::Models::WebWebScrapeSitemapResponse,
|
|
786
|
-
options: options
|
|
787
|
-
)
|
|
788
|
-
end
|
|
789
|
-
|
|
790
410
|
# @api private
|
|
791
411
|
#
|
|
792
412
|
# @param client [ContextDev::Client]
|