context.dev 2.17.1 → 2.19.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +33 -0
- data/README.md +1 -1
- data/lib/context_dev/client.rb +0 -4
- data/lib/context_dev/models/industry_retrieve_naics_params.rb +28 -1
- data/lib/context_dev/models/industry_retrieve_sic_params.rb +28 -1
- data/lib/context_dev/models/monitor_retrieve_run_params.rb +26 -0
- data/lib/context_dev/models/monitor_retrieve_run_response.rb +232 -0
- data/lib/context_dev/models/monitor_rotate_webhook_secret_params.rb +20 -0
- data/lib/context_dev/models/monitor_rotate_webhook_secret_response.rb +765 -0
- data/lib/context_dev/models/parse_handle_params.rb +8 -6
- data/lib/context_dev/models/person_enrich_params.rb +28 -1
- data/lib/context_dev/models/web_answers_params.rb +28 -1
- data/lib/context_dev/models/web_extract_competitors_params.rb +28 -1
- data/lib/context_dev/models/web_extract_styleguide_params.rb +30 -3
- data/lib/context_dev/models/{web_web_scrape_sitemap_params.rb → web_map_urls_params.rb} +22 -20
- data/lib/context_dev/models/web_map_urls_response.rb +123 -0
- data/lib/context_dev/models/web_scrape_params.rb +906 -0
- data/lib/context_dev/models/web_scrape_response.rb +659 -0
- data/lib/context_dev/models/web_screenshot_params.rb +25 -9
- data/lib/context_dev/models/web_screenshot_response.rb +3 -3
- data/lib/context_dev/models/web_search_params.rb +30 -3
- data/lib/context_dev/models.rb +8 -20
- data/lib/context_dev/resources/brand.rb +0 -36
- data/lib/context_dev/resources/industry.rb +6 -2
- data/lib/context_dev/resources/monitors.rb +47 -0
- data/lib/context_dev/resources/people.rb +3 -1
- data/lib/context_dev/resources/web.rb +116 -413
- data/lib/context_dev/version.rb +1 -1
- data/lib/context_dev.rb +8 -21
- data/rbi/context_dev/client.rbi +0 -3
- data/rbi/context_dev/models/industry_retrieve_naics_params.rbi +59 -0
- data/rbi/context_dev/models/industry_retrieve_sic_params.rbi +57 -0
- data/rbi/context_dev/models/monitor_retrieve_run_params.rbi +46 -0
- data/rbi/context_dev/models/monitor_retrieve_run_response.rbi +452 -0
- data/rbi/context_dev/models/monitor_rotate_webhook_secret_params.rbi +38 -0
- data/rbi/context_dev/models/monitor_rotate_webhook_secret_response.rbi +1367 -0
- data/rbi/context_dev/models/parse_handle_params.rbi +12 -9
- data/rbi/context_dev/models/person_enrich_params.rbi +45 -0
- data/rbi/context_dev/models/web_answers_params.rbi +45 -0
- data/rbi/context_dev/models/web_extract_competitors_params.rbi +59 -0
- data/rbi/context_dev/models/web_extract_styleguide_params.rbi +62 -3
- data/rbi/context_dev/models/{web_web_scrape_sitemap_params.rbi → web_map_urls_params.rbi} +35 -54
- data/rbi/context_dev/models/web_map_urls_response.rbi +226 -0
- data/rbi/context_dev/models/web_scrape_params.rbi +2000 -0
- data/rbi/context_dev/models/{web_web_scrape_html_response.rbi → web_scrape_response.rbi} +550 -418
- data/rbi/context_dev/models/web_screenshot_params.rbi +38 -12
- data/rbi/context_dev/models/web_screenshot_response.rbi +4 -4
- data/rbi/context_dev/models/web_search_params.rbi +48 -3
- data/rbi/context_dev/models.rbi +9 -21
- data/rbi/context_dev/resources/brand.rbi +0 -35
- data/rbi/context_dev/resources/industry.rbi +14 -0
- data/rbi/context_dev/resources/monitors.rbi +24 -0
- data/rbi/context_dev/resources/parse.rbi +4 -4
- data/rbi/context_dev/resources/people.rbi +7 -0
- data/rbi/context_dev/resources/web.rbi +166 -533
- data/sig/context_dev/client.rbs +0 -2
- data/sig/context_dev/models/industry_retrieve_naics_params.rbs +21 -1
- data/sig/context_dev/models/industry_retrieve_sic_params.rbs +21 -1
- data/sig/context_dev/models/monitor_retrieve_run_params.rbs +28 -0
- data/sig/context_dev/models/monitor_retrieve_run_response.rbs +182 -0
- data/sig/context_dev/models/monitor_rotate_webhook_secret_params.rbs +23 -0
- data/sig/context_dev/models/monitor_rotate_webhook_secret_response.rbs +545 -0
- data/sig/context_dev/models/person_enrich_params.rbs +21 -1
- data/sig/context_dev/models/web_answers_params.rbs +21 -1
- data/sig/context_dev/models/web_extract_competitors_params.rbs +21 -1
- data/sig/context_dev/models/web_extract_styleguide_params.rbs +21 -1
- data/sig/context_dev/models/{web_web_scrape_sitemap_params.rbs → web_map_urls_params.rbs} +22 -22
- data/sig/context_dev/models/web_map_urls_response.rbs +125 -0
- data/sig/context_dev/models/web_scrape_params.rbs +834 -0
- data/sig/context_dev/models/web_scrape_response.rbs +552 -0
- data/sig/context_dev/models/web_screenshot_params.rbs +7 -0
- data/sig/context_dev/models/web_search_params.rbs +21 -1
- data/sig/context_dev/models.rbs +8 -20
- data/sig/context_dev/resources/brand.rbs +0 -9
- data/sig/context_dev/resources/industry.rbs +2 -0
- data/sig/context_dev/resources/monitors.rbs +11 -0
- data/sig/context_dev/resources/people.rbs +1 -0
- data/sig/context_dev/resources/web.rbs +34 -108
- metadata +26 -65
- data/lib/context_dev/models/ai_extract_product_params.rb +0 -98
- data/lib/context_dev/models/ai_extract_product_response.rb +0 -352
- data/lib/context_dev/models/ai_extract_products_params.rb +0 -233
- data/lib/context_dev/models/ai_extract_products_response.rb +0 -320
- data/lib/context_dev/models/brand_retrieve_simplified_params.rb +0 -120
- data/lib/context_dev/models/brand_retrieve_simplified_response.rb +0 -441
- data/lib/context_dev/models/web_extract_fonts_params.rb +0 -114
- data/lib/context_dev/models/web_extract_fonts_response.rb +0 -291
- data/lib/context_dev/models/web_extract_params.rb +0 -392
- data/lib/context_dev/models/web_extract_response.rb +0 -287
- data/lib/context_dev/models/web_web_scrape_bytes_params.rb +0 -344
- data/lib/context_dev/models/web_web_scrape_bytes_response.rb +0 -135
- data/lib/context_dev/models/web_web_scrape_html_params.rb +0 -633
- data/lib/context_dev/models/web_web_scrape_html_response.rb +0 -588
- data/lib/context_dev/models/web_web_scrape_images_params.rb +0 -346
- data/lib/context_dev/models/web_web_scrape_images_response.rb +0 -409
- data/lib/context_dev/models/web_web_scrape_md_params.rb +0 -668
- data/lib/context_dev/models/web_web_scrape_md_response.rb +0 -567
- data/lib/context_dev/models/web_web_scrape_sitemap_response.rb +0 -144
- data/lib/context_dev/resources/ai.rb +0 -69
- data/rbi/context_dev/models/ai_extract_product_params.rbi +0 -199
- data/rbi/context_dev/models/ai_extract_product_response.rbi +0 -696
- data/rbi/context_dev/models/ai_extract_products_params.rbi +0 -490
- data/rbi/context_dev/models/ai_extract_products_response.rbi +0 -623
- data/rbi/context_dev/models/brand_retrieve_simplified_params.rbi +0 -256
- data/rbi/context_dev/models/brand_retrieve_simplified_response.rbi +0 -1040
- data/rbi/context_dev/models/web_extract_fonts_params.rbi +0 -223
- data/rbi/context_dev/models/web_extract_fonts_response.rbi +0 -563
- data/rbi/context_dev/models/web_extract_params.rbi +0 -736
- data/rbi/context_dev/models/web_extract_response.rbi +0 -512
- data/rbi/context_dev/models/web_web_scrape_bytes_params.rbi +0 -699
- data/rbi/context_dev/models/web_web_scrape_bytes_response.rbi +0 -238
- data/rbi/context_dev/models/web_web_scrape_html_params.rbi +0 -1217
- data/rbi/context_dev/models/web_web_scrape_images_params.rbi +0 -671
- data/rbi/context_dev/models/web_web_scrape_images_response.rbi +0 -910
- data/rbi/context_dev/models/web_web_scrape_md_params.rbi +0 -1049
- data/rbi/context_dev/models/web_web_scrape_md_response.rbi +0 -1088
- data/rbi/context_dev/models/web_web_scrape_sitemap_response.rbi +0 -258
- data/rbi/context_dev/resources/ai.rbi +0 -56
- data/sig/context_dev/models/ai_extract_product_params.rbs +0 -86
- data/sig/context_dev/models/ai_extract_product_response.rbs +0 -273
- data/sig/context_dev/models/ai_extract_products_params.rbs +0 -202
- data/sig/context_dev/models/ai_extract_products_response.rbs +0 -252
- data/sig/context_dev/models/brand_retrieve_simplified_params.rbs +0 -104
- data/sig/context_dev/models/brand_retrieve_simplified_response.rbs +0 -413
- data/sig/context_dev/models/web_extract_fonts_params.rbs +0 -93
- data/sig/context_dev/models/web_extract_fonts_response.rbs +0 -230
- data/sig/context_dev/models/web_extract_params.rbs +0 -304
- data/sig/context_dev/models/web_extract_response.rbs +0 -239
- data/sig/context_dev/models/web_web_scrape_bytes_params.rbs +0 -531
- data/sig/context_dev/models/web_web_scrape_bytes_response.rbs +0 -108
- data/sig/context_dev/models/web_web_scrape_html_params.rbs +0 -730
- data/sig/context_dev/models/web_web_scrape_html_response.rbs +0 -495
- data/sig/context_dev/models/web_web_scrape_images_params.rbs +0 -261
- data/sig/context_dev/models/web_web_scrape_images_response.rbs +0 -371
- data/sig/context_dev/models/web_web_scrape_md_params.rbs +0 -758
- data/sig/context_dev/models/web_web_scrape_md_response.rbs +0 -465
- data/sig/context_dev/models/web_web_scrape_sitemap_response.rbs +0 -117
- data/sig/context_dev/resources/ai.rbs +0 -20
|
@@ -12,7 +12,7 @@ module ContextDev
|
|
|
12
12
|
# 30 seconds and ultra to 50 seconds; timeoutOpts.milliseconds can shorten either
|
|
13
13
|
# deadline.
|
|
14
14
|
#
|
|
15
|
-
# @overload answers(task:, json_format: nil, mode: nil, tags: nil, timeout_opts: nil, request_options: {})
|
|
15
|
+
# @overload answers(task:, json_format: nil, mode: nil, tags: nil, timeout_opts: nil, zdr: nil, request_options: {})
|
|
16
16
|
#
|
|
17
17
|
# @param task [String] What to research and answer, in plain language. Naming a domain in the task (for
|
|
18
18
|
#
|
|
@@ -24,6 +24,8 @@ module ContextDev
|
|
|
24
24
|
#
|
|
25
25
|
# @param timeout_opts [ContextDev::Models::WebAnswersParams::TimeoutOpts] Optional request deadline and behavior on timeout. For GET requests, use timeout
|
|
26
26
|
#
|
|
27
|
+
# @param zdr [Symbol, ContextDev::Models::WebAnswersParams::Zdr] Set to enabled to bypass shared caches and omit request and response content fro
|
|
28
|
+
#
|
|
27
29
|
# @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
|
|
28
30
|
#
|
|
29
31
|
# @return [ContextDev::Models::WebAnswersResponse]
|
|
@@ -40,69 +42,13 @@ module ContextDev
|
|
|
40
42
|
)
|
|
41
43
|
end
|
|
42
44
|
|
|
43
|
-
# Some parameter documentations has been truncated, see
|
|
44
|
-
# {ContextDev::Models::WebExtractParams} for more details.
|
|
45
|
-
#
|
|
46
|
-
# Crawl a website, use the provided JSON Schema and instructions to prioritize
|
|
47
|
-
# relevant internal links, and extract structured data from the selected pages.
|
|
48
|
-
#
|
|
49
|
-
# @overload extract(schema:, url:, actions: nil, fact_check: nil, follow_subdomains: nil, include_frames: nil, instructions: nil, max_age_ms: nil, max_depth: nil, max_pages: nil, pdf: nil, settle_animations: nil, stop_after_ms: nil, tags: nil, timeout_opts: nil, wait_for_ms: nil, request_options: {})
|
|
50
|
-
#
|
|
51
|
-
# @param schema [Hash{Symbol=>Object}] JSON Schema for the returned data object. Image fields such as `image_urls` or `
|
|
52
|
-
#
|
|
53
|
-
# @param url [String] The starting website URL to crawl and extract from. Must include http:// or http
|
|
54
|
-
#
|
|
55
|
-
# @param actions [Array<ContextDev::Models::WebExtractParams::Action::Wait, ContextDev::Models::WebExtractParams::Action::Perform, ContextDev::Models::WebExtractParams::Action::Scroll>] Optional browser actions executed in order on the requested page after it loads,
|
|
56
|
-
#
|
|
57
|
-
# @param fact_check [Boolean] When true, every returned value must be grounded in facts stated on the page; fi
|
|
58
|
-
#
|
|
59
|
-
# @param follow_subdomains [Boolean] When true, follow links on subdomains of the starting URL's domain.
|
|
60
|
-
#
|
|
61
|
-
# @param include_frames [Boolean] When true, iframe contents are included in Markdown before extraction.
|
|
62
|
-
#
|
|
63
|
-
# @param instructions [String] Optional extraction guidance, such as which facts to prioritize or how to interp
|
|
64
|
-
#
|
|
65
|
-
# @param max_age_ms [Integer] Return cached scrape results if a prior scrape for the same parameters is younge
|
|
66
|
-
#
|
|
67
|
-
# @param max_depth [Integer] Optional maximum link depth from the starting URL (0 = only the starting page).
|
|
68
|
-
#
|
|
69
|
-
# @param max_pages [Integer] Maximum number of pages to analyze for extraction. Hard cap: 50. Defaults to 5.
|
|
70
|
-
#
|
|
71
|
-
# @param pdf [ContextDev::Models::WebExtractParams::Pdf]
|
|
72
|
-
#
|
|
73
|
-
# @param settle_animations [Boolean] When true, waits briefly for CSS and transition animations to settle before extr
|
|
74
|
-
#
|
|
75
|
-
# @param stop_after_ms [Integer] Soft time budget for the crawl in milliseconds. Min: 10000 (10s). Max: 110000 (1
|
|
76
|
-
#
|
|
77
|
-
# @param tags [Array<String>] Optional tags for tracking usage. Up to 20 tags, each 1 to 50 characters.
|
|
78
|
-
#
|
|
79
|
-
# @param timeout_opts [ContextDev::Models::WebExtractParams::TimeoutOpts] Optional request deadline and behavior on timeout. For GET requests, use timeout
|
|
80
|
-
#
|
|
81
|
-
# @param wait_for_ms [Integer] Optional browser wait time in milliseconds after initial page load for each craw
|
|
82
|
-
#
|
|
83
|
-
# @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
|
|
84
|
-
#
|
|
85
|
-
# @return [ContextDev::Models::WebExtractResponse]
|
|
86
|
-
#
|
|
87
|
-
# @see ContextDev::Models::WebExtractParams
|
|
88
|
-
def extract(params)
|
|
89
|
-
parsed, options = ContextDev::WebExtractParams.dump_request(params)
|
|
90
|
-
@client.request(
|
|
91
|
-
method: :post,
|
|
92
|
-
path: "web/extract",
|
|
93
|
-
body: parsed,
|
|
94
|
-
model: ContextDev::Models::WebExtractResponse,
|
|
95
|
-
options: options
|
|
96
|
-
)
|
|
97
|
-
end
|
|
98
|
-
|
|
99
45
|
# Some parameter documentations has been truncated, see
|
|
100
46
|
# {ContextDev::Models::WebExtractCompetitorsParams} for more details.
|
|
101
47
|
#
|
|
102
48
|
# Analyze a company's landing page and web search evidence to return direct
|
|
103
49
|
# competitors for the same product or market.
|
|
104
50
|
#
|
|
105
|
-
# @overload extract_competitors(domain:, num_competitors: nil, tags: nil, timeout_opts: nil, request_options: {})
|
|
51
|
+
# @overload extract_competitors(domain:, num_competitors: nil, tags: nil, timeout_opts: nil, zdr: nil, request_options: {})
|
|
106
52
|
#
|
|
107
53
|
# @param domain [String] Company domain to analyze, such as `stripe.com`. Full http(s) URLs are accepted
|
|
108
54
|
#
|
|
@@ -112,6 +58,8 @@ module ContextDev
|
|
|
112
58
|
#
|
|
113
59
|
# @param timeout_opts [ContextDev::Models::WebExtractCompetitorsParams::TimeoutOpts] Optional request deadline and behavior on timeout. For GET requests, use timeout
|
|
114
60
|
#
|
|
61
|
+
# @param zdr [Symbol, ContextDev::Models::WebExtractCompetitorsParams::Zdr] Set to enabled to bypass shared caches and omit request and response content fro
|
|
62
|
+
#
|
|
115
63
|
# @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
|
|
116
64
|
#
|
|
117
65
|
# @return [ContextDev::Models::WebExtractCompetitorsResponse]
|
|
@@ -130,82 +78,154 @@ module ContextDev
|
|
|
130
78
|
end
|
|
131
79
|
|
|
132
80
|
# Some parameter documentations has been truncated, see
|
|
133
|
-
# {ContextDev::Models::
|
|
81
|
+
# {ContextDev::Models::WebExtractStyleguideParams} for more details.
|
|
134
82
|
#
|
|
135
|
-
#
|
|
136
|
-
#
|
|
83
|
+
# Extract a comprehensive design system from a website including colors,
|
|
84
|
+
# typography, spacing, shadows, and UI components.
|
|
137
85
|
#
|
|
138
|
-
# @overload
|
|
86
|
+
# @overload extract_styleguide(color_scheme: nil, direct_url: nil, domain: nil, max_age_ms: nil, tags: nil, timeout_opts: nil, zdr: nil, request_options: {})
|
|
139
87
|
#
|
|
140
|
-
# @param
|
|
88
|
+
# @param color_scheme [Symbol, ContextDev::Models::WebExtractStyleguideParams::ColorScheme] Optional browser color scheme to emulate for websites that respond to prefers-co
|
|
141
89
|
#
|
|
142
|
-
# @param
|
|
90
|
+
# @param direct_url [String] A specific URL to fetch the styleguide from directly, bypassing domain resolutio
|
|
91
|
+
#
|
|
92
|
+
# @param domain [String] Domain name to extract styleguide from (e.g., 'example.com', 'google.com'). The
|
|
143
93
|
#
|
|
144
94
|
# @param max_age_ms [Integer, nil] Maximum age in milliseconds for cached brand data before the API performs a hard
|
|
145
95
|
#
|
|
146
96
|
# @param tags [Array<String>] Comma-separated tags for tracking request usage. Up to 20 tags, each 1-50 charac
|
|
147
97
|
#
|
|
148
|
-
# @param timeout_opts [ContextDev::Models::
|
|
98
|
+
# @param timeout_opts [ContextDev::Models::WebExtractStyleguideParams::TimeoutOpts] Optional request deadline and behavior on timeout. For GET requests, use timeout
|
|
99
|
+
#
|
|
100
|
+
# @param zdr [Symbol, ContextDev::Models::WebExtractStyleguideParams::Zdr] Set to enabled to bypass shared caches and omit request and response content fro
|
|
149
101
|
#
|
|
150
102
|
# @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
|
|
151
103
|
#
|
|
152
|
-
# @return [ContextDev::Models::
|
|
104
|
+
# @return [ContextDev::Models::WebExtractStyleguideResponse]
|
|
153
105
|
#
|
|
154
|
-
# @see ContextDev::Models::
|
|
155
|
-
def
|
|
156
|
-
parsed, options = ContextDev::
|
|
106
|
+
# @see ContextDev::Models::WebExtractStyleguideParams
|
|
107
|
+
def extract_styleguide(params = {})
|
|
108
|
+
parsed, options = ContextDev::WebExtractStyleguideParams.dump_request(params)
|
|
157
109
|
query = ContextDev::Internal::Util.encode_query_params(parsed)
|
|
158
110
|
@client.request(
|
|
159
111
|
method: :get,
|
|
160
|
-
path: "web/
|
|
112
|
+
path: "web/styleguide",
|
|
161
113
|
query: query.transform_keys(
|
|
114
|
+
color_scheme: "colorScheme",
|
|
162
115
|
direct_url: "directUrl",
|
|
163
116
|
max_age_ms: "maxAgeMs",
|
|
164
117
|
timeout_opts: "timeoutOpts"
|
|
165
118
|
),
|
|
166
|
-
model: ContextDev::Models::
|
|
119
|
+
model: ContextDev::Models::WebExtractStyleguideResponse,
|
|
167
120
|
options: options
|
|
168
121
|
)
|
|
169
122
|
end
|
|
170
123
|
|
|
171
124
|
# Some parameter documentations has been truncated, see
|
|
172
|
-
# {ContextDev::Models::
|
|
125
|
+
# {ContextDev::Models::WebMapURLsParams} for more details.
|
|
173
126
|
#
|
|
174
|
-
#
|
|
175
|
-
#
|
|
127
|
+
# Discovers URLs using the same sitemap crawl, filters, and limits as
|
|
128
|
+
# /web/scrape/sitemap. Each URL includes its available title, description,
|
|
129
|
+
# keywords, and language. URLs without stored enrichment are returned immediately
|
|
130
|
+
# with only the URL and queued for background HTML scraping, so later requests can
|
|
131
|
+
# include their metadata. Responses are never cached as a whole; every request
|
|
132
|
+
# reads the current per-URL enrichment. Zero data retention and credential-bearing
|
|
133
|
+
# discovery requests return URLs without reading or storing shared enrichment or
|
|
134
|
+
# queuing background scrapes. Costs 1 credit, or 2 credits with search.
|
|
176
135
|
#
|
|
177
|
-
# @overload
|
|
136
|
+
# @overload map_urls(domain:, headers: nil, include_subdomains: nil, max_links: nil, search: nil, sitemap_url: nil, tags: nil, timeout_opts: nil, url_regex: nil, zdr: nil, request_options: {})
|
|
178
137
|
#
|
|
179
|
-
# @param
|
|
138
|
+
# @param domain [String] Domain to build a sitemap for
|
|
180
139
|
#
|
|
181
|
-
# @param
|
|
140
|
+
# @param headers [Hash{Symbol=>String}] Optional outbound HTTP headers forwarded only to the target URL, sent as deep-ob
|
|
182
141
|
#
|
|
183
|
-
# @param
|
|
142
|
+
# @param include_subdomains [Boolean] When true, discover and include public pages and sitemaps on subdomains of the r
|
|
184
143
|
#
|
|
185
|
-
# @param
|
|
144
|
+
# @param max_links [Integer] Maximum number of links to return from the sitemap crawl. Defaults to 10,000. Mi
|
|
145
|
+
#
|
|
146
|
+
# @param search [String] Optional search phrase. When provided, the crawled sitemap is filtered to the pa
|
|
147
|
+
#
|
|
148
|
+
# @param sitemap_url [String] Optional explicit sitemap URL. When provided, exactly this sitemap is crawled in
|
|
186
149
|
#
|
|
187
150
|
# @param tags [Array<String>] Comma-separated tags for tracking request usage. Up to 20 tags, each 1-50 charac
|
|
188
151
|
#
|
|
189
|
-
# @param timeout_opts [ContextDev::Models::
|
|
152
|
+
# @param timeout_opts [ContextDev::Models::WebMapURLsParams::TimeoutOpts] Optional request deadline and behavior on timeout. For GET requests, use timeout
|
|
153
|
+
#
|
|
154
|
+
# @param url_regex [String] Optional RE2-compatible regex pattern. Only URLs matching this pattern are retur
|
|
155
|
+
#
|
|
156
|
+
# @param zdr [Symbol, ContextDev::Models::WebMapURLsParams::Zdr] Set to enabled to bypass shared caches and omit request and response content fro
|
|
190
157
|
#
|
|
191
158
|
# @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
|
|
192
159
|
#
|
|
193
|
-
# @return [ContextDev::Models::
|
|
160
|
+
# @return [ContextDev::Models::WebMapURLsResponse]
|
|
194
161
|
#
|
|
195
|
-
# @see ContextDev::Models::
|
|
196
|
-
def
|
|
197
|
-
parsed, options = ContextDev::
|
|
162
|
+
# @see ContextDev::Models::WebMapURLsParams
|
|
163
|
+
def map_urls(params)
|
|
164
|
+
parsed, options = ContextDev::WebMapURLsParams.dump_request(params)
|
|
198
165
|
query = ContextDev::Internal::Util.encode_query_params(parsed)
|
|
199
166
|
@client.request(
|
|
200
167
|
method: :get,
|
|
201
|
-
path: "web/
|
|
168
|
+
path: "web/urls",
|
|
202
169
|
query: query.transform_keys(
|
|
203
|
-
|
|
204
|
-
|
|
205
|
-
|
|
206
|
-
timeout_opts: "timeoutOpts"
|
|
170
|
+
include_subdomains: "includeSubdomains",
|
|
171
|
+
max_links: "maxLinks",
|
|
172
|
+
sitemap_url: "sitemapUrl",
|
|
173
|
+
timeout_opts: "timeoutOpts",
|
|
174
|
+
url_regex: "urlRegex"
|
|
207
175
|
),
|
|
208
|
-
model: ContextDev::Models::
|
|
176
|
+
model: ContextDev::Models::WebMapURLsResponse,
|
|
177
|
+
options: options
|
|
178
|
+
)
|
|
179
|
+
end
|
|
180
|
+
|
|
181
|
+
# Some parameter documentations has been truncated, see
|
|
182
|
+
# {ContextDev::Models::WebScrapeParams} for more details.
|
|
183
|
+
#
|
|
184
|
+
# Reuse cached outputs independently and capture missing formats in one page
|
|
185
|
+
# visit. Each cache key includes only the settings that affect that output. HTML
|
|
186
|
+
# is shared with Markdown and parsed fields. Cached outputs can come from
|
|
187
|
+
# different visits within maxAgeMs; use 0 for a fresh capture. HTML-only requests
|
|
188
|
+
# use the existing fast acquisition path. One credit per request, including cache
|
|
189
|
+
# hits, or two with browser actions; PDF OCR adds one credit per recovered page on
|
|
190
|
+
# fresh extraction. Original response bytes and screenshots are limited to 20 MiB
|
|
191
|
+
# each, screenshots to 40 megapixels, and the combined browser capture to 60 MiB.
|
|
192
|
+
#
|
|
193
|
+
# @overload scrape(formats:, url:, image_params: nil, markdown_params: nil, max_age_ms: nil, parse_params: nil, screenshot_params: nil, shared_params: nil, tags: nil, timeout_opts: nil, zdr: nil, request_options: {})
|
|
194
|
+
#
|
|
195
|
+
# @param formats [ContextDev::Models::WebScrapeParams::Formats] Outputs to return. Enable at least one; omitted formats are false.
|
|
196
|
+
#
|
|
197
|
+
# @param url [String] The URL to scrape.
|
|
198
|
+
#
|
|
199
|
+
# @param image_params [ContextDev::Models::WebScrapeParams::ImageParams] Image options. Requires formats.images: true.
|
|
200
|
+
#
|
|
201
|
+
# @param markdown_params [ContextDev::Models::WebScrapeParams::MarkdownParams] Markdown options. Requires formats.markdown: true.
|
|
202
|
+
#
|
|
203
|
+
# @param max_age_ms [Integer] Maximum age of each cached output. Defaults to 1 day; 0 fetches fresh and update
|
|
204
|
+
#
|
|
205
|
+
# @param parse_params [ContextDev::Models::WebScrapeParams::ParseParams] Required when formats.parse is true.
|
|
206
|
+
#
|
|
207
|
+
# @param screenshot_params [ContextDev::Models::WebScrapeParams::ScreenshotParams] Screenshot options. Requires formats.screenshot: true.
|
|
208
|
+
#
|
|
209
|
+
# @param shared_params [ContextDev::Models::WebScrapeParams::SharedParams] Shared browser and content settings. Content filters leave screenshots and origi
|
|
210
|
+
#
|
|
211
|
+
# @param tags [Array<String>] Labels for tracking request usage. Not retained when zdr is enabled.
|
|
212
|
+
#
|
|
213
|
+
# @param timeout_opts [ContextDev::Models::WebScrapeParams::TimeoutOpts] Total deadline, including navigation, actions, waiting, and all outputs. Default
|
|
214
|
+
#
|
|
215
|
+
# @param zdr [Symbol, ContextDev::Models::WebScrapeParams::Zdr] Zero data retention. Bypasses caches and uploads; excludes request/response cont
|
|
216
|
+
#
|
|
217
|
+
# @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
|
|
218
|
+
#
|
|
219
|
+
# @return [ContextDev::Models::WebScrapeResponse]
|
|
220
|
+
#
|
|
221
|
+
# @see ContextDev::Models::WebScrapeParams
|
|
222
|
+
def scrape(params)
|
|
223
|
+
parsed, options = ContextDev::WebScrapeParams.dump_request(params)
|
|
224
|
+
@client.request(
|
|
225
|
+
method: :post,
|
|
226
|
+
path: "web/scrape",
|
|
227
|
+
body: parsed,
|
|
228
|
+
model: ContextDev::Models::WebScrapeResponse,
|
|
209
229
|
options: options
|
|
210
230
|
)
|
|
211
231
|
end
|
|
@@ -215,7 +235,7 @@ module ContextDev
|
|
|
215
235
|
#
|
|
216
236
|
# Capture a screenshot of a website.
|
|
217
237
|
#
|
|
218
|
-
# @overload screenshot(clear_popups: nil, color_scheme: nil, country: nil, direct_url: nil, domain: nil, full_screenshot: nil, handle_cookie_popup: nil, max_age_ms: nil, page: nil, scroll_offset: nil, tags: nil, timeout_opts: nil, viewport: nil, wait_for_ms: nil, zdr: nil, request_options: {})
|
|
238
|
+
# @overload screenshot(clear_popups: nil, color_scheme: nil, country: nil, direct_url: nil, domain: nil, full_screenshot: nil, handle_cookie_popup: nil, headers: nil, max_age_ms: nil, page: nil, scroll_offset: nil, tags: nil, timeout_opts: nil, viewport: nil, wait_for_ms: nil, zdr: nil, request_options: {})
|
|
219
239
|
#
|
|
220
240
|
# @param clear_popups [Boolean] Optional parameter for comprehensive popup cleanup. If 'true', the browser dismi
|
|
221
241
|
#
|
|
@@ -231,6 +251,8 @@ module ContextDev
|
|
|
231
251
|
#
|
|
232
252
|
# @param handle_cookie_popup [Boolean] Optional parameter to control cookie/consent popup handling. If 'true', we dismi
|
|
233
253
|
#
|
|
254
|
+
# @param headers [Hash{Symbol=>String}] Optional outbound HTTP headers, using the same JSON object or deep-object query
|
|
255
|
+
#
|
|
234
256
|
# @param max_age_ms [Integer, nil] Return a cached screenshot if a prior screenshot for the same parameters exists
|
|
235
257
|
#
|
|
236
258
|
# @param page [Symbol, ContextDev::Models::WebScreenshotParams::Page] Optional parameter to specify which page type to screenshot. If provided, the sy
|
|
@@ -279,7 +301,7 @@ module ContextDev
|
|
|
279
301
|
#
|
|
280
302
|
# Search the web and optionally scrape each result to Markdown in one round-trip.
|
|
281
303
|
#
|
|
282
|
-
# @overload search(query:, country: nil, exclude_domains: nil, freshness: nil, include_domains: nil, markdown_options: nil, num_results: nil, query_fanout: nil, tags: nil, timeout_opts: nil, request_options: {})
|
|
304
|
+
# @overload search(query:, country: nil, exclude_domains: nil, freshness: nil, include_domains: nil, markdown_options: nil, num_results: nil, query_fanout: nil, tags: nil, timeout_opts: nil, zdr: nil, request_options: {})
|
|
283
305
|
#
|
|
284
306
|
# @param query [String] Search query. Accepts natural language as well as Google-style search operators
|
|
285
307
|
#
|
|
@@ -301,6 +323,8 @@ module ContextDev
|
|
|
301
323
|
#
|
|
302
324
|
# @param timeout_opts [ContextDev::Models::WebSearchParams::TimeoutOpts] Optional request deadline and behavior on timeout. For GET requests, use timeout
|
|
303
325
|
#
|
|
326
|
+
# @param zdr [Symbol, ContextDev::Models::WebSearchParams::Zdr] Set to enabled to bypass shared caches and omit request and response content fro
|
|
327
|
+
#
|
|
304
328
|
# @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
|
|
305
329
|
#
|
|
306
330
|
# @return [ContextDev::Models::WebSearchResponse]
|
|
@@ -383,327 +407,6 @@ module ContextDev
|
|
|
383
407
|
)
|
|
384
408
|
end
|
|
385
409
|
|
|
386
|
-
# Some parameter documentations has been truncated, see
|
|
387
|
-
# {ContextDev::Models::WebWebScrapeBytesParams} for more details.
|
|
388
|
-
#
|
|
389
|
-
# Downloads a resource and returns its bytes as base64. Supports images, PDFs,
|
|
390
|
-
# HTML pages, and any other content type without image conversion, text
|
|
391
|
-
# extraction, or character-encoding changes. HTTP compression is decoded before
|
|
392
|
-
# base64 encoding. HTML is the original HTTP response; JavaScript is not rendered.
|
|
393
|
-
# Follows public redirects and retries failed downloads through ISP and
|
|
394
|
-
# residential proxies, with a direct fallback. When country is specified, only a
|
|
395
|
-
# residential proxy in that country is used. Supply headers such as Referer for
|
|
396
|
-
# images that require a referring page. Downloads are not cached. Maximum decoded
|
|
397
|
-
# resource size: 20 MiB (20971520 bytes), before base64 encoding. Successful
|
|
398
|
-
# requests cost 1 credit; errors are not billed.
|
|
399
|
-
#
|
|
400
|
-
# @overload web_scrape_bytes(url:, country: nil, headers: nil, tags: nil, timeout_opts: nil, zdr: nil, request_options: {})
|
|
401
|
-
#
|
|
402
|
-
# @param url [String] Full HTTP(S) URL of the resource to download, such as an image, PDF, or page.
|
|
403
|
-
#
|
|
404
|
-
# @param country [Symbol, ContextDev::Models::WebWebScrapeBytesParams::Country] Fetch the target page through a residential proxy in this country (ISO 3166-1 al
|
|
405
|
-
#
|
|
406
|
-
# @param headers [Hash{Symbol=>String}] Optional outbound HTTP headers, such as Referer, Cookie, or Authorization. Send
|
|
407
|
-
#
|
|
408
|
-
# @param tags [Array<String>] Comma-separated tags for tracking request usage. Up to 20 tags, each 1-50 charac
|
|
409
|
-
#
|
|
410
|
-
# @param timeout_opts [ContextDev::Models::WebWebScrapeBytesParams::TimeoutOpts] Optional request deadline and behavior on timeout. For GET requests, use timeout
|
|
411
|
-
#
|
|
412
|
-
# @param zdr [Symbol, ContextDev::Models::WebWebScrapeBytesParams::Zdr] Set to enabled to bypass shared caches and omit request and response content fro
|
|
413
|
-
#
|
|
414
|
-
# @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
|
|
415
|
-
#
|
|
416
|
-
# @return [ContextDev::Models::WebWebScrapeBytesResponse]
|
|
417
|
-
#
|
|
418
|
-
# @see ContextDev::Models::WebWebScrapeBytesParams
|
|
419
|
-
def web_scrape_bytes(params)
|
|
420
|
-
parsed, options = ContextDev::WebWebScrapeBytesParams.dump_request(params)
|
|
421
|
-
query = ContextDev::Internal::Util.encode_query_params(parsed)
|
|
422
|
-
@client.request(
|
|
423
|
-
method: :get,
|
|
424
|
-
path: "web/scrape/bytes",
|
|
425
|
-
query: query.transform_keys(timeout_opts: "timeoutOpts"),
|
|
426
|
-
model: ContextDev::Models::WebWebScrapeBytesResponse,
|
|
427
|
-
options: options
|
|
428
|
-
)
|
|
429
|
-
end
|
|
430
|
-
|
|
431
|
-
# Some parameter documentations has been truncated, see
|
|
432
|
-
# {ContextDev::Models::WebWebScrapeHTMLParams} for more details.
|
|
433
|
-
#
|
|
434
|
-
# Scrapes the given URL and returns the raw HTML content of the page. The base
|
|
435
|
-
# request costs 1 credit; requests with browser actions cost 2 credits. A request
|
|
436
|
-
# that hits its timeoutOpts.milliseconds deadline fails with 408 and is not
|
|
437
|
-
# billed, unless timeoutOpts.behavior=return-partial is set — then the page as
|
|
438
|
-
# rendered so far is returned with `finalDOMState: "still-loading"` and billed at
|
|
439
|
-
# the base cost of 1 credit.
|
|
440
|
-
#
|
|
441
|
-
# @overload web_scrape_html(url:, actions: nil, country: nil, exclude_selectors: nil, headers: nil, include_frames: nil, include_selectors: nil, max_age_ms: nil, pdf: nil, settle_animations: nil, tags: nil, timeout_opts: nil, use_main_content_only: nil, wait_for_ms: nil, zdr: nil, request_options: {})
|
|
442
|
-
#
|
|
443
|
-
# @param url [String] Full URL to scrape (must include http:// or https:// protocol)
|
|
444
|
-
#
|
|
445
|
-
# @param actions [Array<ContextDev::Models::WebWebScrapeHTMLParams::Action::Wait, ContextDev::Models::WebWebScrapeHTMLParams::Action::Perform, ContextDev::Models::WebWebScrapeHTMLParams::Action::Scroll>, nil] Optional browser actions executed in array order after the page loads and before
|
|
446
|
-
#
|
|
447
|
-
# @param country [Symbol, ContextDev::Models::WebWebScrapeHTMLParams::Country] Fetch the target page through a residential proxy in this country (ISO 3166-1 al
|
|
448
|
-
#
|
|
449
|
-
# @param exclude_selectors [Array<String>, nil] CSS selectors to remove from the result. Applied after includeSelectors. Exclusi
|
|
450
|
-
#
|
|
451
|
-
# @param headers [Hash{Symbol=>String}] Optional outbound HTTP headers forwarded only to the target URL, sent as deep-ob
|
|
452
|
-
#
|
|
453
|
-
# @param include_frames [Boolean] When true, iframes are rendered inline into the returned HTML.
|
|
454
|
-
#
|
|
455
|
-
# @param include_selectors [Array<String>, nil] CSS selectors. When provided, only matching subtrees (and their descendants) are
|
|
456
|
-
#
|
|
457
|
-
# @param max_age_ms [Integer, nil] Return a cached result if a prior scrape for the same parameters exists and is y
|
|
458
|
-
#
|
|
459
|
-
# @param pdf [ContextDev::Models::WebWebScrapeHTMLParams::Pdf] PDF parsing controls. Use start/end to limit text extraction and embedded-image
|
|
460
|
-
#
|
|
461
|
-
# @param settle_animations [Boolean] When true, waits briefly for CSS and transition animations to settle before extr
|
|
462
|
-
#
|
|
463
|
-
# @param tags [Array<String>] Comma-separated tags for tracking request usage. Up to 20 tags, each 1-50 charac
|
|
464
|
-
#
|
|
465
|
-
# @param timeout_opts [ContextDev::Models::WebWebScrapeHTMLParams::TimeoutOpts] Optional request deadline and behavior on timeout. For GET requests, use timeout
|
|
466
|
-
#
|
|
467
|
-
# @param use_main_content_only [Boolean] When true, return only the page's main content in the HTML response, excluding h
|
|
468
|
-
#
|
|
469
|
-
# @param wait_for_ms [Integer, nil] Optional browser wait time in milliseconds after initial page load. Min: 0. Max:
|
|
470
|
-
#
|
|
471
|
-
# @param zdr [Symbol, ContextDev::Models::WebWebScrapeHTMLParams::Zdr] Set to enabled to bypass shared caches and omit request and response content fro
|
|
472
|
-
#
|
|
473
|
-
# @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
|
|
474
|
-
#
|
|
475
|
-
# @return [ContextDev::Models::WebWebScrapeHTMLResponse]
|
|
476
|
-
#
|
|
477
|
-
# @see ContextDev::Models::WebWebScrapeHTMLParams
|
|
478
|
-
def web_scrape_html(params)
|
|
479
|
-
parsed, options = ContextDev::WebWebScrapeHTMLParams.dump_request(params)
|
|
480
|
-
query = ContextDev::Internal::Util.encode_query_params(parsed)
|
|
481
|
-
@client.request(
|
|
482
|
-
method: :get,
|
|
483
|
-
path: "web/scrape/html",
|
|
484
|
-
query: query.transform_keys(
|
|
485
|
-
exclude_selectors: "excludeSelectors",
|
|
486
|
-
include_frames: "includeFrames",
|
|
487
|
-
include_selectors: "includeSelectors",
|
|
488
|
-
max_age_ms: "maxAgeMs",
|
|
489
|
-
settle_animations: "settleAnimations",
|
|
490
|
-
timeout_opts: "timeoutOpts",
|
|
491
|
-
use_main_content_only: "useMainContentOnly",
|
|
492
|
-
wait_for_ms: "waitForMs"
|
|
493
|
-
),
|
|
494
|
-
model: ContextDev::Models::WebWebScrapeHTMLResponse,
|
|
495
|
-
options: options
|
|
496
|
-
)
|
|
497
|
-
end
|
|
498
|
-
|
|
499
|
-
# Some parameter documentations has been truncated, see
|
|
500
|
-
# {ContextDev::Models::WebWebScrapeImagesParams} for more details.
|
|
501
|
-
#
|
|
502
|
-
# Extract image assets from a web page, including standard URLs, inline SVGs, data
|
|
503
|
-
# URIs, responsive image sources, metadata, CSS backgrounds, video posters, and
|
|
504
|
-
# embeds. The base request costs 1 credit, or 2 credits with browser actions. When
|
|
505
|
-
# enrichment is enabled, the entire call costs 5 credits, including requests that
|
|
506
|
-
# also use actions.
|
|
507
|
-
#
|
|
508
|
-
# @overload web_scrape_images(url:, actions: nil, dedupe: nil, enrichment: nil, headers: nil, max_age_ms: nil, tags: nil, timeout_opts: nil, wait_for_ms: nil, request_options: {})
|
|
509
|
-
#
|
|
510
|
-
# @param url [String] Page URL to inspect. Must include http:// or https://.
|
|
511
|
-
#
|
|
512
|
-
# @param actions [Array<ContextDev::Models::WebWebScrapeImagesParams::Action::Wait, ContextDev::Models::WebWebScrapeImagesParams::Action::Perform, ContextDev::Models::WebWebScrapeImagesParams::Action::Scroll>, nil] Optional browser actions executed in array order after the page loads and before
|
|
513
|
-
#
|
|
514
|
-
# @param dedupe [Boolean] When true, visually duplicate images are removed: every image is loaded and perc
|
|
515
|
-
#
|
|
516
|
-
# @param enrichment [ContextDev::Models::WebWebScrapeImagesParams::Enrichment, nil] Optional per-image processing, sent as deep-object query params such as enrichme
|
|
517
|
-
#
|
|
518
|
-
# @param headers [Hash{Symbol=>String}] Optional outbound HTTP headers forwarded only to the target URL, sent as deep-ob
|
|
519
|
-
#
|
|
520
|
-
# @param max_age_ms [Integer, nil] Reuse a cached result this many milliseconds old or newer. Default: 86400000 (1
|
|
521
|
-
#
|
|
522
|
-
# @param tags [Array<String>] Comma-separated tags for tracking request usage. Up to 20 tags, each 1-50 charac
|
|
523
|
-
#
|
|
524
|
-
# @param timeout_opts [ContextDev::Models::WebWebScrapeImagesParams::TimeoutOpts] Optional request deadline and behavior on timeout. For GET requests, use timeout
|
|
525
|
-
#
|
|
526
|
-
# @param wait_for_ms [Integer, nil] Optional browser wait time in milliseconds after initial page load before collec
|
|
527
|
-
#
|
|
528
|
-
# @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
|
|
529
|
-
#
|
|
530
|
-
# @return [ContextDev::Models::WebWebScrapeImagesResponse]
|
|
531
|
-
#
|
|
532
|
-
# @see ContextDev::Models::WebWebScrapeImagesParams
|
|
533
|
-
def web_scrape_images(params)
|
|
534
|
-
parsed, options = ContextDev::WebWebScrapeImagesParams.dump_request(params)
|
|
535
|
-
query = ContextDev::Internal::Util.encode_query_params(parsed)
|
|
536
|
-
@client.request(
|
|
537
|
-
method: :get,
|
|
538
|
-
path: "web/scrape/images",
|
|
539
|
-
query: query.transform_keys(
|
|
540
|
-
max_age_ms: "maxAgeMs",
|
|
541
|
-
timeout_opts: "timeoutOpts",
|
|
542
|
-
wait_for_ms: "waitForMs"
|
|
543
|
-
),
|
|
544
|
-
model: ContextDev::Models::WebWebScrapeImagesResponse,
|
|
545
|
-
options: options
|
|
546
|
-
)
|
|
547
|
-
end
|
|
548
|
-
|
|
549
|
-
# Some parameter documentations has been truncated, see
|
|
550
|
-
# {ContextDev::Models::WebWebScrapeMdParams} for more details.
|
|
551
|
-
#
|
|
552
|
-
# Scrapes the given URL into LLM usable Markdown. Inspect key_metadata on JSON
|
|
553
|
-
# responses from a recognized API key; use error_code to distinguish stable
|
|
554
|
-
# failure categories.
|
|
555
|
-
#
|
|
556
|
-
# ### YouTube
|
|
557
|
-
#
|
|
558
|
-
# YouTube URLs return the video or channel itself rather than the surrounding
|
|
559
|
-
# player and navigation chrome. A URL addressing a single video (`/watch`,
|
|
560
|
-
# `youtu.be`, `/shorts`, `/embed`, `/live`) returns its title, channel, duration,
|
|
561
|
-
# view count, keywords, full description, and the transcript when the video has
|
|
562
|
-
# captions that can be retrieved; videos without captions return everything except
|
|
563
|
-
# the transcript. A channel URL (`/channel/UC…`, `/@handle`, `/c/…`, `/user/…`)
|
|
564
|
-
# returns its name, handle, subscriber count, video count, and full description.
|
|
565
|
-
# When `includeImages=true`, video responses also include the thumbnail and
|
|
566
|
-
# channel responses include the avatar. Costs the same as any other scrape.
|
|
567
|
-
#
|
|
568
|
-
# ### Billing & errors
|
|
569
|
-
#
|
|
570
|
-
# | HTTP status | Billed? | Meaning |
|
|
571
|
-
# | ----------- | ----------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
|
572
|
-
# | 200 | Yes — 1 credit, or 2 credits with actions | Successful scrape, including a zero-length result when includeSelectors matched nothing. A partial result (`finalDOMState: "still-loading"`, only with timeoutOpts.behavior=return-partial) is billed at the base 1 credit with no OCR or actions surcharge |
|
|
573
|
-
# | 400 | No | Invalid input, skipped PDF, or the page could not be scraped. error_code WEBSITE_BLOCKED specifically means the site answered with an anti-bot challenge, CAPTCHA wall, or login shell instead of the page (even when the site returned HTTP 200) — retrying later or from another country sometimes succeeds |
|
|
574
|
-
# | 401 / 403 | No | Invalid/disabled key, insufficient permissions, or credits exhausted; inspect error_code |
|
|
575
|
-
# | 404 | No | Target page returned or fingerprinted as not found |
|
|
576
|
-
# | 408 | No | Request timed out. With timeoutOpts.behavior=return-partial this only happens when nothing usable had rendered by the deadline |
|
|
577
|
-
# | 413 | No | Target content exceeds the maximum supported size (20 MB) |
|
|
578
|
-
# | 415 | No | Unsupported content type |
|
|
579
|
-
# | 429 | No | Per-minute rate limit exceeded; honor Retry-After |
|
|
580
|
-
# | 500 | No | Internal error |
|
|
581
|
-
#
|
|
582
|
-
# @overload web_scrape_md(url:, actions: nil, country: nil, exclude_selectors: nil, headers: nil, include_frames: nil, include_html: nil, include_images: nil, include_links: nil, include_selectors: nil, max_age_ms: nil, pdf: nil, settle_animations: nil, shorten_base64_images: nil, tags: nil, timeout_opts: nil, use_main_content_only: nil, wait_for_ms: nil, zdr: nil, request_options: {})
|
|
583
|
-
#
|
|
584
|
-
# @param url [String] Full URL to scrape into LLM usable Markdown (must include http:// or https:// pr
|
|
585
|
-
#
|
|
586
|
-
# @param actions [Array<ContextDev::Models::WebWebScrapeMdParams::Action::Wait, ContextDev::Models::WebWebScrapeMdParams::Action::Perform, ContextDev::Models::WebWebScrapeMdParams::Action::Scroll>, nil] Optional browser actions executed in array order after the page loads and before
|
|
587
|
-
#
|
|
588
|
-
# @param country [Symbol, ContextDev::Models::WebWebScrapeMdParams::Country] Fetch the target page through a residential proxy in this country (ISO 3166-1 al
|
|
589
|
-
#
|
|
590
|
-
# @param exclude_selectors [Array<String>, nil] CSS selectors to remove before conversion to Markdown. Applied after includeSele
|
|
591
|
-
#
|
|
592
|
-
# @param headers [Hash{Symbol=>String}] Optional outbound HTTP headers forwarded only to the target URL, sent as deep-ob
|
|
593
|
-
#
|
|
594
|
-
# @param include_frames [Boolean] When true, the contents of iframes are rendered to Markdown.
|
|
595
|
-
#
|
|
596
|
-
# @param include_html [Boolean] When true, the response also includes an `html` field with the page HTML the Mar
|
|
597
|
-
#
|
|
598
|
-
# @param include_images [Boolean] Include image references in Markdown output
|
|
599
|
-
#
|
|
600
|
-
# @param include_links [Boolean] Preserve hyperlinks in Markdown output
|
|
601
|
-
#
|
|
602
|
-
# @param include_selectors [Array<String>, nil] CSS selectors. When provided, only matching HTML subtrees (and their descendants
|
|
603
|
-
#
|
|
604
|
-
# @param max_age_ms [Integer, nil] Return a cached result if a prior scrape for the same parameters exists and is y
|
|
605
|
-
#
|
|
606
|
-
# @param pdf [ContextDev::Models::WebWebScrapeMdParams::Pdf] PDF parsing controls. Use start/end to limit text extraction and embedded-image
|
|
607
|
-
#
|
|
608
|
-
# @param settle_animations [Boolean] When true, waits briefly for CSS and transition animations to settle before conv
|
|
609
|
-
#
|
|
610
|
-
# @param shorten_base64_images [Boolean] Shorten base64-encoded image data in the Markdown output
|
|
611
|
-
#
|
|
612
|
-
# @param tags [Array<String>] Comma-separated tags for tracking request usage. Up to 20 tags, each 1-50 charac
|
|
613
|
-
#
|
|
614
|
-
# @param timeout_opts [ContextDev::Models::WebWebScrapeMdParams::TimeoutOpts] Optional request deadline and behavior on timeout. For GET requests, use timeout
|
|
615
|
-
#
|
|
616
|
-
# @param use_main_content_only [Boolean] Extract only the main content of the page, excluding headers, footers, sidebars,
|
|
617
|
-
#
|
|
618
|
-
# @param wait_for_ms [Integer, nil] Optional browser wait time in milliseconds after initial page load before conver
|
|
619
|
-
#
|
|
620
|
-
# @param zdr [Symbol, ContextDev::Models::WebWebScrapeMdParams::Zdr] Set to enabled to bypass shared caches and omit request and response content fro
|
|
621
|
-
#
|
|
622
|
-
# @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
|
|
623
|
-
#
|
|
624
|
-
# @return [ContextDev::Models::WebWebScrapeMdResponse]
|
|
625
|
-
#
|
|
626
|
-
# @see ContextDev::Models::WebWebScrapeMdParams
|
|
627
|
-
def web_scrape_md(params)
|
|
628
|
-
parsed, options = ContextDev::WebWebScrapeMdParams.dump_request(params)
|
|
629
|
-
query = ContextDev::Internal::Util.encode_query_params(parsed)
|
|
630
|
-
@client.request(
|
|
631
|
-
method: :get,
|
|
632
|
-
path: "web/scrape/markdown",
|
|
633
|
-
query: query.transform_keys(
|
|
634
|
-
exclude_selectors: "excludeSelectors",
|
|
635
|
-
include_frames: "includeFrames",
|
|
636
|
-
include_html: "includeHTML",
|
|
637
|
-
include_images: "includeImages",
|
|
638
|
-
include_links: "includeLinks",
|
|
639
|
-
include_selectors: "includeSelectors",
|
|
640
|
-
max_age_ms: "maxAgeMs",
|
|
641
|
-
settle_animations: "settleAnimations",
|
|
642
|
-
shorten_base64_images: "shortenBase64Images",
|
|
643
|
-
timeout_opts: "timeoutOpts",
|
|
644
|
-
use_main_content_only: "useMainContentOnly",
|
|
645
|
-
wait_for_ms: "waitForMs"
|
|
646
|
-
),
|
|
647
|
-
model: ContextDev::Models::WebWebScrapeMdResponse,
|
|
648
|
-
options: options
|
|
649
|
-
)
|
|
650
|
-
end
|
|
651
|
-
|
|
652
|
-
# Some parameter documentations has been truncated, see
|
|
653
|
-
# {ContextDev::Models::WebWebScrapeSitemapParams} for more details.
|
|
654
|
-
#
|
|
655
|
-
# Crawl an entire website's sitemap and return all discovered page URLs. Set
|
|
656
|
-
# `includeSubdomains=true` to also discover public pages and sitemaps on child
|
|
657
|
-
# hosts such as `docs.example.com` or `brand.example.com`. Pass `search` to have
|
|
658
|
-
# the discovered URLs filtered down to the pages about a phrase (for example
|
|
659
|
-
# `pricing and plans` or `api authentication docs`), most relevant first — a
|
|
660
|
-
# searched crawl scans the whole sitemap and costs 2 credits instead of 1.
|
|
661
|
-
#
|
|
662
|
-
# @overload web_scrape_sitemap(domain:, headers: nil, include_subdomains: nil, max_links: nil, search: nil, sitemap_url: nil, tags: nil, timeout_opts: nil, url_regex: nil, zdr: nil, request_options: {})
|
|
663
|
-
#
|
|
664
|
-
# @param domain [String] Domain to build a sitemap for
|
|
665
|
-
#
|
|
666
|
-
# @param headers [Hash{Symbol=>String}] Optional outbound HTTP headers forwarded only to the target URL, sent as deep-ob
|
|
667
|
-
#
|
|
668
|
-
# @param include_subdomains [Boolean] When true, discover and include public pages and sitemaps on subdomains of the r
|
|
669
|
-
#
|
|
670
|
-
# @param max_links [Integer] Maximum number of links to return from the sitemap crawl. Defaults to 10,000. Mi
|
|
671
|
-
#
|
|
672
|
-
# @param search [String] Optional search phrase. When provided, the crawled sitemap is filtered to the pa
|
|
673
|
-
#
|
|
674
|
-
# @param sitemap_url [String] Optional explicit sitemap URL. When provided, exactly this sitemap is crawled in
|
|
675
|
-
#
|
|
676
|
-
# @param tags [Array<String>] Comma-separated tags for tracking request usage. Up to 20 tags, each 1-50 charac
|
|
677
|
-
#
|
|
678
|
-
# @param timeout_opts [ContextDev::Models::WebWebScrapeSitemapParams::TimeoutOpts] Optional request deadline and behavior on timeout. For GET requests, use timeout
|
|
679
|
-
#
|
|
680
|
-
# @param url_regex [String] Optional RE2-compatible regex pattern. Only URLs matching this pattern are retur
|
|
681
|
-
#
|
|
682
|
-
# @param zdr [Symbol, ContextDev::Models::WebWebScrapeSitemapParams::Zdr] Set to enabled to bypass shared caches and omit request and response content fro
|
|
683
|
-
#
|
|
684
|
-
# @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
|
|
685
|
-
#
|
|
686
|
-
# @return [ContextDev::Models::WebWebScrapeSitemapResponse]
|
|
687
|
-
#
|
|
688
|
-
# @see ContextDev::Models::WebWebScrapeSitemapParams
|
|
689
|
-
def web_scrape_sitemap(params)
|
|
690
|
-
parsed, options = ContextDev::WebWebScrapeSitemapParams.dump_request(params)
|
|
691
|
-
query = ContextDev::Internal::Util.encode_query_params(parsed)
|
|
692
|
-
@client.request(
|
|
693
|
-
method: :get,
|
|
694
|
-
path: "web/scrape/sitemap",
|
|
695
|
-
query: query.transform_keys(
|
|
696
|
-
include_subdomains: "includeSubdomains",
|
|
697
|
-
max_links: "maxLinks",
|
|
698
|
-
sitemap_url: "sitemapUrl",
|
|
699
|
-
timeout_opts: "timeoutOpts",
|
|
700
|
-
url_regex: "urlRegex"
|
|
701
|
-
),
|
|
702
|
-
model: ContextDev::Models::WebWebScrapeSitemapResponse,
|
|
703
|
-
options: options
|
|
704
|
-
)
|
|
705
|
-
end
|
|
706
|
-
|
|
707
410
|
# @api private
|
|
708
411
|
#
|
|
709
412
|
# @param client [ContextDev::Client]
|