context.dev 2.17.1 → 2.19.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +33 -0
- data/README.md +1 -1
- data/lib/context_dev/client.rb +0 -4
- data/lib/context_dev/models/industry_retrieve_naics_params.rb +28 -1
- data/lib/context_dev/models/industry_retrieve_sic_params.rb +28 -1
- data/lib/context_dev/models/monitor_retrieve_run_params.rb +26 -0
- data/lib/context_dev/models/monitor_retrieve_run_response.rb +232 -0
- data/lib/context_dev/models/monitor_rotate_webhook_secret_params.rb +20 -0
- data/lib/context_dev/models/monitor_rotate_webhook_secret_response.rb +765 -0
- data/lib/context_dev/models/parse_handle_params.rb +8 -6
- data/lib/context_dev/models/person_enrich_params.rb +28 -1
- data/lib/context_dev/models/web_answers_params.rb +28 -1
- data/lib/context_dev/models/web_extract_competitors_params.rb +28 -1
- data/lib/context_dev/models/web_extract_styleguide_params.rb +30 -3
- data/lib/context_dev/models/{web_web_scrape_sitemap_params.rb → web_map_urls_params.rb} +22 -20
- data/lib/context_dev/models/web_map_urls_response.rb +123 -0
- data/lib/context_dev/models/web_scrape_params.rb +906 -0
- data/lib/context_dev/models/web_scrape_response.rb +659 -0
- data/lib/context_dev/models/web_screenshot_params.rb +25 -9
- data/lib/context_dev/models/web_screenshot_response.rb +3 -3
- data/lib/context_dev/models/web_search_params.rb +30 -3
- data/lib/context_dev/models.rb +8 -20
- data/lib/context_dev/resources/brand.rb +0 -36
- data/lib/context_dev/resources/industry.rb +6 -2
- data/lib/context_dev/resources/monitors.rb +47 -0
- data/lib/context_dev/resources/people.rb +3 -1
- data/lib/context_dev/resources/web.rb +116 -413
- data/lib/context_dev/version.rb +1 -1
- data/lib/context_dev.rb +8 -21
- data/rbi/context_dev/client.rbi +0 -3
- data/rbi/context_dev/models/industry_retrieve_naics_params.rbi +59 -0
- data/rbi/context_dev/models/industry_retrieve_sic_params.rbi +57 -0
- data/rbi/context_dev/models/monitor_retrieve_run_params.rbi +46 -0
- data/rbi/context_dev/models/monitor_retrieve_run_response.rbi +452 -0
- data/rbi/context_dev/models/monitor_rotate_webhook_secret_params.rbi +38 -0
- data/rbi/context_dev/models/monitor_rotate_webhook_secret_response.rbi +1367 -0
- data/rbi/context_dev/models/parse_handle_params.rbi +12 -9
- data/rbi/context_dev/models/person_enrich_params.rbi +45 -0
- data/rbi/context_dev/models/web_answers_params.rbi +45 -0
- data/rbi/context_dev/models/web_extract_competitors_params.rbi +59 -0
- data/rbi/context_dev/models/web_extract_styleguide_params.rbi +62 -3
- data/rbi/context_dev/models/{web_web_scrape_sitemap_params.rbi → web_map_urls_params.rbi} +35 -54
- data/rbi/context_dev/models/web_map_urls_response.rbi +226 -0
- data/rbi/context_dev/models/web_scrape_params.rbi +2000 -0
- data/rbi/context_dev/models/{web_web_scrape_html_response.rbi → web_scrape_response.rbi} +550 -418
- data/rbi/context_dev/models/web_screenshot_params.rbi +38 -12
- data/rbi/context_dev/models/web_screenshot_response.rbi +4 -4
- data/rbi/context_dev/models/web_search_params.rbi +48 -3
- data/rbi/context_dev/models.rbi +9 -21
- data/rbi/context_dev/resources/brand.rbi +0 -35
- data/rbi/context_dev/resources/industry.rbi +14 -0
- data/rbi/context_dev/resources/monitors.rbi +24 -0
- data/rbi/context_dev/resources/parse.rbi +4 -4
- data/rbi/context_dev/resources/people.rbi +7 -0
- data/rbi/context_dev/resources/web.rbi +166 -533
- data/sig/context_dev/client.rbs +0 -2
- data/sig/context_dev/models/industry_retrieve_naics_params.rbs +21 -1
- data/sig/context_dev/models/industry_retrieve_sic_params.rbs +21 -1
- data/sig/context_dev/models/monitor_retrieve_run_params.rbs +28 -0
- data/sig/context_dev/models/monitor_retrieve_run_response.rbs +182 -0
- data/sig/context_dev/models/monitor_rotate_webhook_secret_params.rbs +23 -0
- data/sig/context_dev/models/monitor_rotate_webhook_secret_response.rbs +545 -0
- data/sig/context_dev/models/person_enrich_params.rbs +21 -1
- data/sig/context_dev/models/web_answers_params.rbs +21 -1
- data/sig/context_dev/models/web_extract_competitors_params.rbs +21 -1
- data/sig/context_dev/models/web_extract_styleguide_params.rbs +21 -1
- data/sig/context_dev/models/{web_web_scrape_sitemap_params.rbs → web_map_urls_params.rbs} +22 -22
- data/sig/context_dev/models/web_map_urls_response.rbs +125 -0
- data/sig/context_dev/models/web_scrape_params.rbs +834 -0
- data/sig/context_dev/models/web_scrape_response.rbs +552 -0
- data/sig/context_dev/models/web_screenshot_params.rbs +7 -0
- data/sig/context_dev/models/web_search_params.rbs +21 -1
- data/sig/context_dev/models.rbs +8 -20
- data/sig/context_dev/resources/brand.rbs +0 -9
- data/sig/context_dev/resources/industry.rbs +2 -0
- data/sig/context_dev/resources/monitors.rbs +11 -0
- data/sig/context_dev/resources/people.rbs +1 -0
- data/sig/context_dev/resources/web.rbs +34 -108
- metadata +26 -65
- data/lib/context_dev/models/ai_extract_product_params.rb +0 -98
- data/lib/context_dev/models/ai_extract_product_response.rb +0 -352
- data/lib/context_dev/models/ai_extract_products_params.rb +0 -233
- data/lib/context_dev/models/ai_extract_products_response.rb +0 -320
- data/lib/context_dev/models/brand_retrieve_simplified_params.rb +0 -120
- data/lib/context_dev/models/brand_retrieve_simplified_response.rb +0 -441
- data/lib/context_dev/models/web_extract_fonts_params.rb +0 -114
- data/lib/context_dev/models/web_extract_fonts_response.rb +0 -291
- data/lib/context_dev/models/web_extract_params.rb +0 -392
- data/lib/context_dev/models/web_extract_response.rb +0 -287
- data/lib/context_dev/models/web_web_scrape_bytes_params.rb +0 -344
- data/lib/context_dev/models/web_web_scrape_bytes_response.rb +0 -135
- data/lib/context_dev/models/web_web_scrape_html_params.rb +0 -633
- data/lib/context_dev/models/web_web_scrape_html_response.rb +0 -588
- data/lib/context_dev/models/web_web_scrape_images_params.rb +0 -346
- data/lib/context_dev/models/web_web_scrape_images_response.rb +0 -409
- data/lib/context_dev/models/web_web_scrape_md_params.rb +0 -668
- data/lib/context_dev/models/web_web_scrape_md_response.rb +0 -567
- data/lib/context_dev/models/web_web_scrape_sitemap_response.rb +0 -144
- data/lib/context_dev/resources/ai.rb +0 -69
- data/rbi/context_dev/models/ai_extract_product_params.rbi +0 -199
- data/rbi/context_dev/models/ai_extract_product_response.rbi +0 -696
- data/rbi/context_dev/models/ai_extract_products_params.rbi +0 -490
- data/rbi/context_dev/models/ai_extract_products_response.rbi +0 -623
- data/rbi/context_dev/models/brand_retrieve_simplified_params.rbi +0 -256
- data/rbi/context_dev/models/brand_retrieve_simplified_response.rbi +0 -1040
- data/rbi/context_dev/models/web_extract_fonts_params.rbi +0 -223
- data/rbi/context_dev/models/web_extract_fonts_response.rbi +0 -563
- data/rbi/context_dev/models/web_extract_params.rbi +0 -736
- data/rbi/context_dev/models/web_extract_response.rbi +0 -512
- data/rbi/context_dev/models/web_web_scrape_bytes_params.rbi +0 -699
- data/rbi/context_dev/models/web_web_scrape_bytes_response.rbi +0 -238
- data/rbi/context_dev/models/web_web_scrape_html_params.rbi +0 -1217
- data/rbi/context_dev/models/web_web_scrape_images_params.rbi +0 -671
- data/rbi/context_dev/models/web_web_scrape_images_response.rbi +0 -910
- data/rbi/context_dev/models/web_web_scrape_md_params.rbi +0 -1049
- data/rbi/context_dev/models/web_web_scrape_md_response.rbi +0 -1088
- data/rbi/context_dev/models/web_web_scrape_sitemap_response.rbi +0 -258
- data/rbi/context_dev/resources/ai.rbi +0 -56
- data/sig/context_dev/models/ai_extract_product_params.rbs +0 -86
- data/sig/context_dev/models/ai_extract_product_response.rbs +0 -273
- data/sig/context_dev/models/ai_extract_products_params.rbs +0 -202
- data/sig/context_dev/models/ai_extract_products_response.rbs +0 -252
- data/sig/context_dev/models/brand_retrieve_simplified_params.rbs +0 -104
- data/sig/context_dev/models/brand_retrieve_simplified_response.rbs +0 -413
- data/sig/context_dev/models/web_extract_fonts_params.rbs +0 -93
- data/sig/context_dev/models/web_extract_fonts_response.rbs +0 -230
- data/sig/context_dev/models/web_extract_params.rbs +0 -304
- data/sig/context_dev/models/web_extract_response.rbs +0 -239
- data/sig/context_dev/models/web_web_scrape_bytes_params.rbs +0 -531
- data/sig/context_dev/models/web_web_scrape_bytes_response.rbs +0 -108
- data/sig/context_dev/models/web_web_scrape_html_params.rbs +0 -730
- data/sig/context_dev/models/web_web_scrape_html_response.rbs +0 -495
- data/sig/context_dev/models/web_web_scrape_images_params.rbs +0 -261
- data/sig/context_dev/models/web_web_scrape_images_response.rbs +0 -371
- data/sig/context_dev/models/web_web_scrape_md_params.rbs +0 -758
- data/sig/context_dev/models/web_web_scrape_md_response.rbs +0 -465
- data/sig/context_dev/models/web_web_scrape_sitemap_response.rbs +0 -117
- data/sig/context_dev/resources/ai.rbs +0 -20
|
@@ -15,6 +15,7 @@ module ContextDev
|
|
|
15
15
|
mode: ContextDev::WebAnswersParams::Mode::OrSymbol,
|
|
16
16
|
tags: T::Array[String],
|
|
17
17
|
timeout_opts: ContextDev::WebAnswersParams::TimeoutOpts::OrHash,
|
|
18
|
+
zdr: ContextDev::WebAnswersParams::Zdr::OrSymbol,
|
|
18
19
|
request_options: ContextDev::RequestOptions::OrHash
|
|
19
20
|
).returns(ContextDev::Models::WebAnswersResponse)
|
|
20
21
|
end
|
|
@@ -38,95 +39,12 @@ module ContextDev
|
|
|
38
39
|
# timeoutOpts[milliseconds]=30000&timeoutOpts[behavior]=fail or a JSON-encoded
|
|
39
40
|
# timeoutOpts object.
|
|
40
41
|
timeout_opts: nil,
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
sig do
|
|
48
|
-
params(
|
|
49
|
-
schema: T::Hash[Symbol, T.anything],
|
|
50
|
-
url: String,
|
|
51
|
-
actions:
|
|
52
|
-
T::Array[
|
|
53
|
-
T.any(
|
|
54
|
-
ContextDev::WebExtractParams::Action::Wait::OrHash,
|
|
55
|
-
ContextDev::WebExtractParams::Action::Perform::OrHash,
|
|
56
|
-
ContextDev::WebExtractParams::Action::Scroll::OrHash
|
|
57
|
-
)
|
|
58
|
-
],
|
|
59
|
-
fact_check: T::Boolean,
|
|
60
|
-
follow_subdomains: T::Boolean,
|
|
61
|
-
include_frames: T::Boolean,
|
|
62
|
-
instructions: String,
|
|
63
|
-
max_age_ms: Integer,
|
|
64
|
-
max_depth: Integer,
|
|
65
|
-
max_pages: Integer,
|
|
66
|
-
pdf: ContextDev::WebExtractParams::Pdf::OrHash,
|
|
67
|
-
settle_animations: T::Boolean,
|
|
68
|
-
stop_after_ms: Integer,
|
|
69
|
-
tags: T::Array[String],
|
|
70
|
-
timeout_opts: ContextDev::WebExtractParams::TimeoutOpts::OrHash,
|
|
71
|
-
wait_for_ms: Integer,
|
|
72
|
-
request_options: ContextDev::RequestOptions::OrHash
|
|
73
|
-
).returns(ContextDev::Models::WebExtractResponse)
|
|
74
|
-
end
|
|
75
|
-
def extract(
|
|
76
|
-
# JSON Schema for the returned data object. Image fields such as `image_urls` or
|
|
77
|
-
# `product_photos` automatically make page image references available to
|
|
78
|
-
# extraction, so product data and photos can be returned in one call. TypeScript
|
|
79
|
-
# Zod users can pass a JSON Schema generated from a Zod object; Python users can
|
|
80
|
-
# pass the equivalent JSON Schema object.
|
|
81
|
-
schema:,
|
|
82
|
-
# The starting website URL to crawl and extract from. Must include http:// or
|
|
83
|
-
# https://.
|
|
84
|
-
url:,
|
|
85
|
-
# Optional browser actions executed in order on the requested page after it loads,
|
|
86
|
-
# before links are discovered or additional pages are crawled. Requires a paid
|
|
87
|
-
# plan. When actions are provided and stopAfterMs is omitted, the crawl budget
|
|
88
|
-
# defaults to 110000 ms.
|
|
89
|
-
actions: nil,
|
|
90
|
-
# When true, every returned value must be grounded in facts stated on the page;
|
|
91
|
-
# fields that cannot be supported by the page are returned as null/empty. When
|
|
92
|
-
# false (default), the model may make reasonable inferences and derivations from
|
|
93
|
-
# the page content (e.g. ideal customer, competitor analysis, recommendations)
|
|
94
|
-
# while keeping verifiable specifics (names, quotes, URLs, dates, metrics)
|
|
95
|
-
# faithful to the source.
|
|
96
|
-
fact_check: nil,
|
|
97
|
-
# When true, follow links on subdomains of the starting URL's domain.
|
|
98
|
-
follow_subdomains: nil,
|
|
99
|
-
# When true, iframe contents are included in Markdown before extraction.
|
|
100
|
-
include_frames: nil,
|
|
101
|
-
# Optional extraction guidance, such as which facts to prioritize or how to
|
|
102
|
-
# interpret fields in the schema.
|
|
103
|
-
instructions: nil,
|
|
104
|
-
# Return cached scrape results if a prior scrape for the same parameters is
|
|
105
|
-
# younger than this many milliseconds. Defaults to 7 days (604800000 ms).
|
|
106
|
-
max_age_ms: nil,
|
|
107
|
-
# Optional maximum link depth from the starting URL (0 = only the starting page).
|
|
108
|
-
# If omitted, there is no crawl depth limit.
|
|
109
|
-
max_depth: nil,
|
|
110
|
-
# Maximum number of pages to analyze for extraction. Hard cap: 50. Defaults to 5.
|
|
111
|
-
max_pages: nil,
|
|
112
|
-
pdf: nil,
|
|
113
|
-
# When true, waits briefly for CSS and transition animations to settle before
|
|
114
|
-
# extracting each crawled page. Defaults to false. This adds a bit of latency in
|
|
115
|
-
# exchange for more stable output on animated pages.
|
|
116
|
-
settle_animations: nil,
|
|
117
|
-
# Soft time budget for the crawl in milliseconds. Min: 10000 (10s). Max: 110000
|
|
118
|
-
# (110s). Defaults to 80000 (80s), or 110000 (110s) when browser actions are
|
|
119
|
-
# provided.
|
|
120
|
-
stop_after_ms: nil,
|
|
121
|
-
# Optional tags for tracking usage. Up to 20 tags, each 1 to 50 characters.
|
|
122
|
-
tags: nil,
|
|
123
|
-
# Optional request deadline and behavior on timeout. For GET requests, use
|
|
124
|
-
# timeoutOpts[milliseconds]=30000&timeoutOpts[behavior]=fail or a JSON-encoded
|
|
125
|
-
# timeoutOpts object.
|
|
126
|
-
timeout_opts: nil,
|
|
127
|
-
# Optional browser wait time in milliseconds after initial page load for each
|
|
128
|
-
# crawled page.
|
|
129
|
-
wait_for_ms: nil,
|
|
42
|
+
# Set to enabled to bypass shared caches and omit request and response content
|
|
43
|
+
# from retained usage logs. Asset uploads are skipped, so hosted image URLs are
|
|
44
|
+
# omitted. Requires zero data retention to be enabled for your organization
|
|
45
|
+
# (contact support@context.dev), otherwise the request fails with ZDR_NOT_ENABLED.
|
|
46
|
+
# Successful ZDR responses include X-Context-ZDR: true.
|
|
47
|
+
zdr: nil,
|
|
130
48
|
request_options: {}
|
|
131
49
|
)
|
|
132
50
|
end
|
|
@@ -140,6 +58,7 @@ module ContextDev
|
|
|
140
58
|
tags: T::Array[String],
|
|
141
59
|
timeout_opts:
|
|
142
60
|
ContextDev::WebExtractCompetitorsParams::TimeoutOpts::OrHash,
|
|
61
|
+
zdr: ContextDev::WebExtractCompetitorsParams::Zdr::OrSymbol,
|
|
143
62
|
request_options: ContextDev::RequestOptions::OrHash
|
|
144
63
|
).returns(ContextDev::Models::WebExtractCompetitorsResponse)
|
|
145
64
|
end
|
|
@@ -156,28 +75,42 @@ module ContextDev
|
|
|
156
75
|
# timeoutOpts[milliseconds]=30000&timeoutOpts[behavior]=fail or a JSON-encoded
|
|
157
76
|
# timeoutOpts object.
|
|
158
77
|
timeout_opts: nil,
|
|
78
|
+
# Set to enabled to bypass shared caches and omit request and response content
|
|
79
|
+
# from retained usage logs. Asset uploads are skipped, so hosted image URLs are
|
|
80
|
+
# omitted. Requires zero data retention to be enabled for your organization
|
|
81
|
+
# (contact support@context.dev), otherwise the request fails with ZDR_NOT_ENABLED.
|
|
82
|
+
# Successful ZDR responses include X-Context-ZDR: true.
|
|
83
|
+
zdr: nil,
|
|
159
84
|
request_options: {}
|
|
160
85
|
)
|
|
161
86
|
end
|
|
162
87
|
|
|
163
|
-
#
|
|
164
|
-
#
|
|
88
|
+
# Extract a comprehensive design system from a website including colors,
|
|
89
|
+
# typography, spacing, shadows, and UI components.
|
|
165
90
|
sig do
|
|
166
91
|
params(
|
|
92
|
+
color_scheme:
|
|
93
|
+
ContextDev::WebExtractStyleguideParams::ColorScheme::OrSymbol,
|
|
167
94
|
direct_url: String,
|
|
168
95
|
domain: String,
|
|
169
96
|
max_age_ms: T.nilable(Integer),
|
|
170
97
|
tags: T::Array[String],
|
|
171
|
-
timeout_opts:
|
|
98
|
+
timeout_opts:
|
|
99
|
+
ContextDev::WebExtractStyleguideParams::TimeoutOpts::OrHash,
|
|
100
|
+
zdr: ContextDev::WebExtractStyleguideParams::Zdr::OrSymbol,
|
|
172
101
|
request_options: ContextDev::RequestOptions::OrHash
|
|
173
|
-
).returns(ContextDev::Models::
|
|
102
|
+
).returns(ContextDev::Models::WebExtractStyleguideResponse)
|
|
174
103
|
end
|
|
175
|
-
def
|
|
176
|
-
#
|
|
177
|
-
#
|
|
178
|
-
|
|
104
|
+
def extract_styleguide(
|
|
105
|
+
# Optional browser color scheme to emulate for websites that respond to
|
|
106
|
+
# prefers-color-scheme. This value is part of the styleguide cache key.
|
|
107
|
+
color_scheme: nil,
|
|
108
|
+
# A specific URL to fetch the styleguide from directly, bypassing domain
|
|
109
|
+
# resolution (e.g., 'https://example.com/design-system'). When provided, the
|
|
110
|
+
# styleguide is extracted from this exact URL. You must provide either 'domain' or
|
|
111
|
+
# 'directUrl', but not both.
|
|
179
112
|
direct_url: nil,
|
|
180
|
-
# Domain name to extract
|
|
113
|
+
# Domain name to extract styleguide from (e.g., 'example.com', 'google.com'). The
|
|
181
114
|
# domain will be automatically normalized and validated. You must provide either
|
|
182
115
|
# 'domain' or 'directUrl', but not both.
|
|
183
116
|
domain: nil,
|
|
@@ -193,43 +126,59 @@ module ContextDev
|
|
|
193
126
|
# timeoutOpts[milliseconds]=30000&timeoutOpts[behavior]=fail or a JSON-encoded
|
|
194
127
|
# timeoutOpts object.
|
|
195
128
|
timeout_opts: nil,
|
|
129
|
+
# Set to enabled to bypass shared caches and omit request and response content
|
|
130
|
+
# from retained usage logs. Asset uploads are skipped, so hosted image URLs are
|
|
131
|
+
# omitted. Requires zero data retention to be enabled for your organization
|
|
132
|
+
# (contact support@context.dev), otherwise the request fails with ZDR_NOT_ENABLED.
|
|
133
|
+
# Successful ZDR responses include X-Context-ZDR: true.
|
|
134
|
+
zdr: nil,
|
|
196
135
|
request_options: {}
|
|
197
136
|
)
|
|
198
137
|
end
|
|
199
138
|
|
|
200
|
-
#
|
|
201
|
-
#
|
|
139
|
+
# Discovers URLs using the same sitemap crawl, filters, and limits as
|
|
140
|
+
# /web/scrape/sitemap. Each URL includes its available title, description,
|
|
141
|
+
# keywords, and language. URLs without stored enrichment are returned immediately
|
|
142
|
+
# with only the URL and queued for background HTML scraping, so later requests can
|
|
143
|
+
# include their metadata. Responses are never cached as a whole; every request
|
|
144
|
+
# reads the current per-URL enrichment. Zero data retention and credential-bearing
|
|
145
|
+
# discovery requests return URLs without reading or storing shared enrichment or
|
|
146
|
+
# queuing background scrapes. Costs 1 credit, or 2 credits with search.
|
|
202
147
|
sig do
|
|
203
148
|
params(
|
|
204
|
-
color_scheme:
|
|
205
|
-
ContextDev::WebExtractStyleguideParams::ColorScheme::OrSymbol,
|
|
206
|
-
direct_url: String,
|
|
207
149
|
domain: String,
|
|
208
|
-
|
|
150
|
+
headers: T::Hash[Symbol, String],
|
|
151
|
+
include_subdomains: T::Boolean,
|
|
152
|
+
max_links: Integer,
|
|
153
|
+
search: String,
|
|
154
|
+
sitemap_url: String,
|
|
209
155
|
tags: T::Array[String],
|
|
210
|
-
timeout_opts:
|
|
211
|
-
|
|
156
|
+
timeout_opts: ContextDev::WebMapURLsParams::TimeoutOpts::OrHash,
|
|
157
|
+
url_regex: String,
|
|
158
|
+
zdr: ContextDev::WebMapURLsParams::Zdr::OrSymbol,
|
|
212
159
|
request_options: ContextDev::RequestOptions::OrHash
|
|
213
|
-
).returns(ContextDev::Models::
|
|
160
|
+
).returns(ContextDev::Models::WebMapURLsResponse)
|
|
214
161
|
end
|
|
215
|
-
def
|
|
216
|
-
#
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
#
|
|
220
|
-
#
|
|
221
|
-
|
|
222
|
-
#
|
|
223
|
-
|
|
224
|
-
|
|
225
|
-
#
|
|
226
|
-
#
|
|
227
|
-
|
|
228
|
-
#
|
|
229
|
-
#
|
|
230
|
-
#
|
|
231
|
-
|
|
232
|
-
|
|
162
|
+
def map_urls(
|
|
163
|
+
# Domain to build a sitemap for
|
|
164
|
+
domain:,
|
|
165
|
+
# Optional outbound HTTP headers forwarded only to the target URL, sent as
|
|
166
|
+
# deep-object query params such as headers[X-Custom]=value. When provided, caching
|
|
167
|
+
# is bypassed: the result is neither read from nor written to cache.
|
|
168
|
+
headers: nil,
|
|
169
|
+
# When true, discover and include public pages and sitemaps on subdomains of the
|
|
170
|
+
# requested domain. Defaults to false.
|
|
171
|
+
include_subdomains: nil,
|
|
172
|
+
# Maximum number of links to return from the sitemap crawl. Defaults to 10,000.
|
|
173
|
+
# Minimum is 1, maximum is 100,000.
|
|
174
|
+
max_links: nil,
|
|
175
|
+
# Optional search phrase. When provided, the crawled sitemap is filtered to the
|
|
176
|
+
# pages whose URLs are about that phrase, most relevant first, and the request
|
|
177
|
+
# costs 2 credits instead of 1.
|
|
178
|
+
search: nil,
|
|
179
|
+
# Optional explicit sitemap URL. When provided, exactly this sitemap is crawled
|
|
180
|
+
# instead of discovering the domain's sitemaps.
|
|
181
|
+
sitemap_url: nil,
|
|
233
182
|
# Comma-separated tags for tracking request usage. Up to 20 tags, each 1-50
|
|
234
183
|
# characters.
|
|
235
184
|
tags: nil,
|
|
@@ -237,6 +186,78 @@ module ContextDev
|
|
|
237
186
|
# timeoutOpts[milliseconds]=30000&timeoutOpts[behavior]=fail or a JSON-encoded
|
|
238
187
|
# timeoutOpts object.
|
|
239
188
|
timeout_opts: nil,
|
|
189
|
+
# Optional RE2-compatible regex pattern. Only URLs matching this pattern are
|
|
190
|
+
# returned and counted against maxLinks.
|
|
191
|
+
url_regex: nil,
|
|
192
|
+
# Set to enabled to bypass shared caches and omit request and response content
|
|
193
|
+
# from retained usage logs. Asset uploads are skipped, so hosted image URLs are
|
|
194
|
+
# omitted. Requires zero data retention to be enabled for your organization
|
|
195
|
+
# (contact support@context.dev), otherwise the request fails with ZDR_NOT_ENABLED.
|
|
196
|
+
# Successful ZDR responses include X-Context-ZDR: true.
|
|
197
|
+
zdr: nil,
|
|
198
|
+
request_options: {}
|
|
199
|
+
)
|
|
200
|
+
end
|
|
201
|
+
|
|
202
|
+
# Reuse cached outputs independently and capture missing formats in one page
|
|
203
|
+
# visit. Each cache key includes only the settings that affect that output. HTML
|
|
204
|
+
# is shared with Markdown and parsed fields. Cached outputs can come from
|
|
205
|
+
# different visits within maxAgeMs; use 0 for a fresh capture. HTML-only requests
|
|
206
|
+
# use the existing fast acquisition path. One credit per request, including cache
|
|
207
|
+
# hits, or two with browser actions; PDF OCR adds one credit per recovered page on
|
|
208
|
+
# fresh extraction. Original response bytes and screenshots are limited to 20 MiB
|
|
209
|
+
# each, screenshots to 40 megapixels, and the combined browser capture to 60 MiB.
|
|
210
|
+
sig do
|
|
211
|
+
params(
|
|
212
|
+
formats: ContextDev::WebScrapeParams::Formats::OrHash,
|
|
213
|
+
url: String,
|
|
214
|
+
image_params: ContextDev::WebScrapeParams::ImageParams::OrHash,
|
|
215
|
+
markdown_params: ContextDev::WebScrapeParams::MarkdownParams::OrHash,
|
|
216
|
+
max_age_ms: Integer,
|
|
217
|
+
parse_params: ContextDev::WebScrapeParams::ParseParams::OrHash,
|
|
218
|
+
screenshot_params:
|
|
219
|
+
ContextDev::WebScrapeParams::ScreenshotParams::OrHash,
|
|
220
|
+
shared_params: ContextDev::WebScrapeParams::SharedParams::OrHash,
|
|
221
|
+
tags: T::Array[String],
|
|
222
|
+
timeout_opts: ContextDev::WebScrapeParams::TimeoutOpts::OrHash,
|
|
223
|
+
zdr: ContextDev::WebScrapeParams::Zdr::OrSymbol,
|
|
224
|
+
request_options: ContextDev::RequestOptions::OrHash
|
|
225
|
+
).returns(ContextDev::Models::WebScrapeResponse)
|
|
226
|
+
end
|
|
227
|
+
def scrape(
|
|
228
|
+
# Outputs to return. Enable at least one; omitted formats are false.
|
|
229
|
+
formats:,
|
|
230
|
+
# The URL to scrape.
|
|
231
|
+
url:,
|
|
232
|
+
# Image options. Requires formats.images: true.
|
|
233
|
+
image_params: nil,
|
|
234
|
+
# Markdown options. Requires formats.markdown: true.
|
|
235
|
+
markdown_params: nil,
|
|
236
|
+
# Maximum age of each cached output. Defaults to 1 day; 0 fetches fresh and
|
|
237
|
+
# updates the requested outputs. Compatible outputs are shared with the individual
|
|
238
|
+
# scrape endpoints. Image results with hosted files refresh after 23 hours; other
|
|
239
|
+
# outputs retain their own freshness.
|
|
240
|
+
max_age_ms: nil,
|
|
241
|
+
# Required when formats.parse is true.
|
|
242
|
+
parse_params: nil,
|
|
243
|
+
# Screenshot options. Requires formats.screenshot: true.
|
|
244
|
+
screenshot_params: nil,
|
|
245
|
+
# Shared browser and content settings. Content filters leave screenshots and
|
|
246
|
+
# original bytes unchanged.
|
|
247
|
+
shared_params: nil,
|
|
248
|
+
# Labels for tracking request usage. Not retained when zdr is enabled.
|
|
249
|
+
tags: nil,
|
|
250
|
+
# Total deadline, including navigation, actions, waiting, and all outputs.
|
|
251
|
+
# Defaults to 60000 milliseconds with behavior fail. Use return-partial to capture
|
|
252
|
+
# the current page state and return captured images if image processing cannot
|
|
253
|
+
# finish before the deadline; these responses set isPartial and are not cached.
|
|
254
|
+
# Every requested format must still be available. Fixed waits must fit before a
|
|
255
|
+
# response reserve of up to 5000 milliseconds (at most one quarter of the timeout)
|
|
256
|
+
# when using return-partial.
|
|
257
|
+
timeout_opts: nil,
|
|
258
|
+
# Zero data retention. Bypasses caches and uploads; excludes request/response
|
|
259
|
+
# content and tags from logs. Must be enabled for your organization.
|
|
260
|
+
zdr: nil,
|
|
240
261
|
request_options: {}
|
|
241
262
|
)
|
|
242
263
|
end
|
|
@@ -252,6 +273,7 @@ module ContextDev
|
|
|
252
273
|
full_screenshot:
|
|
253
274
|
ContextDev::WebScreenshotParams::FullScreenshot::OrSymbol,
|
|
254
275
|
handle_cookie_popup: T::Boolean,
|
|
276
|
+
headers: T::Hash[Symbol, String],
|
|
255
277
|
max_age_ms: T.nilable(Integer),
|
|
256
278
|
page: ContextDev::WebScreenshotParams::Page::OrSymbol,
|
|
257
279
|
scroll_offset: T.nilable(Integer),
|
|
@@ -292,6 +314,14 @@ module ContextDev
|
|
|
292
314
|
# dismiss cookie banner before capture. If 'false' or not provided, captures the
|
|
293
315
|
# page without that step.
|
|
294
316
|
handle_cookie_popup: nil,
|
|
317
|
+
# Optional outbound HTTP headers, using the same JSON object or deep-object query
|
|
318
|
+
# format as other scrape endpoints (for example headers[Authorization]=Bearer
|
|
319
|
+
# token). Headers are scoped to the target origin during capture. For domain/page
|
|
320
|
+
# requests, discovery receives no custom headers and only pages on the resolved
|
|
321
|
+
# origin are eligible. Non-empty headers bypass screenshot caching and return an
|
|
322
|
+
# in-memory data URL; no screenshot is uploaded. Empty objects behave like omitted
|
|
323
|
+
# headers.
|
|
324
|
+
headers: nil,
|
|
295
325
|
# Return a cached screenshot if a prior screenshot for the same parameters exists
|
|
296
326
|
# and is younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
|
|
297
327
|
# omitted. Max is 30 days (2592000000 ms). Set to 0 to always capture fresh.
|
|
@@ -325,9 +355,10 @@ module ContextDev
|
|
|
325
355
|
# TIMEOUT_TOO_SHORT_FOR_WAIT.
|
|
326
356
|
wait_for_ms: nil,
|
|
327
357
|
# Set to enabled to bypass shared caches and omit request and response content
|
|
328
|
-
# from retained usage logs.
|
|
329
|
-
#
|
|
330
|
-
#
|
|
358
|
+
# from retained usage logs. Asset uploads are skipped, so hosted image URLs are
|
|
359
|
+
# omitted. Requires zero data retention to be enabled for your organization
|
|
360
|
+
# (contact support@context.dev), otherwise the request fails with ZDR_NOT_ENABLED.
|
|
361
|
+
# Successful ZDR responses include X-Context-ZDR: true.
|
|
331
362
|
zdr: nil,
|
|
332
363
|
request_options: {}
|
|
333
364
|
)
|
|
@@ -347,6 +378,7 @@ module ContextDev
|
|
|
347
378
|
query_fanout: T::Boolean,
|
|
348
379
|
tags: T::Array[String],
|
|
349
380
|
timeout_opts: ContextDev::WebSearchParams::TimeoutOpts::OrHash,
|
|
381
|
+
zdr: ContextDev::WebSearchParams::Zdr::OrSymbol,
|
|
350
382
|
request_options: ContextDev::RequestOptions::OrHash
|
|
351
383
|
).returns(ContextDev::Models::WebSearchResponse)
|
|
352
384
|
end
|
|
@@ -377,6 +409,12 @@ module ContextDev
|
|
|
377
409
|
# timeoutOpts[milliseconds]=30000&timeoutOpts[behavior]=fail or a JSON-encoded
|
|
378
410
|
# timeoutOpts object.
|
|
379
411
|
timeout_opts: nil,
|
|
412
|
+
# Set to enabled to bypass shared caches and omit request and response content
|
|
413
|
+
# from retained usage logs. Asset uploads are skipped, so hosted image URLs are
|
|
414
|
+
# omitted. Requires zero data retention to be enabled for your organization
|
|
415
|
+
# (contact support@context.dev), otherwise the request fails with ZDR_NOT_ENABLED.
|
|
416
|
+
# Successful ZDR responses include X-Context-ZDR: true.
|
|
417
|
+
zdr: nil,
|
|
380
418
|
request_options: {}
|
|
381
419
|
)
|
|
382
420
|
end
|
|
@@ -482,411 +520,6 @@ module ContextDev
|
|
|
482
520
|
)
|
|
483
521
|
end
|
|
484
522
|
|
|
485
|
-
# Downloads a resource and returns its bytes as base64. Supports images, PDFs,
|
|
486
|
-
# HTML pages, and any other content type without image conversion, text
|
|
487
|
-
# extraction, or character-encoding changes. HTTP compression is decoded before
|
|
488
|
-
# base64 encoding. HTML is the original HTTP response; JavaScript is not rendered.
|
|
489
|
-
# Follows public redirects and retries failed downloads through ISP and
|
|
490
|
-
# residential proxies, with a direct fallback. When country is specified, only a
|
|
491
|
-
# residential proxy in that country is used. Supply headers such as Referer for
|
|
492
|
-
# images that require a referring page. Downloads are not cached. Maximum decoded
|
|
493
|
-
# resource size: 20 MiB (20971520 bytes), before base64 encoding. Successful
|
|
494
|
-
# requests cost 1 credit; errors are not billed.
|
|
495
|
-
sig do
|
|
496
|
-
params(
|
|
497
|
-
url: String,
|
|
498
|
-
country: ContextDev::WebWebScrapeBytesParams::Country::OrSymbol,
|
|
499
|
-
headers: T::Hash[Symbol, String],
|
|
500
|
-
tags: T::Array[String],
|
|
501
|
-
timeout_opts:
|
|
502
|
-
ContextDev::WebWebScrapeBytesParams::TimeoutOpts::OrHash,
|
|
503
|
-
zdr: ContextDev::WebWebScrapeBytesParams::Zdr::OrSymbol,
|
|
504
|
-
request_options: ContextDev::RequestOptions::OrHash
|
|
505
|
-
).returns(ContextDev::Models::WebWebScrapeBytesResponse)
|
|
506
|
-
end
|
|
507
|
-
def web_scrape_bytes(
|
|
508
|
-
# Full HTTP(S) URL of the resource to download, such as an image, PDF, or page.
|
|
509
|
-
url:,
|
|
510
|
-
# Fetch the target page through a residential proxy in this country (ISO 3166-1
|
|
511
|
-
# alpha-2).
|
|
512
|
-
country: nil,
|
|
513
|
-
# Optional outbound HTTP headers, such as Referer, Cookie, or Authorization. Send
|
|
514
|
-
# as a JSON object or deep-object query params such as
|
|
515
|
-
# headers[Referer]=https://example.com/. Host, Content-Length, and hop-by-hop
|
|
516
|
-
# transport headers are rejected. Authorization and cookies are removed when a
|
|
517
|
-
# redirect changes origin.
|
|
518
|
-
headers: nil,
|
|
519
|
-
# Comma-separated tags for tracking request usage. Up to 20 tags, each 1-50
|
|
520
|
-
# characters.
|
|
521
|
-
tags: nil,
|
|
522
|
-
# Optional request deadline and behavior on timeout. For GET requests, use
|
|
523
|
-
# timeoutOpts[milliseconds]=30000&timeoutOpts[behavior]=fail or a JSON-encoded
|
|
524
|
-
# timeoutOpts object.
|
|
525
|
-
timeout_opts: nil,
|
|
526
|
-
# Set to enabled to bypass shared caches and omit request and response content
|
|
527
|
-
# from retained usage logs. Requires zero data retention to be enabled for your
|
|
528
|
-
# organization (contact support@context.dev), otherwise the request fails with
|
|
529
|
-
# ZDR_NOT_ENABLED. Successful ZDR responses include X-Context-ZDR: true.
|
|
530
|
-
zdr: nil,
|
|
531
|
-
request_options: {}
|
|
532
|
-
)
|
|
533
|
-
end
|
|
534
|
-
|
|
535
|
-
# Scrapes the given URL and returns the raw HTML content of the page. The base
|
|
536
|
-
# request costs 1 credit; requests with browser actions cost 2 credits. A request
|
|
537
|
-
# that hits its timeoutOpts.milliseconds deadline fails with 408 and is not
|
|
538
|
-
# billed, unless timeoutOpts.behavior=return-partial is set — then the page as
|
|
539
|
-
# rendered so far is returned with `finalDOMState: "still-loading"` and billed at
|
|
540
|
-
# the base cost of 1 credit.
|
|
541
|
-
sig do
|
|
542
|
-
params(
|
|
543
|
-
url: String,
|
|
544
|
-
actions:
|
|
545
|
-
T.nilable(
|
|
546
|
-
T::Array[
|
|
547
|
-
T.any(
|
|
548
|
-
ContextDev::WebWebScrapeHTMLParams::Action::Wait::OrHash,
|
|
549
|
-
ContextDev::WebWebScrapeHTMLParams::Action::Perform::OrHash,
|
|
550
|
-
ContextDev::WebWebScrapeHTMLParams::Action::Scroll::OrHash
|
|
551
|
-
)
|
|
552
|
-
]
|
|
553
|
-
),
|
|
554
|
-
country: ContextDev::WebWebScrapeHTMLParams::Country::OrSymbol,
|
|
555
|
-
exclude_selectors: T.nilable(T::Array[String]),
|
|
556
|
-
headers: T::Hash[Symbol, String],
|
|
557
|
-
include_frames: T::Boolean,
|
|
558
|
-
include_selectors: T.nilable(T::Array[String]),
|
|
559
|
-
max_age_ms: T.nilable(Integer),
|
|
560
|
-
pdf: ContextDev::WebWebScrapeHTMLParams::Pdf::OrHash,
|
|
561
|
-
settle_animations: T::Boolean,
|
|
562
|
-
tags: T::Array[String],
|
|
563
|
-
timeout_opts: ContextDev::WebWebScrapeHTMLParams::TimeoutOpts::OrHash,
|
|
564
|
-
use_main_content_only: T::Boolean,
|
|
565
|
-
wait_for_ms: T.nilable(Integer),
|
|
566
|
-
zdr: ContextDev::WebWebScrapeHTMLParams::Zdr::OrSymbol,
|
|
567
|
-
request_options: ContextDev::RequestOptions::OrHash
|
|
568
|
-
).returns(ContextDev::Models::WebWebScrapeHTMLResponse)
|
|
569
|
-
end
|
|
570
|
-
def web_scrape_html(
|
|
571
|
-
# Full URL to scrape (must include http:// or https:// protocol)
|
|
572
|
-
url:,
|
|
573
|
-
# Optional browser actions executed in array order after the page loads and before
|
|
574
|
-
# content is captured. Requires a paid plan. Send a JSON array in the query
|
|
575
|
-
# parameter. Maximum: 5 actions.
|
|
576
|
-
actions: nil,
|
|
577
|
-
# Fetch the target page through a residential proxy in this country (ISO 3166-1
|
|
578
|
-
# alpha-2).
|
|
579
|
-
country: nil,
|
|
580
|
-
# CSS selectors to remove from the result. Applied after includeSelectors.
|
|
581
|
-
# Exclusion takes precedence: an element matching both is removed. Examples:
|
|
582
|
-
# "nav", "footer", ".ad-banner", "[aria-hidden=true]".
|
|
583
|
-
exclude_selectors: nil,
|
|
584
|
-
# Optional outbound HTTP headers forwarded only to the target URL, sent as
|
|
585
|
-
# deep-object query params such as headers[X-Custom]=value. When provided, caching
|
|
586
|
-
# is bypassed: the result is neither read from nor written to cache.
|
|
587
|
-
headers: nil,
|
|
588
|
-
# When true, iframes are rendered inline into the returned HTML.
|
|
589
|
-
include_frames: nil,
|
|
590
|
-
# CSS selectors. When provided, only matching subtrees (and their descendants) are
|
|
591
|
-
# kept and everything else is dropped. When omitted, the entire document is kept.
|
|
592
|
-
# Examples: "article.main", "#content", "[role=main]".
|
|
593
|
-
include_selectors: nil,
|
|
594
|
-
# Return a cached result if a prior scrape for the same parameters exists and is
|
|
595
|
-
# younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
|
|
596
|
-
# omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
|
|
597
|
-
max_age_ms: nil,
|
|
598
|
-
# PDF parsing controls. Use start/end to limit text extraction and embedded-image
|
|
599
|
-
# detection/OCR to an inclusive 1-based page range.
|
|
600
|
-
pdf: nil,
|
|
601
|
-
# When true, waits briefly for CSS and transition animations to settle before
|
|
602
|
-
# extracting HTML. Defaults to false. This adds a bit of latency in exchange for
|
|
603
|
-
# more stable output on animated pages.
|
|
604
|
-
settle_animations: nil,
|
|
605
|
-
# Comma-separated tags for tracking request usage. Up to 20 tags, each 1-50
|
|
606
|
-
# characters.
|
|
607
|
-
tags: nil,
|
|
608
|
-
# Optional request deadline and behavior on timeout. For GET requests, use
|
|
609
|
-
# timeoutOpts[milliseconds]=30000&timeoutOpts[behavior]=fail or a JSON-encoded
|
|
610
|
-
# timeoutOpts object.
|
|
611
|
-
timeout_opts: nil,
|
|
612
|
-
# When true, return only the page's main content in the HTML response, excluding
|
|
613
|
-
# headers, footers, sidebars, and navigation when detectable.
|
|
614
|
-
use_main_content_only: nil,
|
|
615
|
-
# Optional browser wait time in milliseconds after initial page load. Min: 0. Max:
|
|
616
|
-
# 30000 (30 seconds). When combined with timeoutOpts, timeoutOpts.milliseconds
|
|
617
|
-
# must be at least waitForMs + 10000 ms; a shorter deadline is rejected with 400
|
|
618
|
-
# TIMEOUT_TOO_SHORT_FOR_WAIT.
|
|
619
|
-
wait_for_ms: nil,
|
|
620
|
-
# Set to enabled to bypass shared caches and omit request and response content
|
|
621
|
-
# from retained usage logs. Requires zero data retention to be enabled for your
|
|
622
|
-
# organization (contact support@context.dev), otherwise the request fails with
|
|
623
|
-
# ZDR_NOT_ENABLED. Successful ZDR responses include X-Context-ZDR: true.
|
|
624
|
-
zdr: nil,
|
|
625
|
-
request_options: {}
|
|
626
|
-
)
|
|
627
|
-
end
|
|
628
|
-
|
|
629
|
-
# Extract image assets from a web page, including standard URLs, inline SVGs, data
|
|
630
|
-
# URIs, responsive image sources, metadata, CSS backgrounds, video posters, and
|
|
631
|
-
# embeds. The base request costs 1 credit, or 2 credits with browser actions. When
|
|
632
|
-
# enrichment is enabled, the entire call costs 5 credits, including requests that
|
|
633
|
-
# also use actions.
|
|
634
|
-
sig do
|
|
635
|
-
params(
|
|
636
|
-
url: String,
|
|
637
|
-
actions:
|
|
638
|
-
T.nilable(
|
|
639
|
-
T::Array[
|
|
640
|
-
T.any(
|
|
641
|
-
ContextDev::WebWebScrapeImagesParams::Action::Wait::OrHash,
|
|
642
|
-
ContextDev::WebWebScrapeImagesParams::Action::Perform::OrHash,
|
|
643
|
-
ContextDev::WebWebScrapeImagesParams::Action::Scroll::OrHash
|
|
644
|
-
)
|
|
645
|
-
]
|
|
646
|
-
),
|
|
647
|
-
dedupe: T::Boolean,
|
|
648
|
-
enrichment:
|
|
649
|
-
T.nilable(ContextDev::WebWebScrapeImagesParams::Enrichment::OrHash),
|
|
650
|
-
headers: T::Hash[Symbol, String],
|
|
651
|
-
max_age_ms: T.nilable(Integer),
|
|
652
|
-
tags: T::Array[String],
|
|
653
|
-
timeout_opts:
|
|
654
|
-
ContextDev::WebWebScrapeImagesParams::TimeoutOpts::OrHash,
|
|
655
|
-
wait_for_ms: T.nilable(Integer),
|
|
656
|
-
request_options: ContextDev::RequestOptions::OrHash
|
|
657
|
-
).returns(ContextDev::Models::WebWebScrapeImagesResponse)
|
|
658
|
-
end
|
|
659
|
-
def web_scrape_images(
|
|
660
|
-
# Page URL to inspect. Must include http:// or https://.
|
|
661
|
-
url:,
|
|
662
|
-
# Optional browser actions executed in array order after the page loads and before
|
|
663
|
-
# content is captured. Requires a paid plan. Send a JSON array in the query
|
|
664
|
-
# parameter. Maximum: 5 actions.
|
|
665
|
-
actions: nil,
|
|
666
|
-
# When true, visually duplicate images are removed: every image is loaded and
|
|
667
|
-
# perceptually hashed, and only the highest-resolution copy of each duplicate
|
|
668
|
-
# group is kept. Images that cannot be downloaded or hashed are kept. Default:
|
|
669
|
-
# false.
|
|
670
|
-
dedupe: nil,
|
|
671
|
-
# Optional per-image processing, sent as deep-object query params such as
|
|
672
|
-
# enrichment[resolution]=true.
|
|
673
|
-
enrichment: nil,
|
|
674
|
-
# Optional outbound HTTP headers forwarded only to the target URL, sent as
|
|
675
|
-
# deep-object query params such as headers[X-Custom]=value. When provided, caching
|
|
676
|
-
# is bypassed: the result is neither read from nor written to cache.
|
|
677
|
-
headers: nil,
|
|
678
|
-
# Reuse a cached result this many milliseconds old or newer. Default: 86400000 (1
|
|
679
|
-
# day). Set to 0 to bypass cache. Maximum: 2592000000 (30 days).
|
|
680
|
-
max_age_ms: nil,
|
|
681
|
-
# Comma-separated tags for tracking request usage. Up to 20 tags, each 1-50
|
|
682
|
-
# characters.
|
|
683
|
-
tags: nil,
|
|
684
|
-
# Optional request deadline and behavior on timeout. For GET requests, use
|
|
685
|
-
# timeoutOpts[milliseconds]=30000&timeoutOpts[behavior]=fail or a JSON-encoded
|
|
686
|
-
# timeoutOpts object.
|
|
687
|
-
timeout_opts: nil,
|
|
688
|
-
# Optional browser wait time in milliseconds after initial page load before
|
|
689
|
-
# collecting images. Min: 0. Max: 30000 (30 seconds). When combined with
|
|
690
|
-
# timeoutOpts, timeoutOpts.milliseconds must be at least waitForMs + 10000 ms; a
|
|
691
|
-
# shorter deadline is rejected with 400 TIMEOUT_TOO_SHORT_FOR_WAIT.
|
|
692
|
-
wait_for_ms: nil,
|
|
693
|
-
request_options: {}
|
|
694
|
-
)
|
|
695
|
-
end
|
|
696
|
-
|
|
697
|
-
# Scrapes the given URL into LLM usable Markdown. Inspect key_metadata on JSON
|
|
698
|
-
# responses from a recognized API key; use error_code to distinguish stable
|
|
699
|
-
# failure categories.
|
|
700
|
-
#
|
|
701
|
-
# ### YouTube
|
|
702
|
-
#
|
|
703
|
-
# YouTube URLs return the video or channel itself rather than the surrounding
|
|
704
|
-
# player and navigation chrome. A URL addressing a single video (`/watch`,
|
|
705
|
-
# `youtu.be`, `/shorts`, `/embed`, `/live`) returns its title, channel, duration,
|
|
706
|
-
# view count, keywords, full description, and the transcript when the video has
|
|
707
|
-
# captions that can be retrieved; videos without captions return everything except
|
|
708
|
-
# the transcript. A channel URL (`/channel/UC…`, `/@handle`, `/c/…`, `/user/…`)
|
|
709
|
-
# returns its name, handle, subscriber count, video count, and full description.
|
|
710
|
-
# When `includeImages=true`, video responses also include the thumbnail and
|
|
711
|
-
# channel responses include the avatar. Costs the same as any other scrape.
|
|
712
|
-
#
|
|
713
|
-
# ### Billing & errors
|
|
714
|
-
#
|
|
715
|
-
# | HTTP status | Billed? | Meaning |
|
|
716
|
-
# | ----------- | ----------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
|
717
|
-
# | 200 | Yes — 1 credit, or 2 credits with actions | Successful scrape, including a zero-length result when includeSelectors matched nothing. A partial result (`finalDOMState: "still-loading"`, only with timeoutOpts.behavior=return-partial) is billed at the base 1 credit with no OCR or actions surcharge |
|
|
718
|
-
# | 400 | No | Invalid input, skipped PDF, or the page could not be scraped. error_code WEBSITE_BLOCKED specifically means the site answered with an anti-bot challenge, CAPTCHA wall, or login shell instead of the page (even when the site returned HTTP 200) — retrying later or from another country sometimes succeeds |
|
|
719
|
-
# | 401 / 403 | No | Invalid/disabled key, insufficient permissions, or credits exhausted; inspect error_code |
|
|
720
|
-
# | 404 | No | Target page returned or fingerprinted as not found |
|
|
721
|
-
# | 408 | No | Request timed out. With timeoutOpts.behavior=return-partial this only happens when nothing usable had rendered by the deadline |
|
|
722
|
-
# | 413 | No | Target content exceeds the maximum supported size (20 MB) |
|
|
723
|
-
# | 415 | No | Unsupported content type |
|
|
724
|
-
# | 429 | No | Per-minute rate limit exceeded; honor Retry-After |
|
|
725
|
-
# | 500 | No | Internal error |
|
|
726
|
-
sig do
|
|
727
|
-
params(
|
|
728
|
-
url: String,
|
|
729
|
-
actions:
|
|
730
|
-
T.nilable(
|
|
731
|
-
T::Array[
|
|
732
|
-
T.any(
|
|
733
|
-
ContextDev::WebWebScrapeMdParams::Action::Wait::OrHash,
|
|
734
|
-
ContextDev::WebWebScrapeMdParams::Action::Perform::OrHash,
|
|
735
|
-
ContextDev::WebWebScrapeMdParams::Action::Scroll::OrHash
|
|
736
|
-
)
|
|
737
|
-
]
|
|
738
|
-
),
|
|
739
|
-
country: ContextDev::WebWebScrapeMdParams::Country::OrSymbol,
|
|
740
|
-
exclude_selectors: T.nilable(T::Array[String]),
|
|
741
|
-
headers: T::Hash[Symbol, String],
|
|
742
|
-
include_frames: T::Boolean,
|
|
743
|
-
include_html: T::Boolean,
|
|
744
|
-
include_images: T::Boolean,
|
|
745
|
-
include_links: T::Boolean,
|
|
746
|
-
include_selectors: T.nilable(T::Array[String]),
|
|
747
|
-
max_age_ms: T.nilable(Integer),
|
|
748
|
-
pdf: ContextDev::WebWebScrapeMdParams::Pdf::OrHash,
|
|
749
|
-
settle_animations: T::Boolean,
|
|
750
|
-
shorten_base64_images: T::Boolean,
|
|
751
|
-
tags: T::Array[String],
|
|
752
|
-
timeout_opts: ContextDev::WebWebScrapeMdParams::TimeoutOpts::OrHash,
|
|
753
|
-
use_main_content_only: T::Boolean,
|
|
754
|
-
wait_for_ms: T.nilable(Integer),
|
|
755
|
-
zdr: ContextDev::WebWebScrapeMdParams::Zdr::OrSymbol,
|
|
756
|
-
request_options: ContextDev::RequestOptions::OrHash
|
|
757
|
-
).returns(ContextDev::Models::WebWebScrapeMdResponse)
|
|
758
|
-
end
|
|
759
|
-
def web_scrape_md(
|
|
760
|
-
# Full URL to scrape into LLM usable Markdown (must include http:// or https://
|
|
761
|
-
# protocol)
|
|
762
|
-
url:,
|
|
763
|
-
# Optional browser actions executed in array order after the page loads and before
|
|
764
|
-
# content is captured. Requires a paid plan. Send a JSON array in the query
|
|
765
|
-
# parameter. Maximum: 5 actions.
|
|
766
|
-
actions: nil,
|
|
767
|
-
# Fetch the target page through a residential proxy in this country (ISO 3166-1
|
|
768
|
-
# alpha-2).
|
|
769
|
-
country: nil,
|
|
770
|
-
# CSS selectors to remove before conversion to Markdown. Applied after
|
|
771
|
-
# includeSelectors. Exclusion takes precedence: an element matching both is
|
|
772
|
-
# removed. Examples: "nav", "footer", ".ad-banner", "[aria-hidden=true]".
|
|
773
|
-
exclude_selectors: nil,
|
|
774
|
-
# Optional outbound HTTP headers forwarded only to the target URL, sent as
|
|
775
|
-
# deep-object query params such as headers[X-Custom]=value. When provided, caching
|
|
776
|
-
# is bypassed: the result is neither read from nor written to cache.
|
|
777
|
-
headers: nil,
|
|
778
|
-
# When true, the contents of iframes are rendered to Markdown.
|
|
779
|
-
include_frames: nil,
|
|
780
|
-
# When true, the response also includes an `html` field with the page HTML the
|
|
781
|
-
# Markdown was converted from — the same body the Scrape HTML endpoint returns for
|
|
782
|
-
# the equivalent request.
|
|
783
|
-
include_html: nil,
|
|
784
|
-
# Include image references in Markdown output
|
|
785
|
-
include_images: nil,
|
|
786
|
-
# Preserve hyperlinks in Markdown output
|
|
787
|
-
include_links: nil,
|
|
788
|
-
# CSS selectors. When provided, only matching HTML subtrees (and their
|
|
789
|
-
# descendants) are kept before conversion to Markdown. When omitted, the entire
|
|
790
|
-
# document is kept. Examples: "article.main", "#content", "[role=main]".
|
|
791
|
-
include_selectors: nil,
|
|
792
|
-
# Return a cached result if a prior scrape for the same parameters exists and is
|
|
793
|
-
# younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
|
|
794
|
-
# omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
|
|
795
|
-
max_age_ms: nil,
|
|
796
|
-
# PDF parsing controls. Use start/end to limit text extraction and embedded-image
|
|
797
|
-
# detection/OCR to an inclusive 1-based page range.
|
|
798
|
-
pdf: nil,
|
|
799
|
-
# When true, waits briefly for CSS and transition animations to settle before
|
|
800
|
-
# converting to Markdown. Defaults to false. This adds a bit of latency in
|
|
801
|
-
# exchange for more stable output on animated pages.
|
|
802
|
-
settle_animations: nil,
|
|
803
|
-
# Shorten base64-encoded image data in the Markdown output
|
|
804
|
-
shorten_base64_images: nil,
|
|
805
|
-
# Comma-separated tags for tracking request usage. Up to 20 tags, each 1-50
|
|
806
|
-
# characters.
|
|
807
|
-
tags: nil,
|
|
808
|
-
# Optional request deadline and behavior on timeout. For GET requests, use
|
|
809
|
-
# timeoutOpts[milliseconds]=30000&timeoutOpts[behavior]=fail or a JSON-encoded
|
|
810
|
-
# timeoutOpts object.
|
|
811
|
-
timeout_opts: nil,
|
|
812
|
-
# Extract only the main content of the page, excluding headers, footers, sidebars,
|
|
813
|
-
# and navigation
|
|
814
|
-
use_main_content_only: nil,
|
|
815
|
-
# Optional browser wait time in milliseconds after initial page load before
|
|
816
|
-
# converting the page to Markdown. Min: 0. Max: 30000 (30 seconds). When combined
|
|
817
|
-
# with timeoutOpts, timeoutOpts.milliseconds must be at least waitForMs + 10000
|
|
818
|
-
# ms; a shorter deadline is rejected with 400 TIMEOUT_TOO_SHORT_FOR_WAIT.
|
|
819
|
-
wait_for_ms: nil,
|
|
820
|
-
# Set to enabled to bypass shared caches and omit request and response content
|
|
821
|
-
# from retained usage logs. Requires zero data retention to be enabled for your
|
|
822
|
-
# organization (contact support@context.dev), otherwise the request fails with
|
|
823
|
-
# ZDR_NOT_ENABLED. Successful ZDR responses include X-Context-ZDR: true.
|
|
824
|
-
zdr: nil,
|
|
825
|
-
request_options: {}
|
|
826
|
-
)
|
|
827
|
-
end
|
|
828
|
-
|
|
829
|
-
# Crawl an entire website's sitemap and return all discovered page URLs. Set
|
|
830
|
-
# `includeSubdomains=true` to also discover public pages and sitemaps on child
|
|
831
|
-
# hosts such as `docs.example.com` or `brand.example.com`. Pass `search` to have
|
|
832
|
-
# the discovered URLs filtered down to the pages about a phrase (for example
|
|
833
|
-
# `pricing and plans` or `api authentication docs`), most relevant first — a
|
|
834
|
-
# searched crawl scans the whole sitemap and costs 2 credits instead of 1.
|
|
835
|
-
sig do
|
|
836
|
-
params(
|
|
837
|
-
domain: String,
|
|
838
|
-
headers: T::Hash[Symbol, String],
|
|
839
|
-
include_subdomains: T::Boolean,
|
|
840
|
-
max_links: Integer,
|
|
841
|
-
search: String,
|
|
842
|
-
sitemap_url: String,
|
|
843
|
-
tags: T::Array[String],
|
|
844
|
-
timeout_opts:
|
|
845
|
-
ContextDev::WebWebScrapeSitemapParams::TimeoutOpts::OrHash,
|
|
846
|
-
url_regex: String,
|
|
847
|
-
zdr: ContextDev::WebWebScrapeSitemapParams::Zdr::OrSymbol,
|
|
848
|
-
request_options: ContextDev::RequestOptions::OrHash
|
|
849
|
-
).returns(ContextDev::Models::WebWebScrapeSitemapResponse)
|
|
850
|
-
end
|
|
851
|
-
def web_scrape_sitemap(
|
|
852
|
-
# Domain to build a sitemap for
|
|
853
|
-
domain:,
|
|
854
|
-
# Optional outbound HTTP headers forwarded only to the target URL, sent as
|
|
855
|
-
# deep-object query params such as headers[X-Custom]=value. When provided, caching
|
|
856
|
-
# is bypassed: the result is neither read from nor written to cache.
|
|
857
|
-
headers: nil,
|
|
858
|
-
# When true, discover and include public pages and sitemaps on subdomains of the
|
|
859
|
-
# requested domain. Defaults to false.
|
|
860
|
-
include_subdomains: nil,
|
|
861
|
-
# Maximum number of links to return from the sitemap crawl. Defaults to 10,000.
|
|
862
|
-
# Minimum is 1, maximum is 100,000.
|
|
863
|
-
max_links: nil,
|
|
864
|
-
# Optional search phrase. When provided, the crawled sitemap is filtered to the
|
|
865
|
-
# pages whose URLs are about that phrase, most relevant first, and the request
|
|
866
|
-
# costs 2 credits instead of 1.
|
|
867
|
-
search: nil,
|
|
868
|
-
# Optional explicit sitemap URL. When provided, exactly this sitemap is crawled
|
|
869
|
-
# instead of discovering the domain's sitemaps.
|
|
870
|
-
sitemap_url: nil,
|
|
871
|
-
# Comma-separated tags for tracking request usage. Up to 20 tags, each 1-50
|
|
872
|
-
# characters.
|
|
873
|
-
tags: nil,
|
|
874
|
-
# Optional request deadline and behavior on timeout. For GET requests, use
|
|
875
|
-
# timeoutOpts[milliseconds]=30000&timeoutOpts[behavior]=fail or a JSON-encoded
|
|
876
|
-
# timeoutOpts object.
|
|
877
|
-
timeout_opts: nil,
|
|
878
|
-
# Optional RE2-compatible regex pattern. Only URLs matching this pattern are
|
|
879
|
-
# returned and counted against maxLinks.
|
|
880
|
-
url_regex: nil,
|
|
881
|
-
# Set to enabled to bypass shared caches and omit request and response content
|
|
882
|
-
# from retained usage logs. Requires zero data retention to be enabled for your
|
|
883
|
-
# organization (contact support@context.dev), otherwise the request fails with
|
|
884
|
-
# ZDR_NOT_ENABLED. Successful ZDR responses include X-Context-ZDR: true.
|
|
885
|
-
zdr: nil,
|
|
886
|
-
request_options: {}
|
|
887
|
-
)
|
|
888
|
-
end
|
|
889
|
-
|
|
890
523
|
# @api private
|
|
891
524
|
sig { params(client: ContextDev::Client).returns(T.attached_class) }
|
|
892
525
|
def self.new(client:)
|