context.dev 1.16.0 → 1.18.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +19 -0
- data/README.md +3 -3
- data/lib/context_dev/models/ai_extract_product_params.rb +4 -3
- data/lib/context_dev/models/ai_extract_products_params.rb +8 -6
- data/lib/context_dev/models/web_screenshot_params.rb +18 -21
- data/lib/context_dev/models/web_web_crawl_md_params.rb +72 -8
- data/lib/context_dev/models/web_web_crawl_md_response.rb +3 -2
- data/lib/context_dev/models/web_web_scrape_html_params.rb +61 -8
- data/lib/context_dev/models/web_web_scrape_images_params.rb +20 -1
- data/lib/context_dev/models/web_web_scrape_md_params.rb +61 -8
- data/lib/context_dev/models/web_web_scrape_sitemap_params.rb +11 -1
- data/lib/context_dev/resources/ai.rb +1 -1
- data/lib/context_dev/resources/web.rb +46 -16
- data/lib/context_dev/version.rb +1 -1
- data/rbi/context_dev/models/ai_extract_product_params.rbi +6 -4
- data/rbi/context_dev/models/ai_extract_products_params.rbi +12 -8
- data/rbi/context_dev/models/web_screenshot_params.rbi +28 -53
- data/rbi/context_dev/models/web_web_crawl_md_params.rbi +118 -13
- data/rbi/context_dev/models/web_web_crawl_md_response.rbi +4 -2
- data/rbi/context_dev/models/web_web_scrape_html_params.rbi +101 -13
- data/rbi/context_dev/models/web_web_scrape_images_params.rbi +28 -0
- data/rbi/context_dev/models/web_web_scrape_md_params.rbi +101 -13
- data/rbi/context_dev/models/web_web_scrape_sitemap_params.rbi +15 -0
- data/rbi/context_dev/resources/ai.rbi +3 -2
- data/rbi/context_dev/resources/web.rbi +69 -20
- data/sig/context_dev/models/web_screenshot_params.rbs +13 -19
- data/sig/context_dev/models/web_web_crawl_md_params.rbs +53 -6
- data/sig/context_dev/models/web_web_scrape_html_params.rbs +45 -5
- data/sig/context_dev/models/web_web_scrape_images_params.rbs +15 -1
- data/sig/context_dev/models/web_web_scrape_md_params.rbs +46 -6
- data/sig/context_dev/models/web_web_scrape_sitemap_params.rbs +12 -1
- data/sig/context_dev/resources/web.rbs +15 -4
- metadata +2 -2
|
@@ -70,7 +70,7 @@ module ContextDev
|
|
|
70
70
|
#
|
|
71
71
|
# Capture a screenshot of a website.
|
|
72
72
|
#
|
|
73
|
-
# @overload screenshot(direct_url: nil, domain: nil, full_screenshot: nil, max_age_ms: nil, page: nil,
|
|
73
|
+
# @overload screenshot(direct_url: nil, domain: nil, full_screenshot: nil, max_age_ms: nil, page: nil, timeout_ms: nil, viewport: nil, wait_for_ms: nil, request_options: {})
|
|
74
74
|
#
|
|
75
75
|
# @param direct_url [String] A specific URL to screenshot directly, bypassing domain resolution (e.g., 'https
|
|
76
76
|
#
|
|
@@ -82,10 +82,12 @@ module ContextDev
|
|
|
82
82
|
#
|
|
83
83
|
# @param page [Symbol, ContextDev::Models::WebScreenshotParams::Page] Optional parameter to specify which page type to screenshot. If provided, the sy
|
|
84
84
|
#
|
|
85
|
-
# @param
|
|
85
|
+
# @param timeout_ms [Integer] Optional timeout in milliseconds for the request. If the request takes longer th
|
|
86
86
|
#
|
|
87
87
|
# @param viewport [ContextDev::Models::WebScreenshotParams::Viewport] Optional browser viewport dimensions for the screenshot. Defaults to 1920x1080.
|
|
88
88
|
#
|
|
89
|
+
# @param wait_for_ms [Integer] Optional browser wait time in milliseconds after initial page load before taking
|
|
90
|
+
#
|
|
89
91
|
# @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
|
|
90
92
|
#
|
|
91
93
|
# @return [ContextDev::Models::WebScreenshotResponse]
|
|
@@ -100,7 +102,9 @@ module ContextDev
|
|
|
100
102
|
query: query.transform_keys(
|
|
101
103
|
direct_url: "directUrl",
|
|
102
104
|
full_screenshot: "fullScreenshot",
|
|
103
|
-
max_age_ms: "maxAgeMs"
|
|
105
|
+
max_age_ms: "maxAgeMs",
|
|
106
|
+
timeout_ms: "timeoutMS",
|
|
107
|
+
wait_for_ms: "waitForMs"
|
|
104
108
|
),
|
|
105
109
|
model: ContextDev::Models::WebScreenshotResponse,
|
|
106
110
|
options: options
|
|
@@ -113,7 +117,7 @@ module ContextDev
|
|
|
113
117
|
# Performs a crawl starting from a given URL, extracts page content as Markdown,
|
|
114
118
|
# and returns results for all crawled pages.
|
|
115
119
|
#
|
|
116
|
-
# @overload web_crawl_md(url:, follow_subdomains: nil, include_frames: nil, include_images: nil, include_links: nil, max_age_ms: nil, max_depth: nil, max_pages: nil,
|
|
120
|
+
# @overload web_crawl_md(url:, follow_subdomains: nil, include_frames: nil, include_images: nil, include_links: nil, max_age_ms: nil, max_depth: nil, max_pages: nil, pdf: nil, shorten_base64_images: nil, stop_after_ms: nil, timeout_ms: nil, url_regex: nil, use_main_content_only: nil, wait_for_ms: nil, request_options: {})
|
|
117
121
|
#
|
|
118
122
|
# @param url [String] The starting URL for the crawl (must include http:// or https:// protocol)
|
|
119
123
|
#
|
|
@@ -131,14 +135,20 @@ module ContextDev
|
|
|
131
135
|
#
|
|
132
136
|
# @param max_pages [Integer] Maximum number of pages to crawl. Hard cap: 500.
|
|
133
137
|
#
|
|
134
|
-
# @param
|
|
138
|
+
# @param pdf [ContextDev::Models::WebWebCrawlMdParams::Pdf] PDF parsing controls. Use start/end to limit text extraction and OCR to an inclu
|
|
135
139
|
#
|
|
136
140
|
# @param shorten_base64_images [Boolean] Truncate base64-encoded image data in the Markdown output
|
|
137
141
|
#
|
|
142
|
+
# @param stop_after_ms [Integer] Soft time budget for the crawl in milliseconds. After each scrape, the crawler c
|
|
143
|
+
#
|
|
144
|
+
# @param timeout_ms [Integer] Optional timeout in milliseconds for the request. If the request takes longer th
|
|
145
|
+
#
|
|
138
146
|
# @param url_regex [String] Regex pattern. Only URLs matching this pattern will be followed and scraped.
|
|
139
147
|
#
|
|
140
148
|
# @param use_main_content_only [Boolean] Extract only the main content, stripping headers, footers, sidebars, and navigat
|
|
141
149
|
#
|
|
150
|
+
# @param wait_for_ms [Integer] Optional browser wait time in milliseconds after initial page load for each craw
|
|
151
|
+
#
|
|
142
152
|
# @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
|
|
143
153
|
#
|
|
144
154
|
# @return [ContextDev::Models::WebWebCrawlMdResponse]
|
|
@@ -160,7 +170,7 @@ module ContextDev
|
|
|
160
170
|
#
|
|
161
171
|
# Scrapes the given URL and returns the raw HTML content of the page.
|
|
162
172
|
#
|
|
163
|
-
# @overload web_scrape_html(url:, include_frames: nil, max_age_ms: nil,
|
|
173
|
+
# @overload web_scrape_html(url:, include_frames: nil, max_age_ms: nil, pdf: nil, timeout_ms: nil, wait_for_ms: nil, request_options: {})
|
|
164
174
|
#
|
|
165
175
|
# @param url [String] Full URL to scrape (must include http:// or https:// protocol)
|
|
166
176
|
#
|
|
@@ -168,7 +178,11 @@ module ContextDev
|
|
|
168
178
|
#
|
|
169
179
|
# @param max_age_ms [Integer] Return a cached result if a prior scrape for the same parameters exists and is y
|
|
170
180
|
#
|
|
171
|
-
# @param
|
|
181
|
+
# @param pdf [ContextDev::Models::WebWebScrapeHTMLParams::Pdf] PDF parsing controls. Use start/end to limit text extraction and OCR to an inclu
|
|
182
|
+
#
|
|
183
|
+
# @param timeout_ms [Integer] Optional timeout in milliseconds for the request. If the request takes longer th
|
|
184
|
+
#
|
|
185
|
+
# @param wait_for_ms [Integer] Optional browser wait time in milliseconds after initial page load. Min: 0. Max:
|
|
172
186
|
#
|
|
173
187
|
# @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
|
|
174
188
|
#
|
|
@@ -184,7 +198,8 @@ module ContextDev
|
|
|
184
198
|
query: query.transform_keys(
|
|
185
199
|
include_frames: "includeFrames",
|
|
186
200
|
max_age_ms: "maxAgeMs",
|
|
187
|
-
|
|
201
|
+
timeout_ms: "timeoutMS",
|
|
202
|
+
wait_for_ms: "waitForMs"
|
|
188
203
|
),
|
|
189
204
|
model: ContextDev::Models::WebWebScrapeHTMLResponse,
|
|
190
205
|
options: options
|
|
@@ -199,7 +214,7 @@ module ContextDev
|
|
|
199
214
|
# embeds. The base request costs 1 credit; enrichment costs 1 credit per returned
|
|
200
215
|
# image.
|
|
201
216
|
#
|
|
202
|
-
# @overload web_scrape_images(url:, enrichment: nil, max_age_ms: nil, request_options: {})
|
|
217
|
+
# @overload web_scrape_images(url:, enrichment: nil, max_age_ms: nil, timeout_ms: nil, wait_for_ms: nil, request_options: {})
|
|
203
218
|
#
|
|
204
219
|
# @param url [String] Page URL to inspect. Must include http:// or https://.
|
|
205
220
|
#
|
|
@@ -207,6 +222,10 @@ module ContextDev
|
|
|
207
222
|
#
|
|
208
223
|
# @param max_age_ms [Integer] Reuse a cached result this many milliseconds old or newer. Default: 86400000 (1
|
|
209
224
|
#
|
|
225
|
+
# @param timeout_ms [Integer] Optional timeout in milliseconds for the request. If the request takes longer th
|
|
226
|
+
#
|
|
227
|
+
# @param wait_for_ms [Integer] Optional browser wait time in milliseconds after initial page load before collec
|
|
228
|
+
#
|
|
210
229
|
# @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
|
|
211
230
|
#
|
|
212
231
|
# @return [ContextDev::Models::WebWebScrapeImagesResponse]
|
|
@@ -218,7 +237,11 @@ module ContextDev
|
|
|
218
237
|
@client.request(
|
|
219
238
|
method: :get,
|
|
220
239
|
path: "web/scrape/images",
|
|
221
|
-
query: query.transform_keys(
|
|
240
|
+
query: query.transform_keys(
|
|
241
|
+
max_age_ms: "maxAgeMs",
|
|
242
|
+
timeout_ms: "timeoutMS",
|
|
243
|
+
wait_for_ms: "waitForMs"
|
|
244
|
+
),
|
|
222
245
|
model: ContextDev::Models::WebWebScrapeImagesResponse,
|
|
223
246
|
options: options
|
|
224
247
|
)
|
|
@@ -229,7 +252,7 @@ module ContextDev
|
|
|
229
252
|
#
|
|
230
253
|
# Scrapes the given URL into LLM usable Markdown.
|
|
231
254
|
#
|
|
232
|
-
# @overload web_scrape_md(url:, include_frames: nil, include_images: nil, include_links: nil, max_age_ms: nil,
|
|
255
|
+
# @overload web_scrape_md(url:, include_frames: nil, include_images: nil, include_links: nil, max_age_ms: nil, pdf: nil, shorten_base64_images: nil, timeout_ms: nil, use_main_content_only: nil, wait_for_ms: nil, request_options: {})
|
|
233
256
|
#
|
|
234
257
|
# @param url [String] Full URL to scrape into LLM usable Markdown (must include http:// or https:// pr
|
|
235
258
|
#
|
|
@@ -241,12 +264,16 @@ module ContextDev
|
|
|
241
264
|
#
|
|
242
265
|
# @param max_age_ms [Integer] Return a cached result if a prior scrape for the same parameters exists and is y
|
|
243
266
|
#
|
|
244
|
-
# @param
|
|
267
|
+
# @param pdf [ContextDev::Models::WebWebScrapeMdParams::Pdf] PDF parsing controls. Use start/end to limit text extraction and OCR to an inclu
|
|
245
268
|
#
|
|
246
269
|
# @param shorten_base64_images [Boolean] Shorten base64-encoded image data in the Markdown output
|
|
247
270
|
#
|
|
271
|
+
# @param timeout_ms [Integer] Optional timeout in milliseconds for the request. If the request takes longer th
|
|
272
|
+
#
|
|
248
273
|
# @param use_main_content_only [Boolean] Extract only the main content of the page, excluding headers, footers, sidebars,
|
|
249
274
|
#
|
|
275
|
+
# @param wait_for_ms [Integer] Optional browser wait time in milliseconds after initial page load before conver
|
|
276
|
+
#
|
|
250
277
|
# @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
|
|
251
278
|
#
|
|
252
279
|
# @return [ContextDev::Models::WebWebScrapeMdResponse]
|
|
@@ -263,9 +290,10 @@ module ContextDev
|
|
|
263
290
|
include_images: "includeImages",
|
|
264
291
|
include_links: "includeLinks",
|
|
265
292
|
max_age_ms: "maxAgeMs",
|
|
266
|
-
parse_pdf: "parsePDF",
|
|
267
293
|
shorten_base64_images: "shortenBase64Images",
|
|
268
|
-
|
|
294
|
+
timeout_ms: "timeoutMS",
|
|
295
|
+
use_main_content_only: "useMainContentOnly",
|
|
296
|
+
wait_for_ms: "waitForMs"
|
|
269
297
|
),
|
|
270
298
|
model: ContextDev::Models::WebWebScrapeMdResponse,
|
|
271
299
|
options: options
|
|
@@ -277,12 +305,14 @@ module ContextDev
|
|
|
277
305
|
#
|
|
278
306
|
# Crawl an entire website's sitemap and return all discovered page URLs.
|
|
279
307
|
#
|
|
280
|
-
# @overload web_scrape_sitemap(domain:, max_links: nil, url_regex: nil, request_options: {})
|
|
308
|
+
# @overload web_scrape_sitemap(domain:, max_links: nil, timeout_ms: nil, url_regex: nil, request_options: {})
|
|
281
309
|
#
|
|
282
310
|
# @param domain [String] Domain to build a sitemap for
|
|
283
311
|
#
|
|
284
312
|
# @param max_links [Integer] Maximum number of links to return from the sitemap crawl. Defaults to 10,000. Mi
|
|
285
313
|
#
|
|
314
|
+
# @param timeout_ms [Integer] Optional timeout in milliseconds for the request. If the request takes longer th
|
|
315
|
+
#
|
|
286
316
|
# @param url_regex [String] Optional RE2-compatible regex pattern. Only URLs matching this pattern are retur
|
|
287
317
|
#
|
|
288
318
|
# @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
|
|
@@ -296,7 +326,7 @@ module ContextDev
|
|
|
296
326
|
@client.request(
|
|
297
327
|
method: :get,
|
|
298
328
|
path: "web/scrape/sitemap",
|
|
299
|
-
query: query.transform_keys(max_links: "maxLinks", url_regex: "urlRegex"),
|
|
329
|
+
query: query.transform_keys(max_links: "maxLinks", timeout_ms: "timeoutMS", url_regex: "urlRegex"),
|
|
300
330
|
model: ContextDev::Models::WebWebScrapeSitemapResponse,
|
|
301
331
|
options: options
|
|
302
332
|
)
|
data/lib/context_dev/version.rb
CHANGED
|
@@ -27,8 +27,9 @@ module ContextDev
|
|
|
27
27
|
sig { params(max_age_ms: Integer).void }
|
|
28
28
|
attr_writer :max_age_ms
|
|
29
29
|
|
|
30
|
-
# Optional timeout in milliseconds for the request.
|
|
31
|
-
#
|
|
30
|
+
# Optional timeout in milliseconds for the request. If the request takes longer
|
|
31
|
+
# than this value, it will be aborted with a 408 status code. Maximum allowed
|
|
32
|
+
# value is 300000ms (5 minutes).
|
|
32
33
|
sig { returns(T.nilable(Integer)) }
|
|
33
34
|
attr_reader :timeout_ms
|
|
34
35
|
|
|
@@ -50,8 +51,9 @@ module ContextDev
|
|
|
50
51
|
# younger than this many milliseconds. Defaults to 7 days (604800000 ms) when
|
|
51
52
|
# omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
|
|
52
53
|
max_age_ms: nil,
|
|
53
|
-
# Optional timeout in milliseconds for the request.
|
|
54
|
-
#
|
|
54
|
+
# Optional timeout in milliseconds for the request. If the request takes longer
|
|
55
|
+
# than this value, it will be aborted with a 408 status code. Maximum allowed
|
|
56
|
+
# value is 300000ms (5 minutes).
|
|
55
57
|
timeout_ms: nil,
|
|
56
58
|
request_options: {}
|
|
57
59
|
)
|
|
@@ -92,8 +92,9 @@ module ContextDev
|
|
|
92
92
|
sig { params(max_products: Integer).void }
|
|
93
93
|
attr_writer :max_products
|
|
94
94
|
|
|
95
|
-
# Optional timeout in milliseconds for the request.
|
|
96
|
-
#
|
|
95
|
+
# Optional timeout in milliseconds for the request. If the request takes longer
|
|
96
|
+
# than this value, it will be aborted with a 408 status code. Maximum allowed
|
|
97
|
+
# value is 300000ms (5 minutes).
|
|
97
98
|
sig { returns(T.nilable(Integer)) }
|
|
98
99
|
attr_reader :timeout_ms
|
|
99
100
|
|
|
@@ -117,8 +118,9 @@ module ContextDev
|
|
|
117
118
|
max_age_ms: nil,
|
|
118
119
|
# Maximum number of products to extract.
|
|
119
120
|
max_products: nil,
|
|
120
|
-
# Optional timeout in milliseconds for the request.
|
|
121
|
-
#
|
|
121
|
+
# Optional timeout in milliseconds for the request. If the request takes longer
|
|
122
|
+
# than this value, it will be aborted with a 408 status code. Maximum allowed
|
|
123
|
+
# value is 300000ms (5 minutes).
|
|
122
124
|
timeout_ms: nil
|
|
123
125
|
)
|
|
124
126
|
end
|
|
@@ -167,8 +169,9 @@ module ContextDev
|
|
|
167
169
|
sig { params(max_products: Integer).void }
|
|
168
170
|
attr_writer :max_products
|
|
169
171
|
|
|
170
|
-
# Optional timeout in milliseconds for the request.
|
|
171
|
-
#
|
|
172
|
+
# Optional timeout in milliseconds for the request. If the request takes longer
|
|
173
|
+
# than this value, it will be aborted with a 408 status code. Maximum allowed
|
|
174
|
+
# value is 300000ms (5 minutes).
|
|
172
175
|
sig { returns(T.nilable(Integer)) }
|
|
173
176
|
attr_reader :timeout_ms
|
|
174
177
|
|
|
@@ -193,8 +196,9 @@ module ContextDev
|
|
|
193
196
|
max_age_ms: nil,
|
|
194
197
|
# Maximum number of products to extract.
|
|
195
198
|
max_products: nil,
|
|
196
|
-
# Optional timeout in milliseconds for the request.
|
|
197
|
-
#
|
|
199
|
+
# Optional timeout in milliseconds for the request. If the request takes longer
|
|
200
|
+
# than this value, it will be aborted with a 408 status code. Maximum allowed
|
|
201
|
+
# value is 300000ms (5 minutes).
|
|
198
202
|
timeout_ms: nil
|
|
199
203
|
)
|
|
200
204
|
end
|
|
@@ -69,22 +69,14 @@ module ContextDev
|
|
|
69
69
|
sig { params(page: ContextDev::WebScreenshotParams::Page::OrSymbol).void }
|
|
70
70
|
attr_writer :page
|
|
71
71
|
|
|
72
|
-
# Optional
|
|
73
|
-
#
|
|
74
|
-
#
|
|
75
|
-
sig
|
|
76
|
-
|
|
77
|
-
T.nilable(ContextDev::WebScreenshotParams::Prioritize::OrSymbol)
|
|
78
|
-
)
|
|
79
|
-
end
|
|
80
|
-
attr_reader :prioritize
|
|
72
|
+
# Optional timeout in milliseconds for the request. If the request takes longer
|
|
73
|
+
# than this value, it will be aborted with a 408 status code. Maximum allowed
|
|
74
|
+
# value is 300000ms (5 minutes).
|
|
75
|
+
sig { returns(T.nilable(Integer)) }
|
|
76
|
+
attr_reader :timeout_ms
|
|
81
77
|
|
|
82
|
-
sig
|
|
83
|
-
|
|
84
|
-
prioritize: ContextDev::WebScreenshotParams::Prioritize::OrSymbol
|
|
85
|
-
).void
|
|
86
|
-
end
|
|
87
|
-
attr_writer :prioritize
|
|
78
|
+
sig { params(timeout_ms: Integer).void }
|
|
79
|
+
attr_writer :timeout_ms
|
|
88
80
|
|
|
89
81
|
# Optional browser viewport dimensions for the screenshot. Defaults to 1920x1080.
|
|
90
82
|
sig { returns(T.nilable(ContextDev::WebScreenshotParams::Viewport)) }
|
|
@@ -95,6 +87,15 @@ module ContextDev
|
|
|
95
87
|
end
|
|
96
88
|
attr_writer :viewport
|
|
97
89
|
|
|
90
|
+
# Optional browser wait time in milliseconds after initial page load before taking
|
|
91
|
+
# the screenshot. Min: 0. Max: 30000 (30 seconds). Defaults to 3000 ms when
|
|
92
|
+
# omitted.
|
|
93
|
+
sig { returns(T.nilable(Integer)) }
|
|
94
|
+
attr_reader :wait_for_ms
|
|
95
|
+
|
|
96
|
+
sig { params(wait_for_ms: Integer).void }
|
|
97
|
+
attr_writer :wait_for_ms
|
|
98
|
+
|
|
98
99
|
sig do
|
|
99
100
|
params(
|
|
100
101
|
direct_url: String,
|
|
@@ -103,8 +104,9 @@ module ContextDev
|
|
|
103
104
|
ContextDev::WebScreenshotParams::FullScreenshot::OrSymbol,
|
|
104
105
|
max_age_ms: Integer,
|
|
105
106
|
page: ContextDev::WebScreenshotParams::Page::OrSymbol,
|
|
106
|
-
|
|
107
|
+
timeout_ms: Integer,
|
|
107
108
|
viewport: ContextDev::WebScreenshotParams::Viewport::OrHash,
|
|
109
|
+
wait_for_ms: Integer,
|
|
108
110
|
request_options: ContextDev::RequestOptions::OrHash
|
|
109
111
|
).returns(T.attached_class)
|
|
110
112
|
end
|
|
@@ -131,12 +133,16 @@ module ContextDev
|
|
|
131
133
|
# provided, screenshots the main domain landing page. Only applicable when using
|
|
132
134
|
# 'domain', not 'directUrl'.
|
|
133
135
|
page: nil,
|
|
134
|
-
# Optional
|
|
135
|
-
#
|
|
136
|
-
#
|
|
137
|
-
|
|
136
|
+
# Optional timeout in milliseconds for the request. If the request takes longer
|
|
137
|
+
# than this value, it will be aborted with a 408 status code. Maximum allowed
|
|
138
|
+
# value is 300000ms (5 minutes).
|
|
139
|
+
timeout_ms: nil,
|
|
138
140
|
# Optional browser viewport dimensions for the screenshot. Defaults to 1920x1080.
|
|
139
141
|
viewport: nil,
|
|
142
|
+
# Optional browser wait time in milliseconds after initial page load before taking
|
|
143
|
+
# the screenshot. Min: 0. Max: 30000 (30 seconds). Defaults to 3000 ms when
|
|
144
|
+
# omitted.
|
|
145
|
+
wait_for_ms: nil,
|
|
140
146
|
request_options: {}
|
|
141
147
|
)
|
|
142
148
|
end
|
|
@@ -150,8 +156,9 @@ module ContextDev
|
|
|
150
156
|
ContextDev::WebScreenshotParams::FullScreenshot::OrSymbol,
|
|
151
157
|
max_age_ms: Integer,
|
|
152
158
|
page: ContextDev::WebScreenshotParams::Page::OrSymbol,
|
|
153
|
-
|
|
159
|
+
timeout_ms: Integer,
|
|
154
160
|
viewport: ContextDev::WebScreenshotParams::Viewport,
|
|
161
|
+
wait_for_ms: Integer,
|
|
155
162
|
request_options: ContextDev::RequestOptions
|
|
156
163
|
}
|
|
157
164
|
)
|
|
@@ -230,38 +237,6 @@ module ContextDev
|
|
|
230
237
|
end
|
|
231
238
|
end
|
|
232
239
|
|
|
233
|
-
# Optional parameter to prioritize screenshot capture. If 'speed', optimizes for
|
|
234
|
-
# faster capture with basic quality. If 'quality', optimizes for higher quality
|
|
235
|
-
# with longer wait times. Defaults to 'quality' if not provided.
|
|
236
|
-
module Prioritize
|
|
237
|
-
extend ContextDev::Internal::Type::Enum
|
|
238
|
-
|
|
239
|
-
TaggedSymbol =
|
|
240
|
-
T.type_alias do
|
|
241
|
-
T.all(Symbol, ContextDev::WebScreenshotParams::Prioritize)
|
|
242
|
-
end
|
|
243
|
-
OrSymbol = T.type_alias { T.any(Symbol, String) }
|
|
244
|
-
|
|
245
|
-
SPEED =
|
|
246
|
-
T.let(
|
|
247
|
-
:speed,
|
|
248
|
-
ContextDev::WebScreenshotParams::Prioritize::TaggedSymbol
|
|
249
|
-
)
|
|
250
|
-
QUALITY =
|
|
251
|
-
T.let(
|
|
252
|
-
:quality,
|
|
253
|
-
ContextDev::WebScreenshotParams::Prioritize::TaggedSymbol
|
|
254
|
-
)
|
|
255
|
-
|
|
256
|
-
sig do
|
|
257
|
-
override.returns(
|
|
258
|
-
T::Array[ContextDev::WebScreenshotParams::Prioritize::TaggedSymbol]
|
|
259
|
-
)
|
|
260
|
-
end
|
|
261
|
-
def self.values
|
|
262
|
-
end
|
|
263
|
-
end
|
|
264
|
-
|
|
265
240
|
class Viewport < ContextDev::Internal::Type::BaseModel
|
|
266
241
|
OrHash =
|
|
267
242
|
T.type_alias do
|
|
@@ -69,14 +69,13 @@ module ContextDev
|
|
|
69
69
|
sig { params(max_pages: Integer).void }
|
|
70
70
|
attr_writer :max_pages
|
|
71
71
|
|
|
72
|
-
#
|
|
73
|
-
#
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
attr_reader :parse_pdf
|
|
72
|
+
# PDF parsing controls. Use start/end to limit text extraction and OCR to an
|
|
73
|
+
# inclusive 1-based page range.
|
|
74
|
+
sig { returns(T.nilable(ContextDev::WebWebCrawlMdParams::Pdf)) }
|
|
75
|
+
attr_reader :pdf
|
|
77
76
|
|
|
78
|
-
sig { params(
|
|
79
|
-
attr_writer :
|
|
77
|
+
sig { params(pdf: ContextDev::WebWebCrawlMdParams::Pdf::OrHash).void }
|
|
78
|
+
attr_writer :pdf
|
|
80
79
|
|
|
81
80
|
# Truncate base64-encoded image data in the Markdown output
|
|
82
81
|
sig { returns(T.nilable(T::Boolean)) }
|
|
@@ -85,6 +84,25 @@ module ContextDev
|
|
|
85
84
|
sig { params(shorten_base64_images: T::Boolean).void }
|
|
86
85
|
attr_writer :shorten_base64_images
|
|
87
86
|
|
|
87
|
+
# Soft time budget for the crawl in milliseconds. After each scrape, the crawler
|
|
88
|
+
# checks the elapsed time and, if exceeded, returns the pages collected so far
|
|
89
|
+
# instead of continuing. Min: 10000 (10s). Max: 240000 (4 min). Default: 120000 (2
|
|
90
|
+
# min).
|
|
91
|
+
sig { returns(T.nilable(Integer)) }
|
|
92
|
+
attr_reader :stop_after_ms
|
|
93
|
+
|
|
94
|
+
sig { params(stop_after_ms: Integer).void }
|
|
95
|
+
attr_writer :stop_after_ms
|
|
96
|
+
|
|
97
|
+
# Optional timeout in milliseconds for the request. If the request takes longer
|
|
98
|
+
# than this value, it will be aborted with a 408 status code. Maximum allowed
|
|
99
|
+
# value is 300000ms (5 minutes).
|
|
100
|
+
sig { returns(T.nilable(Integer)) }
|
|
101
|
+
attr_reader :timeout_ms
|
|
102
|
+
|
|
103
|
+
sig { params(timeout_ms: Integer).void }
|
|
104
|
+
attr_writer :timeout_ms
|
|
105
|
+
|
|
88
106
|
# Regex pattern. Only URLs matching this pattern will be followed and scraped.
|
|
89
107
|
sig { returns(T.nilable(String)) }
|
|
90
108
|
attr_reader :url_regex
|
|
@@ -100,6 +118,14 @@ module ContextDev
|
|
|
100
118
|
sig { params(use_main_content_only: T::Boolean).void }
|
|
101
119
|
attr_writer :use_main_content_only
|
|
102
120
|
|
|
121
|
+
# Optional browser wait time in milliseconds after initial page load for each
|
|
122
|
+
# crawled page. Min: 0. Max: 30000 (30 seconds).
|
|
123
|
+
sig { returns(T.nilable(Integer)) }
|
|
124
|
+
attr_reader :wait_for_ms
|
|
125
|
+
|
|
126
|
+
sig { params(wait_for_ms: Integer).void }
|
|
127
|
+
attr_writer :wait_for_ms
|
|
128
|
+
|
|
103
129
|
sig do
|
|
104
130
|
params(
|
|
105
131
|
url: String,
|
|
@@ -110,10 +136,13 @@ module ContextDev
|
|
|
110
136
|
max_age_ms: Integer,
|
|
111
137
|
max_depth: Integer,
|
|
112
138
|
max_pages: Integer,
|
|
113
|
-
|
|
139
|
+
pdf: ContextDev::WebWebCrawlMdParams::Pdf::OrHash,
|
|
114
140
|
shorten_base64_images: T::Boolean,
|
|
141
|
+
stop_after_ms: Integer,
|
|
142
|
+
timeout_ms: Integer,
|
|
115
143
|
url_regex: String,
|
|
116
144
|
use_main_content_only: T::Boolean,
|
|
145
|
+
wait_for_ms: Integer,
|
|
117
146
|
request_options: ContextDev::RequestOptions::OrHash
|
|
118
147
|
).returns(T.attached_class)
|
|
119
148
|
end
|
|
@@ -139,17 +168,28 @@ module ContextDev
|
|
|
139
168
|
max_depth: nil,
|
|
140
169
|
# Maximum number of pages to crawl. Hard cap: 500.
|
|
141
170
|
max_pages: nil,
|
|
142
|
-
#
|
|
143
|
-
#
|
|
144
|
-
|
|
145
|
-
parse_pdf: nil,
|
|
171
|
+
# PDF parsing controls. Use start/end to limit text extraction and OCR to an
|
|
172
|
+
# inclusive 1-based page range.
|
|
173
|
+
pdf: nil,
|
|
146
174
|
# Truncate base64-encoded image data in the Markdown output
|
|
147
175
|
shorten_base64_images: nil,
|
|
176
|
+
# Soft time budget for the crawl in milliseconds. After each scrape, the crawler
|
|
177
|
+
# checks the elapsed time and, if exceeded, returns the pages collected so far
|
|
178
|
+
# instead of continuing. Min: 10000 (10s). Max: 240000 (4 min). Default: 120000 (2
|
|
179
|
+
# min).
|
|
180
|
+
stop_after_ms: nil,
|
|
181
|
+
# Optional timeout in milliseconds for the request. If the request takes longer
|
|
182
|
+
# than this value, it will be aborted with a 408 status code. Maximum allowed
|
|
183
|
+
# value is 300000ms (5 minutes).
|
|
184
|
+
timeout_ms: nil,
|
|
148
185
|
# Regex pattern. Only URLs matching this pattern will be followed and scraped.
|
|
149
186
|
url_regex: nil,
|
|
150
187
|
# Extract only the main content, stripping headers, footers, sidebars, and
|
|
151
188
|
# navigation
|
|
152
189
|
use_main_content_only: nil,
|
|
190
|
+
# Optional browser wait time in milliseconds after initial page load for each
|
|
191
|
+
# crawled page. Min: 0. Max: 30000 (30 seconds).
|
|
192
|
+
wait_for_ms: nil,
|
|
153
193
|
request_options: {}
|
|
154
194
|
)
|
|
155
195
|
end
|
|
@@ -165,16 +205,81 @@ module ContextDev
|
|
|
165
205
|
max_age_ms: Integer,
|
|
166
206
|
max_depth: Integer,
|
|
167
207
|
max_pages: Integer,
|
|
168
|
-
|
|
208
|
+
pdf: ContextDev::WebWebCrawlMdParams::Pdf,
|
|
169
209
|
shorten_base64_images: T::Boolean,
|
|
210
|
+
stop_after_ms: Integer,
|
|
211
|
+
timeout_ms: Integer,
|
|
170
212
|
url_regex: String,
|
|
171
213
|
use_main_content_only: T::Boolean,
|
|
214
|
+
wait_for_ms: Integer,
|
|
172
215
|
request_options: ContextDev::RequestOptions
|
|
173
216
|
}
|
|
174
217
|
)
|
|
175
218
|
end
|
|
176
219
|
def to_hash
|
|
177
220
|
end
|
|
221
|
+
|
|
222
|
+
class Pdf < ContextDev::Internal::Type::BaseModel
|
|
223
|
+
OrHash =
|
|
224
|
+
T.type_alias do
|
|
225
|
+
T.any(
|
|
226
|
+
ContextDev::WebWebCrawlMdParams::Pdf,
|
|
227
|
+
ContextDev::Internal::AnyHash
|
|
228
|
+
)
|
|
229
|
+
end
|
|
230
|
+
|
|
231
|
+
# Last 1-based PDF page to parse. When omitted, parsing ends at the last page.
|
|
232
|
+
# Must be greater than or equal to start when both are provided.
|
|
233
|
+
sig { returns(T.nilable(Integer)) }
|
|
234
|
+
attr_reader :end_
|
|
235
|
+
|
|
236
|
+
sig { params(end_: Integer).void }
|
|
237
|
+
attr_writer :end_
|
|
238
|
+
|
|
239
|
+
# When true, PDF pages are fetched and parsed. When false, PDF pages are skipped
|
|
240
|
+
# entirely (not included in results and not counted as failures).
|
|
241
|
+
sig { returns(T.nilable(T::Boolean)) }
|
|
242
|
+
attr_reader :should_parse
|
|
243
|
+
|
|
244
|
+
sig { params(should_parse: T::Boolean).void }
|
|
245
|
+
attr_writer :should_parse
|
|
246
|
+
|
|
247
|
+
# First 1-based PDF page to parse. When omitted, parsing starts at the first page.
|
|
248
|
+
sig { returns(T.nilable(Integer)) }
|
|
249
|
+
attr_reader :start
|
|
250
|
+
|
|
251
|
+
sig { params(start: Integer).void }
|
|
252
|
+
attr_writer :start
|
|
253
|
+
|
|
254
|
+
# PDF parsing controls. Use start/end to limit text extraction and OCR to an
|
|
255
|
+
# inclusive 1-based page range.
|
|
256
|
+
sig do
|
|
257
|
+
params(
|
|
258
|
+
end_: Integer,
|
|
259
|
+
should_parse: T::Boolean,
|
|
260
|
+
start: Integer
|
|
261
|
+
).returns(T.attached_class)
|
|
262
|
+
end
|
|
263
|
+
def self.new(
|
|
264
|
+
# Last 1-based PDF page to parse. When omitted, parsing ends at the last page.
|
|
265
|
+
# Must be greater than or equal to start when both are provided.
|
|
266
|
+
end_: nil,
|
|
267
|
+
# When true, PDF pages are fetched and parsed. When false, PDF pages are skipped
|
|
268
|
+
# entirely (not included in results and not counted as failures).
|
|
269
|
+
should_parse: nil,
|
|
270
|
+
# First 1-based PDF page to parse. When omitted, parsing starts at the first page.
|
|
271
|
+
start: nil
|
|
272
|
+
)
|
|
273
|
+
end
|
|
274
|
+
|
|
275
|
+
sig do
|
|
276
|
+
override.returns(
|
|
277
|
+
{ end_: Integer, should_parse: T::Boolean, start: Integer }
|
|
278
|
+
)
|
|
279
|
+
end
|
|
280
|
+
def to_hash
|
|
281
|
+
end
|
|
282
|
+
end
|
|
178
283
|
end
|
|
179
284
|
end
|
|
180
285
|
end
|
|
@@ -64,7 +64,8 @@ module ContextDev
|
|
|
64
64
|
sig { returns(Integer) }
|
|
65
65
|
attr_accessor :num_failed
|
|
66
66
|
|
|
67
|
-
# Number of URLs skipped (PDFs when
|
|
67
|
+
# Number of URLs skipped (PDFs when pdf.shouldParse=false, or URLs not matching
|
|
68
|
+
# urlRegex)
|
|
68
69
|
sig { returns(Integer) }
|
|
69
70
|
attr_accessor :num_skipped
|
|
70
71
|
|
|
@@ -90,7 +91,8 @@ module ContextDev
|
|
|
90
91
|
max_crawl_depth:,
|
|
91
92
|
# Number of pages that failed to crawl
|
|
92
93
|
num_failed:,
|
|
93
|
-
# Number of URLs skipped (PDFs when
|
|
94
|
+
# Number of URLs skipped (PDFs when pdf.shouldParse=false, or URLs not matching
|
|
95
|
+
# urlRegex)
|
|
94
96
|
num_skipped:,
|
|
95
97
|
# Number of pages successfully crawled
|
|
96
98
|
num_succeeded:,
|