context.dev 1.16.0 → 1.18.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (34) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +19 -0
  3. data/README.md +3 -3
  4. data/lib/context_dev/models/ai_extract_product_params.rb +4 -3
  5. data/lib/context_dev/models/ai_extract_products_params.rb +8 -6
  6. data/lib/context_dev/models/web_screenshot_params.rb +18 -21
  7. data/lib/context_dev/models/web_web_crawl_md_params.rb +72 -8
  8. data/lib/context_dev/models/web_web_crawl_md_response.rb +3 -2
  9. data/lib/context_dev/models/web_web_scrape_html_params.rb +61 -8
  10. data/lib/context_dev/models/web_web_scrape_images_params.rb +20 -1
  11. data/lib/context_dev/models/web_web_scrape_md_params.rb +61 -8
  12. data/lib/context_dev/models/web_web_scrape_sitemap_params.rb +11 -1
  13. data/lib/context_dev/resources/ai.rb +1 -1
  14. data/lib/context_dev/resources/web.rb +46 -16
  15. data/lib/context_dev/version.rb +1 -1
  16. data/rbi/context_dev/models/ai_extract_product_params.rbi +6 -4
  17. data/rbi/context_dev/models/ai_extract_products_params.rbi +12 -8
  18. data/rbi/context_dev/models/web_screenshot_params.rbi +28 -53
  19. data/rbi/context_dev/models/web_web_crawl_md_params.rbi +118 -13
  20. data/rbi/context_dev/models/web_web_crawl_md_response.rbi +4 -2
  21. data/rbi/context_dev/models/web_web_scrape_html_params.rbi +101 -13
  22. data/rbi/context_dev/models/web_web_scrape_images_params.rbi +28 -0
  23. data/rbi/context_dev/models/web_web_scrape_md_params.rbi +101 -13
  24. data/rbi/context_dev/models/web_web_scrape_sitemap_params.rbi +15 -0
  25. data/rbi/context_dev/resources/ai.rbi +3 -2
  26. data/rbi/context_dev/resources/web.rbi +69 -20
  27. data/sig/context_dev/models/web_screenshot_params.rbs +13 -19
  28. data/sig/context_dev/models/web_web_crawl_md_params.rbs +53 -6
  29. data/sig/context_dev/models/web_web_scrape_html_params.rbs +45 -5
  30. data/sig/context_dev/models/web_web_scrape_images_params.rbs +15 -1
  31. data/sig/context_dev/models/web_web_scrape_md_params.rbs +46 -6
  32. data/sig/context_dev/models/web_web_scrape_sitemap_params.rbs +12 -1
  33. data/sig/context_dev/resources/web.rbs +15 -4
  34. metadata +2 -2
@@ -70,7 +70,7 @@ module ContextDev
70
70
  #
71
71
  # Capture a screenshot of a website.
72
72
  #
73
- # @overload screenshot(direct_url: nil, domain: nil, full_screenshot: nil, max_age_ms: nil, page: nil, prioritize: nil, viewport: nil, request_options: {})
73
+ # @overload screenshot(direct_url: nil, domain: nil, full_screenshot: nil, max_age_ms: nil, page: nil, timeout_ms: nil, viewport: nil, wait_for_ms: nil, request_options: {})
74
74
  #
75
75
  # @param direct_url [String] A specific URL to screenshot directly, bypassing domain resolution (e.g., 'https
76
76
  #
@@ -82,10 +82,12 @@ module ContextDev
82
82
  #
83
83
  # @param page [Symbol, ContextDev::Models::WebScreenshotParams::Page] Optional parameter to specify which page type to screenshot. If provided, the sy
84
84
  #
85
- # @param prioritize [Symbol, ContextDev::Models::WebScreenshotParams::Prioritize] Optional parameter to prioritize screenshot capture. If 'speed', optimizes for f
85
+ # @param timeout_ms [Integer] Optional timeout in milliseconds for the request. If the request takes longer th
86
86
  #
87
87
  # @param viewport [ContextDev::Models::WebScreenshotParams::Viewport] Optional browser viewport dimensions for the screenshot. Defaults to 1920x1080.
88
88
  #
89
+ # @param wait_for_ms [Integer] Optional browser wait time in milliseconds after initial page load before taking
90
+ #
89
91
  # @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
90
92
  #
91
93
  # @return [ContextDev::Models::WebScreenshotResponse]
@@ -100,7 +102,9 @@ module ContextDev
100
102
  query: query.transform_keys(
101
103
  direct_url: "directUrl",
102
104
  full_screenshot: "fullScreenshot",
103
- max_age_ms: "maxAgeMs"
105
+ max_age_ms: "maxAgeMs",
106
+ timeout_ms: "timeoutMS",
107
+ wait_for_ms: "waitForMs"
104
108
  ),
105
109
  model: ContextDev::Models::WebScreenshotResponse,
106
110
  options: options
@@ -113,7 +117,7 @@ module ContextDev
113
117
  # Performs a crawl starting from a given URL, extracts page content as Markdown,
114
118
  # and returns results for all crawled pages.
115
119
  #
116
- # @overload web_crawl_md(url:, follow_subdomains: nil, include_frames: nil, include_images: nil, include_links: nil, max_age_ms: nil, max_depth: nil, max_pages: nil, parse_pdf: nil, shorten_base64_images: nil, url_regex: nil, use_main_content_only: nil, request_options: {})
120
+ # @overload web_crawl_md(url:, follow_subdomains: nil, include_frames: nil, include_images: nil, include_links: nil, max_age_ms: nil, max_depth: nil, max_pages: nil, pdf: nil, shorten_base64_images: nil, stop_after_ms: nil, timeout_ms: nil, url_regex: nil, use_main_content_only: nil, wait_for_ms: nil, request_options: {})
117
121
  #
118
122
  # @param url [String] The starting URL for the crawl (must include http:// or https:// protocol)
119
123
  #
@@ -131,14 +135,20 @@ module ContextDev
131
135
  #
132
136
  # @param max_pages [Integer] Maximum number of pages to crawl. Hard cap: 500.
133
137
  #
134
- # @param parse_pdf [Boolean] When true (default), PDF pages are fetched and their text layer is extracted and
138
+ # @param pdf [ContextDev::Models::WebWebCrawlMdParams::Pdf] PDF parsing controls. Use start/end to limit text extraction and OCR to an inclu
135
139
  #
136
140
  # @param shorten_base64_images [Boolean] Truncate base64-encoded image data in the Markdown output
137
141
  #
142
+ # @param stop_after_ms [Integer] Soft time budget for the crawl in milliseconds. After each scrape, the crawler c
143
+ #
144
+ # @param timeout_ms [Integer] Optional timeout in milliseconds for the request. If the request takes longer th
145
+ #
138
146
  # @param url_regex [String] Regex pattern. Only URLs matching this pattern will be followed and scraped.
139
147
  #
140
148
  # @param use_main_content_only [Boolean] Extract only the main content, stripping headers, footers, sidebars, and navigat
141
149
  #
150
+ # @param wait_for_ms [Integer] Optional browser wait time in milliseconds after initial page load for each craw
151
+ #
142
152
  # @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
143
153
  #
144
154
  # @return [ContextDev::Models::WebWebCrawlMdResponse]
@@ -160,7 +170,7 @@ module ContextDev
160
170
  #
161
171
  # Scrapes the given URL and returns the raw HTML content of the page.
162
172
  #
163
- # @overload web_scrape_html(url:, include_frames: nil, max_age_ms: nil, parse_pdf: nil, request_options: {})
173
+ # @overload web_scrape_html(url:, include_frames: nil, max_age_ms: nil, pdf: nil, timeout_ms: nil, wait_for_ms: nil, request_options: {})
164
174
  #
165
175
  # @param url [String] Full URL to scrape (must include http:// or https:// protocol)
166
176
  #
@@ -168,7 +178,11 @@ module ContextDev
168
178
  #
169
179
  # @param max_age_ms [Integer] Return a cached result if a prior scrape for the same parameters exists and is y
170
180
  #
171
- # @param parse_pdf [Boolean] When true (default), PDF URLs are fetched and their text layer is extracted and
181
+ # @param pdf [ContextDev::Models::WebWebScrapeHTMLParams::Pdf] PDF parsing controls. Use start/end to limit text extraction and OCR to an inclu
182
+ #
183
+ # @param timeout_ms [Integer] Optional timeout in milliseconds for the request. If the request takes longer th
184
+ #
185
+ # @param wait_for_ms [Integer] Optional browser wait time in milliseconds after initial page load. Min: 0. Max:
172
186
  #
173
187
  # @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
174
188
  #
@@ -184,7 +198,8 @@ module ContextDev
184
198
  query: query.transform_keys(
185
199
  include_frames: "includeFrames",
186
200
  max_age_ms: "maxAgeMs",
187
- parse_pdf: "parsePDF"
201
+ timeout_ms: "timeoutMS",
202
+ wait_for_ms: "waitForMs"
188
203
  ),
189
204
  model: ContextDev::Models::WebWebScrapeHTMLResponse,
190
205
  options: options
@@ -199,7 +214,7 @@ module ContextDev
199
214
  # embeds. The base request costs 1 credit; enrichment costs 1 credit per returned
200
215
  # image.
201
216
  #
202
- # @overload web_scrape_images(url:, enrichment: nil, max_age_ms: nil, request_options: {})
217
+ # @overload web_scrape_images(url:, enrichment: nil, max_age_ms: nil, timeout_ms: nil, wait_for_ms: nil, request_options: {})
203
218
  #
204
219
  # @param url [String] Page URL to inspect. Must include http:// or https://.
205
220
  #
@@ -207,6 +222,10 @@ module ContextDev
207
222
  #
208
223
  # @param max_age_ms [Integer] Reuse a cached result this many milliseconds old or newer. Default: 86400000 (1
209
224
  #
225
+ # @param timeout_ms [Integer] Optional timeout in milliseconds for the request. If the request takes longer th
226
+ #
227
+ # @param wait_for_ms [Integer] Optional browser wait time in milliseconds after initial page load before collec
228
+ #
210
229
  # @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
211
230
  #
212
231
  # @return [ContextDev::Models::WebWebScrapeImagesResponse]
@@ -218,7 +237,11 @@ module ContextDev
218
237
  @client.request(
219
238
  method: :get,
220
239
  path: "web/scrape/images",
221
- query: query.transform_keys(max_age_ms: "maxAgeMs"),
240
+ query: query.transform_keys(
241
+ max_age_ms: "maxAgeMs",
242
+ timeout_ms: "timeoutMS",
243
+ wait_for_ms: "waitForMs"
244
+ ),
222
245
  model: ContextDev::Models::WebWebScrapeImagesResponse,
223
246
  options: options
224
247
  )
@@ -229,7 +252,7 @@ module ContextDev
229
252
  #
230
253
  # Scrapes the given URL into LLM usable Markdown.
231
254
  #
232
- # @overload web_scrape_md(url:, include_frames: nil, include_images: nil, include_links: nil, max_age_ms: nil, parse_pdf: nil, shorten_base64_images: nil, use_main_content_only: nil, request_options: {})
255
+ # @overload web_scrape_md(url:, include_frames: nil, include_images: nil, include_links: nil, max_age_ms: nil, pdf: nil, shorten_base64_images: nil, timeout_ms: nil, use_main_content_only: nil, wait_for_ms: nil, request_options: {})
233
256
  #
234
257
  # @param url [String] Full URL to scrape into LLM usable Markdown (must include http:// or https:// pr
235
258
  #
@@ -241,12 +264,16 @@ module ContextDev
241
264
  #
242
265
  # @param max_age_ms [Integer] Return a cached result if a prior scrape for the same parameters exists and is y
243
266
  #
244
- # @param parse_pdf [Boolean] When true (default), PDF URLs are fetched and their text layer is extracted and
267
+ # @param pdf [ContextDev::Models::WebWebScrapeMdParams::Pdf] PDF parsing controls. Use start/end to limit text extraction and OCR to an inclu
245
268
  #
246
269
  # @param shorten_base64_images [Boolean] Shorten base64-encoded image data in the Markdown output
247
270
  #
271
+ # @param timeout_ms [Integer] Optional timeout in milliseconds for the request. If the request takes longer th
272
+ #
248
273
  # @param use_main_content_only [Boolean] Extract only the main content of the page, excluding headers, footers, sidebars,
249
274
  #
275
+ # @param wait_for_ms [Integer] Optional browser wait time in milliseconds after initial page load before conver
276
+ #
250
277
  # @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
251
278
  #
252
279
  # @return [ContextDev::Models::WebWebScrapeMdResponse]
@@ -263,9 +290,10 @@ module ContextDev
263
290
  include_images: "includeImages",
264
291
  include_links: "includeLinks",
265
292
  max_age_ms: "maxAgeMs",
266
- parse_pdf: "parsePDF",
267
293
  shorten_base64_images: "shortenBase64Images",
268
- use_main_content_only: "useMainContentOnly"
294
+ timeout_ms: "timeoutMS",
295
+ use_main_content_only: "useMainContentOnly",
296
+ wait_for_ms: "waitForMs"
269
297
  ),
270
298
  model: ContextDev::Models::WebWebScrapeMdResponse,
271
299
  options: options
@@ -277,12 +305,14 @@ module ContextDev
277
305
  #
278
306
  # Crawl an entire website's sitemap and return all discovered page URLs.
279
307
  #
280
- # @overload web_scrape_sitemap(domain:, max_links: nil, url_regex: nil, request_options: {})
308
+ # @overload web_scrape_sitemap(domain:, max_links: nil, timeout_ms: nil, url_regex: nil, request_options: {})
281
309
  #
282
310
  # @param domain [String] Domain to build a sitemap for
283
311
  #
284
312
  # @param max_links [Integer] Maximum number of links to return from the sitemap crawl. Defaults to 10,000. Mi
285
313
  #
314
+ # @param timeout_ms [Integer] Optional timeout in milliseconds for the request. If the request takes longer th
315
+ #
286
316
  # @param url_regex [String] Optional RE2-compatible regex pattern. Only URLs matching this pattern are retur
287
317
  #
288
318
  # @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
@@ -296,7 +326,7 @@ module ContextDev
296
326
  @client.request(
297
327
  method: :get,
298
328
  path: "web/scrape/sitemap",
299
- query: query.transform_keys(max_links: "maxLinks", url_regex: "urlRegex"),
329
+ query: query.transform_keys(max_links: "maxLinks", timeout_ms: "timeoutMS", url_regex: "urlRegex"),
300
330
  model: ContextDev::Models::WebWebScrapeSitemapResponse,
301
331
  options: options
302
332
  )
@@ -1,5 +1,5 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  module ContextDev
4
- VERSION = "1.16.0"
4
+ VERSION = "1.18.0"
5
5
  end
@@ -27,8 +27,9 @@ module ContextDev
27
27
  sig { params(max_age_ms: Integer).void }
28
28
  attr_writer :max_age_ms
29
29
 
30
- # Optional timeout in milliseconds for the request. Maximum allowed value is
31
- # 300000ms (5 minutes).
30
+ # Optional timeout in milliseconds for the request. If the request takes longer
31
+ # than this value, it will be aborted with a 408 status code. Maximum allowed
32
+ # value is 300000ms (5 minutes).
32
33
  sig { returns(T.nilable(Integer)) }
33
34
  attr_reader :timeout_ms
34
35
 
@@ -50,8 +51,9 @@ module ContextDev
50
51
  # younger than this many milliseconds. Defaults to 7 days (604800000 ms) when
51
52
  # omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
52
53
  max_age_ms: nil,
53
- # Optional timeout in milliseconds for the request. Maximum allowed value is
54
- # 300000ms (5 minutes).
54
+ # Optional timeout in milliseconds for the request. If the request takes longer
55
+ # than this value, it will be aborted with a 408 status code. Maximum allowed
56
+ # value is 300000ms (5 minutes).
55
57
  timeout_ms: nil,
56
58
  request_options: {}
57
59
  )
@@ -92,8 +92,9 @@ module ContextDev
92
92
  sig { params(max_products: Integer).void }
93
93
  attr_writer :max_products
94
94
 
95
- # Optional timeout in milliseconds for the request. Maximum allowed value is
96
- # 300000ms (5 minutes).
95
+ # Optional timeout in milliseconds for the request. If the request takes longer
96
+ # than this value, it will be aborted with a 408 status code. Maximum allowed
97
+ # value is 300000ms (5 minutes).
97
98
  sig { returns(T.nilable(Integer)) }
98
99
  attr_reader :timeout_ms
99
100
 
@@ -117,8 +118,9 @@ module ContextDev
117
118
  max_age_ms: nil,
118
119
  # Maximum number of products to extract.
119
120
  max_products: nil,
120
- # Optional timeout in milliseconds for the request. Maximum allowed value is
121
- # 300000ms (5 minutes).
121
+ # Optional timeout in milliseconds for the request. If the request takes longer
122
+ # than this value, it will be aborted with a 408 status code. Maximum allowed
123
+ # value is 300000ms (5 minutes).
122
124
  timeout_ms: nil
123
125
  )
124
126
  end
@@ -167,8 +169,9 @@ module ContextDev
167
169
  sig { params(max_products: Integer).void }
168
170
  attr_writer :max_products
169
171
 
170
- # Optional timeout in milliseconds for the request. Maximum allowed value is
171
- # 300000ms (5 minutes).
172
+ # Optional timeout in milliseconds for the request. If the request takes longer
173
+ # than this value, it will be aborted with a 408 status code. Maximum allowed
174
+ # value is 300000ms (5 minutes).
172
175
  sig { returns(T.nilable(Integer)) }
173
176
  attr_reader :timeout_ms
174
177
 
@@ -193,8 +196,9 @@ module ContextDev
193
196
  max_age_ms: nil,
194
197
  # Maximum number of products to extract.
195
198
  max_products: nil,
196
- # Optional timeout in milliseconds for the request. Maximum allowed value is
197
- # 300000ms (5 minutes).
199
+ # Optional timeout in milliseconds for the request. If the request takes longer
200
+ # than this value, it will be aborted with a 408 status code. Maximum allowed
201
+ # value is 300000ms (5 minutes).
198
202
  timeout_ms: nil
199
203
  )
200
204
  end
@@ -69,22 +69,14 @@ module ContextDev
69
69
  sig { params(page: ContextDev::WebScreenshotParams::Page::OrSymbol).void }
70
70
  attr_writer :page
71
71
 
72
- # Optional parameter to prioritize screenshot capture. If 'speed', optimizes for
73
- # faster capture with basic quality. If 'quality', optimizes for higher quality
74
- # with longer wait times. Defaults to 'quality' if not provided.
75
- sig do
76
- returns(
77
- T.nilable(ContextDev::WebScreenshotParams::Prioritize::OrSymbol)
78
- )
79
- end
80
- attr_reader :prioritize
72
+ # Optional timeout in milliseconds for the request. If the request takes longer
73
+ # than this value, it will be aborted with a 408 status code. Maximum allowed
74
+ # value is 300000ms (5 minutes).
75
+ sig { returns(T.nilable(Integer)) }
76
+ attr_reader :timeout_ms
81
77
 
82
- sig do
83
- params(
84
- prioritize: ContextDev::WebScreenshotParams::Prioritize::OrSymbol
85
- ).void
86
- end
87
- attr_writer :prioritize
78
+ sig { params(timeout_ms: Integer).void }
79
+ attr_writer :timeout_ms
88
80
 
89
81
  # Optional browser viewport dimensions for the screenshot. Defaults to 1920x1080.
90
82
  sig { returns(T.nilable(ContextDev::WebScreenshotParams::Viewport)) }
@@ -95,6 +87,15 @@ module ContextDev
95
87
  end
96
88
  attr_writer :viewport
97
89
 
90
+ # Optional browser wait time in milliseconds after initial page load before taking
91
+ # the screenshot. Min: 0. Max: 30000 (30 seconds). Defaults to 3000 ms when
92
+ # omitted.
93
+ sig { returns(T.nilable(Integer)) }
94
+ attr_reader :wait_for_ms
95
+
96
+ sig { params(wait_for_ms: Integer).void }
97
+ attr_writer :wait_for_ms
98
+
98
99
  sig do
99
100
  params(
100
101
  direct_url: String,
@@ -103,8 +104,9 @@ module ContextDev
103
104
  ContextDev::WebScreenshotParams::FullScreenshot::OrSymbol,
104
105
  max_age_ms: Integer,
105
106
  page: ContextDev::WebScreenshotParams::Page::OrSymbol,
106
- prioritize: ContextDev::WebScreenshotParams::Prioritize::OrSymbol,
107
+ timeout_ms: Integer,
107
108
  viewport: ContextDev::WebScreenshotParams::Viewport::OrHash,
109
+ wait_for_ms: Integer,
108
110
  request_options: ContextDev::RequestOptions::OrHash
109
111
  ).returns(T.attached_class)
110
112
  end
@@ -131,12 +133,16 @@ module ContextDev
131
133
  # provided, screenshots the main domain landing page. Only applicable when using
132
134
  # 'domain', not 'directUrl'.
133
135
  page: nil,
134
- # Optional parameter to prioritize screenshot capture. If 'speed', optimizes for
135
- # faster capture with basic quality. If 'quality', optimizes for higher quality
136
- # with longer wait times. Defaults to 'quality' if not provided.
137
- prioritize: nil,
136
+ # Optional timeout in milliseconds for the request. If the request takes longer
137
+ # than this value, it will be aborted with a 408 status code. Maximum allowed
138
+ # value is 300000ms (5 minutes).
139
+ timeout_ms: nil,
138
140
  # Optional browser viewport dimensions for the screenshot. Defaults to 1920x1080.
139
141
  viewport: nil,
142
+ # Optional browser wait time in milliseconds after initial page load before taking
143
+ # the screenshot. Min: 0. Max: 30000 (30 seconds). Defaults to 3000 ms when
144
+ # omitted.
145
+ wait_for_ms: nil,
140
146
  request_options: {}
141
147
  )
142
148
  end
@@ -150,8 +156,9 @@ module ContextDev
150
156
  ContextDev::WebScreenshotParams::FullScreenshot::OrSymbol,
151
157
  max_age_ms: Integer,
152
158
  page: ContextDev::WebScreenshotParams::Page::OrSymbol,
153
- prioritize: ContextDev::WebScreenshotParams::Prioritize::OrSymbol,
159
+ timeout_ms: Integer,
154
160
  viewport: ContextDev::WebScreenshotParams::Viewport,
161
+ wait_for_ms: Integer,
155
162
  request_options: ContextDev::RequestOptions
156
163
  }
157
164
  )
@@ -230,38 +237,6 @@ module ContextDev
230
237
  end
231
238
  end
232
239
 
233
- # Optional parameter to prioritize screenshot capture. If 'speed', optimizes for
234
- # faster capture with basic quality. If 'quality', optimizes for higher quality
235
- # with longer wait times. Defaults to 'quality' if not provided.
236
- module Prioritize
237
- extend ContextDev::Internal::Type::Enum
238
-
239
- TaggedSymbol =
240
- T.type_alias do
241
- T.all(Symbol, ContextDev::WebScreenshotParams::Prioritize)
242
- end
243
- OrSymbol = T.type_alias { T.any(Symbol, String) }
244
-
245
- SPEED =
246
- T.let(
247
- :speed,
248
- ContextDev::WebScreenshotParams::Prioritize::TaggedSymbol
249
- )
250
- QUALITY =
251
- T.let(
252
- :quality,
253
- ContextDev::WebScreenshotParams::Prioritize::TaggedSymbol
254
- )
255
-
256
- sig do
257
- override.returns(
258
- T::Array[ContextDev::WebScreenshotParams::Prioritize::TaggedSymbol]
259
- )
260
- end
261
- def self.values
262
- end
263
- end
264
-
265
240
  class Viewport < ContextDev::Internal::Type::BaseModel
266
241
  OrHash =
267
242
  T.type_alias do
@@ -69,14 +69,13 @@ module ContextDev
69
69
  sig { params(max_pages: Integer).void }
70
70
  attr_writer :max_pages
71
71
 
72
- # When true (default), PDF pages are fetched and their text layer is extracted and
73
- # converted to Markdown alongside HTML pages. When false, PDF pages are skipped
74
- # entirely (not included in results and not counted as failures).
75
- sig { returns(T.nilable(T::Boolean)) }
76
- attr_reader :parse_pdf
72
+ # PDF parsing controls. Use start/end to limit text extraction and OCR to an
73
+ # inclusive 1-based page range.
74
+ sig { returns(T.nilable(ContextDev::WebWebCrawlMdParams::Pdf)) }
75
+ attr_reader :pdf
77
76
 
78
- sig { params(parse_pdf: T::Boolean).void }
79
- attr_writer :parse_pdf
77
+ sig { params(pdf: ContextDev::WebWebCrawlMdParams::Pdf::OrHash).void }
78
+ attr_writer :pdf
80
79
 
81
80
  # Truncate base64-encoded image data in the Markdown output
82
81
  sig { returns(T.nilable(T::Boolean)) }
@@ -85,6 +84,25 @@ module ContextDev
85
84
  sig { params(shorten_base64_images: T::Boolean).void }
86
85
  attr_writer :shorten_base64_images
87
86
 
87
+ # Soft time budget for the crawl in milliseconds. After each scrape, the crawler
88
+ # checks the elapsed time and, if exceeded, returns the pages collected so far
89
+ # instead of continuing. Min: 10000 (10s). Max: 240000 (4 min). Default: 120000 (2
90
+ # min).
91
+ sig { returns(T.nilable(Integer)) }
92
+ attr_reader :stop_after_ms
93
+
94
+ sig { params(stop_after_ms: Integer).void }
95
+ attr_writer :stop_after_ms
96
+
97
+ # Optional timeout in milliseconds for the request. If the request takes longer
98
+ # than this value, it will be aborted with a 408 status code. Maximum allowed
99
+ # value is 300000ms (5 minutes).
100
+ sig { returns(T.nilable(Integer)) }
101
+ attr_reader :timeout_ms
102
+
103
+ sig { params(timeout_ms: Integer).void }
104
+ attr_writer :timeout_ms
105
+
88
106
  # Regex pattern. Only URLs matching this pattern will be followed and scraped.
89
107
  sig { returns(T.nilable(String)) }
90
108
  attr_reader :url_regex
@@ -100,6 +118,14 @@ module ContextDev
100
118
  sig { params(use_main_content_only: T::Boolean).void }
101
119
  attr_writer :use_main_content_only
102
120
 
121
+ # Optional browser wait time in milliseconds after initial page load for each
122
+ # crawled page. Min: 0. Max: 30000 (30 seconds).
123
+ sig { returns(T.nilable(Integer)) }
124
+ attr_reader :wait_for_ms
125
+
126
+ sig { params(wait_for_ms: Integer).void }
127
+ attr_writer :wait_for_ms
128
+
103
129
  sig do
104
130
  params(
105
131
  url: String,
@@ -110,10 +136,13 @@ module ContextDev
110
136
  max_age_ms: Integer,
111
137
  max_depth: Integer,
112
138
  max_pages: Integer,
113
- parse_pdf: T::Boolean,
139
+ pdf: ContextDev::WebWebCrawlMdParams::Pdf::OrHash,
114
140
  shorten_base64_images: T::Boolean,
141
+ stop_after_ms: Integer,
142
+ timeout_ms: Integer,
115
143
  url_regex: String,
116
144
  use_main_content_only: T::Boolean,
145
+ wait_for_ms: Integer,
117
146
  request_options: ContextDev::RequestOptions::OrHash
118
147
  ).returns(T.attached_class)
119
148
  end
@@ -139,17 +168,28 @@ module ContextDev
139
168
  max_depth: nil,
140
169
  # Maximum number of pages to crawl. Hard cap: 500.
141
170
  max_pages: nil,
142
- # When true (default), PDF pages are fetched and their text layer is extracted and
143
- # converted to Markdown alongside HTML pages. When false, PDF pages are skipped
144
- # entirely (not included in results and not counted as failures).
145
- parse_pdf: nil,
171
+ # PDF parsing controls. Use start/end to limit text extraction and OCR to an
172
+ # inclusive 1-based page range.
173
+ pdf: nil,
146
174
  # Truncate base64-encoded image data in the Markdown output
147
175
  shorten_base64_images: nil,
176
+ # Soft time budget for the crawl in milliseconds. After each scrape, the crawler
177
+ # checks the elapsed time and, if exceeded, returns the pages collected so far
178
+ # instead of continuing. Min: 10000 (10s). Max: 240000 (4 min). Default: 120000 (2
179
+ # min).
180
+ stop_after_ms: nil,
181
+ # Optional timeout in milliseconds for the request. If the request takes longer
182
+ # than this value, it will be aborted with a 408 status code. Maximum allowed
183
+ # value is 300000ms (5 minutes).
184
+ timeout_ms: nil,
148
185
  # Regex pattern. Only URLs matching this pattern will be followed and scraped.
149
186
  url_regex: nil,
150
187
  # Extract only the main content, stripping headers, footers, sidebars, and
151
188
  # navigation
152
189
  use_main_content_only: nil,
190
+ # Optional browser wait time in milliseconds after initial page load for each
191
+ # crawled page. Min: 0. Max: 30000 (30 seconds).
192
+ wait_for_ms: nil,
153
193
  request_options: {}
154
194
  )
155
195
  end
@@ -165,16 +205,81 @@ module ContextDev
165
205
  max_age_ms: Integer,
166
206
  max_depth: Integer,
167
207
  max_pages: Integer,
168
- parse_pdf: T::Boolean,
208
+ pdf: ContextDev::WebWebCrawlMdParams::Pdf,
169
209
  shorten_base64_images: T::Boolean,
210
+ stop_after_ms: Integer,
211
+ timeout_ms: Integer,
170
212
  url_regex: String,
171
213
  use_main_content_only: T::Boolean,
214
+ wait_for_ms: Integer,
172
215
  request_options: ContextDev::RequestOptions
173
216
  }
174
217
  )
175
218
  end
176
219
  def to_hash
177
220
  end
221
+
222
+ class Pdf < ContextDev::Internal::Type::BaseModel
223
+ OrHash =
224
+ T.type_alias do
225
+ T.any(
226
+ ContextDev::WebWebCrawlMdParams::Pdf,
227
+ ContextDev::Internal::AnyHash
228
+ )
229
+ end
230
+
231
+ # Last 1-based PDF page to parse. When omitted, parsing ends at the last page.
232
+ # Must be greater than or equal to start when both are provided.
233
+ sig { returns(T.nilable(Integer)) }
234
+ attr_reader :end_
235
+
236
+ sig { params(end_: Integer).void }
237
+ attr_writer :end_
238
+
239
+ # When true, PDF pages are fetched and parsed. When false, PDF pages are skipped
240
+ # entirely (not included in results and not counted as failures).
241
+ sig { returns(T.nilable(T::Boolean)) }
242
+ attr_reader :should_parse
243
+
244
+ sig { params(should_parse: T::Boolean).void }
245
+ attr_writer :should_parse
246
+
247
+ # First 1-based PDF page to parse. When omitted, parsing starts at the first page.
248
+ sig { returns(T.nilable(Integer)) }
249
+ attr_reader :start
250
+
251
+ sig { params(start: Integer).void }
252
+ attr_writer :start
253
+
254
+ # PDF parsing controls. Use start/end to limit text extraction and OCR to an
255
+ # inclusive 1-based page range.
256
+ sig do
257
+ params(
258
+ end_: Integer,
259
+ should_parse: T::Boolean,
260
+ start: Integer
261
+ ).returns(T.attached_class)
262
+ end
263
+ def self.new(
264
+ # Last 1-based PDF page to parse. When omitted, parsing ends at the last page.
265
+ # Must be greater than or equal to start when both are provided.
266
+ end_: nil,
267
+ # When true, PDF pages are fetched and parsed. When false, PDF pages are skipped
268
+ # entirely (not included in results and not counted as failures).
269
+ should_parse: nil,
270
+ # First 1-based PDF page to parse. When omitted, parsing starts at the first page.
271
+ start: nil
272
+ )
273
+ end
274
+
275
+ sig do
276
+ override.returns(
277
+ { end_: Integer, should_parse: T::Boolean, start: Integer }
278
+ )
279
+ end
280
+ def to_hash
281
+ end
282
+ end
178
283
  end
179
284
  end
180
285
  end
@@ -64,7 +64,8 @@ module ContextDev
64
64
  sig { returns(Integer) }
65
65
  attr_accessor :num_failed
66
66
 
67
- # Number of URLs skipped (PDFs when parsePDF=false, or URLs not matching urlRegex)
67
+ # Number of URLs skipped (PDFs when pdf.shouldParse=false, or URLs not matching
68
+ # urlRegex)
68
69
  sig { returns(Integer) }
69
70
  attr_accessor :num_skipped
70
71
 
@@ -90,7 +91,8 @@ module ContextDev
90
91
  max_crawl_depth:,
91
92
  # Number of pages that failed to crawl
92
93
  num_failed:,
93
- # Number of URLs skipped (PDFs when parsePDF=false, or URLs not matching urlRegex)
94
+ # Number of URLs skipped (PDFs when pdf.shouldParse=false, or URLs not matching
95
+ # urlRegex)
94
96
  num_skipped:,
95
97
  # Number of pages successfully crawled
96
98
  num_succeeded:,