context.dev 1.16.0 → 1.18.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (34) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +19 -0
  3. data/README.md +3 -3
  4. data/lib/context_dev/models/ai_extract_product_params.rb +4 -3
  5. data/lib/context_dev/models/ai_extract_products_params.rb +8 -6
  6. data/lib/context_dev/models/web_screenshot_params.rb +18 -21
  7. data/lib/context_dev/models/web_web_crawl_md_params.rb +72 -8
  8. data/lib/context_dev/models/web_web_crawl_md_response.rb +3 -2
  9. data/lib/context_dev/models/web_web_scrape_html_params.rb +61 -8
  10. data/lib/context_dev/models/web_web_scrape_images_params.rb +20 -1
  11. data/lib/context_dev/models/web_web_scrape_md_params.rb +61 -8
  12. data/lib/context_dev/models/web_web_scrape_sitemap_params.rb +11 -1
  13. data/lib/context_dev/resources/ai.rb +1 -1
  14. data/lib/context_dev/resources/web.rb +46 -16
  15. data/lib/context_dev/version.rb +1 -1
  16. data/rbi/context_dev/models/ai_extract_product_params.rbi +6 -4
  17. data/rbi/context_dev/models/ai_extract_products_params.rbi +12 -8
  18. data/rbi/context_dev/models/web_screenshot_params.rbi +28 -53
  19. data/rbi/context_dev/models/web_web_crawl_md_params.rbi +118 -13
  20. data/rbi/context_dev/models/web_web_crawl_md_response.rbi +4 -2
  21. data/rbi/context_dev/models/web_web_scrape_html_params.rbi +101 -13
  22. data/rbi/context_dev/models/web_web_scrape_images_params.rbi +28 -0
  23. data/rbi/context_dev/models/web_web_scrape_md_params.rbi +101 -13
  24. data/rbi/context_dev/models/web_web_scrape_sitemap_params.rbi +15 -0
  25. data/rbi/context_dev/resources/ai.rbi +3 -2
  26. data/rbi/context_dev/resources/web.rbi +69 -20
  27. data/sig/context_dev/models/web_screenshot_params.rbs +13 -19
  28. data/sig/context_dev/models/web_web_crawl_md_params.rbs +53 -6
  29. data/sig/context_dev/models/web_web_scrape_html_params.rbs +45 -5
  30. data/sig/context_dev/models/web_web_scrape_images_params.rbs +15 -1
  31. data/sig/context_dev/models/web_web_scrape_md_params.rbs +46 -6
  32. data/sig/context_dev/models/web_web_scrape_sitemap_params.rbs +12 -1
  33. data/sig/context_dev/resources/web.rbs +15 -4
  34. metadata +2 -2
@@ -34,21 +34,39 @@ module ContextDev
34
34
  sig { params(max_age_ms: Integer).void }
35
35
  attr_writer :max_age_ms
36
36
 
37
- # When true (default), PDF URLs are fetched and their text layer is extracted and
38
- # returned wrapped in <html><pdf>…</pdf></html>. When false, PDF URLs are skipped
39
- # and a 400 WEBSITE_ACCESS_ERROR is returned.
40
- sig { returns(T.nilable(T::Boolean)) }
41
- attr_reader :parse_pdf
37
+ # PDF parsing controls. Use start/end to limit text extraction and OCR to an
38
+ # inclusive 1-based page range.
39
+ sig { returns(T.nilable(ContextDev::WebWebScrapeHTMLParams::Pdf)) }
40
+ attr_reader :pdf
41
+
42
+ sig { params(pdf: ContextDev::WebWebScrapeHTMLParams::Pdf::OrHash).void }
43
+ attr_writer :pdf
44
+
45
+ # Optional timeout in milliseconds for the request. If the request takes longer
46
+ # than this value, it will be aborted with a 408 status code. Maximum allowed
47
+ # value is 300000ms (5 minutes).
48
+ sig { returns(T.nilable(Integer)) }
49
+ attr_reader :timeout_ms
50
+
51
+ sig { params(timeout_ms: Integer).void }
52
+ attr_writer :timeout_ms
53
+
54
+ # Optional browser wait time in milliseconds after initial page load. Min: 0. Max:
55
+ # 30000 (30 seconds).
56
+ sig { returns(T.nilable(Integer)) }
57
+ attr_reader :wait_for_ms
42
58
 
43
- sig { params(parse_pdf: T::Boolean).void }
44
- attr_writer :parse_pdf
59
+ sig { params(wait_for_ms: Integer).void }
60
+ attr_writer :wait_for_ms
45
61
 
46
62
  sig do
47
63
  params(
48
64
  url: String,
49
65
  include_frames: T::Boolean,
50
66
  max_age_ms: Integer,
51
- parse_pdf: T::Boolean,
67
+ pdf: ContextDev::WebWebScrapeHTMLParams::Pdf::OrHash,
68
+ timeout_ms: Integer,
69
+ wait_for_ms: Integer,
52
70
  request_options: ContextDev::RequestOptions::OrHash
53
71
  ).returns(T.attached_class)
54
72
  end
@@ -61,10 +79,16 @@ module ContextDev
61
79
  # younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
62
80
  # omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
63
81
  max_age_ms: nil,
64
- # When true (default), PDF URLs are fetched and their text layer is extracted and
65
- # returned wrapped in <html><pdf>…</pdf></html>. When false, PDF URLs are skipped
66
- # and a 400 WEBSITE_ACCESS_ERROR is returned.
67
- parse_pdf: nil,
82
+ # PDF parsing controls. Use start/end to limit text extraction and OCR to an
83
+ # inclusive 1-based page range.
84
+ pdf: nil,
85
+ # Optional timeout in milliseconds for the request. If the request takes longer
86
+ # than this value, it will be aborted with a 408 status code. Maximum allowed
87
+ # value is 300000ms (5 minutes).
88
+ timeout_ms: nil,
89
+ # Optional browser wait time in milliseconds after initial page load. Min: 0. Max:
90
+ # 30000 (30 seconds).
91
+ wait_for_ms: nil,
68
92
  request_options: {}
69
93
  )
70
94
  end
@@ -75,13 +99,77 @@ module ContextDev
75
99
  url: String,
76
100
  include_frames: T::Boolean,
77
101
  max_age_ms: Integer,
78
- parse_pdf: T::Boolean,
102
+ pdf: ContextDev::WebWebScrapeHTMLParams::Pdf,
103
+ timeout_ms: Integer,
104
+ wait_for_ms: Integer,
79
105
  request_options: ContextDev::RequestOptions
80
106
  }
81
107
  )
82
108
  end
83
109
  def to_hash
84
110
  end
111
+
112
+ class Pdf < ContextDev::Internal::Type::BaseModel
113
+ OrHash =
114
+ T.type_alias do
115
+ T.any(
116
+ ContextDev::WebWebScrapeHTMLParams::Pdf,
117
+ ContextDev::Internal::AnyHash
118
+ )
119
+ end
120
+
121
+ # Last 1-based PDF page to parse. When omitted, parsing ends at the last page.
122
+ # Must be greater than or equal to start when both are provided.
123
+ sig { returns(T.nilable(Integer)) }
124
+ attr_reader :end_
125
+
126
+ sig { params(end_: Integer).void }
127
+ attr_writer :end_
128
+
129
+ # When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
130
+ # a 400 WEBSITE_ACCESS_ERROR is returned.
131
+ sig { returns(T.nilable(T::Boolean)) }
132
+ attr_reader :should_parse
133
+
134
+ sig { params(should_parse: T::Boolean).void }
135
+ attr_writer :should_parse
136
+
137
+ # First 1-based PDF page to parse. When omitted, parsing starts at the first page.
138
+ sig { returns(T.nilable(Integer)) }
139
+ attr_reader :start
140
+
141
+ sig { params(start: Integer).void }
142
+ attr_writer :start
143
+
144
+ # PDF parsing controls. Use start/end to limit text extraction and OCR to an
145
+ # inclusive 1-based page range.
146
+ sig do
147
+ params(
148
+ end_: Integer,
149
+ should_parse: T::Boolean,
150
+ start: Integer
151
+ ).returns(T.attached_class)
152
+ end
153
+ def self.new(
154
+ # Last 1-based PDF page to parse. When omitted, parsing ends at the last page.
155
+ # Must be greater than or equal to start when both are provided.
156
+ end_: nil,
157
+ # When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
158
+ # a 400 WEBSITE_ACCESS_ERROR is returned.
159
+ should_parse: nil,
160
+ # First 1-based PDF page to parse. When omitted, parsing starts at the first page.
161
+ start: nil
162
+ )
163
+ end
164
+
165
+ sig do
166
+ override.returns(
167
+ { end_: Integer, should_parse: T::Boolean, start: Integer }
168
+ )
169
+ end
170
+ def to_hash
171
+ end
172
+ end
85
173
  end
86
174
  end
87
175
  end
@@ -40,11 +40,30 @@ module ContextDev
40
40
  sig { params(max_age_ms: Integer).void }
41
41
  attr_writer :max_age_ms
42
42
 
43
+ # Optional timeout in milliseconds for the request. If the request takes longer
44
+ # than this value, it will be aborted with a 408 status code. Maximum allowed
45
+ # value is 300000ms (5 minutes).
46
+ sig { returns(T.nilable(Integer)) }
47
+ attr_reader :timeout_ms
48
+
49
+ sig { params(timeout_ms: Integer).void }
50
+ attr_writer :timeout_ms
51
+
52
+ # Optional browser wait time in milliseconds after initial page load before
53
+ # collecting images. Min: 0. Max: 30000 (30 seconds).
54
+ sig { returns(T.nilable(Integer)) }
55
+ attr_reader :wait_for_ms
56
+
57
+ sig { params(wait_for_ms: Integer).void }
58
+ attr_writer :wait_for_ms
59
+
43
60
  sig do
44
61
  params(
45
62
  url: String,
46
63
  enrichment: ContextDev::WebWebScrapeImagesParams::Enrichment::OrHash,
47
64
  max_age_ms: Integer,
65
+ timeout_ms: Integer,
66
+ wait_for_ms: Integer,
48
67
  request_options: ContextDev::RequestOptions::OrHash
49
68
  ).returns(T.attached_class)
50
69
  end
@@ -57,6 +76,13 @@ module ContextDev
57
76
  # Reuse a cached result this many milliseconds old or newer. Default: 86400000 (1
58
77
  # day). Set to 0 to bypass cache. Maximum: 2592000000 (30 days).
59
78
  max_age_ms: nil,
79
+ # Optional timeout in milliseconds for the request. If the request takes longer
80
+ # than this value, it will be aborted with a 408 status code. Maximum allowed
81
+ # value is 300000ms (5 minutes).
82
+ timeout_ms: nil,
83
+ # Optional browser wait time in milliseconds after initial page load before
84
+ # collecting images. Min: 0. Max: 30000 (30 seconds).
85
+ wait_for_ms: nil,
60
86
  request_options: {}
61
87
  )
62
88
  end
@@ -67,6 +93,8 @@ module ContextDev
67
93
  url: String,
68
94
  enrichment: ContextDev::WebWebScrapeImagesParams::Enrichment,
69
95
  max_age_ms: Integer,
96
+ timeout_ms: Integer,
97
+ wait_for_ms: Integer,
70
98
  request_options: ContextDev::RequestOptions
71
99
  }
72
100
  )
@@ -46,14 +46,13 @@ module ContextDev
46
46
  sig { params(max_age_ms: Integer).void }
47
47
  attr_writer :max_age_ms
48
48
 
49
- # When true (default), PDF URLs are fetched and their text layer is extracted and
50
- # converted to Markdown. When false, PDF URLs are skipped and a 400
51
- # WEBSITE_ACCESS_ERROR is returned.
52
- sig { returns(T.nilable(T::Boolean)) }
53
- attr_reader :parse_pdf
49
+ # PDF parsing controls. Use start/end to limit text extraction and OCR to an
50
+ # inclusive 1-based page range.
51
+ sig { returns(T.nilable(ContextDev::WebWebScrapeMdParams::Pdf)) }
52
+ attr_reader :pdf
54
53
 
55
- sig { params(parse_pdf: T::Boolean).void }
56
- attr_writer :parse_pdf
54
+ sig { params(pdf: ContextDev::WebWebScrapeMdParams::Pdf::OrHash).void }
55
+ attr_writer :pdf
57
56
 
58
57
  # Shorten base64-encoded image data in the Markdown output
59
58
  sig { returns(T.nilable(T::Boolean)) }
@@ -62,6 +61,15 @@ module ContextDev
62
61
  sig { params(shorten_base64_images: T::Boolean).void }
63
62
  attr_writer :shorten_base64_images
64
63
 
64
+ # Optional timeout in milliseconds for the request. If the request takes longer
65
+ # than this value, it will be aborted with a 408 status code. Maximum allowed
66
+ # value is 300000ms (5 minutes).
67
+ sig { returns(T.nilable(Integer)) }
68
+ attr_reader :timeout_ms
69
+
70
+ sig { params(timeout_ms: Integer).void }
71
+ attr_writer :timeout_ms
72
+
65
73
  # Extract only the main content of the page, excluding headers, footers, sidebars,
66
74
  # and navigation
67
75
  sig { returns(T.nilable(T::Boolean)) }
@@ -70,6 +78,14 @@ module ContextDev
70
78
  sig { params(use_main_content_only: T::Boolean).void }
71
79
  attr_writer :use_main_content_only
72
80
 
81
+ # Optional browser wait time in milliseconds after initial page load before
82
+ # converting the page to Markdown. Min: 0. Max: 30000 (30 seconds).
83
+ sig { returns(T.nilable(Integer)) }
84
+ attr_reader :wait_for_ms
85
+
86
+ sig { params(wait_for_ms: Integer).void }
87
+ attr_writer :wait_for_ms
88
+
73
89
  sig do
74
90
  params(
75
91
  url: String,
@@ -77,9 +93,11 @@ module ContextDev
77
93
  include_images: T::Boolean,
78
94
  include_links: T::Boolean,
79
95
  max_age_ms: Integer,
80
- parse_pdf: T::Boolean,
96
+ pdf: ContextDev::WebWebScrapeMdParams::Pdf::OrHash,
81
97
  shorten_base64_images: T::Boolean,
98
+ timeout_ms: Integer,
82
99
  use_main_content_only: T::Boolean,
100
+ wait_for_ms: Integer,
83
101
  request_options: ContextDev::RequestOptions::OrHash
84
102
  ).returns(T.attached_class)
85
103
  end
@@ -97,15 +115,21 @@ module ContextDev
97
115
  # younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
98
116
  # omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
99
117
  max_age_ms: nil,
100
- # When true (default), PDF URLs are fetched and their text layer is extracted and
101
- # converted to Markdown. When false, PDF URLs are skipped and a 400
102
- # WEBSITE_ACCESS_ERROR is returned.
103
- parse_pdf: nil,
118
+ # PDF parsing controls. Use start/end to limit text extraction and OCR to an
119
+ # inclusive 1-based page range.
120
+ pdf: nil,
104
121
  # Shorten base64-encoded image data in the Markdown output
105
122
  shorten_base64_images: nil,
123
+ # Optional timeout in milliseconds for the request. If the request takes longer
124
+ # than this value, it will be aborted with a 408 status code. Maximum allowed
125
+ # value is 300000ms (5 minutes).
126
+ timeout_ms: nil,
106
127
  # Extract only the main content of the page, excluding headers, footers, sidebars,
107
128
  # and navigation
108
129
  use_main_content_only: nil,
130
+ # Optional browser wait time in milliseconds after initial page load before
131
+ # converting the page to Markdown. Min: 0. Max: 30000 (30 seconds).
132
+ wait_for_ms: nil,
109
133
  request_options: {}
110
134
  )
111
135
  end
@@ -118,15 +142,79 @@ module ContextDev
118
142
  include_images: T::Boolean,
119
143
  include_links: T::Boolean,
120
144
  max_age_ms: Integer,
121
- parse_pdf: T::Boolean,
145
+ pdf: ContextDev::WebWebScrapeMdParams::Pdf,
122
146
  shorten_base64_images: T::Boolean,
147
+ timeout_ms: Integer,
123
148
  use_main_content_only: T::Boolean,
149
+ wait_for_ms: Integer,
124
150
  request_options: ContextDev::RequestOptions
125
151
  }
126
152
  )
127
153
  end
128
154
  def to_hash
129
155
  end
156
+
157
+ class Pdf < ContextDev::Internal::Type::BaseModel
158
+ OrHash =
159
+ T.type_alias do
160
+ T.any(
161
+ ContextDev::WebWebScrapeMdParams::Pdf,
162
+ ContextDev::Internal::AnyHash
163
+ )
164
+ end
165
+
166
+ # Last 1-based PDF page to parse. When omitted, parsing ends at the last page.
167
+ # Must be greater than or equal to start when both are provided.
168
+ sig { returns(T.nilable(Integer)) }
169
+ attr_reader :end_
170
+
171
+ sig { params(end_: Integer).void }
172
+ attr_writer :end_
173
+
174
+ # When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
175
+ # a 400 WEBSITE_ACCESS_ERROR is returned.
176
+ sig { returns(T.nilable(T::Boolean)) }
177
+ attr_reader :should_parse
178
+
179
+ sig { params(should_parse: T::Boolean).void }
180
+ attr_writer :should_parse
181
+
182
+ # First 1-based PDF page to parse. When omitted, parsing starts at the first page.
183
+ sig { returns(T.nilable(Integer)) }
184
+ attr_reader :start
185
+
186
+ sig { params(start: Integer).void }
187
+ attr_writer :start
188
+
189
+ # PDF parsing controls. Use start/end to limit text extraction and OCR to an
190
+ # inclusive 1-based page range.
191
+ sig do
192
+ params(
193
+ end_: Integer,
194
+ should_parse: T::Boolean,
195
+ start: Integer
196
+ ).returns(T.attached_class)
197
+ end
198
+ def self.new(
199
+ # Last 1-based PDF page to parse. When omitted, parsing ends at the last page.
200
+ # Must be greater than or equal to start when both are provided.
201
+ end_: nil,
202
+ # When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
203
+ # a 400 WEBSITE_ACCESS_ERROR is returned.
204
+ should_parse: nil,
205
+ # First 1-based PDF page to parse. When omitted, parsing starts at the first page.
206
+ start: nil
207
+ )
208
+ end
209
+
210
+ sig do
211
+ override.returns(
212
+ { end_: Integer, should_parse: T::Boolean, start: Integer }
213
+ )
214
+ end
215
+ def to_hash
216
+ end
217
+ end
130
218
  end
131
219
  end
132
220
  end
@@ -26,6 +26,15 @@ module ContextDev
26
26
  sig { params(max_links: Integer).void }
27
27
  attr_writer :max_links
28
28
 
29
+ # Optional timeout in milliseconds for the request. If the request takes longer
30
+ # than this value, it will be aborted with a 408 status code. Maximum allowed
31
+ # value is 300000ms (5 minutes).
32
+ sig { returns(T.nilable(Integer)) }
33
+ attr_reader :timeout_ms
34
+
35
+ sig { params(timeout_ms: Integer).void }
36
+ attr_writer :timeout_ms
37
+
29
38
  # Optional RE2-compatible regex pattern. Only URLs matching this pattern are
30
39
  # returned and counted against maxLinks.
31
40
  sig { returns(T.nilable(String)) }
@@ -38,6 +47,7 @@ module ContextDev
38
47
  params(
39
48
  domain: String,
40
49
  max_links: Integer,
50
+ timeout_ms: Integer,
41
51
  url_regex: String,
42
52
  request_options: ContextDev::RequestOptions::OrHash
43
53
  ).returns(T.attached_class)
@@ -48,6 +58,10 @@ module ContextDev
48
58
  # Maximum number of links to return from the sitemap crawl. Defaults to 10,000.
49
59
  # Minimum is 1, maximum is 100,000.
50
60
  max_links: nil,
61
+ # Optional timeout in milliseconds for the request. If the request takes longer
62
+ # than this value, it will be aborted with a 408 status code. Maximum allowed
63
+ # value is 300000ms (5 minutes).
64
+ timeout_ms: nil,
51
65
  # Optional RE2-compatible regex pattern. Only URLs matching this pattern are
52
66
  # returned and counted against maxLinks.
53
67
  url_regex: nil,
@@ -60,6 +74,7 @@ module ContextDev
60
74
  {
61
75
  domain: String,
62
76
  max_links: Integer,
77
+ timeout_ms: Integer,
63
78
  url_regex: String,
64
79
  request_options: ContextDev::RequestOptions
65
80
  }
@@ -48,8 +48,9 @@ module ContextDev
48
48
  # younger than this many milliseconds. Defaults to 7 days (604800000 ms) when
49
49
  # omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
50
50
  max_age_ms: nil,
51
- # Optional timeout in milliseconds for the request. Maximum allowed value is
52
- # 300000ms (5 minutes).
51
+ # Optional timeout in milliseconds for the request. If the request takes longer
52
+ # than this value, it will be aborted with a 408 status code. Maximum allowed
53
+ # value is 300000ms (5 minutes).
53
54
  timeout_ms: nil,
54
55
  request_options: {}
55
56
  )
@@ -67,8 +67,9 @@ module ContextDev
67
67
  ContextDev::WebScreenshotParams::FullScreenshot::OrSymbol,
68
68
  max_age_ms: Integer,
69
69
  page: ContextDev::WebScreenshotParams::Page::OrSymbol,
70
- prioritize: ContextDev::WebScreenshotParams::Prioritize::OrSymbol,
70
+ timeout_ms: Integer,
71
71
  viewport: ContextDev::WebScreenshotParams::Viewport::OrHash,
72
+ wait_for_ms: Integer,
72
73
  request_options: ContextDev::RequestOptions::OrHash
73
74
  ).returns(ContextDev::Models::WebScreenshotResponse)
74
75
  end
@@ -95,12 +96,16 @@ module ContextDev
95
96
  # provided, screenshots the main domain landing page. Only applicable when using
96
97
  # 'domain', not 'directUrl'.
97
98
  page: nil,
98
- # Optional parameter to prioritize screenshot capture. If 'speed', optimizes for
99
- # faster capture with basic quality. If 'quality', optimizes for higher quality
100
- # with longer wait times. Defaults to 'quality' if not provided.
101
- prioritize: nil,
99
+ # Optional timeout in milliseconds for the request. If the request takes longer
100
+ # than this value, it will be aborted with a 408 status code. Maximum allowed
101
+ # value is 300000ms (5 minutes).
102
+ timeout_ms: nil,
102
103
  # Optional browser viewport dimensions for the screenshot. Defaults to 1920x1080.
103
104
  viewport: nil,
105
+ # Optional browser wait time in milliseconds after initial page load before taking
106
+ # the screenshot. Min: 0. Max: 30000 (30 seconds). Defaults to 3000 ms when
107
+ # omitted.
108
+ wait_for_ms: nil,
104
109
  request_options: {}
105
110
  )
106
111
  end
@@ -117,10 +122,13 @@ module ContextDev
117
122
  max_age_ms: Integer,
118
123
  max_depth: Integer,
119
124
  max_pages: Integer,
120
- parse_pdf: T::Boolean,
125
+ pdf: ContextDev::WebWebCrawlMdParams::Pdf::OrHash,
121
126
  shorten_base64_images: T::Boolean,
127
+ stop_after_ms: Integer,
128
+ timeout_ms: Integer,
122
129
  url_regex: String,
123
130
  use_main_content_only: T::Boolean,
131
+ wait_for_ms: Integer,
124
132
  request_options: ContextDev::RequestOptions::OrHash
125
133
  ).returns(ContextDev::Models::WebWebCrawlMdResponse)
126
134
  end
@@ -146,17 +154,28 @@ module ContextDev
146
154
  max_depth: nil,
147
155
  # Maximum number of pages to crawl. Hard cap: 500.
148
156
  max_pages: nil,
149
- # When true (default), PDF pages are fetched and their text layer is extracted and
150
- # converted to Markdown alongside HTML pages. When false, PDF pages are skipped
151
- # entirely (not included in results and not counted as failures).
152
- parse_pdf: nil,
157
+ # PDF parsing controls. Use start/end to limit text extraction and OCR to an
158
+ # inclusive 1-based page range.
159
+ pdf: nil,
153
160
  # Truncate base64-encoded image data in the Markdown output
154
161
  shorten_base64_images: nil,
162
+ # Soft time budget for the crawl in milliseconds. After each scrape, the crawler
163
+ # checks the elapsed time and, if exceeded, returns the pages collected so far
164
+ # instead of continuing. Min: 10000 (10s). Max: 240000 (4 min). Default: 120000 (2
165
+ # min).
166
+ stop_after_ms: nil,
167
+ # Optional timeout in milliseconds for the request. If the request takes longer
168
+ # than this value, it will be aborted with a 408 status code. Maximum allowed
169
+ # value is 300000ms (5 minutes).
170
+ timeout_ms: nil,
155
171
  # Regex pattern. Only URLs matching this pattern will be followed and scraped.
156
172
  url_regex: nil,
157
173
  # Extract only the main content, stripping headers, footers, sidebars, and
158
174
  # navigation
159
175
  use_main_content_only: nil,
176
+ # Optional browser wait time in milliseconds after initial page load for each
177
+ # crawled page. Min: 0. Max: 30000 (30 seconds).
178
+ wait_for_ms: nil,
160
179
  request_options: {}
161
180
  )
162
181
  end
@@ -167,7 +186,9 @@ module ContextDev
167
186
  url: String,
168
187
  include_frames: T::Boolean,
169
188
  max_age_ms: Integer,
170
- parse_pdf: T::Boolean,
189
+ pdf: ContextDev::WebWebScrapeHTMLParams::Pdf::OrHash,
190
+ timeout_ms: Integer,
191
+ wait_for_ms: Integer,
171
192
  request_options: ContextDev::RequestOptions::OrHash
172
193
  ).returns(ContextDev::Models::WebWebScrapeHTMLResponse)
173
194
  end
@@ -180,10 +201,16 @@ module ContextDev
180
201
  # younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
181
202
  # omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
182
203
  max_age_ms: nil,
183
- # When true (default), PDF URLs are fetched and their text layer is extracted and
184
- # returned wrapped in <html><pdf>…</pdf></html>. When false, PDF URLs are skipped
185
- # and a 400 WEBSITE_ACCESS_ERROR is returned.
186
- parse_pdf: nil,
204
+ # PDF parsing controls. Use start/end to limit text extraction and OCR to an
205
+ # inclusive 1-based page range.
206
+ pdf: nil,
207
+ # Optional timeout in milliseconds for the request. If the request takes longer
208
+ # than this value, it will be aborted with a 408 status code. Maximum allowed
209
+ # value is 300000ms (5 minutes).
210
+ timeout_ms: nil,
211
+ # Optional browser wait time in milliseconds after initial page load. Min: 0. Max:
212
+ # 30000 (30 seconds).
213
+ wait_for_ms: nil,
187
214
  request_options: {}
188
215
  )
189
216
  end
@@ -197,6 +224,8 @@ module ContextDev
197
224
  url: String,
198
225
  enrichment: ContextDev::WebWebScrapeImagesParams::Enrichment::OrHash,
199
226
  max_age_ms: Integer,
227
+ timeout_ms: Integer,
228
+ wait_for_ms: Integer,
200
229
  request_options: ContextDev::RequestOptions::OrHash
201
230
  ).returns(ContextDev::Models::WebWebScrapeImagesResponse)
202
231
  end
@@ -209,6 +238,13 @@ module ContextDev
209
238
  # Reuse a cached result this many milliseconds old or newer. Default: 86400000 (1
210
239
  # day). Set to 0 to bypass cache. Maximum: 2592000000 (30 days).
211
240
  max_age_ms: nil,
241
+ # Optional timeout in milliseconds for the request. If the request takes longer
242
+ # than this value, it will be aborted with a 408 status code. Maximum allowed
243
+ # value is 300000ms (5 minutes).
244
+ timeout_ms: nil,
245
+ # Optional browser wait time in milliseconds after initial page load before
246
+ # collecting images. Min: 0. Max: 30000 (30 seconds).
247
+ wait_for_ms: nil,
212
248
  request_options: {}
213
249
  )
214
250
  end
@@ -221,9 +257,11 @@ module ContextDev
221
257
  include_images: T::Boolean,
222
258
  include_links: T::Boolean,
223
259
  max_age_ms: Integer,
224
- parse_pdf: T::Boolean,
260
+ pdf: ContextDev::WebWebScrapeMdParams::Pdf::OrHash,
225
261
  shorten_base64_images: T::Boolean,
262
+ timeout_ms: Integer,
226
263
  use_main_content_only: T::Boolean,
264
+ wait_for_ms: Integer,
227
265
  request_options: ContextDev::RequestOptions::OrHash
228
266
  ).returns(ContextDev::Models::WebWebScrapeMdResponse)
229
267
  end
@@ -241,15 +279,21 @@ module ContextDev
241
279
  # younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
242
280
  # omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
243
281
  max_age_ms: nil,
244
- # When true (default), PDF URLs are fetched and their text layer is extracted and
245
- # converted to Markdown. When false, PDF URLs are skipped and a 400
246
- # WEBSITE_ACCESS_ERROR is returned.
247
- parse_pdf: nil,
282
+ # PDF parsing controls. Use start/end to limit text extraction and OCR to an
283
+ # inclusive 1-based page range.
284
+ pdf: nil,
248
285
  # Shorten base64-encoded image data in the Markdown output
249
286
  shorten_base64_images: nil,
287
+ # Optional timeout in milliseconds for the request. If the request takes longer
288
+ # than this value, it will be aborted with a 408 status code. Maximum allowed
289
+ # value is 300000ms (5 minutes).
290
+ timeout_ms: nil,
250
291
  # Extract only the main content of the page, excluding headers, footers, sidebars,
251
292
  # and navigation
252
293
  use_main_content_only: nil,
294
+ # Optional browser wait time in milliseconds after initial page load before
295
+ # converting the page to Markdown. Min: 0. Max: 30000 (30 seconds).
296
+ wait_for_ms: nil,
253
297
  request_options: {}
254
298
  )
255
299
  end
@@ -259,6 +303,7 @@ module ContextDev
259
303
  params(
260
304
  domain: String,
261
305
  max_links: Integer,
306
+ timeout_ms: Integer,
262
307
  url_regex: String,
263
308
  request_options: ContextDev::RequestOptions::OrHash
264
309
  ).returns(ContextDev::Models::WebWebScrapeSitemapResponse)
@@ -269,6 +314,10 @@ module ContextDev
269
314
  # Maximum number of links to return from the sitemap crawl. Defaults to 10,000.
270
315
  # Minimum is 1, maximum is 100,000.
271
316
  max_links: nil,
317
+ # Optional timeout in milliseconds for the request. If the request takes longer
318
+ # than this value, it will be aborted with a 408 status code. Maximum allowed
319
+ # value is 300000ms (5 minutes).
320
+ timeout_ms: nil,
272
321
  # Optional RE2-compatible regex pattern. Only URLs matching this pattern are
273
322
  # returned and counted against maxLinks.
274
323
  url_regex: nil,