context.dev 1.17.0 → 1.19.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +19 -0
- data/README.md +3 -3
- data/lib/context_dev/models/web_screenshot_params.rb +24 -1
- data/lib/context_dev/models/web_web_crawl_md_params.rb +53 -8
- data/lib/context_dev/models/web_web_crawl_md_response.rb +3 -2
- data/lib/context_dev/models/web_web_scrape_html_params.rb +42 -8
- data/lib/context_dev/models/web_web_scrape_md_params.rb +42 -8
- data/lib/context_dev/resources/web.rb +12 -9
- data/lib/context_dev/version.rb +1 -1
- data/rbi/context_dev/models/web_screenshot_params.rbi +62 -0
- data/rbi/context_dev/models/web_web_crawl_md_params.rbi +90 -13
- data/rbi/context_dev/models/web_web_crawl_md_response.rbi +4 -2
- data/rbi/context_dev/models/web_web_scrape_html_params.rbi +73 -13
- data/rbi/context_dev/models/web_web_scrape_md_params.rbi +73 -13
- data/rbi/context_dev/resources/web.rbi +24 -15
- data/sig/context_dev/models/web_screenshot_params.rbs +20 -0
- data/sig/context_dev/models/web_web_crawl_md_params.rbs +38 -5
- data/sig/context_dev/models/web_web_scrape_html_params.rbs +31 -5
- data/sig/context_dev/models/web_web_scrape_md_params.rbs +31 -5
- data/sig/context_dev/resources/web.rbs +5 -3
- metadata +2 -2
|
@@ -69,14 +69,13 @@ module ContextDev
|
|
|
69
69
|
sig { params(max_pages: Integer).void }
|
|
70
70
|
attr_writer :max_pages
|
|
71
71
|
|
|
72
|
-
#
|
|
73
|
-
#
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
attr_reader :parse_pdf
|
|
72
|
+
# PDF parsing controls. Use start/end to limit text extraction and OCR to an
|
|
73
|
+
# inclusive 1-based page range.
|
|
74
|
+
sig { returns(T.nilable(ContextDev::WebWebCrawlMdParams::Pdf)) }
|
|
75
|
+
attr_reader :pdf
|
|
77
76
|
|
|
78
|
-
sig { params(
|
|
79
|
-
attr_writer :
|
|
77
|
+
sig { params(pdf: ContextDev::WebWebCrawlMdParams::Pdf::OrHash).void }
|
|
78
|
+
attr_writer :pdf
|
|
80
79
|
|
|
81
80
|
# Truncate base64-encoded image data in the Markdown output
|
|
82
81
|
sig { returns(T.nilable(T::Boolean)) }
|
|
@@ -85,6 +84,16 @@ module ContextDev
|
|
|
85
84
|
sig { params(shorten_base64_images: T::Boolean).void }
|
|
86
85
|
attr_writer :shorten_base64_images
|
|
87
86
|
|
|
87
|
+
# Soft time budget for the crawl in milliseconds. After each scrape, the crawler
|
|
88
|
+
# checks the elapsed time and, if exceeded, returns the pages collected so far
|
|
89
|
+
# instead of continuing. Min: 10000 (10s). Max: 240000 (4 min). Default: 120000 (2
|
|
90
|
+
# min).
|
|
91
|
+
sig { returns(T.nilable(Integer)) }
|
|
92
|
+
attr_reader :stop_after_ms
|
|
93
|
+
|
|
94
|
+
sig { params(stop_after_ms: Integer).void }
|
|
95
|
+
attr_writer :stop_after_ms
|
|
96
|
+
|
|
88
97
|
# Optional timeout in milliseconds for the request. If the request takes longer
|
|
89
98
|
# than this value, it will be aborted with a 408 status code. Maximum allowed
|
|
90
99
|
# value is 300000ms (5 minutes).
|
|
@@ -127,8 +136,9 @@ module ContextDev
|
|
|
127
136
|
max_age_ms: Integer,
|
|
128
137
|
max_depth: Integer,
|
|
129
138
|
max_pages: Integer,
|
|
130
|
-
|
|
139
|
+
pdf: ContextDev::WebWebCrawlMdParams::Pdf::OrHash,
|
|
131
140
|
shorten_base64_images: T::Boolean,
|
|
141
|
+
stop_after_ms: Integer,
|
|
132
142
|
timeout_ms: Integer,
|
|
133
143
|
url_regex: String,
|
|
134
144
|
use_main_content_only: T::Boolean,
|
|
@@ -158,12 +168,16 @@ module ContextDev
|
|
|
158
168
|
max_depth: nil,
|
|
159
169
|
# Maximum number of pages to crawl. Hard cap: 500.
|
|
160
170
|
max_pages: nil,
|
|
161
|
-
#
|
|
162
|
-
#
|
|
163
|
-
|
|
164
|
-
parse_pdf: nil,
|
|
171
|
+
# PDF parsing controls. Use start/end to limit text extraction and OCR to an
|
|
172
|
+
# inclusive 1-based page range.
|
|
173
|
+
pdf: nil,
|
|
165
174
|
# Truncate base64-encoded image data in the Markdown output
|
|
166
175
|
shorten_base64_images: nil,
|
|
176
|
+
# Soft time budget for the crawl in milliseconds. After each scrape, the crawler
|
|
177
|
+
# checks the elapsed time and, if exceeded, returns the pages collected so far
|
|
178
|
+
# instead of continuing. Min: 10000 (10s). Max: 240000 (4 min). Default: 120000 (2
|
|
179
|
+
# min).
|
|
180
|
+
stop_after_ms: nil,
|
|
167
181
|
# Optional timeout in milliseconds for the request. If the request takes longer
|
|
168
182
|
# than this value, it will be aborted with a 408 status code. Maximum allowed
|
|
169
183
|
# value is 300000ms (5 minutes).
|
|
@@ -191,8 +205,9 @@ module ContextDev
|
|
|
191
205
|
max_age_ms: Integer,
|
|
192
206
|
max_depth: Integer,
|
|
193
207
|
max_pages: Integer,
|
|
194
|
-
|
|
208
|
+
pdf: ContextDev::WebWebCrawlMdParams::Pdf,
|
|
195
209
|
shorten_base64_images: T::Boolean,
|
|
210
|
+
stop_after_ms: Integer,
|
|
196
211
|
timeout_ms: Integer,
|
|
197
212
|
url_regex: String,
|
|
198
213
|
use_main_content_only: T::Boolean,
|
|
@@ -203,6 +218,68 @@ module ContextDev
|
|
|
203
218
|
end
|
|
204
219
|
def to_hash
|
|
205
220
|
end
|
|
221
|
+
|
|
222
|
+
class Pdf < ContextDev::Internal::Type::BaseModel
|
|
223
|
+
OrHash =
|
|
224
|
+
T.type_alias do
|
|
225
|
+
T.any(
|
|
226
|
+
ContextDev::WebWebCrawlMdParams::Pdf,
|
|
227
|
+
ContextDev::Internal::AnyHash
|
|
228
|
+
)
|
|
229
|
+
end
|
|
230
|
+
|
|
231
|
+
# Last 1-based PDF page to parse. When omitted, parsing ends at the last page.
|
|
232
|
+
# Must be greater than or equal to start when both are provided.
|
|
233
|
+
sig { returns(T.nilable(Integer)) }
|
|
234
|
+
attr_reader :end_
|
|
235
|
+
|
|
236
|
+
sig { params(end_: Integer).void }
|
|
237
|
+
attr_writer :end_
|
|
238
|
+
|
|
239
|
+
# When true, PDF pages are fetched and parsed. When false, PDF pages are skipped
|
|
240
|
+
# entirely (not included in results and not counted as failures).
|
|
241
|
+
sig { returns(T.nilable(T::Boolean)) }
|
|
242
|
+
attr_reader :should_parse
|
|
243
|
+
|
|
244
|
+
sig { params(should_parse: T::Boolean).void }
|
|
245
|
+
attr_writer :should_parse
|
|
246
|
+
|
|
247
|
+
# First 1-based PDF page to parse. When omitted, parsing starts at the first page.
|
|
248
|
+
sig { returns(T.nilable(Integer)) }
|
|
249
|
+
attr_reader :start
|
|
250
|
+
|
|
251
|
+
sig { params(start: Integer).void }
|
|
252
|
+
attr_writer :start
|
|
253
|
+
|
|
254
|
+
# PDF parsing controls. Use start/end to limit text extraction and OCR to an
|
|
255
|
+
# inclusive 1-based page range.
|
|
256
|
+
sig do
|
|
257
|
+
params(
|
|
258
|
+
end_: Integer,
|
|
259
|
+
should_parse: T::Boolean,
|
|
260
|
+
start: Integer
|
|
261
|
+
).returns(T.attached_class)
|
|
262
|
+
end
|
|
263
|
+
def self.new(
|
|
264
|
+
# Last 1-based PDF page to parse. When omitted, parsing ends at the last page.
|
|
265
|
+
# Must be greater than or equal to start when both are provided.
|
|
266
|
+
end_: nil,
|
|
267
|
+
# When true, PDF pages are fetched and parsed. When false, PDF pages are skipped
|
|
268
|
+
# entirely (not included in results and not counted as failures).
|
|
269
|
+
should_parse: nil,
|
|
270
|
+
# First 1-based PDF page to parse. When omitted, parsing starts at the first page.
|
|
271
|
+
start: nil
|
|
272
|
+
)
|
|
273
|
+
end
|
|
274
|
+
|
|
275
|
+
sig do
|
|
276
|
+
override.returns(
|
|
277
|
+
{ end_: Integer, should_parse: T::Boolean, start: Integer }
|
|
278
|
+
)
|
|
279
|
+
end
|
|
280
|
+
def to_hash
|
|
281
|
+
end
|
|
282
|
+
end
|
|
206
283
|
end
|
|
207
284
|
end
|
|
208
285
|
end
|
|
@@ -64,7 +64,8 @@ module ContextDev
|
|
|
64
64
|
sig { returns(Integer) }
|
|
65
65
|
attr_accessor :num_failed
|
|
66
66
|
|
|
67
|
-
# Number of URLs skipped (PDFs when
|
|
67
|
+
# Number of URLs skipped (PDFs when pdf.shouldParse=false, or URLs not matching
|
|
68
|
+
# urlRegex)
|
|
68
69
|
sig { returns(Integer) }
|
|
69
70
|
attr_accessor :num_skipped
|
|
70
71
|
|
|
@@ -90,7 +91,8 @@ module ContextDev
|
|
|
90
91
|
max_crawl_depth:,
|
|
91
92
|
# Number of pages that failed to crawl
|
|
92
93
|
num_failed:,
|
|
93
|
-
# Number of URLs skipped (PDFs when
|
|
94
|
+
# Number of URLs skipped (PDFs when pdf.shouldParse=false, or URLs not matching
|
|
95
|
+
# urlRegex)
|
|
94
96
|
num_skipped:,
|
|
95
97
|
# Number of pages successfully crawled
|
|
96
98
|
num_succeeded:,
|
|
@@ -34,14 +34,13 @@ module ContextDev
|
|
|
34
34
|
sig { params(max_age_ms: Integer).void }
|
|
35
35
|
attr_writer :max_age_ms
|
|
36
36
|
|
|
37
|
-
#
|
|
38
|
-
#
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
attr_reader :parse_pdf
|
|
37
|
+
# PDF parsing controls. Use start/end to limit text extraction and OCR to an
|
|
38
|
+
# inclusive 1-based page range.
|
|
39
|
+
sig { returns(T.nilable(ContextDev::WebWebScrapeHTMLParams::Pdf)) }
|
|
40
|
+
attr_reader :pdf
|
|
42
41
|
|
|
43
|
-
sig { params(
|
|
44
|
-
attr_writer :
|
|
42
|
+
sig { params(pdf: ContextDev::WebWebScrapeHTMLParams::Pdf::OrHash).void }
|
|
43
|
+
attr_writer :pdf
|
|
45
44
|
|
|
46
45
|
# Optional timeout in milliseconds for the request. If the request takes longer
|
|
47
46
|
# than this value, it will be aborted with a 408 status code. Maximum allowed
|
|
@@ -65,7 +64,7 @@ module ContextDev
|
|
|
65
64
|
url: String,
|
|
66
65
|
include_frames: T::Boolean,
|
|
67
66
|
max_age_ms: Integer,
|
|
68
|
-
|
|
67
|
+
pdf: ContextDev::WebWebScrapeHTMLParams::Pdf::OrHash,
|
|
69
68
|
timeout_ms: Integer,
|
|
70
69
|
wait_for_ms: Integer,
|
|
71
70
|
request_options: ContextDev::RequestOptions::OrHash
|
|
@@ -80,10 +79,9 @@ module ContextDev
|
|
|
80
79
|
# younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
|
|
81
80
|
# omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
|
|
82
81
|
max_age_ms: nil,
|
|
83
|
-
#
|
|
84
|
-
#
|
|
85
|
-
|
|
86
|
-
parse_pdf: nil,
|
|
82
|
+
# PDF parsing controls. Use start/end to limit text extraction and OCR to an
|
|
83
|
+
# inclusive 1-based page range.
|
|
84
|
+
pdf: nil,
|
|
87
85
|
# Optional timeout in milliseconds for the request. If the request takes longer
|
|
88
86
|
# than this value, it will be aborted with a 408 status code. Maximum allowed
|
|
89
87
|
# value is 300000ms (5 minutes).
|
|
@@ -101,7 +99,7 @@ module ContextDev
|
|
|
101
99
|
url: String,
|
|
102
100
|
include_frames: T::Boolean,
|
|
103
101
|
max_age_ms: Integer,
|
|
104
|
-
|
|
102
|
+
pdf: ContextDev::WebWebScrapeHTMLParams::Pdf,
|
|
105
103
|
timeout_ms: Integer,
|
|
106
104
|
wait_for_ms: Integer,
|
|
107
105
|
request_options: ContextDev::RequestOptions
|
|
@@ -110,6 +108,68 @@ module ContextDev
|
|
|
110
108
|
end
|
|
111
109
|
def to_hash
|
|
112
110
|
end
|
|
111
|
+
|
|
112
|
+
class Pdf < ContextDev::Internal::Type::BaseModel
|
|
113
|
+
OrHash =
|
|
114
|
+
T.type_alias do
|
|
115
|
+
T.any(
|
|
116
|
+
ContextDev::WebWebScrapeHTMLParams::Pdf,
|
|
117
|
+
ContextDev::Internal::AnyHash
|
|
118
|
+
)
|
|
119
|
+
end
|
|
120
|
+
|
|
121
|
+
# Last 1-based PDF page to parse. When omitted, parsing ends at the last page.
|
|
122
|
+
# Must be greater than or equal to start when both are provided.
|
|
123
|
+
sig { returns(T.nilable(Integer)) }
|
|
124
|
+
attr_reader :end_
|
|
125
|
+
|
|
126
|
+
sig { params(end_: Integer).void }
|
|
127
|
+
attr_writer :end_
|
|
128
|
+
|
|
129
|
+
# When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
|
|
130
|
+
# a 400 WEBSITE_ACCESS_ERROR is returned.
|
|
131
|
+
sig { returns(T.nilable(T::Boolean)) }
|
|
132
|
+
attr_reader :should_parse
|
|
133
|
+
|
|
134
|
+
sig { params(should_parse: T::Boolean).void }
|
|
135
|
+
attr_writer :should_parse
|
|
136
|
+
|
|
137
|
+
# First 1-based PDF page to parse. When omitted, parsing starts at the first page.
|
|
138
|
+
sig { returns(T.nilable(Integer)) }
|
|
139
|
+
attr_reader :start
|
|
140
|
+
|
|
141
|
+
sig { params(start: Integer).void }
|
|
142
|
+
attr_writer :start
|
|
143
|
+
|
|
144
|
+
# PDF parsing controls. Use start/end to limit text extraction and OCR to an
|
|
145
|
+
# inclusive 1-based page range.
|
|
146
|
+
sig do
|
|
147
|
+
params(
|
|
148
|
+
end_: Integer,
|
|
149
|
+
should_parse: T::Boolean,
|
|
150
|
+
start: Integer
|
|
151
|
+
).returns(T.attached_class)
|
|
152
|
+
end
|
|
153
|
+
def self.new(
|
|
154
|
+
# Last 1-based PDF page to parse. When omitted, parsing ends at the last page.
|
|
155
|
+
# Must be greater than or equal to start when both are provided.
|
|
156
|
+
end_: nil,
|
|
157
|
+
# When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
|
|
158
|
+
# a 400 WEBSITE_ACCESS_ERROR is returned.
|
|
159
|
+
should_parse: nil,
|
|
160
|
+
# First 1-based PDF page to parse. When omitted, parsing starts at the first page.
|
|
161
|
+
start: nil
|
|
162
|
+
)
|
|
163
|
+
end
|
|
164
|
+
|
|
165
|
+
sig do
|
|
166
|
+
override.returns(
|
|
167
|
+
{ end_: Integer, should_parse: T::Boolean, start: Integer }
|
|
168
|
+
)
|
|
169
|
+
end
|
|
170
|
+
def to_hash
|
|
171
|
+
end
|
|
172
|
+
end
|
|
113
173
|
end
|
|
114
174
|
end
|
|
115
175
|
end
|
|
@@ -46,14 +46,13 @@ module ContextDev
|
|
|
46
46
|
sig { params(max_age_ms: Integer).void }
|
|
47
47
|
attr_writer :max_age_ms
|
|
48
48
|
|
|
49
|
-
#
|
|
50
|
-
#
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
attr_reader :parse_pdf
|
|
49
|
+
# PDF parsing controls. Use start/end to limit text extraction and OCR to an
|
|
50
|
+
# inclusive 1-based page range.
|
|
51
|
+
sig { returns(T.nilable(ContextDev::WebWebScrapeMdParams::Pdf)) }
|
|
52
|
+
attr_reader :pdf
|
|
54
53
|
|
|
55
|
-
sig { params(
|
|
56
|
-
attr_writer :
|
|
54
|
+
sig { params(pdf: ContextDev::WebWebScrapeMdParams::Pdf::OrHash).void }
|
|
55
|
+
attr_writer :pdf
|
|
57
56
|
|
|
58
57
|
# Shorten base64-encoded image data in the Markdown output
|
|
59
58
|
sig { returns(T.nilable(T::Boolean)) }
|
|
@@ -94,7 +93,7 @@ module ContextDev
|
|
|
94
93
|
include_images: T::Boolean,
|
|
95
94
|
include_links: T::Boolean,
|
|
96
95
|
max_age_ms: Integer,
|
|
97
|
-
|
|
96
|
+
pdf: ContextDev::WebWebScrapeMdParams::Pdf::OrHash,
|
|
98
97
|
shorten_base64_images: T::Boolean,
|
|
99
98
|
timeout_ms: Integer,
|
|
100
99
|
use_main_content_only: T::Boolean,
|
|
@@ -116,10 +115,9 @@ module ContextDev
|
|
|
116
115
|
# younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
|
|
117
116
|
# omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
|
|
118
117
|
max_age_ms: nil,
|
|
119
|
-
#
|
|
120
|
-
#
|
|
121
|
-
|
|
122
|
-
parse_pdf: nil,
|
|
118
|
+
# PDF parsing controls. Use start/end to limit text extraction and OCR to an
|
|
119
|
+
# inclusive 1-based page range.
|
|
120
|
+
pdf: nil,
|
|
123
121
|
# Shorten base64-encoded image data in the Markdown output
|
|
124
122
|
shorten_base64_images: nil,
|
|
125
123
|
# Optional timeout in milliseconds for the request. If the request takes longer
|
|
@@ -144,7 +142,7 @@ module ContextDev
|
|
|
144
142
|
include_images: T::Boolean,
|
|
145
143
|
include_links: T::Boolean,
|
|
146
144
|
max_age_ms: Integer,
|
|
147
|
-
|
|
145
|
+
pdf: ContextDev::WebWebScrapeMdParams::Pdf,
|
|
148
146
|
shorten_base64_images: T::Boolean,
|
|
149
147
|
timeout_ms: Integer,
|
|
150
148
|
use_main_content_only: T::Boolean,
|
|
@@ -155,6 +153,68 @@ module ContextDev
|
|
|
155
153
|
end
|
|
156
154
|
def to_hash
|
|
157
155
|
end
|
|
156
|
+
|
|
157
|
+
class Pdf < ContextDev::Internal::Type::BaseModel
|
|
158
|
+
OrHash =
|
|
159
|
+
T.type_alias do
|
|
160
|
+
T.any(
|
|
161
|
+
ContextDev::WebWebScrapeMdParams::Pdf,
|
|
162
|
+
ContextDev::Internal::AnyHash
|
|
163
|
+
)
|
|
164
|
+
end
|
|
165
|
+
|
|
166
|
+
# Last 1-based PDF page to parse. When omitted, parsing ends at the last page.
|
|
167
|
+
# Must be greater than or equal to start when both are provided.
|
|
168
|
+
sig { returns(T.nilable(Integer)) }
|
|
169
|
+
attr_reader :end_
|
|
170
|
+
|
|
171
|
+
sig { params(end_: Integer).void }
|
|
172
|
+
attr_writer :end_
|
|
173
|
+
|
|
174
|
+
# When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
|
|
175
|
+
# a 400 WEBSITE_ACCESS_ERROR is returned.
|
|
176
|
+
sig { returns(T.nilable(T::Boolean)) }
|
|
177
|
+
attr_reader :should_parse
|
|
178
|
+
|
|
179
|
+
sig { params(should_parse: T::Boolean).void }
|
|
180
|
+
attr_writer :should_parse
|
|
181
|
+
|
|
182
|
+
# First 1-based PDF page to parse. When omitted, parsing starts at the first page.
|
|
183
|
+
sig { returns(T.nilable(Integer)) }
|
|
184
|
+
attr_reader :start
|
|
185
|
+
|
|
186
|
+
sig { params(start: Integer).void }
|
|
187
|
+
attr_writer :start
|
|
188
|
+
|
|
189
|
+
# PDF parsing controls. Use start/end to limit text extraction and OCR to an
|
|
190
|
+
# inclusive 1-based page range.
|
|
191
|
+
sig do
|
|
192
|
+
params(
|
|
193
|
+
end_: Integer,
|
|
194
|
+
should_parse: T::Boolean,
|
|
195
|
+
start: Integer
|
|
196
|
+
).returns(T.attached_class)
|
|
197
|
+
end
|
|
198
|
+
def self.new(
|
|
199
|
+
# Last 1-based PDF page to parse. When omitted, parsing ends at the last page.
|
|
200
|
+
# Must be greater than or equal to start when both are provided.
|
|
201
|
+
end_: nil,
|
|
202
|
+
# When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
|
|
203
|
+
# a 400 WEBSITE_ACCESS_ERROR is returned.
|
|
204
|
+
should_parse: nil,
|
|
205
|
+
# First 1-based PDF page to parse. When omitted, parsing starts at the first page.
|
|
206
|
+
start: nil
|
|
207
|
+
)
|
|
208
|
+
end
|
|
209
|
+
|
|
210
|
+
sig do
|
|
211
|
+
override.returns(
|
|
212
|
+
{ end_: Integer, should_parse: T::Boolean, start: Integer }
|
|
213
|
+
)
|
|
214
|
+
end
|
|
215
|
+
def to_hash
|
|
216
|
+
end
|
|
217
|
+
end
|
|
158
218
|
end
|
|
159
219
|
end
|
|
160
220
|
end
|
|
@@ -65,6 +65,8 @@ module ContextDev
|
|
|
65
65
|
domain: String,
|
|
66
66
|
full_screenshot:
|
|
67
67
|
ContextDev::WebScreenshotParams::FullScreenshot::OrSymbol,
|
|
68
|
+
handle_cookie_popup:
|
|
69
|
+
ContextDev::WebScreenshotParams::HandleCookiePopup::OrSymbol,
|
|
68
70
|
max_age_ms: Integer,
|
|
69
71
|
page: ContextDev::WebScreenshotParams::Page::OrSymbol,
|
|
70
72
|
timeout_ms: Integer,
|
|
@@ -86,6 +88,10 @@ module ContextDev
|
|
|
86
88
|
# screenshot capturing all content. If 'false' or not provided, takes a viewport
|
|
87
89
|
# screenshot (standard browser view).
|
|
88
90
|
full_screenshot: nil,
|
|
91
|
+
# Optional parameter to control cookie/consent popup handling. If 'true', we
|
|
92
|
+
# dismiss cookie banner before capture. If 'false' or not provided, captures the
|
|
93
|
+
# page without that step.
|
|
94
|
+
handle_cookie_popup: nil,
|
|
89
95
|
# Return a cached screenshot if a prior screenshot for the same parameters exists
|
|
90
96
|
# and is younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
|
|
91
97
|
# omitted. Max is 30 days (2592000000 ms). Set to 0 to always capture fresh.
|
|
@@ -122,8 +128,9 @@ module ContextDev
|
|
|
122
128
|
max_age_ms: Integer,
|
|
123
129
|
max_depth: Integer,
|
|
124
130
|
max_pages: Integer,
|
|
125
|
-
|
|
131
|
+
pdf: ContextDev::WebWebCrawlMdParams::Pdf::OrHash,
|
|
126
132
|
shorten_base64_images: T::Boolean,
|
|
133
|
+
stop_after_ms: Integer,
|
|
127
134
|
timeout_ms: Integer,
|
|
128
135
|
url_regex: String,
|
|
129
136
|
use_main_content_only: T::Boolean,
|
|
@@ -153,12 +160,16 @@ module ContextDev
|
|
|
153
160
|
max_depth: nil,
|
|
154
161
|
# Maximum number of pages to crawl. Hard cap: 500.
|
|
155
162
|
max_pages: nil,
|
|
156
|
-
#
|
|
157
|
-
#
|
|
158
|
-
|
|
159
|
-
parse_pdf: nil,
|
|
163
|
+
# PDF parsing controls. Use start/end to limit text extraction and OCR to an
|
|
164
|
+
# inclusive 1-based page range.
|
|
165
|
+
pdf: nil,
|
|
160
166
|
# Truncate base64-encoded image data in the Markdown output
|
|
161
167
|
shorten_base64_images: nil,
|
|
168
|
+
# Soft time budget for the crawl in milliseconds. After each scrape, the crawler
|
|
169
|
+
# checks the elapsed time and, if exceeded, returns the pages collected so far
|
|
170
|
+
# instead of continuing. Min: 10000 (10s). Max: 240000 (4 min). Default: 120000 (2
|
|
171
|
+
# min).
|
|
172
|
+
stop_after_ms: nil,
|
|
162
173
|
# Optional timeout in milliseconds for the request. If the request takes longer
|
|
163
174
|
# than this value, it will be aborted with a 408 status code. Maximum allowed
|
|
164
175
|
# value is 300000ms (5 minutes).
|
|
@@ -181,7 +192,7 @@ module ContextDev
|
|
|
181
192
|
url: String,
|
|
182
193
|
include_frames: T::Boolean,
|
|
183
194
|
max_age_ms: Integer,
|
|
184
|
-
|
|
195
|
+
pdf: ContextDev::WebWebScrapeHTMLParams::Pdf::OrHash,
|
|
185
196
|
timeout_ms: Integer,
|
|
186
197
|
wait_for_ms: Integer,
|
|
187
198
|
request_options: ContextDev::RequestOptions::OrHash
|
|
@@ -196,10 +207,9 @@ module ContextDev
|
|
|
196
207
|
# younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
|
|
197
208
|
# omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
|
|
198
209
|
max_age_ms: nil,
|
|
199
|
-
#
|
|
200
|
-
#
|
|
201
|
-
|
|
202
|
-
parse_pdf: nil,
|
|
210
|
+
# PDF parsing controls. Use start/end to limit text extraction and OCR to an
|
|
211
|
+
# inclusive 1-based page range.
|
|
212
|
+
pdf: nil,
|
|
203
213
|
# Optional timeout in milliseconds for the request. If the request takes longer
|
|
204
214
|
# than this value, it will be aborted with a 408 status code. Maximum allowed
|
|
205
215
|
# value is 300000ms (5 minutes).
|
|
@@ -253,7 +263,7 @@ module ContextDev
|
|
|
253
263
|
include_images: T::Boolean,
|
|
254
264
|
include_links: T::Boolean,
|
|
255
265
|
max_age_ms: Integer,
|
|
256
|
-
|
|
266
|
+
pdf: ContextDev::WebWebScrapeMdParams::Pdf::OrHash,
|
|
257
267
|
shorten_base64_images: T::Boolean,
|
|
258
268
|
timeout_ms: Integer,
|
|
259
269
|
use_main_content_only: T::Boolean,
|
|
@@ -275,10 +285,9 @@ module ContextDev
|
|
|
275
285
|
# younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
|
|
276
286
|
# omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
|
|
277
287
|
max_age_ms: nil,
|
|
278
|
-
#
|
|
279
|
-
#
|
|
280
|
-
|
|
281
|
-
parse_pdf: nil,
|
|
288
|
+
# PDF parsing controls. Use start/end to limit text extraction and OCR to an
|
|
289
|
+
# inclusive 1-based page range.
|
|
290
|
+
pdf: nil,
|
|
282
291
|
# Shorten base64-encoded image data in the Markdown output
|
|
283
292
|
shorten_base64_images: nil,
|
|
284
293
|
# Optional timeout in milliseconds for the request. If the request takes longer
|
|
@@ -5,6 +5,7 @@ module ContextDev
|
|
|
5
5
|
direct_url: String,
|
|
6
6
|
domain: String,
|
|
7
7
|
full_screenshot: ContextDev::Models::WebScreenshotParams::full_screenshot,
|
|
8
|
+
handle_cookie_popup: ContextDev::Models::WebScreenshotParams::handle_cookie_popup,
|
|
8
9
|
max_age_ms: Integer,
|
|
9
10
|
page: ContextDev::Models::WebScreenshotParams::page,
|
|
10
11
|
timeout_ms: Integer,
|
|
@@ -31,6 +32,12 @@ module ContextDev
|
|
|
31
32
|
ContextDev::Models::WebScreenshotParams::full_screenshot
|
|
32
33
|
) -> ContextDev::Models::WebScreenshotParams::full_screenshot
|
|
33
34
|
|
|
35
|
+
attr_reader handle_cookie_popup: ContextDev::Models::WebScreenshotParams::handle_cookie_popup?
|
|
36
|
+
|
|
37
|
+
def handle_cookie_popup=: (
|
|
38
|
+
ContextDev::Models::WebScreenshotParams::handle_cookie_popup
|
|
39
|
+
) -> ContextDev::Models::WebScreenshotParams::handle_cookie_popup
|
|
40
|
+
|
|
34
41
|
attr_reader max_age_ms: Integer?
|
|
35
42
|
|
|
36
43
|
def max_age_ms=: (Integer) -> Integer
|
|
@@ -59,6 +66,7 @@ module ContextDev
|
|
|
59
66
|
?direct_url: String,
|
|
60
67
|
?domain: String,
|
|
61
68
|
?full_screenshot: ContextDev::Models::WebScreenshotParams::full_screenshot,
|
|
69
|
+
?handle_cookie_popup: ContextDev::Models::WebScreenshotParams::handle_cookie_popup,
|
|
62
70
|
?max_age_ms: Integer,
|
|
63
71
|
?page: ContextDev::Models::WebScreenshotParams::page,
|
|
64
72
|
?timeout_ms: Integer,
|
|
@@ -71,6 +79,7 @@ module ContextDev
|
|
|
71
79
|
direct_url: String,
|
|
72
80
|
domain: String,
|
|
73
81
|
full_screenshot: ContextDev::Models::WebScreenshotParams::full_screenshot,
|
|
82
|
+
handle_cookie_popup: ContextDev::Models::WebScreenshotParams::handle_cookie_popup,
|
|
74
83
|
max_age_ms: Integer,
|
|
75
84
|
page: ContextDev::Models::WebScreenshotParams::page,
|
|
76
85
|
timeout_ms: Integer,
|
|
@@ -90,6 +99,17 @@ module ContextDev
|
|
|
90
99
|
def self?.values: -> ::Array[ContextDev::Models::WebScreenshotParams::full_screenshot]
|
|
91
100
|
end
|
|
92
101
|
|
|
102
|
+
type handle_cookie_popup = :true | :false
|
|
103
|
+
|
|
104
|
+
module HandleCookiePopup
|
|
105
|
+
extend ContextDev::Internal::Type::Enum
|
|
106
|
+
|
|
107
|
+
TRUE: :true
|
|
108
|
+
FALSE: :false
|
|
109
|
+
|
|
110
|
+
def self?.values: -> ::Array[ContextDev::Models::WebScreenshotParams::handle_cookie_popup]
|
|
111
|
+
end
|
|
112
|
+
|
|
93
113
|
type page =
|
|
94
114
|
:login
|
|
95
115
|
| :signup
|
|
@@ -10,8 +10,9 @@ module ContextDev
|
|
|
10
10
|
max_age_ms: Integer,
|
|
11
11
|
max_depth: Integer,
|
|
12
12
|
max_pages: Integer,
|
|
13
|
-
|
|
13
|
+
pdf: ContextDev::WebWebCrawlMdParams::Pdf,
|
|
14
14
|
:shorten_base64_images => bool,
|
|
15
|
+
stop_after_ms: Integer,
|
|
15
16
|
timeout_ms: Integer,
|
|
16
17
|
url_regex: String,
|
|
17
18
|
use_main_content_only: bool,
|
|
@@ -53,14 +54,20 @@ module ContextDev
|
|
|
53
54
|
|
|
54
55
|
def max_pages=: (Integer) -> Integer
|
|
55
56
|
|
|
56
|
-
attr_reader
|
|
57
|
+
attr_reader pdf: ContextDev::WebWebCrawlMdParams::Pdf?
|
|
57
58
|
|
|
58
|
-
def
|
|
59
|
+
def pdf=: (
|
|
60
|
+
ContextDev::WebWebCrawlMdParams::Pdf
|
|
61
|
+
) -> ContextDev::WebWebCrawlMdParams::Pdf
|
|
59
62
|
|
|
60
63
|
attr_reader shorten_base64_images: bool?
|
|
61
64
|
|
|
62
65
|
def shorten_base64_images=: (bool) -> bool
|
|
63
66
|
|
|
67
|
+
attr_reader stop_after_ms: Integer?
|
|
68
|
+
|
|
69
|
+
def stop_after_ms=: (Integer) -> Integer
|
|
70
|
+
|
|
64
71
|
attr_reader timeout_ms: Integer?
|
|
65
72
|
|
|
66
73
|
def timeout_ms=: (Integer) -> Integer
|
|
@@ -86,8 +93,9 @@ module ContextDev
|
|
|
86
93
|
?max_age_ms: Integer,
|
|
87
94
|
?max_depth: Integer,
|
|
88
95
|
?max_pages: Integer,
|
|
89
|
-
?
|
|
96
|
+
?pdf: ContextDev::WebWebCrawlMdParams::Pdf,
|
|
90
97
|
?shorten_base64_images: bool,
|
|
98
|
+
?stop_after_ms: Integer,
|
|
91
99
|
?timeout_ms: Integer,
|
|
92
100
|
?url_regex: String,
|
|
93
101
|
?use_main_content_only: bool,
|
|
@@ -104,14 +112,39 @@ module ContextDev
|
|
|
104
112
|
max_age_ms: Integer,
|
|
105
113
|
max_depth: Integer,
|
|
106
114
|
max_pages: Integer,
|
|
107
|
-
|
|
115
|
+
pdf: ContextDev::WebWebCrawlMdParams::Pdf,
|
|
108
116
|
:shorten_base64_images => bool,
|
|
117
|
+
stop_after_ms: Integer,
|
|
109
118
|
timeout_ms: Integer,
|
|
110
119
|
url_regex: String,
|
|
111
120
|
use_main_content_only: bool,
|
|
112
121
|
wait_for_ms: Integer,
|
|
113
122
|
request_options: ContextDev::RequestOptions
|
|
114
123
|
}
|
|
124
|
+
|
|
125
|
+
type pdf = { end_: Integer, should_parse: bool, start: Integer }
|
|
126
|
+
|
|
127
|
+
class Pdf < ContextDev::Internal::Type::BaseModel
|
|
128
|
+
attr_reader end_: Integer?
|
|
129
|
+
|
|
130
|
+
def end_=: (Integer) -> Integer
|
|
131
|
+
|
|
132
|
+
attr_reader should_parse: bool?
|
|
133
|
+
|
|
134
|
+
def should_parse=: (bool) -> bool
|
|
135
|
+
|
|
136
|
+
attr_reader start: Integer?
|
|
137
|
+
|
|
138
|
+
def start=: (Integer) -> Integer
|
|
139
|
+
|
|
140
|
+
def initialize: (
|
|
141
|
+
?end_: Integer,
|
|
142
|
+
?should_parse: bool,
|
|
143
|
+
?start: Integer
|
|
144
|
+
) -> void
|
|
145
|
+
|
|
146
|
+
def to_hash: -> { end_: Integer, should_parse: bool, start: Integer }
|
|
147
|
+
end
|
|
115
148
|
end
|
|
116
149
|
end
|
|
117
150
|
end
|