context.dev 1.16.0 → 1.18.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +19 -0
- data/README.md +3 -3
- data/lib/context_dev/models/ai_extract_product_params.rb +4 -3
- data/lib/context_dev/models/ai_extract_products_params.rb +8 -6
- data/lib/context_dev/models/web_screenshot_params.rb +18 -21
- data/lib/context_dev/models/web_web_crawl_md_params.rb +72 -8
- data/lib/context_dev/models/web_web_crawl_md_response.rb +3 -2
- data/lib/context_dev/models/web_web_scrape_html_params.rb +61 -8
- data/lib/context_dev/models/web_web_scrape_images_params.rb +20 -1
- data/lib/context_dev/models/web_web_scrape_md_params.rb +61 -8
- data/lib/context_dev/models/web_web_scrape_sitemap_params.rb +11 -1
- data/lib/context_dev/resources/ai.rb +1 -1
- data/lib/context_dev/resources/web.rb +46 -16
- data/lib/context_dev/version.rb +1 -1
- data/rbi/context_dev/models/ai_extract_product_params.rbi +6 -4
- data/rbi/context_dev/models/ai_extract_products_params.rbi +12 -8
- data/rbi/context_dev/models/web_screenshot_params.rbi +28 -53
- data/rbi/context_dev/models/web_web_crawl_md_params.rbi +118 -13
- data/rbi/context_dev/models/web_web_crawl_md_response.rbi +4 -2
- data/rbi/context_dev/models/web_web_scrape_html_params.rbi +101 -13
- data/rbi/context_dev/models/web_web_scrape_images_params.rbi +28 -0
- data/rbi/context_dev/models/web_web_scrape_md_params.rbi +101 -13
- data/rbi/context_dev/models/web_web_scrape_sitemap_params.rbi +15 -0
- data/rbi/context_dev/resources/ai.rbi +3 -2
- data/rbi/context_dev/resources/web.rbi +69 -20
- data/sig/context_dev/models/web_screenshot_params.rbs +13 -19
- data/sig/context_dev/models/web_web_crawl_md_params.rbs +53 -6
- data/sig/context_dev/models/web_web_scrape_html_params.rbs +45 -5
- data/sig/context_dev/models/web_web_scrape_images_params.rbs +15 -1
- data/sig/context_dev/models/web_web_scrape_md_params.rbs +46 -6
- data/sig/context_dev/models/web_web_scrape_sitemap_params.rbs +12 -1
- data/sig/context_dev/resources/web.rbs +15 -4
- metadata +2 -2
|
@@ -34,21 +34,39 @@ module ContextDev
|
|
|
34
34
|
sig { params(max_age_ms: Integer).void }
|
|
35
35
|
attr_writer :max_age_ms
|
|
36
36
|
|
|
37
|
-
#
|
|
38
|
-
#
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
37
|
+
# PDF parsing controls. Use start/end to limit text extraction and OCR to an
|
|
38
|
+
# inclusive 1-based page range.
|
|
39
|
+
sig { returns(T.nilable(ContextDev::WebWebScrapeHTMLParams::Pdf)) }
|
|
40
|
+
attr_reader :pdf
|
|
41
|
+
|
|
42
|
+
sig { params(pdf: ContextDev::WebWebScrapeHTMLParams::Pdf::OrHash).void }
|
|
43
|
+
attr_writer :pdf
|
|
44
|
+
|
|
45
|
+
# Optional timeout in milliseconds for the request. If the request takes longer
|
|
46
|
+
# than this value, it will be aborted with a 408 status code. Maximum allowed
|
|
47
|
+
# value is 300000ms (5 minutes).
|
|
48
|
+
sig { returns(T.nilable(Integer)) }
|
|
49
|
+
attr_reader :timeout_ms
|
|
50
|
+
|
|
51
|
+
sig { params(timeout_ms: Integer).void }
|
|
52
|
+
attr_writer :timeout_ms
|
|
53
|
+
|
|
54
|
+
# Optional browser wait time in milliseconds after initial page load. Min: 0. Max:
|
|
55
|
+
# 30000 (30 seconds).
|
|
56
|
+
sig { returns(T.nilable(Integer)) }
|
|
57
|
+
attr_reader :wait_for_ms
|
|
42
58
|
|
|
43
|
-
sig { params(
|
|
44
|
-
attr_writer :
|
|
59
|
+
sig { params(wait_for_ms: Integer).void }
|
|
60
|
+
attr_writer :wait_for_ms
|
|
45
61
|
|
|
46
62
|
sig do
|
|
47
63
|
params(
|
|
48
64
|
url: String,
|
|
49
65
|
include_frames: T::Boolean,
|
|
50
66
|
max_age_ms: Integer,
|
|
51
|
-
|
|
67
|
+
pdf: ContextDev::WebWebScrapeHTMLParams::Pdf::OrHash,
|
|
68
|
+
timeout_ms: Integer,
|
|
69
|
+
wait_for_ms: Integer,
|
|
52
70
|
request_options: ContextDev::RequestOptions::OrHash
|
|
53
71
|
).returns(T.attached_class)
|
|
54
72
|
end
|
|
@@ -61,10 +79,16 @@ module ContextDev
|
|
|
61
79
|
# younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
|
|
62
80
|
# omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
|
|
63
81
|
max_age_ms: nil,
|
|
64
|
-
#
|
|
65
|
-
#
|
|
66
|
-
|
|
67
|
-
|
|
82
|
+
# PDF parsing controls. Use start/end to limit text extraction and OCR to an
|
|
83
|
+
# inclusive 1-based page range.
|
|
84
|
+
pdf: nil,
|
|
85
|
+
# Optional timeout in milliseconds for the request. If the request takes longer
|
|
86
|
+
# than this value, it will be aborted with a 408 status code. Maximum allowed
|
|
87
|
+
# value is 300000ms (5 minutes).
|
|
88
|
+
timeout_ms: nil,
|
|
89
|
+
# Optional browser wait time in milliseconds after initial page load. Min: 0. Max:
|
|
90
|
+
# 30000 (30 seconds).
|
|
91
|
+
wait_for_ms: nil,
|
|
68
92
|
request_options: {}
|
|
69
93
|
)
|
|
70
94
|
end
|
|
@@ -75,13 +99,77 @@ module ContextDev
|
|
|
75
99
|
url: String,
|
|
76
100
|
include_frames: T::Boolean,
|
|
77
101
|
max_age_ms: Integer,
|
|
78
|
-
|
|
102
|
+
pdf: ContextDev::WebWebScrapeHTMLParams::Pdf,
|
|
103
|
+
timeout_ms: Integer,
|
|
104
|
+
wait_for_ms: Integer,
|
|
79
105
|
request_options: ContextDev::RequestOptions
|
|
80
106
|
}
|
|
81
107
|
)
|
|
82
108
|
end
|
|
83
109
|
def to_hash
|
|
84
110
|
end
|
|
111
|
+
|
|
112
|
+
class Pdf < ContextDev::Internal::Type::BaseModel
|
|
113
|
+
OrHash =
|
|
114
|
+
T.type_alias do
|
|
115
|
+
T.any(
|
|
116
|
+
ContextDev::WebWebScrapeHTMLParams::Pdf,
|
|
117
|
+
ContextDev::Internal::AnyHash
|
|
118
|
+
)
|
|
119
|
+
end
|
|
120
|
+
|
|
121
|
+
# Last 1-based PDF page to parse. When omitted, parsing ends at the last page.
|
|
122
|
+
# Must be greater than or equal to start when both are provided.
|
|
123
|
+
sig { returns(T.nilable(Integer)) }
|
|
124
|
+
attr_reader :end_
|
|
125
|
+
|
|
126
|
+
sig { params(end_: Integer).void }
|
|
127
|
+
attr_writer :end_
|
|
128
|
+
|
|
129
|
+
# When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
|
|
130
|
+
# a 400 WEBSITE_ACCESS_ERROR is returned.
|
|
131
|
+
sig { returns(T.nilable(T::Boolean)) }
|
|
132
|
+
attr_reader :should_parse
|
|
133
|
+
|
|
134
|
+
sig { params(should_parse: T::Boolean).void }
|
|
135
|
+
attr_writer :should_parse
|
|
136
|
+
|
|
137
|
+
# First 1-based PDF page to parse. When omitted, parsing starts at the first page.
|
|
138
|
+
sig { returns(T.nilable(Integer)) }
|
|
139
|
+
attr_reader :start
|
|
140
|
+
|
|
141
|
+
sig { params(start: Integer).void }
|
|
142
|
+
attr_writer :start
|
|
143
|
+
|
|
144
|
+
# PDF parsing controls. Use start/end to limit text extraction and OCR to an
|
|
145
|
+
# inclusive 1-based page range.
|
|
146
|
+
sig do
|
|
147
|
+
params(
|
|
148
|
+
end_: Integer,
|
|
149
|
+
should_parse: T::Boolean,
|
|
150
|
+
start: Integer
|
|
151
|
+
).returns(T.attached_class)
|
|
152
|
+
end
|
|
153
|
+
def self.new(
|
|
154
|
+
# Last 1-based PDF page to parse. When omitted, parsing ends at the last page.
|
|
155
|
+
# Must be greater than or equal to start when both are provided.
|
|
156
|
+
end_: nil,
|
|
157
|
+
# When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
|
|
158
|
+
# a 400 WEBSITE_ACCESS_ERROR is returned.
|
|
159
|
+
should_parse: nil,
|
|
160
|
+
# First 1-based PDF page to parse. When omitted, parsing starts at the first page.
|
|
161
|
+
start: nil
|
|
162
|
+
)
|
|
163
|
+
end
|
|
164
|
+
|
|
165
|
+
sig do
|
|
166
|
+
override.returns(
|
|
167
|
+
{ end_: Integer, should_parse: T::Boolean, start: Integer }
|
|
168
|
+
)
|
|
169
|
+
end
|
|
170
|
+
def to_hash
|
|
171
|
+
end
|
|
172
|
+
end
|
|
85
173
|
end
|
|
86
174
|
end
|
|
87
175
|
end
|
|
@@ -40,11 +40,30 @@ module ContextDev
|
|
|
40
40
|
sig { params(max_age_ms: Integer).void }
|
|
41
41
|
attr_writer :max_age_ms
|
|
42
42
|
|
|
43
|
+
# Optional timeout in milliseconds for the request. If the request takes longer
|
|
44
|
+
# than this value, it will be aborted with a 408 status code. Maximum allowed
|
|
45
|
+
# value is 300000ms (5 minutes).
|
|
46
|
+
sig { returns(T.nilable(Integer)) }
|
|
47
|
+
attr_reader :timeout_ms
|
|
48
|
+
|
|
49
|
+
sig { params(timeout_ms: Integer).void }
|
|
50
|
+
attr_writer :timeout_ms
|
|
51
|
+
|
|
52
|
+
# Optional browser wait time in milliseconds after initial page load before
|
|
53
|
+
# collecting images. Min: 0. Max: 30000 (30 seconds).
|
|
54
|
+
sig { returns(T.nilable(Integer)) }
|
|
55
|
+
attr_reader :wait_for_ms
|
|
56
|
+
|
|
57
|
+
sig { params(wait_for_ms: Integer).void }
|
|
58
|
+
attr_writer :wait_for_ms
|
|
59
|
+
|
|
43
60
|
sig do
|
|
44
61
|
params(
|
|
45
62
|
url: String,
|
|
46
63
|
enrichment: ContextDev::WebWebScrapeImagesParams::Enrichment::OrHash,
|
|
47
64
|
max_age_ms: Integer,
|
|
65
|
+
timeout_ms: Integer,
|
|
66
|
+
wait_for_ms: Integer,
|
|
48
67
|
request_options: ContextDev::RequestOptions::OrHash
|
|
49
68
|
).returns(T.attached_class)
|
|
50
69
|
end
|
|
@@ -57,6 +76,13 @@ module ContextDev
|
|
|
57
76
|
# Reuse a cached result this many milliseconds old or newer. Default: 86400000 (1
|
|
58
77
|
# day). Set to 0 to bypass cache. Maximum: 2592000000 (30 days).
|
|
59
78
|
max_age_ms: nil,
|
|
79
|
+
# Optional timeout in milliseconds for the request. If the request takes longer
|
|
80
|
+
# than this value, it will be aborted with a 408 status code. Maximum allowed
|
|
81
|
+
# value is 300000ms (5 minutes).
|
|
82
|
+
timeout_ms: nil,
|
|
83
|
+
# Optional browser wait time in milliseconds after initial page load before
|
|
84
|
+
# collecting images. Min: 0. Max: 30000 (30 seconds).
|
|
85
|
+
wait_for_ms: nil,
|
|
60
86
|
request_options: {}
|
|
61
87
|
)
|
|
62
88
|
end
|
|
@@ -67,6 +93,8 @@ module ContextDev
|
|
|
67
93
|
url: String,
|
|
68
94
|
enrichment: ContextDev::WebWebScrapeImagesParams::Enrichment,
|
|
69
95
|
max_age_ms: Integer,
|
|
96
|
+
timeout_ms: Integer,
|
|
97
|
+
wait_for_ms: Integer,
|
|
70
98
|
request_options: ContextDev::RequestOptions
|
|
71
99
|
}
|
|
72
100
|
)
|
|
@@ -46,14 +46,13 @@ module ContextDev
|
|
|
46
46
|
sig { params(max_age_ms: Integer).void }
|
|
47
47
|
attr_writer :max_age_ms
|
|
48
48
|
|
|
49
|
-
#
|
|
50
|
-
#
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
attr_reader :parse_pdf
|
|
49
|
+
# PDF parsing controls. Use start/end to limit text extraction and OCR to an
|
|
50
|
+
# inclusive 1-based page range.
|
|
51
|
+
sig { returns(T.nilable(ContextDev::WebWebScrapeMdParams::Pdf)) }
|
|
52
|
+
attr_reader :pdf
|
|
54
53
|
|
|
55
|
-
sig { params(
|
|
56
|
-
attr_writer :
|
|
54
|
+
sig { params(pdf: ContextDev::WebWebScrapeMdParams::Pdf::OrHash).void }
|
|
55
|
+
attr_writer :pdf
|
|
57
56
|
|
|
58
57
|
# Shorten base64-encoded image data in the Markdown output
|
|
59
58
|
sig { returns(T.nilable(T::Boolean)) }
|
|
@@ -62,6 +61,15 @@ module ContextDev
|
|
|
62
61
|
sig { params(shorten_base64_images: T::Boolean).void }
|
|
63
62
|
attr_writer :shorten_base64_images
|
|
64
63
|
|
|
64
|
+
# Optional timeout in milliseconds for the request. If the request takes longer
|
|
65
|
+
# than this value, it will be aborted with a 408 status code. Maximum allowed
|
|
66
|
+
# value is 300000ms (5 minutes).
|
|
67
|
+
sig { returns(T.nilable(Integer)) }
|
|
68
|
+
attr_reader :timeout_ms
|
|
69
|
+
|
|
70
|
+
sig { params(timeout_ms: Integer).void }
|
|
71
|
+
attr_writer :timeout_ms
|
|
72
|
+
|
|
65
73
|
# Extract only the main content of the page, excluding headers, footers, sidebars,
|
|
66
74
|
# and navigation
|
|
67
75
|
sig { returns(T.nilable(T::Boolean)) }
|
|
@@ -70,6 +78,14 @@ module ContextDev
|
|
|
70
78
|
sig { params(use_main_content_only: T::Boolean).void }
|
|
71
79
|
attr_writer :use_main_content_only
|
|
72
80
|
|
|
81
|
+
# Optional browser wait time in milliseconds after initial page load before
|
|
82
|
+
# converting the page to Markdown. Min: 0. Max: 30000 (30 seconds).
|
|
83
|
+
sig { returns(T.nilable(Integer)) }
|
|
84
|
+
attr_reader :wait_for_ms
|
|
85
|
+
|
|
86
|
+
sig { params(wait_for_ms: Integer).void }
|
|
87
|
+
attr_writer :wait_for_ms
|
|
88
|
+
|
|
73
89
|
sig do
|
|
74
90
|
params(
|
|
75
91
|
url: String,
|
|
@@ -77,9 +93,11 @@ module ContextDev
|
|
|
77
93
|
include_images: T::Boolean,
|
|
78
94
|
include_links: T::Boolean,
|
|
79
95
|
max_age_ms: Integer,
|
|
80
|
-
|
|
96
|
+
pdf: ContextDev::WebWebScrapeMdParams::Pdf::OrHash,
|
|
81
97
|
shorten_base64_images: T::Boolean,
|
|
98
|
+
timeout_ms: Integer,
|
|
82
99
|
use_main_content_only: T::Boolean,
|
|
100
|
+
wait_for_ms: Integer,
|
|
83
101
|
request_options: ContextDev::RequestOptions::OrHash
|
|
84
102
|
).returns(T.attached_class)
|
|
85
103
|
end
|
|
@@ -97,15 +115,21 @@ module ContextDev
|
|
|
97
115
|
# younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
|
|
98
116
|
# omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
|
|
99
117
|
max_age_ms: nil,
|
|
100
|
-
#
|
|
101
|
-
#
|
|
102
|
-
|
|
103
|
-
parse_pdf: nil,
|
|
118
|
+
# PDF parsing controls. Use start/end to limit text extraction and OCR to an
|
|
119
|
+
# inclusive 1-based page range.
|
|
120
|
+
pdf: nil,
|
|
104
121
|
# Shorten base64-encoded image data in the Markdown output
|
|
105
122
|
shorten_base64_images: nil,
|
|
123
|
+
# Optional timeout in milliseconds for the request. If the request takes longer
|
|
124
|
+
# than this value, it will be aborted with a 408 status code. Maximum allowed
|
|
125
|
+
# value is 300000ms (5 minutes).
|
|
126
|
+
timeout_ms: nil,
|
|
106
127
|
# Extract only the main content of the page, excluding headers, footers, sidebars,
|
|
107
128
|
# and navigation
|
|
108
129
|
use_main_content_only: nil,
|
|
130
|
+
# Optional browser wait time in milliseconds after initial page load before
|
|
131
|
+
# converting the page to Markdown. Min: 0. Max: 30000 (30 seconds).
|
|
132
|
+
wait_for_ms: nil,
|
|
109
133
|
request_options: {}
|
|
110
134
|
)
|
|
111
135
|
end
|
|
@@ -118,15 +142,79 @@ module ContextDev
|
|
|
118
142
|
include_images: T::Boolean,
|
|
119
143
|
include_links: T::Boolean,
|
|
120
144
|
max_age_ms: Integer,
|
|
121
|
-
|
|
145
|
+
pdf: ContextDev::WebWebScrapeMdParams::Pdf,
|
|
122
146
|
shorten_base64_images: T::Boolean,
|
|
147
|
+
timeout_ms: Integer,
|
|
123
148
|
use_main_content_only: T::Boolean,
|
|
149
|
+
wait_for_ms: Integer,
|
|
124
150
|
request_options: ContextDev::RequestOptions
|
|
125
151
|
}
|
|
126
152
|
)
|
|
127
153
|
end
|
|
128
154
|
def to_hash
|
|
129
155
|
end
|
|
156
|
+
|
|
157
|
+
class Pdf < ContextDev::Internal::Type::BaseModel
|
|
158
|
+
OrHash =
|
|
159
|
+
T.type_alias do
|
|
160
|
+
T.any(
|
|
161
|
+
ContextDev::WebWebScrapeMdParams::Pdf,
|
|
162
|
+
ContextDev::Internal::AnyHash
|
|
163
|
+
)
|
|
164
|
+
end
|
|
165
|
+
|
|
166
|
+
# Last 1-based PDF page to parse. When omitted, parsing ends at the last page.
|
|
167
|
+
# Must be greater than or equal to start when both are provided.
|
|
168
|
+
sig { returns(T.nilable(Integer)) }
|
|
169
|
+
attr_reader :end_
|
|
170
|
+
|
|
171
|
+
sig { params(end_: Integer).void }
|
|
172
|
+
attr_writer :end_
|
|
173
|
+
|
|
174
|
+
# When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
|
|
175
|
+
# a 400 WEBSITE_ACCESS_ERROR is returned.
|
|
176
|
+
sig { returns(T.nilable(T::Boolean)) }
|
|
177
|
+
attr_reader :should_parse
|
|
178
|
+
|
|
179
|
+
sig { params(should_parse: T::Boolean).void }
|
|
180
|
+
attr_writer :should_parse
|
|
181
|
+
|
|
182
|
+
# First 1-based PDF page to parse. When omitted, parsing starts at the first page.
|
|
183
|
+
sig { returns(T.nilable(Integer)) }
|
|
184
|
+
attr_reader :start
|
|
185
|
+
|
|
186
|
+
sig { params(start: Integer).void }
|
|
187
|
+
attr_writer :start
|
|
188
|
+
|
|
189
|
+
# PDF parsing controls. Use start/end to limit text extraction and OCR to an
|
|
190
|
+
# inclusive 1-based page range.
|
|
191
|
+
sig do
|
|
192
|
+
params(
|
|
193
|
+
end_: Integer,
|
|
194
|
+
should_parse: T::Boolean,
|
|
195
|
+
start: Integer
|
|
196
|
+
).returns(T.attached_class)
|
|
197
|
+
end
|
|
198
|
+
def self.new(
|
|
199
|
+
# Last 1-based PDF page to parse. When omitted, parsing ends at the last page.
|
|
200
|
+
# Must be greater than or equal to start when both are provided.
|
|
201
|
+
end_: nil,
|
|
202
|
+
# When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
|
|
203
|
+
# a 400 WEBSITE_ACCESS_ERROR is returned.
|
|
204
|
+
should_parse: nil,
|
|
205
|
+
# First 1-based PDF page to parse. When omitted, parsing starts at the first page.
|
|
206
|
+
start: nil
|
|
207
|
+
)
|
|
208
|
+
end
|
|
209
|
+
|
|
210
|
+
sig do
|
|
211
|
+
override.returns(
|
|
212
|
+
{ end_: Integer, should_parse: T::Boolean, start: Integer }
|
|
213
|
+
)
|
|
214
|
+
end
|
|
215
|
+
def to_hash
|
|
216
|
+
end
|
|
217
|
+
end
|
|
130
218
|
end
|
|
131
219
|
end
|
|
132
220
|
end
|
|
@@ -26,6 +26,15 @@ module ContextDev
|
|
|
26
26
|
sig { params(max_links: Integer).void }
|
|
27
27
|
attr_writer :max_links
|
|
28
28
|
|
|
29
|
+
# Optional timeout in milliseconds for the request. If the request takes longer
|
|
30
|
+
# than this value, it will be aborted with a 408 status code. Maximum allowed
|
|
31
|
+
# value is 300000ms (5 minutes).
|
|
32
|
+
sig { returns(T.nilable(Integer)) }
|
|
33
|
+
attr_reader :timeout_ms
|
|
34
|
+
|
|
35
|
+
sig { params(timeout_ms: Integer).void }
|
|
36
|
+
attr_writer :timeout_ms
|
|
37
|
+
|
|
29
38
|
# Optional RE2-compatible regex pattern. Only URLs matching this pattern are
|
|
30
39
|
# returned and counted against maxLinks.
|
|
31
40
|
sig { returns(T.nilable(String)) }
|
|
@@ -38,6 +47,7 @@ module ContextDev
|
|
|
38
47
|
params(
|
|
39
48
|
domain: String,
|
|
40
49
|
max_links: Integer,
|
|
50
|
+
timeout_ms: Integer,
|
|
41
51
|
url_regex: String,
|
|
42
52
|
request_options: ContextDev::RequestOptions::OrHash
|
|
43
53
|
).returns(T.attached_class)
|
|
@@ -48,6 +58,10 @@ module ContextDev
|
|
|
48
58
|
# Maximum number of links to return from the sitemap crawl. Defaults to 10,000.
|
|
49
59
|
# Minimum is 1, maximum is 100,000.
|
|
50
60
|
max_links: nil,
|
|
61
|
+
# Optional timeout in milliseconds for the request. If the request takes longer
|
|
62
|
+
# than this value, it will be aborted with a 408 status code. Maximum allowed
|
|
63
|
+
# value is 300000ms (5 minutes).
|
|
64
|
+
timeout_ms: nil,
|
|
51
65
|
# Optional RE2-compatible regex pattern. Only URLs matching this pattern are
|
|
52
66
|
# returned and counted against maxLinks.
|
|
53
67
|
url_regex: nil,
|
|
@@ -60,6 +74,7 @@ module ContextDev
|
|
|
60
74
|
{
|
|
61
75
|
domain: String,
|
|
62
76
|
max_links: Integer,
|
|
77
|
+
timeout_ms: Integer,
|
|
63
78
|
url_regex: String,
|
|
64
79
|
request_options: ContextDev::RequestOptions
|
|
65
80
|
}
|
|
@@ -48,8 +48,9 @@ module ContextDev
|
|
|
48
48
|
# younger than this many milliseconds. Defaults to 7 days (604800000 ms) when
|
|
49
49
|
# omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
|
|
50
50
|
max_age_ms: nil,
|
|
51
|
-
# Optional timeout in milliseconds for the request.
|
|
52
|
-
#
|
|
51
|
+
# Optional timeout in milliseconds for the request. If the request takes longer
|
|
52
|
+
# than this value, it will be aborted with a 408 status code. Maximum allowed
|
|
53
|
+
# value is 300000ms (5 minutes).
|
|
53
54
|
timeout_ms: nil,
|
|
54
55
|
request_options: {}
|
|
55
56
|
)
|
|
@@ -67,8 +67,9 @@ module ContextDev
|
|
|
67
67
|
ContextDev::WebScreenshotParams::FullScreenshot::OrSymbol,
|
|
68
68
|
max_age_ms: Integer,
|
|
69
69
|
page: ContextDev::WebScreenshotParams::Page::OrSymbol,
|
|
70
|
-
|
|
70
|
+
timeout_ms: Integer,
|
|
71
71
|
viewport: ContextDev::WebScreenshotParams::Viewport::OrHash,
|
|
72
|
+
wait_for_ms: Integer,
|
|
72
73
|
request_options: ContextDev::RequestOptions::OrHash
|
|
73
74
|
).returns(ContextDev::Models::WebScreenshotResponse)
|
|
74
75
|
end
|
|
@@ -95,12 +96,16 @@ module ContextDev
|
|
|
95
96
|
# provided, screenshots the main domain landing page. Only applicable when using
|
|
96
97
|
# 'domain', not 'directUrl'.
|
|
97
98
|
page: nil,
|
|
98
|
-
# Optional
|
|
99
|
-
#
|
|
100
|
-
#
|
|
101
|
-
|
|
99
|
+
# Optional timeout in milliseconds for the request. If the request takes longer
|
|
100
|
+
# than this value, it will be aborted with a 408 status code. Maximum allowed
|
|
101
|
+
# value is 300000ms (5 minutes).
|
|
102
|
+
timeout_ms: nil,
|
|
102
103
|
# Optional browser viewport dimensions for the screenshot. Defaults to 1920x1080.
|
|
103
104
|
viewport: nil,
|
|
105
|
+
# Optional browser wait time in milliseconds after initial page load before taking
|
|
106
|
+
# the screenshot. Min: 0. Max: 30000 (30 seconds). Defaults to 3000 ms when
|
|
107
|
+
# omitted.
|
|
108
|
+
wait_for_ms: nil,
|
|
104
109
|
request_options: {}
|
|
105
110
|
)
|
|
106
111
|
end
|
|
@@ -117,10 +122,13 @@ module ContextDev
|
|
|
117
122
|
max_age_ms: Integer,
|
|
118
123
|
max_depth: Integer,
|
|
119
124
|
max_pages: Integer,
|
|
120
|
-
|
|
125
|
+
pdf: ContextDev::WebWebCrawlMdParams::Pdf::OrHash,
|
|
121
126
|
shorten_base64_images: T::Boolean,
|
|
127
|
+
stop_after_ms: Integer,
|
|
128
|
+
timeout_ms: Integer,
|
|
122
129
|
url_regex: String,
|
|
123
130
|
use_main_content_only: T::Boolean,
|
|
131
|
+
wait_for_ms: Integer,
|
|
124
132
|
request_options: ContextDev::RequestOptions::OrHash
|
|
125
133
|
).returns(ContextDev::Models::WebWebCrawlMdResponse)
|
|
126
134
|
end
|
|
@@ -146,17 +154,28 @@ module ContextDev
|
|
|
146
154
|
max_depth: nil,
|
|
147
155
|
# Maximum number of pages to crawl. Hard cap: 500.
|
|
148
156
|
max_pages: nil,
|
|
149
|
-
#
|
|
150
|
-
#
|
|
151
|
-
|
|
152
|
-
parse_pdf: nil,
|
|
157
|
+
# PDF parsing controls. Use start/end to limit text extraction and OCR to an
|
|
158
|
+
# inclusive 1-based page range.
|
|
159
|
+
pdf: nil,
|
|
153
160
|
# Truncate base64-encoded image data in the Markdown output
|
|
154
161
|
shorten_base64_images: nil,
|
|
162
|
+
# Soft time budget for the crawl in milliseconds. After each scrape, the crawler
|
|
163
|
+
# checks the elapsed time and, if exceeded, returns the pages collected so far
|
|
164
|
+
# instead of continuing. Min: 10000 (10s). Max: 240000 (4 min). Default: 120000 (2
|
|
165
|
+
# min).
|
|
166
|
+
stop_after_ms: nil,
|
|
167
|
+
# Optional timeout in milliseconds for the request. If the request takes longer
|
|
168
|
+
# than this value, it will be aborted with a 408 status code. Maximum allowed
|
|
169
|
+
# value is 300000ms (5 minutes).
|
|
170
|
+
timeout_ms: nil,
|
|
155
171
|
# Regex pattern. Only URLs matching this pattern will be followed and scraped.
|
|
156
172
|
url_regex: nil,
|
|
157
173
|
# Extract only the main content, stripping headers, footers, sidebars, and
|
|
158
174
|
# navigation
|
|
159
175
|
use_main_content_only: nil,
|
|
176
|
+
# Optional browser wait time in milliseconds after initial page load for each
|
|
177
|
+
# crawled page. Min: 0. Max: 30000 (30 seconds).
|
|
178
|
+
wait_for_ms: nil,
|
|
160
179
|
request_options: {}
|
|
161
180
|
)
|
|
162
181
|
end
|
|
@@ -167,7 +186,9 @@ module ContextDev
|
|
|
167
186
|
url: String,
|
|
168
187
|
include_frames: T::Boolean,
|
|
169
188
|
max_age_ms: Integer,
|
|
170
|
-
|
|
189
|
+
pdf: ContextDev::WebWebScrapeHTMLParams::Pdf::OrHash,
|
|
190
|
+
timeout_ms: Integer,
|
|
191
|
+
wait_for_ms: Integer,
|
|
171
192
|
request_options: ContextDev::RequestOptions::OrHash
|
|
172
193
|
).returns(ContextDev::Models::WebWebScrapeHTMLResponse)
|
|
173
194
|
end
|
|
@@ -180,10 +201,16 @@ module ContextDev
|
|
|
180
201
|
# younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
|
|
181
202
|
# omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
|
|
182
203
|
max_age_ms: nil,
|
|
183
|
-
#
|
|
184
|
-
#
|
|
185
|
-
|
|
186
|
-
|
|
204
|
+
# PDF parsing controls. Use start/end to limit text extraction and OCR to an
|
|
205
|
+
# inclusive 1-based page range.
|
|
206
|
+
pdf: nil,
|
|
207
|
+
# Optional timeout in milliseconds for the request. If the request takes longer
|
|
208
|
+
# than this value, it will be aborted with a 408 status code. Maximum allowed
|
|
209
|
+
# value is 300000ms (5 minutes).
|
|
210
|
+
timeout_ms: nil,
|
|
211
|
+
# Optional browser wait time in milliseconds after initial page load. Min: 0. Max:
|
|
212
|
+
# 30000 (30 seconds).
|
|
213
|
+
wait_for_ms: nil,
|
|
187
214
|
request_options: {}
|
|
188
215
|
)
|
|
189
216
|
end
|
|
@@ -197,6 +224,8 @@ module ContextDev
|
|
|
197
224
|
url: String,
|
|
198
225
|
enrichment: ContextDev::WebWebScrapeImagesParams::Enrichment::OrHash,
|
|
199
226
|
max_age_ms: Integer,
|
|
227
|
+
timeout_ms: Integer,
|
|
228
|
+
wait_for_ms: Integer,
|
|
200
229
|
request_options: ContextDev::RequestOptions::OrHash
|
|
201
230
|
).returns(ContextDev::Models::WebWebScrapeImagesResponse)
|
|
202
231
|
end
|
|
@@ -209,6 +238,13 @@ module ContextDev
|
|
|
209
238
|
# Reuse a cached result this many milliseconds old or newer. Default: 86400000 (1
|
|
210
239
|
# day). Set to 0 to bypass cache. Maximum: 2592000000 (30 days).
|
|
211
240
|
max_age_ms: nil,
|
|
241
|
+
# Optional timeout in milliseconds for the request. If the request takes longer
|
|
242
|
+
# than this value, it will be aborted with a 408 status code. Maximum allowed
|
|
243
|
+
# value is 300000ms (5 minutes).
|
|
244
|
+
timeout_ms: nil,
|
|
245
|
+
# Optional browser wait time in milliseconds after initial page load before
|
|
246
|
+
# collecting images. Min: 0. Max: 30000 (30 seconds).
|
|
247
|
+
wait_for_ms: nil,
|
|
212
248
|
request_options: {}
|
|
213
249
|
)
|
|
214
250
|
end
|
|
@@ -221,9 +257,11 @@ module ContextDev
|
|
|
221
257
|
include_images: T::Boolean,
|
|
222
258
|
include_links: T::Boolean,
|
|
223
259
|
max_age_ms: Integer,
|
|
224
|
-
|
|
260
|
+
pdf: ContextDev::WebWebScrapeMdParams::Pdf::OrHash,
|
|
225
261
|
shorten_base64_images: T::Boolean,
|
|
262
|
+
timeout_ms: Integer,
|
|
226
263
|
use_main_content_only: T::Boolean,
|
|
264
|
+
wait_for_ms: Integer,
|
|
227
265
|
request_options: ContextDev::RequestOptions::OrHash
|
|
228
266
|
).returns(ContextDev::Models::WebWebScrapeMdResponse)
|
|
229
267
|
end
|
|
@@ -241,15 +279,21 @@ module ContextDev
|
|
|
241
279
|
# younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
|
|
242
280
|
# omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
|
|
243
281
|
max_age_ms: nil,
|
|
244
|
-
#
|
|
245
|
-
#
|
|
246
|
-
|
|
247
|
-
parse_pdf: nil,
|
|
282
|
+
# PDF parsing controls. Use start/end to limit text extraction and OCR to an
|
|
283
|
+
# inclusive 1-based page range.
|
|
284
|
+
pdf: nil,
|
|
248
285
|
# Shorten base64-encoded image data in the Markdown output
|
|
249
286
|
shorten_base64_images: nil,
|
|
287
|
+
# Optional timeout in milliseconds for the request. If the request takes longer
|
|
288
|
+
# than this value, it will be aborted with a 408 status code. Maximum allowed
|
|
289
|
+
# value is 300000ms (5 minutes).
|
|
290
|
+
timeout_ms: nil,
|
|
250
291
|
# Extract only the main content of the page, excluding headers, footers, sidebars,
|
|
251
292
|
# and navigation
|
|
252
293
|
use_main_content_only: nil,
|
|
294
|
+
# Optional browser wait time in milliseconds after initial page load before
|
|
295
|
+
# converting the page to Markdown. Min: 0. Max: 30000 (30 seconds).
|
|
296
|
+
wait_for_ms: nil,
|
|
253
297
|
request_options: {}
|
|
254
298
|
)
|
|
255
299
|
end
|
|
@@ -259,6 +303,7 @@ module ContextDev
|
|
|
259
303
|
params(
|
|
260
304
|
domain: String,
|
|
261
305
|
max_links: Integer,
|
|
306
|
+
timeout_ms: Integer,
|
|
262
307
|
url_regex: String,
|
|
263
308
|
request_options: ContextDev::RequestOptions::OrHash
|
|
264
309
|
).returns(ContextDev::Models::WebWebScrapeSitemapResponse)
|
|
@@ -269,6 +314,10 @@ module ContextDev
|
|
|
269
314
|
# Maximum number of links to return from the sitemap crawl. Defaults to 10,000.
|
|
270
315
|
# Minimum is 1, maximum is 100,000.
|
|
271
316
|
max_links: nil,
|
|
317
|
+
# Optional timeout in milliseconds for the request. If the request takes longer
|
|
318
|
+
# than this value, it will be aborted with a 408 status code. Maximum allowed
|
|
319
|
+
# value is 300000ms (5 minutes).
|
|
320
|
+
timeout_ms: nil,
|
|
272
321
|
# Optional RE2-compatible regex pattern. Only URLs matching this pattern are
|
|
273
322
|
# returned and counted against maxLinks.
|
|
274
323
|
url_regex: nil,
|