context.dev 1.23.0 → 1.25.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +16 -0
- data/README.md +1 -1
- data/lib/context_dev/internal/type/base_model.rb +5 -5
- data/lib/context_dev/models/web_extract_fonts_params.rb +12 -1
- data/lib/context_dev/models/web_extract_params.rb +148 -0
- data/lib/context_dev/models/web_extract_response.rb +83 -0
- data/lib/context_dev/models/web_extract_styleguide_params.rb +12 -1
- data/lib/context_dev/models.rb +2 -0
- data/lib/context_dev/resources/web.rb +64 -4
- data/lib/context_dev/version.rb +1 -1
- data/lib/context_dev.rb +2 -0
- data/rbi/context_dev/models/web_extract_fonts_params.rbi +17 -0
- data/rbi/context_dev/models/web_extract_params.rbi +232 -0
- data/rbi/context_dev/models/web_extract_response.rbi +134 -0
- data/rbi/context_dev/models/web_extract_styleguide_params.rbi +17 -0
- data/rbi/context_dev/models.rbi +2 -0
- data/rbi/context_dev/resources/web.rbi +71 -0
- data/sig/context_dev/models/web_extract_fonts_params.rbs +12 -1
- data/sig/context_dev/models/web_extract_params.rbs +120 -0
- data/sig/context_dev/models/web_extract_response.rbs +77 -0
- data/sig/context_dev/models/web_extract_styleguide_params.rbs +12 -1
- data/sig/context_dev/models.rbs +2 -0
- data/sig/context_dev/resources/web.rbs +17 -0
- metadata +8 -2
|
@@ -0,0 +1,232 @@
|
|
|
1
|
+
# typed: strong
|
|
2
|
+
|
|
3
|
+
module ContextDev
|
|
4
|
+
module Models
|
|
5
|
+
class WebExtractParams < ContextDev::Internal::Type::BaseModel
|
|
6
|
+
extend ContextDev::Internal::Type::RequestParameters::Converter
|
|
7
|
+
include ContextDev::Internal::Type::RequestParameters
|
|
8
|
+
|
|
9
|
+
OrHash =
|
|
10
|
+
T.type_alias do
|
|
11
|
+
T.any(ContextDev::WebExtractParams, ContextDev::Internal::AnyHash)
|
|
12
|
+
end
|
|
13
|
+
|
|
14
|
+
# JSON Schema for the returned data object. TypeScript Zod users can pass a JSON
|
|
15
|
+
# Schema generated from a Zod object; Python users can pass the equivalent JSON
|
|
16
|
+
# Schema object.
|
|
17
|
+
sig { returns(T::Hash[Symbol, T.anything]) }
|
|
18
|
+
attr_accessor :schema
|
|
19
|
+
|
|
20
|
+
# The starting website URL to crawl and extract from. Must include http:// or
|
|
21
|
+
# https://.
|
|
22
|
+
sig { returns(String) }
|
|
23
|
+
attr_accessor :url
|
|
24
|
+
|
|
25
|
+
# When true (default), every returned value must be grounded in facts stated on
|
|
26
|
+
# the page; fields that cannot be supported by the page are returned as
|
|
27
|
+
# null/empty. When false, the model may make reasonable inferences and derivations
|
|
28
|
+
# from the page content (e.g. ideal customer, competitor analysis,
|
|
29
|
+
# recommendations) while keeping verifiable specifics (names, quotes, URLs, dates,
|
|
30
|
+
# metrics) faithful to the source.
|
|
31
|
+
sig { returns(T.nilable(T::Boolean)) }
|
|
32
|
+
attr_reader :fact_check
|
|
33
|
+
|
|
34
|
+
sig { params(fact_check: T::Boolean).void }
|
|
35
|
+
attr_writer :fact_check
|
|
36
|
+
|
|
37
|
+
# When true, follow links on subdomains of the starting URL's domain.
|
|
38
|
+
sig { returns(T.nilable(T::Boolean)) }
|
|
39
|
+
attr_reader :follow_subdomains
|
|
40
|
+
|
|
41
|
+
sig { params(follow_subdomains: T::Boolean).void }
|
|
42
|
+
attr_writer :follow_subdomains
|
|
43
|
+
|
|
44
|
+
# When true, iframe contents are included in Markdown before extraction.
|
|
45
|
+
sig { returns(T.nilable(T::Boolean)) }
|
|
46
|
+
attr_reader :include_frames
|
|
47
|
+
|
|
48
|
+
sig { params(include_frames: T::Boolean).void }
|
|
49
|
+
attr_writer :include_frames
|
|
50
|
+
|
|
51
|
+
# Optional extraction guidance, such as which facts to prioritize or how to
|
|
52
|
+
# interpret fields in the schema.
|
|
53
|
+
sig { returns(T.nilable(String)) }
|
|
54
|
+
attr_reader :instructions
|
|
55
|
+
|
|
56
|
+
sig { params(instructions: String).void }
|
|
57
|
+
attr_writer :instructions
|
|
58
|
+
|
|
59
|
+
# Return cached scrape results if a prior scrape for the same parameters is
|
|
60
|
+
# younger than this many milliseconds.
|
|
61
|
+
sig { returns(T.nilable(Integer)) }
|
|
62
|
+
attr_reader :max_age_ms
|
|
63
|
+
|
|
64
|
+
sig { params(max_age_ms: Integer).void }
|
|
65
|
+
attr_writer :max_age_ms
|
|
66
|
+
|
|
67
|
+
sig { returns(T.nilable(ContextDev::WebExtractParams::Pdf)) }
|
|
68
|
+
attr_reader :pdf
|
|
69
|
+
|
|
70
|
+
sig { params(pdf: ContextDev::WebExtractParams::Pdf::OrHash).void }
|
|
71
|
+
attr_writer :pdf
|
|
72
|
+
|
|
73
|
+
# Soft time budget for the crawl in milliseconds.
|
|
74
|
+
sig { returns(T.nilable(Integer)) }
|
|
75
|
+
attr_reader :stop_after_ms
|
|
76
|
+
|
|
77
|
+
sig { params(stop_after_ms: Integer).void }
|
|
78
|
+
attr_writer :stop_after_ms
|
|
79
|
+
|
|
80
|
+
# Optional timeout in milliseconds for the request. If the request takes longer
|
|
81
|
+
# than this value, it will be aborted with a 408 status code. Maximum allowed
|
|
82
|
+
# value is 300000ms (5 minutes).
|
|
83
|
+
sig { returns(T.nilable(Integer)) }
|
|
84
|
+
attr_reader :timeout_ms
|
|
85
|
+
|
|
86
|
+
sig { params(timeout_ms: Integer).void }
|
|
87
|
+
attr_writer :timeout_ms
|
|
88
|
+
|
|
89
|
+
# Optional browser wait time in milliseconds after initial page load for each
|
|
90
|
+
# crawled page.
|
|
91
|
+
sig { returns(T.nilable(Integer)) }
|
|
92
|
+
attr_reader :wait_for_ms
|
|
93
|
+
|
|
94
|
+
sig { params(wait_for_ms: Integer).void }
|
|
95
|
+
attr_writer :wait_for_ms
|
|
96
|
+
|
|
97
|
+
sig do
|
|
98
|
+
params(
|
|
99
|
+
schema: T::Hash[Symbol, T.anything],
|
|
100
|
+
url: String,
|
|
101
|
+
fact_check: T::Boolean,
|
|
102
|
+
follow_subdomains: T::Boolean,
|
|
103
|
+
include_frames: T::Boolean,
|
|
104
|
+
instructions: String,
|
|
105
|
+
max_age_ms: Integer,
|
|
106
|
+
pdf: ContextDev::WebExtractParams::Pdf::OrHash,
|
|
107
|
+
stop_after_ms: Integer,
|
|
108
|
+
timeout_ms: Integer,
|
|
109
|
+
wait_for_ms: Integer,
|
|
110
|
+
request_options: ContextDev::RequestOptions::OrHash
|
|
111
|
+
).returns(T.attached_class)
|
|
112
|
+
end
|
|
113
|
+
def self.new(
|
|
114
|
+
# JSON Schema for the returned data object. TypeScript Zod users can pass a JSON
|
|
115
|
+
# Schema generated from a Zod object; Python users can pass the equivalent JSON
|
|
116
|
+
# Schema object.
|
|
117
|
+
schema:,
|
|
118
|
+
# The starting website URL to crawl and extract from. Must include http:// or
|
|
119
|
+
# https://.
|
|
120
|
+
url:,
|
|
121
|
+
# When true (default), every returned value must be grounded in facts stated on
|
|
122
|
+
# the page; fields that cannot be supported by the page are returned as
|
|
123
|
+
# null/empty. When false, the model may make reasonable inferences and derivations
|
|
124
|
+
# from the page content (e.g. ideal customer, competitor analysis,
|
|
125
|
+
# recommendations) while keeping verifiable specifics (names, quotes, URLs, dates,
|
|
126
|
+
# metrics) faithful to the source.
|
|
127
|
+
fact_check: nil,
|
|
128
|
+
# When true, follow links on subdomains of the starting URL's domain.
|
|
129
|
+
follow_subdomains: nil,
|
|
130
|
+
# When true, iframe contents are included in Markdown before extraction.
|
|
131
|
+
include_frames: nil,
|
|
132
|
+
# Optional extraction guidance, such as which facts to prioritize or how to
|
|
133
|
+
# interpret fields in the schema.
|
|
134
|
+
instructions: nil,
|
|
135
|
+
# Return cached scrape results if a prior scrape for the same parameters is
|
|
136
|
+
# younger than this many milliseconds.
|
|
137
|
+
max_age_ms: nil,
|
|
138
|
+
pdf: nil,
|
|
139
|
+
# Soft time budget for the crawl in milliseconds.
|
|
140
|
+
stop_after_ms: nil,
|
|
141
|
+
# Optional timeout in milliseconds for the request. If the request takes longer
|
|
142
|
+
# than this value, it will be aborted with a 408 status code. Maximum allowed
|
|
143
|
+
# value is 300000ms (5 minutes).
|
|
144
|
+
timeout_ms: nil,
|
|
145
|
+
# Optional browser wait time in milliseconds after initial page load for each
|
|
146
|
+
# crawled page.
|
|
147
|
+
wait_for_ms: nil,
|
|
148
|
+
request_options: {}
|
|
149
|
+
)
|
|
150
|
+
end
|
|
151
|
+
|
|
152
|
+
sig do
|
|
153
|
+
override.returns(
|
|
154
|
+
{
|
|
155
|
+
schema: T::Hash[Symbol, T.anything],
|
|
156
|
+
url: String,
|
|
157
|
+
fact_check: T::Boolean,
|
|
158
|
+
follow_subdomains: T::Boolean,
|
|
159
|
+
include_frames: T::Boolean,
|
|
160
|
+
instructions: String,
|
|
161
|
+
max_age_ms: Integer,
|
|
162
|
+
pdf: ContextDev::WebExtractParams::Pdf,
|
|
163
|
+
stop_after_ms: Integer,
|
|
164
|
+
timeout_ms: Integer,
|
|
165
|
+
wait_for_ms: Integer,
|
|
166
|
+
request_options: ContextDev::RequestOptions
|
|
167
|
+
}
|
|
168
|
+
)
|
|
169
|
+
end
|
|
170
|
+
def to_hash
|
|
171
|
+
end
|
|
172
|
+
|
|
173
|
+
class Pdf < ContextDev::Internal::Type::BaseModel
|
|
174
|
+
OrHash =
|
|
175
|
+
T.type_alias do
|
|
176
|
+
T.any(
|
|
177
|
+
ContextDev::WebExtractParams::Pdf,
|
|
178
|
+
ContextDev::Internal::AnyHash
|
|
179
|
+
)
|
|
180
|
+
end
|
|
181
|
+
|
|
182
|
+
# Last 1-based PDF page to parse. Must be greater than or equal to start when both
|
|
183
|
+
# are provided.
|
|
184
|
+
sig { returns(T.nilable(Integer)) }
|
|
185
|
+
attr_reader :end_
|
|
186
|
+
|
|
187
|
+
sig { params(end_: Integer).void }
|
|
188
|
+
attr_writer :end_
|
|
189
|
+
|
|
190
|
+
# When true, PDF pages are fetched and parsed. When false, PDF pages are skipped.
|
|
191
|
+
sig { returns(T.nilable(T::Boolean)) }
|
|
192
|
+
attr_reader :should_parse
|
|
193
|
+
|
|
194
|
+
sig { params(should_parse: T::Boolean).void }
|
|
195
|
+
attr_writer :should_parse
|
|
196
|
+
|
|
197
|
+
# First 1-based PDF page to parse.
|
|
198
|
+
sig { returns(T.nilable(Integer)) }
|
|
199
|
+
attr_reader :start
|
|
200
|
+
|
|
201
|
+
sig { params(start: Integer).void }
|
|
202
|
+
attr_writer :start
|
|
203
|
+
|
|
204
|
+
sig do
|
|
205
|
+
params(
|
|
206
|
+
end_: Integer,
|
|
207
|
+
should_parse: T::Boolean,
|
|
208
|
+
start: Integer
|
|
209
|
+
).returns(T.attached_class)
|
|
210
|
+
end
|
|
211
|
+
def self.new(
|
|
212
|
+
# Last 1-based PDF page to parse. Must be greater than or equal to start when both
|
|
213
|
+
# are provided.
|
|
214
|
+
end_: nil,
|
|
215
|
+
# When true, PDF pages are fetched and parsed. When false, PDF pages are skipped.
|
|
216
|
+
should_parse: nil,
|
|
217
|
+
# First 1-based PDF page to parse.
|
|
218
|
+
start: nil
|
|
219
|
+
)
|
|
220
|
+
end
|
|
221
|
+
|
|
222
|
+
sig do
|
|
223
|
+
override.returns(
|
|
224
|
+
{ end_: Integer, should_parse: T::Boolean, start: Integer }
|
|
225
|
+
)
|
|
226
|
+
end
|
|
227
|
+
def to_hash
|
|
228
|
+
end
|
|
229
|
+
end
|
|
230
|
+
end
|
|
231
|
+
end
|
|
232
|
+
end
|
|
@@ -0,0 +1,134 @@
|
|
|
1
|
+
# typed: strong
|
|
2
|
+
|
|
3
|
+
module ContextDev
|
|
4
|
+
module Models
|
|
5
|
+
class WebExtractResponse < ContextDev::Internal::Type::BaseModel
|
|
6
|
+
OrHash =
|
|
7
|
+
T.type_alias do
|
|
8
|
+
T.any(
|
|
9
|
+
ContextDev::Models::WebExtractResponse,
|
|
10
|
+
ContextDev::Internal::AnyHash
|
|
11
|
+
)
|
|
12
|
+
end
|
|
13
|
+
|
|
14
|
+
# Extracted data matching the request schema
|
|
15
|
+
sig { returns(T::Hash[Symbol, T.anything]) }
|
|
16
|
+
attr_accessor :data
|
|
17
|
+
|
|
18
|
+
sig { returns(ContextDev::Models::WebExtractResponse::Metadata) }
|
|
19
|
+
attr_reader :metadata
|
|
20
|
+
|
|
21
|
+
sig do
|
|
22
|
+
params(
|
|
23
|
+
metadata: ContextDev::Models::WebExtractResponse::Metadata::OrHash
|
|
24
|
+
).void
|
|
25
|
+
end
|
|
26
|
+
attr_writer :metadata
|
|
27
|
+
|
|
28
|
+
# Status of the response, e.g., 'ok'
|
|
29
|
+
sig { returns(String) }
|
|
30
|
+
attr_accessor :status
|
|
31
|
+
|
|
32
|
+
# The starting URL that was analyzed
|
|
33
|
+
sig { returns(String) }
|
|
34
|
+
attr_accessor :url
|
|
35
|
+
|
|
36
|
+
# List of URLs whose Markdown was used for extraction
|
|
37
|
+
sig { returns(T::Array[String]) }
|
|
38
|
+
attr_accessor :urls_analyzed
|
|
39
|
+
|
|
40
|
+
sig do
|
|
41
|
+
params(
|
|
42
|
+
data: T::Hash[Symbol, T.anything],
|
|
43
|
+
metadata: ContextDev::Models::WebExtractResponse::Metadata::OrHash,
|
|
44
|
+
status: String,
|
|
45
|
+
url: String,
|
|
46
|
+
urls_analyzed: T::Array[String]
|
|
47
|
+
).returns(T.attached_class)
|
|
48
|
+
end
|
|
49
|
+
def self.new(
|
|
50
|
+
# Extracted data matching the request schema
|
|
51
|
+
data:,
|
|
52
|
+
metadata:,
|
|
53
|
+
# Status of the response, e.g., 'ok'
|
|
54
|
+
status:,
|
|
55
|
+
# The starting URL that was analyzed
|
|
56
|
+
url:,
|
|
57
|
+
# List of URLs whose Markdown was used for extraction
|
|
58
|
+
urls_analyzed:
|
|
59
|
+
)
|
|
60
|
+
end
|
|
61
|
+
|
|
62
|
+
sig do
|
|
63
|
+
override.returns(
|
|
64
|
+
{
|
|
65
|
+
data: T::Hash[Symbol, T.anything],
|
|
66
|
+
metadata: ContextDev::Models::WebExtractResponse::Metadata,
|
|
67
|
+
status: String,
|
|
68
|
+
url: String,
|
|
69
|
+
urls_analyzed: T::Array[String]
|
|
70
|
+
}
|
|
71
|
+
)
|
|
72
|
+
end
|
|
73
|
+
def to_hash
|
|
74
|
+
end
|
|
75
|
+
|
|
76
|
+
class Metadata < ContextDev::Internal::Type::BaseModel
|
|
77
|
+
OrHash =
|
|
78
|
+
T.type_alias do
|
|
79
|
+
T.any(
|
|
80
|
+
ContextDev::Models::WebExtractResponse::Metadata,
|
|
81
|
+
ContextDev::Internal::AnyHash
|
|
82
|
+
)
|
|
83
|
+
end
|
|
84
|
+
|
|
85
|
+
sig { returns(Integer) }
|
|
86
|
+
attr_accessor :max_crawl_depth
|
|
87
|
+
|
|
88
|
+
sig { returns(Integer) }
|
|
89
|
+
attr_accessor :num_failed
|
|
90
|
+
|
|
91
|
+
sig { returns(Integer) }
|
|
92
|
+
attr_accessor :num_skipped
|
|
93
|
+
|
|
94
|
+
sig { returns(Integer) }
|
|
95
|
+
attr_accessor :num_succeeded
|
|
96
|
+
|
|
97
|
+
sig { returns(Integer) }
|
|
98
|
+
attr_accessor :num_urls
|
|
99
|
+
|
|
100
|
+
sig do
|
|
101
|
+
params(
|
|
102
|
+
max_crawl_depth: Integer,
|
|
103
|
+
num_failed: Integer,
|
|
104
|
+
num_skipped: Integer,
|
|
105
|
+
num_succeeded: Integer,
|
|
106
|
+
num_urls: Integer
|
|
107
|
+
).returns(T.attached_class)
|
|
108
|
+
end
|
|
109
|
+
def self.new(
|
|
110
|
+
max_crawl_depth:,
|
|
111
|
+
num_failed:,
|
|
112
|
+
num_skipped:,
|
|
113
|
+
num_succeeded:,
|
|
114
|
+
num_urls:
|
|
115
|
+
)
|
|
116
|
+
end
|
|
117
|
+
|
|
118
|
+
sig do
|
|
119
|
+
override.returns(
|
|
120
|
+
{
|
|
121
|
+
max_crawl_depth: Integer,
|
|
122
|
+
num_failed: Integer,
|
|
123
|
+
num_skipped: Integer,
|
|
124
|
+
num_succeeded: Integer,
|
|
125
|
+
num_urls: Integer
|
|
126
|
+
}
|
|
127
|
+
)
|
|
128
|
+
end
|
|
129
|
+
def to_hash
|
|
130
|
+
end
|
|
131
|
+
end
|
|
132
|
+
end
|
|
133
|
+
end
|
|
134
|
+
end
|
|
@@ -33,6 +33,16 @@ module ContextDev
|
|
|
33
33
|
sig { params(domain: String).void }
|
|
34
34
|
attr_writer :domain
|
|
35
35
|
|
|
36
|
+
# Maximum age in milliseconds for cached data before the API performs a hard
|
|
37
|
+
# refresh. Defaults to 3 months (7776000000 ms). Values below 1 day (86400000 ms)
|
|
38
|
+
# are clamped to 1 day; values above 1 year (31536000000 ms) are clamped to 1
|
|
39
|
+
# year.
|
|
40
|
+
sig { returns(T.nilable(Integer)) }
|
|
41
|
+
attr_reader :max_age_ms
|
|
42
|
+
|
|
43
|
+
sig { params(max_age_ms: Integer).void }
|
|
44
|
+
attr_writer :max_age_ms
|
|
45
|
+
|
|
36
46
|
# Optional timeout in milliseconds for the request. If the request takes longer
|
|
37
47
|
# than this value, it will be aborted with a 408 status code. Maximum allowed
|
|
38
48
|
# value is 300000ms (5 minutes).
|
|
@@ -46,6 +56,7 @@ module ContextDev
|
|
|
46
56
|
params(
|
|
47
57
|
direct_url: String,
|
|
48
58
|
domain: String,
|
|
59
|
+
max_age_ms: Integer,
|
|
49
60
|
timeout_ms: Integer,
|
|
50
61
|
request_options: ContextDev::RequestOptions::OrHash
|
|
51
62
|
).returns(T.attached_class)
|
|
@@ -60,6 +71,11 @@ module ContextDev
|
|
|
60
71
|
# domain will be automatically normalized and validated. You must provide either
|
|
61
72
|
# 'domain' or 'directUrl', but not both.
|
|
62
73
|
domain: nil,
|
|
74
|
+
# Maximum age in milliseconds for cached data before the API performs a hard
|
|
75
|
+
# refresh. Defaults to 3 months (7776000000 ms). Values below 1 day (86400000 ms)
|
|
76
|
+
# are clamped to 1 day; values above 1 year (31536000000 ms) are clamped to 1
|
|
77
|
+
# year.
|
|
78
|
+
max_age_ms: nil,
|
|
63
79
|
# Optional timeout in milliseconds for the request. If the request takes longer
|
|
64
80
|
# than this value, it will be aborted with a 408 status code. Maximum allowed
|
|
65
81
|
# value is 300000ms (5 minutes).
|
|
@@ -73,6 +89,7 @@ module ContextDev
|
|
|
73
89
|
{
|
|
74
90
|
direct_url: String,
|
|
75
91
|
domain: String,
|
|
92
|
+
max_age_ms: Integer,
|
|
76
93
|
timeout_ms: Integer,
|
|
77
94
|
request_options: ContextDev::RequestOptions
|
|
78
95
|
}
|
data/rbi/context_dev/models.rbi
CHANGED
|
@@ -34,6 +34,8 @@ module ContextDev
|
|
|
34
34
|
|
|
35
35
|
WebExtractFontsParams = ContextDev::Models::WebExtractFontsParams
|
|
36
36
|
|
|
37
|
+
WebExtractParams = ContextDev::Models::WebExtractParams
|
|
38
|
+
|
|
37
39
|
WebExtractStyleguideParams = ContextDev::Models::WebExtractStyleguideParams
|
|
38
40
|
|
|
39
41
|
WebScreenshotParams = ContextDev::Models::WebScreenshotParams
|
|
@@ -3,12 +3,72 @@
|
|
|
3
3
|
module ContextDev
|
|
4
4
|
module Resources
|
|
5
5
|
class Web
|
|
6
|
+
# Crawl a website, convert pages to Markdown using the scrape cache, and extract
|
|
7
|
+
# structured data into the provided JSON Schema. The schema must describe the
|
|
8
|
+
# response data object. This endpoint does not accept targeted page-type
|
|
9
|
+
# selection.
|
|
10
|
+
sig do
|
|
11
|
+
params(
|
|
12
|
+
schema: T::Hash[Symbol, T.anything],
|
|
13
|
+
url: String,
|
|
14
|
+
fact_check: T::Boolean,
|
|
15
|
+
follow_subdomains: T::Boolean,
|
|
16
|
+
include_frames: T::Boolean,
|
|
17
|
+
instructions: String,
|
|
18
|
+
max_age_ms: Integer,
|
|
19
|
+
pdf: ContextDev::WebExtractParams::Pdf::OrHash,
|
|
20
|
+
stop_after_ms: Integer,
|
|
21
|
+
timeout_ms: Integer,
|
|
22
|
+
wait_for_ms: Integer,
|
|
23
|
+
request_options: ContextDev::RequestOptions::OrHash
|
|
24
|
+
).returns(ContextDev::Models::WebExtractResponse)
|
|
25
|
+
end
|
|
26
|
+
def extract(
|
|
27
|
+
# JSON Schema for the returned data object. TypeScript Zod users can pass a JSON
|
|
28
|
+
# Schema generated from a Zod object; Python users can pass the equivalent JSON
|
|
29
|
+
# Schema object.
|
|
30
|
+
schema:,
|
|
31
|
+
# The starting website URL to crawl and extract from. Must include http:// or
|
|
32
|
+
# https://.
|
|
33
|
+
url:,
|
|
34
|
+
# When true (default), every returned value must be grounded in facts stated on
|
|
35
|
+
# the page; fields that cannot be supported by the page are returned as
|
|
36
|
+
# null/empty. When false, the model may make reasonable inferences and derivations
|
|
37
|
+
# from the page content (e.g. ideal customer, competitor analysis,
|
|
38
|
+
# recommendations) while keeping verifiable specifics (names, quotes, URLs, dates,
|
|
39
|
+
# metrics) faithful to the source.
|
|
40
|
+
fact_check: nil,
|
|
41
|
+
# When true, follow links on subdomains of the starting URL's domain.
|
|
42
|
+
follow_subdomains: nil,
|
|
43
|
+
# When true, iframe contents are included in Markdown before extraction.
|
|
44
|
+
include_frames: nil,
|
|
45
|
+
# Optional extraction guidance, such as which facts to prioritize or how to
|
|
46
|
+
# interpret fields in the schema.
|
|
47
|
+
instructions: nil,
|
|
48
|
+
# Return cached scrape results if a prior scrape for the same parameters is
|
|
49
|
+
# younger than this many milliseconds.
|
|
50
|
+
max_age_ms: nil,
|
|
51
|
+
pdf: nil,
|
|
52
|
+
# Soft time budget for the crawl in milliseconds.
|
|
53
|
+
stop_after_ms: nil,
|
|
54
|
+
# Optional timeout in milliseconds for the request. If the request takes longer
|
|
55
|
+
# than this value, it will be aborted with a 408 status code. Maximum allowed
|
|
56
|
+
# value is 300000ms (5 minutes).
|
|
57
|
+
timeout_ms: nil,
|
|
58
|
+
# Optional browser wait time in milliseconds after initial page load for each
|
|
59
|
+
# crawled page.
|
|
60
|
+
wait_for_ms: nil,
|
|
61
|
+
request_options: {}
|
|
62
|
+
)
|
|
63
|
+
end
|
|
64
|
+
|
|
6
65
|
# Scrape font information from a website including font families, usage
|
|
7
66
|
# statistics, fallbacks, and element/word counts.
|
|
8
67
|
sig do
|
|
9
68
|
params(
|
|
10
69
|
direct_url: String,
|
|
11
70
|
domain: String,
|
|
71
|
+
max_age_ms: Integer,
|
|
12
72
|
timeout_ms: Integer,
|
|
13
73
|
request_options: ContextDev::RequestOptions::OrHash
|
|
14
74
|
).returns(ContextDev::Models::WebExtractFontsResponse)
|
|
@@ -22,6 +82,11 @@ module ContextDev
|
|
|
22
82
|
# domain will be automatically normalized and validated. You must provide either
|
|
23
83
|
# 'domain' or 'directUrl', but not both.
|
|
24
84
|
domain: nil,
|
|
85
|
+
# Maximum age in milliseconds for cached data before the API performs a hard
|
|
86
|
+
# refresh. Defaults to 3 months (7776000000 ms). Values below 1 day (86400000 ms)
|
|
87
|
+
# are clamped to 1 day; values above 1 year (31536000000 ms) are clamped to 1
|
|
88
|
+
# year.
|
|
89
|
+
max_age_ms: nil,
|
|
25
90
|
# Optional timeout in milliseconds for the request. If the request takes longer
|
|
26
91
|
# than this value, it will be aborted with a 408 status code. Maximum allowed
|
|
27
92
|
# value is 300000ms (5 minutes).
|
|
@@ -36,6 +101,7 @@ module ContextDev
|
|
|
36
101
|
params(
|
|
37
102
|
direct_url: String,
|
|
38
103
|
domain: String,
|
|
104
|
+
max_age_ms: Integer,
|
|
39
105
|
timeout_ms: Integer,
|
|
40
106
|
request_options: ContextDev::RequestOptions::OrHash
|
|
41
107
|
).returns(ContextDev::Models::WebExtractStyleguideResponse)
|
|
@@ -50,6 +116,11 @@ module ContextDev
|
|
|
50
116
|
# domain will be automatically normalized and validated. You must provide either
|
|
51
117
|
# 'domain' or 'directUrl', but not both.
|
|
52
118
|
domain: nil,
|
|
119
|
+
# Maximum age in milliseconds for cached data before the API performs a hard
|
|
120
|
+
# refresh. Defaults to 3 months (7776000000 ms). Values below 1 day (86400000 ms)
|
|
121
|
+
# are clamped to 1 day; values above 1 year (31536000000 ms) are clamped to 1
|
|
122
|
+
# year.
|
|
123
|
+
max_age_ms: nil,
|
|
53
124
|
# Optional timeout in milliseconds for the request. If the request takes longer
|
|
54
125
|
# than this value, it will be aborted with a 408 status code. Maximum allowed
|
|
55
126
|
# value is 300000ms (5 minutes).
|
|
@@ -1,7 +1,12 @@
|
|
|
1
1
|
module ContextDev
|
|
2
2
|
module Models
|
|
3
3
|
type web_extract_fonts_params =
|
|
4
|
-
{
|
|
4
|
+
{
|
|
5
|
+
direct_url: String,
|
|
6
|
+
domain: String,
|
|
7
|
+
max_age_ms: Integer,
|
|
8
|
+
timeout_ms: Integer
|
|
9
|
+
}
|
|
5
10
|
& ContextDev::Internal::Type::request_parameters
|
|
6
11
|
|
|
7
12
|
class WebExtractFontsParams < ContextDev::Internal::Type::BaseModel
|
|
@@ -16,6 +21,10 @@ module ContextDev
|
|
|
16
21
|
|
|
17
22
|
def domain=: (String) -> String
|
|
18
23
|
|
|
24
|
+
attr_reader max_age_ms: Integer?
|
|
25
|
+
|
|
26
|
+
def max_age_ms=: (Integer) -> Integer
|
|
27
|
+
|
|
19
28
|
attr_reader timeout_ms: Integer?
|
|
20
29
|
|
|
21
30
|
def timeout_ms=: (Integer) -> Integer
|
|
@@ -23,6 +32,7 @@ module ContextDev
|
|
|
23
32
|
def initialize: (
|
|
24
33
|
?direct_url: String,
|
|
25
34
|
?domain: String,
|
|
35
|
+
?max_age_ms: Integer,
|
|
26
36
|
?timeout_ms: Integer,
|
|
27
37
|
?request_options: ContextDev::request_opts
|
|
28
38
|
) -> void
|
|
@@ -30,6 +40,7 @@ module ContextDev
|
|
|
30
40
|
def to_hash: -> {
|
|
31
41
|
direct_url: String,
|
|
32
42
|
domain: String,
|
|
43
|
+
max_age_ms: Integer,
|
|
33
44
|
timeout_ms: Integer,
|
|
34
45
|
request_options: ContextDev::RequestOptions
|
|
35
46
|
}
|
|
@@ -0,0 +1,120 @@
|
|
|
1
|
+
module ContextDev
|
|
2
|
+
module Models
|
|
3
|
+
type web_extract_params =
|
|
4
|
+
{
|
|
5
|
+
schema: ::Hash[Symbol, top],
|
|
6
|
+
url: String,
|
|
7
|
+
fact_check: bool,
|
|
8
|
+
follow_subdomains: bool,
|
|
9
|
+
include_frames: bool,
|
|
10
|
+
instructions: String,
|
|
11
|
+
max_age_ms: Integer,
|
|
12
|
+
pdf: ContextDev::WebExtractParams::Pdf,
|
|
13
|
+
stop_after_ms: Integer,
|
|
14
|
+
timeout_ms: Integer,
|
|
15
|
+
wait_for_ms: Integer
|
|
16
|
+
}
|
|
17
|
+
& ContextDev::Internal::Type::request_parameters
|
|
18
|
+
|
|
19
|
+
class WebExtractParams < ContextDev::Internal::Type::BaseModel
|
|
20
|
+
extend ContextDev::Internal::Type::RequestParameters::Converter
|
|
21
|
+
include ContextDev::Internal::Type::RequestParameters
|
|
22
|
+
|
|
23
|
+
attr_accessor schema: ::Hash[Symbol, top]
|
|
24
|
+
|
|
25
|
+
attr_accessor url: String
|
|
26
|
+
|
|
27
|
+
attr_reader fact_check: bool?
|
|
28
|
+
|
|
29
|
+
def fact_check=: (bool) -> bool
|
|
30
|
+
|
|
31
|
+
attr_reader follow_subdomains: bool?
|
|
32
|
+
|
|
33
|
+
def follow_subdomains=: (bool) -> bool
|
|
34
|
+
|
|
35
|
+
attr_reader include_frames: bool?
|
|
36
|
+
|
|
37
|
+
def include_frames=: (bool) -> bool
|
|
38
|
+
|
|
39
|
+
attr_reader instructions: String?
|
|
40
|
+
|
|
41
|
+
def instructions=: (String) -> String
|
|
42
|
+
|
|
43
|
+
attr_reader max_age_ms: Integer?
|
|
44
|
+
|
|
45
|
+
def max_age_ms=: (Integer) -> Integer
|
|
46
|
+
|
|
47
|
+
attr_reader pdf: ContextDev::WebExtractParams::Pdf?
|
|
48
|
+
|
|
49
|
+
def pdf=: (
|
|
50
|
+
ContextDev::WebExtractParams::Pdf
|
|
51
|
+
) -> ContextDev::WebExtractParams::Pdf
|
|
52
|
+
|
|
53
|
+
attr_reader stop_after_ms: Integer?
|
|
54
|
+
|
|
55
|
+
def stop_after_ms=: (Integer) -> Integer
|
|
56
|
+
|
|
57
|
+
attr_reader timeout_ms: Integer?
|
|
58
|
+
|
|
59
|
+
def timeout_ms=: (Integer) -> Integer
|
|
60
|
+
|
|
61
|
+
attr_reader wait_for_ms: Integer?
|
|
62
|
+
|
|
63
|
+
def wait_for_ms=: (Integer) -> Integer
|
|
64
|
+
|
|
65
|
+
def initialize: (
|
|
66
|
+
schema: ::Hash[Symbol, top],
|
|
67
|
+
url: String,
|
|
68
|
+
?fact_check: bool,
|
|
69
|
+
?follow_subdomains: bool,
|
|
70
|
+
?include_frames: bool,
|
|
71
|
+
?instructions: String,
|
|
72
|
+
?max_age_ms: Integer,
|
|
73
|
+
?pdf: ContextDev::WebExtractParams::Pdf,
|
|
74
|
+
?stop_after_ms: Integer,
|
|
75
|
+
?timeout_ms: Integer,
|
|
76
|
+
?wait_for_ms: Integer,
|
|
77
|
+
?request_options: ContextDev::request_opts
|
|
78
|
+
) -> void
|
|
79
|
+
|
|
80
|
+
def to_hash: -> {
|
|
81
|
+
schema: ::Hash[Symbol, top],
|
|
82
|
+
url: String,
|
|
83
|
+
fact_check: bool,
|
|
84
|
+
follow_subdomains: bool,
|
|
85
|
+
include_frames: bool,
|
|
86
|
+
instructions: String,
|
|
87
|
+
max_age_ms: Integer,
|
|
88
|
+
pdf: ContextDev::WebExtractParams::Pdf,
|
|
89
|
+
stop_after_ms: Integer,
|
|
90
|
+
timeout_ms: Integer,
|
|
91
|
+
wait_for_ms: Integer,
|
|
92
|
+
request_options: ContextDev::RequestOptions
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
type pdf = { end_: Integer, should_parse: bool, start: Integer }
|
|
96
|
+
|
|
97
|
+
class Pdf < ContextDev::Internal::Type::BaseModel
|
|
98
|
+
attr_reader end_: Integer?
|
|
99
|
+
|
|
100
|
+
def end_=: (Integer) -> Integer
|
|
101
|
+
|
|
102
|
+
attr_reader should_parse: bool?
|
|
103
|
+
|
|
104
|
+
def should_parse=: (bool) -> bool
|
|
105
|
+
|
|
106
|
+
attr_reader start: Integer?
|
|
107
|
+
|
|
108
|
+
def start=: (Integer) -> Integer
|
|
109
|
+
|
|
110
|
+
def initialize: (
|
|
111
|
+
?end_: Integer,
|
|
112
|
+
?should_parse: bool,
|
|
113
|
+
?start: Integer
|
|
114
|
+
) -> void
|
|
115
|
+
|
|
116
|
+
def to_hash: -> { end_: Integer, should_parse: bool, start: Integer }
|
|
117
|
+
end
|
|
118
|
+
end
|
|
119
|
+
end
|
|
120
|
+
end
|