context.dev 2.8.0 → 2.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +28 -0
- data/README.md +1 -1
- data/lib/context_dev/client.rb +10 -0
- data/lib/context_dev/models/batch_delete_params.rb +22 -0
- data/lib/context_dev/models/batch_delete_response.rb +60 -0
- data/lib/context_dev/models/batch_get_results_response.rb +46 -4
- data/lib/context_dev/models/batch_list_response.rb +13 -4
- data/lib/context_dev/models/batch_retrieve_response.rb +13 -4
- data/lib/context_dev/models/batch_submit_params.rb +2080 -26
- data/lib/context_dev/models/batch_submit_response.rb +125 -531
- data/lib/context_dev/models/brand_retrieve_response.rb +50 -1
- data/lib/context_dev/models/brand_search_params.rb +41 -3
- data/lib/context_dev/models/brand_search_response.rb +3 -2
- data/lib/context_dev/models/crawl_controls.rb +21 -15
- data/lib/context_dev/models/news_search_params.rb +467 -0
- data/lib/context_dev/models/news_search_response.rb +238 -0
- data/lib/context_dev/models/parse_handle_params.rb +20 -147
- data/lib/context_dev/models/person_enrich_params.rb +176 -0
- data/lib/context_dev/models/person_enrich_response.rb +641 -0
- data/lib/context_dev/models/utility_prefetch_params.rb +19 -16
- data/lib/context_dev/models/utility_prefetch_response.rb +4 -5
- data/lib/context_dev/models/web_screenshot_params.rb +3 -30
- data/lib/context_dev/models/web_search_response.rb +1 -0
- data/lib/context_dev/models/web_web_crawl_md_params.rb +5 -4
- data/lib/context_dev/models/web_web_crawl_md_response.rb +30 -1
- data/lib/context_dev/models/web_web_scrape_html_params.rb +20 -156
- data/lib/context_dev/models/web_web_scrape_html_response.rb +30 -1
- data/lib/context_dev/models/web_web_scrape_images_params.rb +12 -124
- data/lib/context_dev/models/web_web_scrape_md_params.rb +40 -241
- data/lib/context_dev/models/web_web_scrape_md_response.rb +41 -2
- data/lib/context_dev/models/web_web_scrape_sitemap_params.rb +11 -1
- data/lib/context_dev/models/web_web_scrape_sitemap_response.rb +3 -2
- data/lib/context_dev/models.rb +6 -0
- data/lib/context_dev/resources/batch.rb +33 -7
- data/lib/context_dev/resources/brand.rb +10 -10
- data/lib/context_dev/resources/news.rb +51 -0
- data/lib/context_dev/resources/parse.rb +5 -5
- data/lib/context_dev/resources/people.rb +56 -0
- data/lib/context_dev/resources/utility.rb +7 -6
- data/lib/context_dev/resources/web.rb +46 -24
- data/lib/context_dev/version.rb +1 -1
- data/lib/context_dev.rb +8 -0
- data/rbi/context_dev/client.rbi +8 -0
- data/rbi/context_dev/models/batch_delete_params.rbi +40 -0
- data/rbi/context_dev/models/batch_delete_response.rbi +116 -0
- data/rbi/context_dev/models/batch_get_results_response.rbi +85 -3
- data/rbi/context_dev/models/batch_list_response.rbi +24 -8
- data/rbi/context_dev/models/batch_retrieve_response.rbi +24 -8
- data/rbi/context_dev/models/batch_submit_params.rbi +6421 -44
- data/rbi/context_dev/models/batch_submit_response.rbi +184 -1151
- data/rbi/context_dev/models/brand_retrieve_response.rbi +152 -0
- data/rbi/context_dev/models/brand_search_params.rbi +71 -2
- data/rbi/context_dev/models/brand_search_response.rbi +4 -2
- data/rbi/context_dev/models/crawl_controls.rbi +22 -28
- data/rbi/context_dev/models/news_search_params.rbi +1294 -0
- data/rbi/context_dev/models/news_search_response.rbi +423 -0
- data/rbi/context_dev/models/parse_handle_params.rbi +30 -323
- data/rbi/context_dev/models/person_enrich_params.rbi +332 -0
- data/rbi/context_dev/models/person_enrich_response.rbi +1295 -0
- data/rbi/context_dev/models/utility_prefetch_params.rbi +24 -16
- data/rbi/context_dev/models/utility_prefetch_response.rbi +8 -6
- data/rbi/context_dev/models/web_screenshot_params.rbi +4 -71
- data/rbi/context_dev/models/web_search_response.rbi +5 -0
- data/rbi/context_dev/models/web_web_crawl_md_params.rbi +8 -6
- data/rbi/context_dev/models/web_web_crawl_md_response.rbi +67 -0
- data/rbi/context_dev/models/web_web_scrape_html_params.rbi +30 -363
- data/rbi/context_dev/models/web_web_scrape_html_response.rbi +65 -0
- data/rbi/context_dev/models/web_web_scrape_images_params.rbi +16 -287
- data/rbi/context_dev/models/web_web_scrape_md_params.rbi +57 -558
- data/rbi/context_dev/models/web_web_scrape_md_response.rbi +80 -0
- data/rbi/context_dev/models/web_web_scrape_sitemap_params.rbi +15 -0
- data/rbi/context_dev/models/web_web_scrape_sitemap_response.rbi +4 -2
- data/rbi/context_dev/models.rbi +6 -0
- data/rbi/context_dev/resources/batch.rbi +32 -10
- data/rbi/context_dev/resources/brand.rbi +14 -8
- data/rbi/context_dev/resources/news.rbi +46 -0
- data/rbi/context_dev/resources/parse.rbi +10 -26
- data/rbi/context_dev/resources/people.rbi +47 -0
- data/rbi/context_dev/resources/utility.rbi +8 -6
- data/rbi/context_dev/resources/web.rbi +49 -66
- data/sig/context_dev/client.rbs +4 -0
- data/sig/context_dev/models/batch_delete_params.rbs +23 -0
- data/sig/context_dev/models/batch_delete_response.rbs +57 -0
- data/sig/context_dev/models/batch_get_results_response.rbs +30 -2
- data/sig/context_dev/models/batch_list_response.rbs +16 -2
- data/sig/context_dev/models/batch_retrieve_response.rbs +16 -2
- data/sig/context_dev/models/batch_submit_params.rbs +2666 -15
- data/sig/context_dev/models/batch_submit_response.rbs +78 -466
- data/sig/context_dev/models/brand_retrieve_response.rbs +62 -0
- data/sig/context_dev/models/brand_search_params.rbs +38 -1
- data/sig/context_dev/models/crawl_controls.rbs +16 -16
- data/sig/context_dev/models/news_search_params.rbs +532 -0
- data/sig/context_dev/models/news_search_response.rbs +206 -0
- data/sig/context_dev/models/parse_handle_params.rbs +25 -90
- data/sig/context_dev/models/person_enrich_params.rbs +199 -0
- data/sig/context_dev/models/person_enrich_response.rbs +638 -0
- data/sig/context_dev/models/utility_prefetch_params.rbs +2 -1
- data/sig/context_dev/models/utility_prefetch_response.rbs +2 -1
- data/sig/context_dev/models/web_screenshot_params.rbs +5 -18
- data/sig/context_dev/models/web_search_response.rbs +2 -0
- data/sig/context_dev/models/web_web_crawl_md_response.rbs +21 -0
- data/sig/context_dev/models/web_web_scrape_html_params.rbs +24 -94
- data/sig/context_dev/models/web_web_scrape_html_response.rbs +21 -0
- data/sig/context_dev/models/web_web_scrape_images_params.rbs +20 -72
- data/sig/context_dev/models/web_web_scrape_md_params.rbs +46 -148
- data/sig/context_dev/models/web_web_scrape_md_response.rbs +28 -0
- data/sig/context_dev/models/web_web_scrape_sitemap_params.rbs +7 -0
- data/sig/context_dev/models.rbs +6 -0
- data/sig/context_dev/resources/batch.rbs +8 -2
- data/sig/context_dev/resources/brand.rbs +3 -0
- data/sig/context_dev/resources/news.rbs +17 -0
- data/sig/context_dev/resources/parse.rbs +5 -5
- data/sig/context_dev/resources/people.rbs +19 -0
- data/sig/context_dev/resources/web.rbs +13 -11
- metadata +26 -2
|
@@ -7,49 +7,2103 @@ module ContextDev
|
|
|
7
7
|
extend ContextDev::Internal::Type::RequestParameters::Converter
|
|
8
8
|
include ContextDev::Internal::Type::RequestParameters
|
|
9
9
|
|
|
10
|
-
# @!attribute
|
|
11
|
-
#
|
|
10
|
+
# @!attribute input
|
|
11
|
+
# Choose a URL list or a site crawl.
|
|
12
12
|
#
|
|
13
|
-
# @return [ContextDev::Models::BatchSubmitParams::
|
|
14
|
-
required :
|
|
13
|
+
# @return [ContextDev::Models::BatchSubmitParams::Input::Scrape, ContextDev::Models::BatchSubmitParams::Input::Crawl]
|
|
14
|
+
required :input, union: -> { ContextDev::BatchSubmitParams::Input }
|
|
15
15
|
|
|
16
16
|
# @!attribute tags
|
|
17
|
-
#
|
|
17
|
+
# Tags stored on the batch. Filter the batch list by them later.
|
|
18
18
|
#
|
|
19
19
|
# @return [Array<String>, nil]
|
|
20
20
|
optional :tags, ContextDev::Internal::Type::ArrayOf[String]
|
|
21
21
|
|
|
22
|
-
# @!attribute
|
|
23
|
-
#
|
|
24
|
-
# than this value, it will be aborted with a 408 status code. Maximum allowed
|
|
25
|
-
# value is 300000ms (5 minutes).
|
|
22
|
+
# @!attribute webhook_url
|
|
23
|
+
# URL notified when the batch finishes.
|
|
26
24
|
#
|
|
27
|
-
# @return [
|
|
28
|
-
optional :
|
|
25
|
+
# @return [String, nil]
|
|
26
|
+
optional :webhook_url, String, api_name: :webhookUrl
|
|
29
27
|
|
|
30
|
-
# @!
|
|
28
|
+
# @!attribute idempotency_key
|
|
29
|
+
# Any string unique to this submission. Retries with the same key return the
|
|
30
|
+
# original batch.
|
|
31
|
+
#
|
|
32
|
+
# @return [String, nil]
|
|
33
|
+
optional :idempotency_key, String
|
|
34
|
+
|
|
35
|
+
# @!method initialize(input:, tags: nil, webhook_url: nil, idempotency_key: nil, request_options: {})
|
|
31
36
|
# Some parameter documentations has been truncated, see
|
|
32
37
|
# {ContextDev::Models::BatchSubmitParams} for more details.
|
|
33
38
|
#
|
|
34
|
-
# @param
|
|
39
|
+
# @param input [ContextDev::Models::BatchSubmitParams::Input::Scrape, ContextDev::Models::BatchSubmitParams::Input::Crawl] Choose a URL list or a site crawl.
|
|
40
|
+
#
|
|
41
|
+
# @param tags [Array<String>] Tags stored on the batch. Filter the batch list by them later.
|
|
35
42
|
#
|
|
36
|
-
# @param
|
|
43
|
+
# @param webhook_url [String] URL notified when the batch finishes.
|
|
37
44
|
#
|
|
38
|
-
# @param
|
|
45
|
+
# @param idempotency_key [String] Any string unique to this submission. Retries with the same key return the origi
|
|
39
46
|
#
|
|
40
47
|
# @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}]
|
|
41
48
|
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
#
|
|
52
|
-
|
|
49
|
+
# Choose a URL list or a site crawl.
|
|
50
|
+
module Input
|
|
51
|
+
extend ContextDev::Internal::Type::Union
|
|
52
|
+
|
|
53
|
+
discriminator :mode
|
|
54
|
+
|
|
55
|
+
# Scrape up to 25K URLs in one batch.
|
|
56
|
+
variant :scrape, -> { ContextDev::BatchSubmitParams::Input::Scrape }
|
|
57
|
+
|
|
58
|
+
# Crawl pages starting from a URL or from a domain's sitemap.
|
|
59
|
+
variant :crawl, -> { ContextDev::BatchSubmitParams::Input::Crawl }
|
|
60
|
+
|
|
61
|
+
class Scrape < ContextDev::Internal::Type::BaseModel
|
|
62
|
+
# @!attribute data
|
|
63
|
+
# Pages to scrape and their output format.
|
|
64
|
+
#
|
|
65
|
+
# @return [ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown, ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML]
|
|
66
|
+
required :data, union: -> { ContextDev::BatchSubmitParams::Input::Scrape::Data }
|
|
67
|
+
|
|
68
|
+
# @!attribute mode
|
|
69
|
+
# Scrape the pages in `data.urls`.
|
|
70
|
+
#
|
|
71
|
+
# @return [Symbol, :scrape]
|
|
72
|
+
required :mode, const: :scrape
|
|
73
|
+
|
|
74
|
+
# @!method initialize(data:, mode: :scrape)
|
|
75
|
+
# Scrape up to 25K URLs in one batch.
|
|
76
|
+
#
|
|
77
|
+
# @param data [ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown, ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML] Pages to scrape and their output format.
|
|
78
|
+
#
|
|
79
|
+
# @param mode [Symbol, :scrape] Scrape the pages in `data.urls`.
|
|
80
|
+
|
|
81
|
+
# Pages to scrape and their output format.
|
|
82
|
+
#
|
|
83
|
+
# @see ContextDev::Models::BatchSubmitParams::Input::Scrape#data
|
|
84
|
+
module Data
|
|
85
|
+
extend ContextDev::Internal::Type::Union
|
|
86
|
+
|
|
87
|
+
discriminator :format
|
|
88
|
+
|
|
89
|
+
# Scrape the listed pages as Markdown.
|
|
90
|
+
variant :markdown, -> { ContextDev::BatchSubmitParams::Input::Scrape::Data::Markdown }
|
|
91
|
+
|
|
92
|
+
# Scrape the listed pages as HTML.
|
|
93
|
+
variant :html, -> { ContextDev::BatchSubmitParams::Input::Scrape::Data::HTML }
|
|
94
|
+
|
|
95
|
+
class Markdown < ContextDev::Internal::Type::BaseModel
|
|
96
|
+
# @!attribute format_
|
|
97
|
+
# Return page content as Markdown.
|
|
98
|
+
#
|
|
99
|
+
# @return [Symbol, :markdown]
|
|
100
|
+
required :format_, const: :markdown, api_name: :format
|
|
101
|
+
|
|
102
|
+
# @!attribute urls
|
|
103
|
+
# Pages to scrape. Maximum 25000.
|
|
104
|
+
#
|
|
105
|
+
# @return [Array<ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::URL>]
|
|
106
|
+
required :urls,
|
|
107
|
+
-> { ContextDev::Internal::Type::ArrayOf[ContextDev::BatchSubmitParams::Input::Scrape::Data::Markdown::URL] }
|
|
108
|
+
|
|
109
|
+
# @!attribute options
|
|
110
|
+
# Options for Markdown output.
|
|
111
|
+
#
|
|
112
|
+
# @return [ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::Options, nil]
|
|
113
|
+
optional :options, -> { ContextDev::BatchSubmitParams::Input::Scrape::Data::Markdown::Options }
|
|
114
|
+
|
|
115
|
+
# @!method initialize(urls:, options: nil, format_: :markdown)
|
|
116
|
+
# Scrape the listed pages as Markdown.
|
|
117
|
+
#
|
|
118
|
+
# @param urls [Array<ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::URL>] Pages to scrape. Maximum 25000.
|
|
119
|
+
#
|
|
120
|
+
# @param options [ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::Options] Options for Markdown output.
|
|
121
|
+
#
|
|
122
|
+
# @param format_ [Symbol, :markdown] Return page content as Markdown.
|
|
123
|
+
|
|
124
|
+
class URL < ContextDev::Internal::Type::BaseModel
|
|
125
|
+
# @!attribute url
|
|
126
|
+
# Page URL to scrape.
|
|
127
|
+
#
|
|
128
|
+
# @return [String]
|
|
129
|
+
required :url, String
|
|
130
|
+
|
|
131
|
+
# @!attribute item_id
|
|
132
|
+
# Your ID for this page, returned with its result. The same URL can use different
|
|
133
|
+
# IDs.
|
|
134
|
+
#
|
|
135
|
+
# @return [String, nil]
|
|
136
|
+
optional :item_id, String, api_name: :itemId
|
|
137
|
+
|
|
138
|
+
# @!attribute meta
|
|
139
|
+
# Custom JSON returned unchanged with this page result.
|
|
140
|
+
#
|
|
141
|
+
# @return [Hash{Symbol=>Object}, nil]
|
|
142
|
+
optional :meta, ContextDev::Internal::Type::HashOf[ContextDev::Internal::Type::Unknown]
|
|
143
|
+
|
|
144
|
+
# @!method initialize(url:, item_id: nil, meta: nil)
|
|
145
|
+
# Some parameter documentations has been truncated, see
|
|
146
|
+
# {ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::URL} for
|
|
147
|
+
# more details.
|
|
148
|
+
#
|
|
149
|
+
# A page to scrape, with optional data for matching results.
|
|
150
|
+
#
|
|
151
|
+
# @param url [String] Page URL to scrape.
|
|
152
|
+
#
|
|
153
|
+
# @param item_id [String] Your ID for this page, returned with its result. The same URL can use different
|
|
154
|
+
#
|
|
155
|
+
# @param meta [Hash{Symbol=>Object}] Custom JSON returned unchanged with this page result.
|
|
156
|
+
end
|
|
157
|
+
|
|
158
|
+
# @see ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown#options
|
|
159
|
+
class Options < ContextDev::Internal::Type::BaseModel
|
|
160
|
+
# @!attribute country
|
|
161
|
+
# Fetch the target page through a residential proxy in this country (ISO 3166-1
|
|
162
|
+
# alpha-2).
|
|
163
|
+
#
|
|
164
|
+
# @return [Symbol, ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Country, nil]
|
|
165
|
+
optional :country,
|
|
166
|
+
enum: -> { ContextDev::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Country }
|
|
167
|
+
|
|
168
|
+
# @!attribute exclude_selectors
|
|
169
|
+
# Remove elements matching these CSS selectors. Applied after `includeSelectors`,
|
|
170
|
+
# so an element matching both is removed.
|
|
171
|
+
#
|
|
172
|
+
# @return [Array<String>, nil]
|
|
173
|
+
optional :exclude_selectors,
|
|
174
|
+
ContextDev::Internal::Type::ArrayOf[String],
|
|
175
|
+
api_name: :excludeSelectors,
|
|
176
|
+
nil?: true
|
|
177
|
+
|
|
178
|
+
# @!attribute include_html
|
|
179
|
+
# Also include each page's HTML in its result record, as an `html` field alongside
|
|
180
|
+
# the Markdown.
|
|
181
|
+
#
|
|
182
|
+
# @return [Boolean, nil]
|
|
183
|
+
optional :include_html, ContextDev::Internal::Type::Boolean, api_name: :includeHTML
|
|
184
|
+
|
|
185
|
+
# @!attribute include_images
|
|
186
|
+
# Include image references in the Markdown.
|
|
187
|
+
#
|
|
188
|
+
# @return [Boolean, nil]
|
|
189
|
+
optional :include_images, ContextDev::Internal::Type::Boolean, api_name: :includeImages
|
|
190
|
+
|
|
191
|
+
# @!attribute include_links
|
|
192
|
+
# Include links in the Markdown.
|
|
193
|
+
#
|
|
194
|
+
# @return [Boolean, nil]
|
|
195
|
+
optional :include_links, ContextDev::Internal::Type::Boolean, api_name: :includeLinks
|
|
196
|
+
|
|
197
|
+
# @!attribute include_selectors
|
|
198
|
+
# Keep only the subtrees matching these CSS selectors. Filtered pages are always
|
|
199
|
+
# fetched fresh, ignoring `maxAgeMs`.
|
|
200
|
+
#
|
|
201
|
+
# @return [Array<String>, nil]
|
|
202
|
+
optional :include_selectors,
|
|
203
|
+
ContextDev::Internal::Type::ArrayOf[String],
|
|
204
|
+
api_name: :includeSelectors,
|
|
205
|
+
nil?: true
|
|
206
|
+
|
|
207
|
+
# @!attribute max_age_ms
|
|
208
|
+
# Return a cached result if a prior scrape for the same parameters exists and is
|
|
209
|
+
# younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
|
|
210
|
+
# omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
|
|
211
|
+
#
|
|
212
|
+
# @return [Integer, nil]
|
|
213
|
+
optional :max_age_ms, Integer, api_name: :maxAgeMs, nil?: true
|
|
214
|
+
|
|
215
|
+
# @!attribute pdf
|
|
216
|
+
# PDF parsing controls. Use start/end to limit text extraction and embedded-image
|
|
217
|
+
# detection/OCR to an inclusive 1-based page range.
|
|
218
|
+
#
|
|
219
|
+
# @return [ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf, nil]
|
|
220
|
+
optional :pdf, -> { ContextDev::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf }
|
|
221
|
+
|
|
222
|
+
# @!attribute settle_animations
|
|
223
|
+
# Wait briefly for CSS and transition animations to settle before extraction, on
|
|
224
|
+
# pages that render in a browser.
|
|
225
|
+
#
|
|
226
|
+
# @return [Boolean, nil]
|
|
227
|
+
optional :settle_animations, ContextDev::Internal::Type::Boolean, api_name: :settleAnimations
|
|
228
|
+
|
|
229
|
+
# @!attribute shorten_base64_images
|
|
230
|
+
# Shorten inline base64 image data.
|
|
231
|
+
#
|
|
232
|
+
# @return [Boolean, nil]
|
|
233
|
+
optional :shorten_base64_images,
|
|
234
|
+
ContextDev::Internal::Type::Boolean,
|
|
235
|
+
api_name: :shortenBase64Images
|
|
236
|
+
|
|
237
|
+
# @!attribute use_main_content_only
|
|
238
|
+
# Return the main content without navigation or footers.
|
|
239
|
+
#
|
|
240
|
+
# @return [Boolean, nil]
|
|
241
|
+
optional :use_main_content_only,
|
|
242
|
+
ContextDev::Internal::Type::Boolean,
|
|
243
|
+
api_name: :useMainContentOnly
|
|
244
|
+
|
|
245
|
+
# @!attribute wait_for_ms
|
|
246
|
+
# How long to wait after initial page load, in milliseconds. `0` waits 500 ms.
|
|
247
|
+
#
|
|
248
|
+
# @return [Integer, nil]
|
|
249
|
+
optional :wait_for_ms, Integer, api_name: :waitForMs
|
|
250
|
+
|
|
251
|
+
# @!method initialize(country: nil, exclude_selectors: nil, include_html: nil, include_images: nil, include_links: nil, include_selectors: nil, max_age_ms: nil, pdf: nil, settle_animations: nil, shorten_base64_images: nil, use_main_content_only: nil, wait_for_ms: nil)
|
|
252
|
+
# Some parameter documentations has been truncated, see
|
|
253
|
+
# {ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::Options}
|
|
254
|
+
# for more details.
|
|
255
|
+
#
|
|
256
|
+
# Options for Markdown output.
|
|
257
|
+
#
|
|
258
|
+
# @param country [Symbol, ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Country] Fetch the target page through a residential proxy in this country (ISO 3166-1 al
|
|
259
|
+
#
|
|
260
|
+
# @param exclude_selectors [Array<String>, nil] Remove elements matching these CSS selectors. Applied after `includeSelectors`,
|
|
261
|
+
#
|
|
262
|
+
# @param include_html [Boolean] Also include each page's HTML in its result record, as an `html` field alongside
|
|
263
|
+
#
|
|
264
|
+
# @param include_images [Boolean] Include image references in the Markdown.
|
|
265
|
+
#
|
|
266
|
+
# @param include_links [Boolean] Include links in the Markdown.
|
|
267
|
+
#
|
|
268
|
+
# @param include_selectors [Array<String>, nil] Keep only the subtrees matching these CSS selectors. Filtered pages are always f
|
|
269
|
+
#
|
|
270
|
+
# @param max_age_ms [Integer, nil] Return a cached result if a prior scrape for the same parameters exists and is y
|
|
271
|
+
#
|
|
272
|
+
# @param pdf [ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf] PDF parsing controls. Use start/end to limit text extraction and embedded-image
|
|
273
|
+
#
|
|
274
|
+
# @param settle_animations [Boolean] Wait briefly for CSS and transition animations to settle before extraction, on p
|
|
275
|
+
#
|
|
276
|
+
# @param shorten_base64_images [Boolean] Shorten inline base64 image data.
|
|
277
|
+
#
|
|
278
|
+
# @param use_main_content_only [Boolean] Return the main content without navigation or footers.
|
|
279
|
+
#
|
|
280
|
+
# @param wait_for_ms [Integer] How long to wait after initial page load, in milliseconds. `0` waits 500 ms.
|
|
281
|
+
|
|
282
|
+
# Fetch the target page through a residential proxy in this country (ISO 3166-1
|
|
283
|
+
# alpha-2).
|
|
284
|
+
#
|
|
285
|
+
# @see ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::Options#country
|
|
286
|
+
module Country
|
|
287
|
+
extend ContextDev::Internal::Type::Enum
|
|
288
|
+
|
|
289
|
+
AD = :ad
|
|
290
|
+
AE = :ae
|
|
291
|
+
AF = :af
|
|
292
|
+
AG = :ag
|
|
293
|
+
AI = :ai
|
|
294
|
+
AL = :al
|
|
295
|
+
AM = :am
|
|
296
|
+
AO = :ao
|
|
297
|
+
AR = :ar
|
|
298
|
+
AT = :at
|
|
299
|
+
AU = :au
|
|
300
|
+
AW = :aw
|
|
301
|
+
AZ = :az
|
|
302
|
+
BA = :ba
|
|
303
|
+
BB = :bb
|
|
304
|
+
BD = :bd
|
|
305
|
+
BE = :be
|
|
306
|
+
BF = :bf
|
|
307
|
+
BG = :bg
|
|
308
|
+
BH = :bh
|
|
309
|
+
BI = :bi
|
|
310
|
+
BJ = :bj
|
|
311
|
+
BM = :bm
|
|
312
|
+
BN = :bn
|
|
313
|
+
BO = :bo
|
|
314
|
+
BQ = :bq
|
|
315
|
+
BR = :br
|
|
316
|
+
BS = :bs
|
|
317
|
+
BW = :bw
|
|
318
|
+
BY = :by
|
|
319
|
+
BZ = :bz
|
|
320
|
+
CA = :ca
|
|
321
|
+
CD = :cd
|
|
322
|
+
CF = :cf
|
|
323
|
+
CG = :cg
|
|
324
|
+
CH = :ch
|
|
325
|
+
CI = :ci
|
|
326
|
+
CL = :cl
|
|
327
|
+
CM = :cm
|
|
328
|
+
CN = :cn
|
|
329
|
+
CO = :co
|
|
330
|
+
CR = :cr
|
|
331
|
+
CV = :cv
|
|
332
|
+
CW = :cw
|
|
333
|
+
CY = :cy
|
|
334
|
+
CZ = :cz
|
|
335
|
+
DE = :de
|
|
336
|
+
DJ = :dj
|
|
337
|
+
DK = :dk
|
|
338
|
+
DM = :dm
|
|
339
|
+
DO = :do
|
|
340
|
+
DZ = :dz
|
|
341
|
+
EC = :ec
|
|
342
|
+
EE = :ee
|
|
343
|
+
EG = :eg
|
|
344
|
+
ES = :es
|
|
345
|
+
ET = :et
|
|
346
|
+
FI = :fi
|
|
347
|
+
FJ = :fj
|
|
348
|
+
FR = :fr
|
|
349
|
+
GA = :ga
|
|
350
|
+
GB = :gb
|
|
351
|
+
GD = :gd
|
|
352
|
+
GE = :ge
|
|
353
|
+
GF = :gf
|
|
354
|
+
GG = :gg
|
|
355
|
+
GH = :gh
|
|
356
|
+
GM = :gm
|
|
357
|
+
GN = :gn
|
|
358
|
+
GP = :gp
|
|
359
|
+
GQ = :gq
|
|
360
|
+
GR = :gr
|
|
361
|
+
GT = :gt
|
|
362
|
+
GU = :gu
|
|
363
|
+
GW = :gw
|
|
364
|
+
GY = :gy
|
|
365
|
+
HK = :hk
|
|
366
|
+
HN = :hn
|
|
367
|
+
HR = :hr
|
|
368
|
+
HT = :ht
|
|
369
|
+
HU = :hu
|
|
370
|
+
ID = :id
|
|
371
|
+
IE = :ie
|
|
372
|
+
IL = :il
|
|
373
|
+
IM = :im
|
|
374
|
+
IN = :in
|
|
375
|
+
IQ = :iq
|
|
376
|
+
IR = :ir
|
|
377
|
+
IS = :is
|
|
378
|
+
IT = :it
|
|
379
|
+
JE = :je
|
|
380
|
+
JM = :jm
|
|
381
|
+
JO = :jo
|
|
382
|
+
JP = :jp
|
|
383
|
+
KE = :ke
|
|
384
|
+
KG = :kg
|
|
385
|
+
KH = :kh
|
|
386
|
+
KN = :kn
|
|
387
|
+
KR = :kr
|
|
388
|
+
KW = :kw
|
|
389
|
+
KY = :ky
|
|
390
|
+
KZ = :kz
|
|
391
|
+
LA = :la
|
|
392
|
+
LB = :lb
|
|
393
|
+
LC = :lc
|
|
394
|
+
LK = :lk
|
|
395
|
+
LR = :lr
|
|
396
|
+
LS = :ls
|
|
397
|
+
LT = :lt
|
|
398
|
+
LU = :lu
|
|
399
|
+
LV = :lv
|
|
400
|
+
LY = :ly
|
|
401
|
+
MA = :ma
|
|
402
|
+
MC = :mc
|
|
403
|
+
MD = :md
|
|
404
|
+
ME = :me
|
|
405
|
+
MF = :mf
|
|
406
|
+
MG = :mg
|
|
407
|
+
MK = :mk
|
|
408
|
+
ML = :ml
|
|
409
|
+
MM = :mm
|
|
410
|
+
MN = :mn
|
|
411
|
+
MO = :mo
|
|
412
|
+
MQ = :mq
|
|
413
|
+
MR = :mr
|
|
414
|
+
MT = :mt
|
|
415
|
+
MU = :mu
|
|
416
|
+
MV = :mv
|
|
417
|
+
MW = :mw
|
|
418
|
+
MX = :mx
|
|
419
|
+
MY = :my
|
|
420
|
+
MZ = :mz
|
|
421
|
+
NA = :na
|
|
422
|
+
NC = :nc
|
|
423
|
+
NE = :ne
|
|
424
|
+
NG = :ng
|
|
425
|
+
NI = :ni
|
|
426
|
+
NL = :nl
|
|
427
|
+
NO = :no
|
|
428
|
+
NP = :np
|
|
429
|
+
NZ = :nz
|
|
430
|
+
OM = :om
|
|
431
|
+
PA = :pa
|
|
432
|
+
PE = :pe
|
|
433
|
+
PF = :pf
|
|
434
|
+
PG = :pg
|
|
435
|
+
PH = :ph
|
|
436
|
+
PK = :pk
|
|
437
|
+
PL = :pl
|
|
438
|
+
PR = :pr
|
|
439
|
+
PS = :ps
|
|
440
|
+
PT = :pt
|
|
441
|
+
PY = :py
|
|
442
|
+
QA = :qa
|
|
443
|
+
RE = :re
|
|
444
|
+
RO = :ro
|
|
445
|
+
RS = :rs
|
|
446
|
+
RU = :ru
|
|
447
|
+
RW = :rw
|
|
448
|
+
SA = :sa
|
|
449
|
+
SC = :sc
|
|
450
|
+
SD = :sd
|
|
451
|
+
SE = :se
|
|
452
|
+
SG = :sg
|
|
453
|
+
SI = :si
|
|
454
|
+
SK = :sk
|
|
455
|
+
SL = :sl
|
|
456
|
+
SM = :sm
|
|
457
|
+
SN = :sn
|
|
458
|
+
SO = :so
|
|
459
|
+
SR = :sr
|
|
460
|
+
SS = :ss
|
|
461
|
+
ST = :st
|
|
462
|
+
SV = :sv
|
|
463
|
+
SX = :sx
|
|
464
|
+
SY = :sy
|
|
465
|
+
SZ = :sz
|
|
466
|
+
TC = :tc
|
|
467
|
+
TD = :td
|
|
468
|
+
TG = :tg
|
|
469
|
+
TH = :th
|
|
470
|
+
TJ = :tj
|
|
471
|
+
TL = :tl
|
|
472
|
+
TM = :tm
|
|
473
|
+
TN = :tn
|
|
474
|
+
TR = :tr
|
|
475
|
+
TT = :tt
|
|
476
|
+
TW = :tw
|
|
477
|
+
TZ = :tz
|
|
478
|
+
UA = :ua
|
|
479
|
+
UG = :ug
|
|
480
|
+
US = :us
|
|
481
|
+
UY = :uy
|
|
482
|
+
UZ = :uz
|
|
483
|
+
VC = :vc
|
|
484
|
+
VE = :ve
|
|
485
|
+
VG = :vg
|
|
486
|
+
VI = :vi
|
|
487
|
+
VN = :vn
|
|
488
|
+
YE = :ye
|
|
489
|
+
YT = :yt
|
|
490
|
+
ZA = :za
|
|
491
|
+
ZM = :zm
|
|
492
|
+
ZW = :zw
|
|
493
|
+
|
|
494
|
+
# @!method self.values
|
|
495
|
+
# @return [Array<Symbol>]
|
|
496
|
+
end
|
|
497
|
+
|
|
498
|
+
# @see ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::Options#pdf
|
|
499
|
+
class Pdf < ContextDev::Internal::Type::BaseModel
|
|
500
|
+
# @!attribute end_
|
|
501
|
+
# Last 1-based PDF page to parse. When omitted, parsing ends at the last page.
|
|
502
|
+
# Must be greater than or equal to start when both are provided.
|
|
503
|
+
#
|
|
504
|
+
# @return [Integer, nil]
|
|
505
|
+
optional :end_, Integer, api_name: :end
|
|
506
|
+
|
|
507
|
+
# @!attribute ocr
|
|
508
|
+
# When true, OCR the selected PDF pages that have no usable text layer (scans),
|
|
509
|
+
# replacing each recovered page's text with the OCR result while pages with a real
|
|
510
|
+
# text layer keep it. Billed at 1 credit per page OCR actually recovered, on top
|
|
511
|
+
# of the base request cost. When false, no OCR runs.
|
|
512
|
+
#
|
|
513
|
+
# @return [Boolean, nil]
|
|
514
|
+
optional :ocr, ContextDev::Internal::Type::Boolean
|
|
515
|
+
|
|
516
|
+
# @!attribute should_parse
|
|
517
|
+
# When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
|
|
518
|
+
# a 400 PDF_SKIPPED is returned.
|
|
519
|
+
#
|
|
520
|
+
# @return [Boolean, nil]
|
|
521
|
+
optional :should_parse, ContextDev::Internal::Type::Boolean, api_name: :shouldParse
|
|
522
|
+
|
|
523
|
+
# @!attribute start
|
|
524
|
+
# First 1-based PDF page to parse. When omitted, parsing starts at the first page.
|
|
525
|
+
#
|
|
526
|
+
# @return [Integer, nil]
|
|
527
|
+
optional :start, Integer
|
|
528
|
+
|
|
529
|
+
# @!method initialize(end_: nil, ocr: nil, should_parse: nil, start: nil)
|
|
530
|
+
# Some parameter documentations has been truncated, see
|
|
531
|
+
# {ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf}
|
|
532
|
+
# for more details.
|
|
533
|
+
#
|
|
534
|
+
# PDF parsing controls. Use start/end to limit text extraction and embedded-image
|
|
535
|
+
# detection/OCR to an inclusive 1-based page range.
|
|
536
|
+
#
|
|
537
|
+
# @param end_ [Integer] Last 1-based PDF page to parse. When omitted, parsing ends at the last page. Mus
|
|
538
|
+
#
|
|
539
|
+
# @param ocr [Boolean] When true, OCR the selected PDF pages that have no usable text layer (scans), re
|
|
540
|
+
#
|
|
541
|
+
# @param should_parse [Boolean] When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
|
|
542
|
+
#
|
|
543
|
+
# @param start [Integer] First 1-based PDF page to parse. When omitted, parsing starts at the first page.
|
|
544
|
+
end
|
|
545
|
+
end
|
|
546
|
+
end
|
|
547
|
+
|
|
548
|
+
class HTML < ContextDev::Internal::Type::BaseModel
|
|
549
|
+
# @!attribute format_
|
|
550
|
+
# Return page content as HTML.
|
|
551
|
+
#
|
|
552
|
+
# @return [Symbol, :html]
|
|
553
|
+
required :format_, const: :html, api_name: :format
|
|
554
|
+
|
|
555
|
+
# @!attribute urls
|
|
556
|
+
# Pages to scrape. Maximum 25000.
|
|
557
|
+
#
|
|
558
|
+
# @return [Array<ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::URL>]
|
|
559
|
+
required :urls,
|
|
560
|
+
-> { ContextDev::Internal::Type::ArrayOf[ContextDev::BatchSubmitParams::Input::Scrape::Data::HTML::URL] }
|
|
561
|
+
|
|
562
|
+
# @!attribute options
|
|
563
|
+
# Options for HTML output.
|
|
564
|
+
#
|
|
565
|
+
# @return [ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::Options, nil]
|
|
566
|
+
optional :options, -> { ContextDev::BatchSubmitParams::Input::Scrape::Data::HTML::Options }
|
|
567
|
+
|
|
568
|
+
# @!method initialize(urls:, options: nil, format_: :html)
|
|
569
|
+
# Scrape the listed pages as HTML.
|
|
570
|
+
#
|
|
571
|
+
# @param urls [Array<ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::URL>] Pages to scrape. Maximum 25000.
|
|
572
|
+
#
|
|
573
|
+
# @param options [ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::Options] Options for HTML output.
|
|
574
|
+
#
|
|
575
|
+
# @param format_ [Symbol, :html] Return page content as HTML.
|
|
576
|
+
|
|
577
|
+
class URL < ContextDev::Internal::Type::BaseModel
|
|
578
|
+
# @!attribute url
|
|
579
|
+
# Page URL to scrape.
|
|
580
|
+
#
|
|
581
|
+
# @return [String]
|
|
582
|
+
required :url, String
|
|
583
|
+
|
|
584
|
+
# @!attribute item_id
|
|
585
|
+
# Your ID for this page, returned with its result. The same URL can use different
|
|
586
|
+
# IDs.
|
|
587
|
+
#
|
|
588
|
+
# @return [String, nil]
|
|
589
|
+
optional :item_id, String, api_name: :itemId
|
|
590
|
+
|
|
591
|
+
# @!attribute meta
|
|
592
|
+
# Custom JSON returned unchanged with this page result.
|
|
593
|
+
#
|
|
594
|
+
# @return [Hash{Symbol=>Object}, nil]
|
|
595
|
+
optional :meta, ContextDev::Internal::Type::HashOf[ContextDev::Internal::Type::Unknown]
|
|
596
|
+
|
|
597
|
+
# @!method initialize(url:, item_id: nil, meta: nil)
|
|
598
|
+
# Some parameter documentations has been truncated, see
|
|
599
|
+
# {ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::URL} for more
|
|
600
|
+
# details.
|
|
601
|
+
#
|
|
602
|
+
# A page to scrape, with optional data for matching results.
|
|
603
|
+
#
|
|
604
|
+
# @param url [String] Page URL to scrape.
|
|
605
|
+
#
|
|
606
|
+
# @param item_id [String] Your ID for this page, returned with its result. The same URL can use different
|
|
607
|
+
#
|
|
608
|
+
# @param meta [Hash{Symbol=>Object}] Custom JSON returned unchanged with this page result.
|
|
609
|
+
end
|
|
610
|
+
|
|
611
|
+
# @see ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML#options
|
|
612
|
+
class Options < ContextDev::Internal::Type::BaseModel
|
|
613
|
+
# @!attribute country
|
|
614
|
+
# Fetch the target page through a residential proxy in this country (ISO 3166-1
|
|
615
|
+
# alpha-2).
|
|
616
|
+
#
|
|
617
|
+
# @return [Symbol, ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Country, nil]
|
|
618
|
+
optional :country, enum: -> { ContextDev::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Country }
|
|
619
|
+
|
|
620
|
+
# @!attribute exclude_selectors
|
|
621
|
+
# Remove elements matching these CSS selectors. Applied after `includeSelectors`,
|
|
622
|
+
# so an element matching both is removed.
|
|
623
|
+
#
|
|
624
|
+
# @return [Array<String>, nil]
|
|
625
|
+
optional :exclude_selectors,
|
|
626
|
+
ContextDev::Internal::Type::ArrayOf[String],
|
|
627
|
+
api_name: :excludeSelectors,
|
|
628
|
+
nil?: true
|
|
629
|
+
|
|
630
|
+
# @!attribute include_selectors
|
|
631
|
+
# Keep only the subtrees matching these CSS selectors. Filtered pages are always
|
|
632
|
+
# fetched fresh, ignoring `maxAgeMs`.
|
|
633
|
+
#
|
|
634
|
+
# @return [Array<String>, nil]
|
|
635
|
+
optional :include_selectors,
|
|
636
|
+
ContextDev::Internal::Type::ArrayOf[String],
|
|
637
|
+
api_name: :includeSelectors,
|
|
638
|
+
nil?: true
|
|
639
|
+
|
|
640
|
+
# @!attribute max_age_ms
|
|
641
|
+
# Return a cached result if a prior scrape for the same parameters exists and is
|
|
642
|
+
# younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
|
|
643
|
+
# omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
|
|
644
|
+
#
|
|
645
|
+
# @return [Integer, nil]
|
|
646
|
+
optional :max_age_ms, Integer, api_name: :maxAgeMs, nil?: true
|
|
647
|
+
|
|
648
|
+
# @!attribute pdf
|
|
649
|
+
# PDF parsing controls. Use start/end to limit text extraction and embedded-image
|
|
650
|
+
# detection/OCR to an inclusive 1-based page range.
|
|
651
|
+
#
|
|
652
|
+
# @return [ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf, nil]
|
|
653
|
+
optional :pdf, -> { ContextDev::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf }
|
|
654
|
+
|
|
655
|
+
# @!attribute settle_animations
|
|
656
|
+
# Wait briefly for CSS and transition animations to settle before extraction, on
|
|
657
|
+
# pages that render in a browser.
|
|
658
|
+
#
|
|
659
|
+
# @return [Boolean, nil]
|
|
660
|
+
optional :settle_animations, ContextDev::Internal::Type::Boolean, api_name: :settleAnimations
|
|
661
|
+
|
|
662
|
+
# @!attribute use_main_content_only
|
|
663
|
+
# Return the main content without navigation or footers.
|
|
664
|
+
#
|
|
665
|
+
# @return [Boolean, nil]
|
|
666
|
+
optional :use_main_content_only,
|
|
667
|
+
ContextDev::Internal::Type::Boolean,
|
|
668
|
+
api_name: :useMainContentOnly
|
|
669
|
+
|
|
670
|
+
# @!attribute wait_for_ms
|
|
671
|
+
# How long to wait after initial page load, in milliseconds. `0` waits 500 ms.
|
|
672
|
+
#
|
|
673
|
+
# @return [Integer, nil]
|
|
674
|
+
optional :wait_for_ms, Integer, api_name: :waitForMs
|
|
675
|
+
|
|
676
|
+
# @!method initialize(country: nil, exclude_selectors: nil, include_selectors: nil, max_age_ms: nil, pdf: nil, settle_animations: nil, use_main_content_only: nil, wait_for_ms: nil)
|
|
677
|
+
# Some parameter documentations has been truncated, see
|
|
678
|
+
# {ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::Options} for
|
|
679
|
+
# more details.
|
|
680
|
+
#
|
|
681
|
+
# Options for HTML output.
|
|
682
|
+
#
|
|
683
|
+
# @param country [Symbol, ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Country] Fetch the target page through a residential proxy in this country (ISO 3166-1 al
|
|
684
|
+
#
|
|
685
|
+
# @param exclude_selectors [Array<String>, nil] Remove elements matching these CSS selectors. Applied after `includeSelectors`,
|
|
686
|
+
#
|
|
687
|
+
# @param include_selectors [Array<String>, nil] Keep only the subtrees matching these CSS selectors. Filtered pages are always f
|
|
688
|
+
#
|
|
689
|
+
# @param max_age_ms [Integer, nil] Return a cached result if a prior scrape for the same parameters exists and is y
|
|
690
|
+
#
|
|
691
|
+
# @param pdf [ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf] PDF parsing controls. Use start/end to limit text extraction and embedded-image
|
|
692
|
+
#
|
|
693
|
+
# @param settle_animations [Boolean] Wait briefly for CSS and transition animations to settle before extraction, on p
|
|
694
|
+
#
|
|
695
|
+
# @param use_main_content_only [Boolean] Return the main content without navigation or footers.
|
|
696
|
+
#
|
|
697
|
+
# @param wait_for_ms [Integer] How long to wait after initial page load, in milliseconds. `0` waits 500 ms.
|
|
698
|
+
|
|
699
|
+
# Fetch the target page through a residential proxy in this country (ISO 3166-1
|
|
700
|
+
# alpha-2).
|
|
701
|
+
#
|
|
702
|
+
# @see ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::Options#country
|
|
703
|
+
module Country
|
|
704
|
+
extend ContextDev::Internal::Type::Enum
|
|
705
|
+
|
|
706
|
+
AD = :ad
|
|
707
|
+
AE = :ae
|
|
708
|
+
AF = :af
|
|
709
|
+
AG = :ag
|
|
710
|
+
AI = :ai
|
|
711
|
+
AL = :al
|
|
712
|
+
AM = :am
|
|
713
|
+
AO = :ao
|
|
714
|
+
AR = :ar
|
|
715
|
+
AT = :at
|
|
716
|
+
AU = :au
|
|
717
|
+
AW = :aw
|
|
718
|
+
AZ = :az
|
|
719
|
+
BA = :ba
|
|
720
|
+
BB = :bb
|
|
721
|
+
BD = :bd
|
|
722
|
+
BE = :be
|
|
723
|
+
BF = :bf
|
|
724
|
+
BG = :bg
|
|
725
|
+
BH = :bh
|
|
726
|
+
BI = :bi
|
|
727
|
+
BJ = :bj
|
|
728
|
+
BM = :bm
|
|
729
|
+
BN = :bn
|
|
730
|
+
BO = :bo
|
|
731
|
+
BQ = :bq
|
|
732
|
+
BR = :br
|
|
733
|
+
BS = :bs
|
|
734
|
+
BW = :bw
|
|
735
|
+
BY = :by
|
|
736
|
+
BZ = :bz
|
|
737
|
+
CA = :ca
|
|
738
|
+
CD = :cd
|
|
739
|
+
CF = :cf
|
|
740
|
+
CG = :cg
|
|
741
|
+
CH = :ch
|
|
742
|
+
CI = :ci
|
|
743
|
+
CL = :cl
|
|
744
|
+
CM = :cm
|
|
745
|
+
CN = :cn
|
|
746
|
+
CO = :co
|
|
747
|
+
CR = :cr
|
|
748
|
+
CV = :cv
|
|
749
|
+
CW = :cw
|
|
750
|
+
CY = :cy
|
|
751
|
+
CZ = :cz
|
|
752
|
+
DE = :de
|
|
753
|
+
DJ = :dj
|
|
754
|
+
DK = :dk
|
|
755
|
+
DM = :dm
|
|
756
|
+
DO = :do
|
|
757
|
+
DZ = :dz
|
|
758
|
+
EC = :ec
|
|
759
|
+
EE = :ee
|
|
760
|
+
EG = :eg
|
|
761
|
+
ES = :es
|
|
762
|
+
ET = :et
|
|
763
|
+
FI = :fi
|
|
764
|
+
FJ = :fj
|
|
765
|
+
FR = :fr
|
|
766
|
+
GA = :ga
|
|
767
|
+
GB = :gb
|
|
768
|
+
GD = :gd
|
|
769
|
+
GE = :ge
|
|
770
|
+
GF = :gf
|
|
771
|
+
GG = :gg
|
|
772
|
+
GH = :gh
|
|
773
|
+
GM = :gm
|
|
774
|
+
GN = :gn
|
|
775
|
+
GP = :gp
|
|
776
|
+
GQ = :gq
|
|
777
|
+
GR = :gr
|
|
778
|
+
GT = :gt
|
|
779
|
+
GU = :gu
|
|
780
|
+
GW = :gw
|
|
781
|
+
GY = :gy
|
|
782
|
+
HK = :hk
|
|
783
|
+
HN = :hn
|
|
784
|
+
HR = :hr
|
|
785
|
+
HT = :ht
|
|
786
|
+
HU = :hu
|
|
787
|
+
ID = :id
|
|
788
|
+
IE = :ie
|
|
789
|
+
IL = :il
|
|
790
|
+
IM = :im
|
|
791
|
+
IN = :in
|
|
792
|
+
IQ = :iq
|
|
793
|
+
IR = :ir
|
|
794
|
+
IS = :is
|
|
795
|
+
IT = :it
|
|
796
|
+
JE = :je
|
|
797
|
+
JM = :jm
|
|
798
|
+
JO = :jo
|
|
799
|
+
JP = :jp
|
|
800
|
+
KE = :ke
|
|
801
|
+
KG = :kg
|
|
802
|
+
KH = :kh
|
|
803
|
+
KN = :kn
|
|
804
|
+
KR = :kr
|
|
805
|
+
KW = :kw
|
|
806
|
+
KY = :ky
|
|
807
|
+
KZ = :kz
|
|
808
|
+
LA = :la
|
|
809
|
+
LB = :lb
|
|
810
|
+
LC = :lc
|
|
811
|
+
LK = :lk
|
|
812
|
+
LR = :lr
|
|
813
|
+
LS = :ls
|
|
814
|
+
LT = :lt
|
|
815
|
+
LU = :lu
|
|
816
|
+
LV = :lv
|
|
817
|
+
LY = :ly
|
|
818
|
+
MA = :ma
|
|
819
|
+
MC = :mc
|
|
820
|
+
MD = :md
|
|
821
|
+
ME = :me
|
|
822
|
+
MF = :mf
|
|
823
|
+
MG = :mg
|
|
824
|
+
MK = :mk
|
|
825
|
+
ML = :ml
|
|
826
|
+
MM = :mm
|
|
827
|
+
MN = :mn
|
|
828
|
+
MO = :mo
|
|
829
|
+
MQ = :mq
|
|
830
|
+
MR = :mr
|
|
831
|
+
MT = :mt
|
|
832
|
+
MU = :mu
|
|
833
|
+
MV = :mv
|
|
834
|
+
MW = :mw
|
|
835
|
+
MX = :mx
|
|
836
|
+
MY = :my
|
|
837
|
+
MZ = :mz
|
|
838
|
+
NA = :na
|
|
839
|
+
NC = :nc
|
|
840
|
+
NE = :ne
|
|
841
|
+
NG = :ng
|
|
842
|
+
NI = :ni
|
|
843
|
+
NL = :nl
|
|
844
|
+
NO = :no
|
|
845
|
+
NP = :np
|
|
846
|
+
NZ = :nz
|
|
847
|
+
OM = :om
|
|
848
|
+
PA = :pa
|
|
849
|
+
PE = :pe
|
|
850
|
+
PF = :pf
|
|
851
|
+
PG = :pg
|
|
852
|
+
PH = :ph
|
|
853
|
+
PK = :pk
|
|
854
|
+
PL = :pl
|
|
855
|
+
PR = :pr
|
|
856
|
+
PS = :ps
|
|
857
|
+
PT = :pt
|
|
858
|
+
PY = :py
|
|
859
|
+
QA = :qa
|
|
860
|
+
RE = :re
|
|
861
|
+
RO = :ro
|
|
862
|
+
RS = :rs
|
|
863
|
+
RU = :ru
|
|
864
|
+
RW = :rw
|
|
865
|
+
SA = :sa
|
|
866
|
+
SC = :sc
|
|
867
|
+
SD = :sd
|
|
868
|
+
SE = :se
|
|
869
|
+
SG = :sg
|
|
870
|
+
SI = :si
|
|
871
|
+
SK = :sk
|
|
872
|
+
SL = :sl
|
|
873
|
+
SM = :sm
|
|
874
|
+
SN = :sn
|
|
875
|
+
SO = :so
|
|
876
|
+
SR = :sr
|
|
877
|
+
SS = :ss
|
|
878
|
+
ST = :st
|
|
879
|
+
SV = :sv
|
|
880
|
+
SX = :sx
|
|
881
|
+
SY = :sy
|
|
882
|
+
SZ = :sz
|
|
883
|
+
TC = :tc
|
|
884
|
+
TD = :td
|
|
885
|
+
TG = :tg
|
|
886
|
+
TH = :th
|
|
887
|
+
TJ = :tj
|
|
888
|
+
TL = :tl
|
|
889
|
+
TM = :tm
|
|
890
|
+
TN = :tn
|
|
891
|
+
TR = :tr
|
|
892
|
+
TT = :tt
|
|
893
|
+
TW = :tw
|
|
894
|
+
TZ = :tz
|
|
895
|
+
UA = :ua
|
|
896
|
+
UG = :ug
|
|
897
|
+
US = :us
|
|
898
|
+
UY = :uy
|
|
899
|
+
UZ = :uz
|
|
900
|
+
VC = :vc
|
|
901
|
+
VE = :ve
|
|
902
|
+
VG = :vg
|
|
903
|
+
VI = :vi
|
|
904
|
+
VN = :vn
|
|
905
|
+
YE = :ye
|
|
906
|
+
YT = :yt
|
|
907
|
+
ZA = :za
|
|
908
|
+
ZM = :zm
|
|
909
|
+
ZW = :zw
|
|
910
|
+
|
|
911
|
+
# @!method self.values
|
|
912
|
+
# @return [Array<Symbol>]
|
|
913
|
+
end
|
|
914
|
+
|
|
915
|
+
# @see ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::Options#pdf
|
|
916
|
+
class Pdf < ContextDev::Internal::Type::BaseModel
|
|
917
|
+
# @!attribute end_
|
|
918
|
+
# Last 1-based PDF page to parse. When omitted, parsing ends at the last page.
|
|
919
|
+
# Must be greater than or equal to start when both are provided.
|
|
920
|
+
#
|
|
921
|
+
# @return [Integer, nil]
|
|
922
|
+
optional :end_, Integer, api_name: :end
|
|
923
|
+
|
|
924
|
+
# @!attribute ocr
|
|
925
|
+
# When true, OCR the selected PDF pages that have no usable text layer (scans),
|
|
926
|
+
# replacing each recovered page's text with the OCR result while pages with a real
|
|
927
|
+
# text layer keep it. Billed at 1 credit per page OCR actually recovered, on top
|
|
928
|
+
# of the base request cost. When false, no OCR runs.
|
|
929
|
+
#
|
|
930
|
+
# @return [Boolean, nil]
|
|
931
|
+
optional :ocr, ContextDev::Internal::Type::Boolean
|
|
932
|
+
|
|
933
|
+
# @!attribute should_parse
|
|
934
|
+
# When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
|
|
935
|
+
# a 400 PDF_SKIPPED is returned.
|
|
936
|
+
#
|
|
937
|
+
# @return [Boolean, nil]
|
|
938
|
+
optional :should_parse, ContextDev::Internal::Type::Boolean, api_name: :shouldParse
|
|
939
|
+
|
|
940
|
+
# @!attribute start
|
|
941
|
+
# First 1-based PDF page to parse. When omitted, parsing starts at the first page.
|
|
942
|
+
#
|
|
943
|
+
# @return [Integer, nil]
|
|
944
|
+
optional :start, Integer
|
|
945
|
+
|
|
946
|
+
# @!method initialize(end_: nil, ocr: nil, should_parse: nil, start: nil)
|
|
947
|
+
# Some parameter documentations has been truncated, see
|
|
948
|
+
# {ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf}
|
|
949
|
+
# for more details.
|
|
950
|
+
#
|
|
951
|
+
# PDF parsing controls. Use start/end to limit text extraction and embedded-image
|
|
952
|
+
# detection/OCR to an inclusive 1-based page range.
|
|
953
|
+
#
|
|
954
|
+
# @param end_ [Integer] Last 1-based PDF page to parse. When omitted, parsing ends at the last page. Mus
|
|
955
|
+
#
|
|
956
|
+
# @param ocr [Boolean] When true, OCR the selected PDF pages that have no usable text layer (scans), re
|
|
957
|
+
#
|
|
958
|
+
# @param should_parse [Boolean] When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
|
|
959
|
+
#
|
|
960
|
+
# @param start [Integer] First 1-based PDF page to parse. When omitted, parsing starts at the first page.
|
|
961
|
+
end
|
|
962
|
+
end
|
|
963
|
+
end
|
|
964
|
+
|
|
965
|
+
# @!method self.variants
|
|
966
|
+
# @return [Array(ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown, ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML)]
|
|
967
|
+
end
|
|
968
|
+
end
|
|
969
|
+
|
|
970
|
+
class Crawl < ContextDev::Internal::Type::BaseModel
|
|
971
|
+
# @!attribute data
|
|
972
|
+
# Crawl source and output format.
|
|
973
|
+
#
|
|
974
|
+
# @return [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown, ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML]
|
|
975
|
+
required :data, union: -> { ContextDev::BatchSubmitParams::Input::Crawl::Data }
|
|
976
|
+
|
|
977
|
+
# @!attribute mode
|
|
978
|
+
# Discover and scrape pages from `data.source`.
|
|
979
|
+
#
|
|
980
|
+
# @return [Symbol, :crawl]
|
|
981
|
+
required :mode, const: :crawl
|
|
982
|
+
|
|
983
|
+
# @!method initialize(data:, mode: :crawl)
|
|
984
|
+
# Crawl pages starting from a URL or from a domain's sitemap.
|
|
985
|
+
#
|
|
986
|
+
# @param data [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown, ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML] Crawl source and output format.
|
|
987
|
+
#
|
|
988
|
+
# @param mode [Symbol, :crawl] Discover and scrape pages from `data.source`.
|
|
989
|
+
|
|
990
|
+
# Crawl source and output format.
|
|
991
|
+
#
|
|
992
|
+
# @see ContextDev::Models::BatchSubmitParams::Input::Crawl#data
|
|
993
|
+
module Data
|
|
994
|
+
extend ContextDev::Internal::Type::Union
|
|
995
|
+
|
|
996
|
+
discriminator :format
|
|
997
|
+
|
|
998
|
+
# Crawl pages and return Markdown.
|
|
999
|
+
variant :markdown, -> { ContextDev::BatchSubmitParams::Input::Crawl::Data::Markdown }
|
|
1000
|
+
|
|
1001
|
+
# Crawl pages and return HTML.
|
|
1002
|
+
variant :html, -> { ContextDev::BatchSubmitParams::Input::Crawl::Data::HTML }
|
|
1003
|
+
|
|
1004
|
+
class Markdown < ContextDev::Internal::Type::BaseModel
|
|
1005
|
+
# @!attribute format_
|
|
1006
|
+
# Return page content as Markdown.
|
|
1007
|
+
#
|
|
1008
|
+
# @return [Symbol, :markdown]
|
|
1009
|
+
required :format_, const: :markdown, api_name: :format
|
|
1010
|
+
|
|
1011
|
+
# @!attribute source
|
|
1012
|
+
# How to find pages to crawl.
|
|
1013
|
+
#
|
|
1014
|
+
# @return [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Source::StartURL, ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Source::Sitemap]
|
|
1015
|
+
required :source, union: -> { ContextDev::BatchSubmitParams::Input::Crawl::Data::Markdown::Source }
|
|
1016
|
+
|
|
1017
|
+
# @!attribute options
|
|
1018
|
+
# Options for Markdown output.
|
|
1019
|
+
#
|
|
1020
|
+
# @return [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Options, nil]
|
|
1021
|
+
optional :options, -> { ContextDev::BatchSubmitParams::Input::Crawl::Data::Markdown::Options }
|
|
1022
|
+
|
|
1023
|
+
# @!method initialize(source:, options: nil, format_: :markdown)
|
|
1024
|
+
# Crawl pages and return Markdown.
|
|
1025
|
+
#
|
|
1026
|
+
# @param source [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Source::StartURL, ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Source::Sitemap] How to find pages to crawl.
|
|
1027
|
+
#
|
|
1028
|
+
# @param options [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Options] Options for Markdown output.
|
|
1029
|
+
#
|
|
1030
|
+
# @param format_ [Symbol, :markdown] Return page content as Markdown.
|
|
1031
|
+
|
|
1032
|
+
# How to find pages to crawl.
|
|
1033
|
+
#
|
|
1034
|
+
# @see ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown#source
|
|
1035
|
+
module Source
|
|
1036
|
+
extend ContextDev::Internal::Type::Union
|
|
1037
|
+
|
|
1038
|
+
discriminator :type
|
|
1039
|
+
|
|
1040
|
+
# Discover pages by following links from one URL.
|
|
1041
|
+
variant :start_url, -> { ContextDev::BatchSubmitParams::Input::Crawl::Data::Markdown::Source::StartURL }
|
|
1042
|
+
|
|
1043
|
+
# Scrape the pages listed in a domain's sitemap. Links on those pages are not followed.
|
|
1044
|
+
variant :sitemap, -> { ContextDev::BatchSubmitParams::Input::Crawl::Data::Markdown::Source::Sitemap }
|
|
1045
|
+
|
|
1046
|
+
class StartURL < ContextDev::Internal::Type::BaseModel
|
|
1047
|
+
# @!attribute type
|
|
1048
|
+
# Start from one page.
|
|
1049
|
+
#
|
|
1050
|
+
# @return [Symbol, :start_url]
|
|
1051
|
+
required :type, const: :start_url
|
|
1052
|
+
|
|
1053
|
+
# @!attribute url
|
|
1054
|
+
# Page where crawling begins. A URL without a scheme is read as https://.
|
|
1055
|
+
#
|
|
1056
|
+
# @return [String]
|
|
1057
|
+
required :url, String
|
|
1058
|
+
|
|
1059
|
+
# @!attribute controls
|
|
1060
|
+
# Limits and filters for page discovery.
|
|
1061
|
+
#
|
|
1062
|
+
# @return [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Source::StartURL::Controls, nil]
|
|
1063
|
+
optional :controls,
|
|
1064
|
+
-> { ContextDev::BatchSubmitParams::Input::Crawl::Data::Markdown::Source::StartURL::Controls }
|
|
1065
|
+
|
|
1066
|
+
# @!method initialize(url:, controls: nil, type: :start_url)
|
|
1067
|
+
# Discover pages by following links from one URL.
|
|
1068
|
+
#
|
|
1069
|
+
# @param url [String] Page where crawling begins. A URL without a scheme is read as https://.
|
|
1070
|
+
#
|
|
1071
|
+
# @param controls [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Source::StartURL::Controls] Limits and filters for page discovery.
|
|
1072
|
+
#
|
|
1073
|
+
# @param type [Symbol, :start_url] Start from one page.
|
|
1074
|
+
|
|
1075
|
+
# @see ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Source::StartURL#controls
|
|
1076
|
+
class Controls < ContextDev::Internal::Type::BaseModel
|
|
1077
|
+
# @!attribute follow_subdomains
|
|
1078
|
+
# Follow links to subdomains.
|
|
1079
|
+
#
|
|
1080
|
+
# @return [Boolean, nil]
|
|
1081
|
+
optional :follow_subdomains,
|
|
1082
|
+
ContextDev::Internal::Type::Boolean,
|
|
1083
|
+
api_name: :followSubdomains
|
|
1084
|
+
|
|
1085
|
+
# @!attribute max_depth
|
|
1086
|
+
# Maximum link depth. Source pages are depth 0. No limit when omitted.
|
|
1087
|
+
#
|
|
1088
|
+
# @return [Integer, nil]
|
|
1089
|
+
optional :max_depth, Integer, api_name: :maxDepth
|
|
1090
|
+
|
|
1091
|
+
# @!attribute max_urls
|
|
1092
|
+
# Maximum pages to fetch. Unused reserved credits are refunded. Maximum 25000.
|
|
1093
|
+
#
|
|
1094
|
+
# @return [Integer, nil]
|
|
1095
|
+
optional :max_urls, Integer, api_name: :maxUrls
|
|
1096
|
+
|
|
1097
|
+
# @!attribute regex
|
|
1098
|
+
# RE2 pattern for URLs to include. The `start_url` itself is always included.
|
|
1099
|
+
#
|
|
1100
|
+
# @return [String, nil]
|
|
1101
|
+
optional :regex, String
|
|
1102
|
+
|
|
1103
|
+
# @!method initialize(follow_subdomains: nil, max_depth: nil, max_urls: nil, regex: nil)
|
|
1104
|
+
# Limits and filters for page discovery.
|
|
1105
|
+
#
|
|
1106
|
+
# @param follow_subdomains [Boolean] Follow links to subdomains.
|
|
1107
|
+
#
|
|
1108
|
+
# @param max_depth [Integer] Maximum link depth. Source pages are depth 0. No limit when omitted.
|
|
1109
|
+
#
|
|
1110
|
+
# @param max_urls [Integer] Maximum pages to fetch. Unused reserved credits are refunded. Maximum 25000.
|
|
1111
|
+
#
|
|
1112
|
+
# @param regex [String] RE2 pattern for URLs to include. The `start_url` itself is always included.
|
|
1113
|
+
end
|
|
1114
|
+
end
|
|
1115
|
+
|
|
1116
|
+
class Sitemap < ContextDev::Internal::Type::BaseModel
|
|
1117
|
+
# @!attribute domain
|
|
1118
|
+
# Domain whose sitemap lists the pages to scrape. A full URL is reduced to its
|
|
1119
|
+
# domain.
|
|
1120
|
+
#
|
|
1121
|
+
# @return [String]
|
|
1122
|
+
required :domain, String
|
|
1123
|
+
|
|
1124
|
+
# @!attribute type
|
|
1125
|
+
# Scrape the URLs in the domain's sitemap.
|
|
1126
|
+
#
|
|
1127
|
+
# @return [Symbol, :sitemap]
|
|
1128
|
+
required :type, const: :sitemap
|
|
1129
|
+
|
|
1130
|
+
# @!attribute controls
|
|
1131
|
+
# Limits and filters for the sitemap URLs. A sitemap batch scrapes exactly those
|
|
1132
|
+
# URLs and never follows links off them, so there is no crawl depth here.
|
|
1133
|
+
#
|
|
1134
|
+
# @return [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Source::Sitemap::Controls, nil]
|
|
1135
|
+
optional :controls,
|
|
1136
|
+
-> { ContextDev::BatchSubmitParams::Input::Crawl::Data::Markdown::Source::Sitemap::Controls }
|
|
1137
|
+
|
|
1138
|
+
# @!method initialize(domain:, controls: nil, type: :sitemap)
|
|
1139
|
+
# Some parameter documentations has been truncated, see
|
|
1140
|
+
# {ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Source::Sitemap}
|
|
1141
|
+
# for more details.
|
|
1142
|
+
#
|
|
1143
|
+
# Scrape the pages listed in a domain's sitemap. Links on those pages are not
|
|
1144
|
+
# followed.
|
|
1145
|
+
#
|
|
1146
|
+
# @param domain [String] Domain whose sitemap lists the pages to scrape. A full URL is reduced to its dom
|
|
1147
|
+
#
|
|
1148
|
+
# @param controls [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Source::Sitemap::Controls] Limits and filters for the sitemap URLs. A sitemap batch scrapes exactly those U
|
|
1149
|
+
#
|
|
1150
|
+
# @param type [Symbol, :sitemap] Scrape the URLs in the domain's sitemap.
|
|
1151
|
+
|
|
1152
|
+
# @see ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Source::Sitemap#controls
|
|
1153
|
+
class Controls < ContextDev::Internal::Type::BaseModel
|
|
1154
|
+
# @!attribute max_urls
|
|
1155
|
+
# Maximum pages to fetch. Unused reserved credits are refunded. Maximum 25000.
|
|
1156
|
+
#
|
|
1157
|
+
# @return [Integer, nil]
|
|
1158
|
+
optional :max_urls, Integer, api_name: :maxUrls
|
|
1159
|
+
|
|
1160
|
+
# @!attribute regex
|
|
1161
|
+
# RE2 pattern; only sitemap URLs matching it are scraped.
|
|
1162
|
+
#
|
|
1163
|
+
# @return [String, nil]
|
|
1164
|
+
optional :regex, String
|
|
1165
|
+
|
|
1166
|
+
# @!method initialize(max_urls: nil, regex: nil)
|
|
1167
|
+
# Limits and filters for the sitemap URLs. A sitemap batch scrapes exactly those
|
|
1168
|
+
# URLs and never follows links off them, so there is no crawl depth here.
|
|
1169
|
+
#
|
|
1170
|
+
# @param max_urls [Integer] Maximum pages to fetch. Unused reserved credits are refunded. Maximum 25000.
|
|
1171
|
+
#
|
|
1172
|
+
# @param regex [String] RE2 pattern; only sitemap URLs matching it are scraped.
|
|
1173
|
+
end
|
|
1174
|
+
end
|
|
1175
|
+
|
|
1176
|
+
# @!method self.variants
|
|
1177
|
+
# @return [Array(ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Source::StartURL, ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Source::Sitemap)]
|
|
1178
|
+
end
|
|
1179
|
+
|
|
1180
|
+
# @see ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown#options
|
|
1181
|
+
class Options < ContextDev::Internal::Type::BaseModel
|
|
1182
|
+
# @!attribute country
|
|
1183
|
+
# Fetch the target page through a residential proxy in this country (ISO 3166-1
|
|
1184
|
+
# alpha-2).
|
|
1185
|
+
#
|
|
1186
|
+
# @return [Symbol, ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Country, nil]
|
|
1187
|
+
optional :country,
|
|
1188
|
+
enum: -> { ContextDev::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Country }
|
|
1189
|
+
|
|
1190
|
+
# @!attribute exclude_selectors
|
|
1191
|
+
# Remove elements matching these CSS selectors. Applied after `includeSelectors`,
|
|
1192
|
+
# so an element matching both is removed.
|
|
1193
|
+
#
|
|
1194
|
+
# @return [Array<String>, nil]
|
|
1195
|
+
optional :exclude_selectors,
|
|
1196
|
+
ContextDev::Internal::Type::ArrayOf[String],
|
|
1197
|
+
api_name: :excludeSelectors,
|
|
1198
|
+
nil?: true
|
|
1199
|
+
|
|
1200
|
+
# @!attribute include_html
|
|
1201
|
+
# Also include each page's HTML in its result record, as an `html` field alongside
|
|
1202
|
+
# the Markdown.
|
|
1203
|
+
#
|
|
1204
|
+
# @return [Boolean, nil]
|
|
1205
|
+
optional :include_html, ContextDev::Internal::Type::Boolean, api_name: :includeHTML
|
|
1206
|
+
|
|
1207
|
+
# @!attribute include_images
|
|
1208
|
+
# Include image references in the Markdown.
|
|
1209
|
+
#
|
|
1210
|
+
# @return [Boolean, nil]
|
|
1211
|
+
optional :include_images, ContextDev::Internal::Type::Boolean, api_name: :includeImages
|
|
1212
|
+
|
|
1213
|
+
# @!attribute include_links
|
|
1214
|
+
# Include links in the Markdown.
|
|
1215
|
+
#
|
|
1216
|
+
# @return [Boolean, nil]
|
|
1217
|
+
optional :include_links, ContextDev::Internal::Type::Boolean, api_name: :includeLinks
|
|
1218
|
+
|
|
1219
|
+
# @!attribute include_selectors
|
|
1220
|
+
# Keep only the subtrees matching these CSS selectors. Filtered pages are always
|
|
1221
|
+
# fetched fresh, ignoring `maxAgeMs`.
|
|
1222
|
+
#
|
|
1223
|
+
# @return [Array<String>, nil]
|
|
1224
|
+
optional :include_selectors,
|
|
1225
|
+
ContextDev::Internal::Type::ArrayOf[String],
|
|
1226
|
+
api_name: :includeSelectors,
|
|
1227
|
+
nil?: true
|
|
1228
|
+
|
|
1229
|
+
# @!attribute max_age_ms
|
|
1230
|
+
# Return a cached result if a prior scrape for the same parameters exists and is
|
|
1231
|
+
# younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
|
|
1232
|
+
# omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
|
|
1233
|
+
#
|
|
1234
|
+
# @return [Integer, nil]
|
|
1235
|
+
optional :max_age_ms, Integer, api_name: :maxAgeMs, nil?: true
|
|
1236
|
+
|
|
1237
|
+
# @!attribute pdf
|
|
1238
|
+
# PDF parsing controls. Use start/end to limit text extraction and embedded-image
|
|
1239
|
+
# detection/OCR to an inclusive 1-based page range.
|
|
1240
|
+
#
|
|
1241
|
+
# @return [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf, nil]
|
|
1242
|
+
optional :pdf, -> { ContextDev::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf }
|
|
1243
|
+
|
|
1244
|
+
# @!attribute settle_animations
|
|
1245
|
+
# Wait briefly for CSS and transition animations to settle before extraction, on
|
|
1246
|
+
# pages that render in a browser.
|
|
1247
|
+
#
|
|
1248
|
+
# @return [Boolean, nil]
|
|
1249
|
+
optional :settle_animations, ContextDev::Internal::Type::Boolean, api_name: :settleAnimations
|
|
1250
|
+
|
|
1251
|
+
# @!attribute shorten_base64_images
|
|
1252
|
+
# Shorten inline base64 image data.
|
|
1253
|
+
#
|
|
1254
|
+
# @return [Boolean, nil]
|
|
1255
|
+
optional :shorten_base64_images,
|
|
1256
|
+
ContextDev::Internal::Type::Boolean,
|
|
1257
|
+
api_name: :shortenBase64Images
|
|
1258
|
+
|
|
1259
|
+
# @!attribute use_main_content_only
|
|
1260
|
+
# Return the main content without navigation or footers.
|
|
1261
|
+
#
|
|
1262
|
+
# @return [Boolean, nil]
|
|
1263
|
+
optional :use_main_content_only,
|
|
1264
|
+
ContextDev::Internal::Type::Boolean,
|
|
1265
|
+
api_name: :useMainContentOnly
|
|
1266
|
+
|
|
1267
|
+
# @!attribute wait_for_ms
|
|
1268
|
+
# How long to wait after initial page load, in milliseconds. `0` waits 500 ms.
|
|
1269
|
+
#
|
|
1270
|
+
# @return [Integer, nil]
|
|
1271
|
+
optional :wait_for_ms, Integer, api_name: :waitForMs
|
|
1272
|
+
|
|
1273
|
+
# @!method initialize(country: nil, exclude_selectors: nil, include_html: nil, include_images: nil, include_links: nil, include_selectors: nil, max_age_ms: nil, pdf: nil, settle_animations: nil, shorten_base64_images: nil, use_main_content_only: nil, wait_for_ms: nil)
|
|
1274
|
+
# Some parameter documentations has been truncated, see
|
|
1275
|
+
# {ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Options}
|
|
1276
|
+
# for more details.
|
|
1277
|
+
#
|
|
1278
|
+
# Options for Markdown output.
|
|
1279
|
+
#
|
|
1280
|
+
# @param country [Symbol, ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Country] Fetch the target page through a residential proxy in this country (ISO 3166-1 al
|
|
1281
|
+
#
|
|
1282
|
+
# @param exclude_selectors [Array<String>, nil] Remove elements matching these CSS selectors. Applied after `includeSelectors`,
|
|
1283
|
+
#
|
|
1284
|
+
# @param include_html [Boolean] Also include each page's HTML in its result record, as an `html` field alongside
|
|
1285
|
+
#
|
|
1286
|
+
# @param include_images [Boolean] Include image references in the Markdown.
|
|
1287
|
+
#
|
|
1288
|
+
# @param include_links [Boolean] Include links in the Markdown.
|
|
1289
|
+
#
|
|
1290
|
+
# @param include_selectors [Array<String>, nil] Keep only the subtrees matching these CSS selectors. Filtered pages are always f
|
|
1291
|
+
#
|
|
1292
|
+
# @param max_age_ms [Integer, nil] Return a cached result if a prior scrape for the same parameters exists and is y
|
|
1293
|
+
#
|
|
1294
|
+
# @param pdf [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf] PDF parsing controls. Use start/end to limit text extraction and embedded-image
|
|
1295
|
+
#
|
|
1296
|
+
# @param settle_animations [Boolean] Wait briefly for CSS and transition animations to settle before extraction, on p
|
|
1297
|
+
#
|
|
1298
|
+
# @param shorten_base64_images [Boolean] Shorten inline base64 image data.
|
|
1299
|
+
#
|
|
1300
|
+
# @param use_main_content_only [Boolean] Return the main content without navigation or footers.
|
|
1301
|
+
#
|
|
1302
|
+
# @param wait_for_ms [Integer] How long to wait after initial page load, in milliseconds. `0` waits 500 ms.
|
|
1303
|
+
|
|
1304
|
+
# Fetch the target page through a residential proxy in this country (ISO 3166-1
|
|
1305
|
+
# alpha-2).
|
|
1306
|
+
#
|
|
1307
|
+
# @see ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Options#country
|
|
1308
|
+
module Country
|
|
1309
|
+
extend ContextDev::Internal::Type::Enum
|
|
1310
|
+
|
|
1311
|
+
AD = :ad
|
|
1312
|
+
AE = :ae
|
|
1313
|
+
AF = :af
|
|
1314
|
+
AG = :ag
|
|
1315
|
+
AI = :ai
|
|
1316
|
+
AL = :al
|
|
1317
|
+
AM = :am
|
|
1318
|
+
AO = :ao
|
|
1319
|
+
AR = :ar
|
|
1320
|
+
AT = :at
|
|
1321
|
+
AU = :au
|
|
1322
|
+
AW = :aw
|
|
1323
|
+
AZ = :az
|
|
1324
|
+
BA = :ba
|
|
1325
|
+
BB = :bb
|
|
1326
|
+
BD = :bd
|
|
1327
|
+
BE = :be
|
|
1328
|
+
BF = :bf
|
|
1329
|
+
BG = :bg
|
|
1330
|
+
BH = :bh
|
|
1331
|
+
BI = :bi
|
|
1332
|
+
BJ = :bj
|
|
1333
|
+
BM = :bm
|
|
1334
|
+
BN = :bn
|
|
1335
|
+
BO = :bo
|
|
1336
|
+
BQ = :bq
|
|
1337
|
+
BR = :br
|
|
1338
|
+
BS = :bs
|
|
1339
|
+
BW = :bw
|
|
1340
|
+
BY = :by
|
|
1341
|
+
BZ = :bz
|
|
1342
|
+
CA = :ca
|
|
1343
|
+
CD = :cd
|
|
1344
|
+
CF = :cf
|
|
1345
|
+
CG = :cg
|
|
1346
|
+
CH = :ch
|
|
1347
|
+
CI = :ci
|
|
1348
|
+
CL = :cl
|
|
1349
|
+
CM = :cm
|
|
1350
|
+
CN = :cn
|
|
1351
|
+
CO = :co
|
|
1352
|
+
CR = :cr
|
|
1353
|
+
CV = :cv
|
|
1354
|
+
CW = :cw
|
|
1355
|
+
CY = :cy
|
|
1356
|
+
CZ = :cz
|
|
1357
|
+
DE = :de
|
|
1358
|
+
DJ = :dj
|
|
1359
|
+
DK = :dk
|
|
1360
|
+
DM = :dm
|
|
1361
|
+
DO = :do
|
|
1362
|
+
DZ = :dz
|
|
1363
|
+
EC = :ec
|
|
1364
|
+
EE = :ee
|
|
1365
|
+
EG = :eg
|
|
1366
|
+
ES = :es
|
|
1367
|
+
ET = :et
|
|
1368
|
+
FI = :fi
|
|
1369
|
+
FJ = :fj
|
|
1370
|
+
FR = :fr
|
|
1371
|
+
GA = :ga
|
|
1372
|
+
GB = :gb
|
|
1373
|
+
GD = :gd
|
|
1374
|
+
GE = :ge
|
|
1375
|
+
GF = :gf
|
|
1376
|
+
GG = :gg
|
|
1377
|
+
GH = :gh
|
|
1378
|
+
GM = :gm
|
|
1379
|
+
GN = :gn
|
|
1380
|
+
GP = :gp
|
|
1381
|
+
GQ = :gq
|
|
1382
|
+
GR = :gr
|
|
1383
|
+
GT = :gt
|
|
1384
|
+
GU = :gu
|
|
1385
|
+
GW = :gw
|
|
1386
|
+
GY = :gy
|
|
1387
|
+
HK = :hk
|
|
1388
|
+
HN = :hn
|
|
1389
|
+
HR = :hr
|
|
1390
|
+
HT = :ht
|
|
1391
|
+
HU = :hu
|
|
1392
|
+
ID = :id
|
|
1393
|
+
IE = :ie
|
|
1394
|
+
IL = :il
|
|
1395
|
+
IM = :im
|
|
1396
|
+
IN = :in
|
|
1397
|
+
IQ = :iq
|
|
1398
|
+
IR = :ir
|
|
1399
|
+
IS = :is
|
|
1400
|
+
IT = :it
|
|
1401
|
+
JE = :je
|
|
1402
|
+
JM = :jm
|
|
1403
|
+
JO = :jo
|
|
1404
|
+
JP = :jp
|
|
1405
|
+
KE = :ke
|
|
1406
|
+
KG = :kg
|
|
1407
|
+
KH = :kh
|
|
1408
|
+
KN = :kn
|
|
1409
|
+
KR = :kr
|
|
1410
|
+
KW = :kw
|
|
1411
|
+
KY = :ky
|
|
1412
|
+
KZ = :kz
|
|
1413
|
+
LA = :la
|
|
1414
|
+
LB = :lb
|
|
1415
|
+
LC = :lc
|
|
1416
|
+
LK = :lk
|
|
1417
|
+
LR = :lr
|
|
1418
|
+
LS = :ls
|
|
1419
|
+
LT = :lt
|
|
1420
|
+
LU = :lu
|
|
1421
|
+
LV = :lv
|
|
1422
|
+
LY = :ly
|
|
1423
|
+
MA = :ma
|
|
1424
|
+
MC = :mc
|
|
1425
|
+
MD = :md
|
|
1426
|
+
ME = :me
|
|
1427
|
+
MF = :mf
|
|
1428
|
+
MG = :mg
|
|
1429
|
+
MK = :mk
|
|
1430
|
+
ML = :ml
|
|
1431
|
+
MM = :mm
|
|
1432
|
+
MN = :mn
|
|
1433
|
+
MO = :mo
|
|
1434
|
+
MQ = :mq
|
|
1435
|
+
MR = :mr
|
|
1436
|
+
MT = :mt
|
|
1437
|
+
MU = :mu
|
|
1438
|
+
MV = :mv
|
|
1439
|
+
MW = :mw
|
|
1440
|
+
MX = :mx
|
|
1441
|
+
MY = :my
|
|
1442
|
+
MZ = :mz
|
|
1443
|
+
NA = :na
|
|
1444
|
+
NC = :nc
|
|
1445
|
+
NE = :ne
|
|
1446
|
+
NG = :ng
|
|
1447
|
+
NI = :ni
|
|
1448
|
+
NL = :nl
|
|
1449
|
+
NO = :no
|
|
1450
|
+
NP = :np
|
|
1451
|
+
NZ = :nz
|
|
1452
|
+
OM = :om
|
|
1453
|
+
PA = :pa
|
|
1454
|
+
PE = :pe
|
|
1455
|
+
PF = :pf
|
|
1456
|
+
PG = :pg
|
|
1457
|
+
PH = :ph
|
|
1458
|
+
PK = :pk
|
|
1459
|
+
PL = :pl
|
|
1460
|
+
PR = :pr
|
|
1461
|
+
PS = :ps
|
|
1462
|
+
PT = :pt
|
|
1463
|
+
PY = :py
|
|
1464
|
+
QA = :qa
|
|
1465
|
+
RE = :re
|
|
1466
|
+
RO = :ro
|
|
1467
|
+
RS = :rs
|
|
1468
|
+
RU = :ru
|
|
1469
|
+
RW = :rw
|
|
1470
|
+
SA = :sa
|
|
1471
|
+
SC = :sc
|
|
1472
|
+
SD = :sd
|
|
1473
|
+
SE = :se
|
|
1474
|
+
SG = :sg
|
|
1475
|
+
SI = :si
|
|
1476
|
+
SK = :sk
|
|
1477
|
+
SL = :sl
|
|
1478
|
+
SM = :sm
|
|
1479
|
+
SN = :sn
|
|
1480
|
+
SO = :so
|
|
1481
|
+
SR = :sr
|
|
1482
|
+
SS = :ss
|
|
1483
|
+
ST = :st
|
|
1484
|
+
SV = :sv
|
|
1485
|
+
SX = :sx
|
|
1486
|
+
SY = :sy
|
|
1487
|
+
SZ = :sz
|
|
1488
|
+
TC = :tc
|
|
1489
|
+
TD = :td
|
|
1490
|
+
TG = :tg
|
|
1491
|
+
TH = :th
|
|
1492
|
+
TJ = :tj
|
|
1493
|
+
TL = :tl
|
|
1494
|
+
TM = :tm
|
|
1495
|
+
TN = :tn
|
|
1496
|
+
TR = :tr
|
|
1497
|
+
TT = :tt
|
|
1498
|
+
TW = :tw
|
|
1499
|
+
TZ = :tz
|
|
1500
|
+
UA = :ua
|
|
1501
|
+
UG = :ug
|
|
1502
|
+
US = :us
|
|
1503
|
+
UY = :uy
|
|
1504
|
+
UZ = :uz
|
|
1505
|
+
VC = :vc
|
|
1506
|
+
VE = :ve
|
|
1507
|
+
VG = :vg
|
|
1508
|
+
VI = :vi
|
|
1509
|
+
VN = :vn
|
|
1510
|
+
YE = :ye
|
|
1511
|
+
YT = :yt
|
|
1512
|
+
ZA = :za
|
|
1513
|
+
ZM = :zm
|
|
1514
|
+
ZW = :zw
|
|
1515
|
+
|
|
1516
|
+
# @!method self.values
|
|
1517
|
+
# @return [Array<Symbol>]
|
|
1518
|
+
end
|
|
1519
|
+
|
|
1520
|
+
# @see ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Options#pdf
|
|
1521
|
+
class Pdf < ContextDev::Internal::Type::BaseModel
|
|
1522
|
+
# @!attribute end_
|
|
1523
|
+
# Last 1-based PDF page to parse. When omitted, parsing ends at the last page.
|
|
1524
|
+
# Must be greater than or equal to start when both are provided.
|
|
1525
|
+
#
|
|
1526
|
+
# @return [Integer, nil]
|
|
1527
|
+
optional :end_, Integer, api_name: :end
|
|
1528
|
+
|
|
1529
|
+
# @!attribute ocr
|
|
1530
|
+
# When true, OCR the selected PDF pages that have no usable text layer (scans),
|
|
1531
|
+
# replacing each recovered page's text with the OCR result while pages with a real
|
|
1532
|
+
# text layer keep it. Billed at 1 credit per page OCR actually recovered, on top
|
|
1533
|
+
# of the base request cost. When false, no OCR runs.
|
|
1534
|
+
#
|
|
1535
|
+
# @return [Boolean, nil]
|
|
1536
|
+
optional :ocr, ContextDev::Internal::Type::Boolean
|
|
1537
|
+
|
|
1538
|
+
# @!attribute should_parse
|
|
1539
|
+
# When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
|
|
1540
|
+
# a 400 PDF_SKIPPED is returned.
|
|
1541
|
+
#
|
|
1542
|
+
# @return [Boolean, nil]
|
|
1543
|
+
optional :should_parse, ContextDev::Internal::Type::Boolean, api_name: :shouldParse
|
|
1544
|
+
|
|
1545
|
+
# @!attribute start
|
|
1546
|
+
# First 1-based PDF page to parse. When omitted, parsing starts at the first page.
|
|
1547
|
+
#
|
|
1548
|
+
# @return [Integer, nil]
|
|
1549
|
+
optional :start, Integer
|
|
1550
|
+
|
|
1551
|
+
# @!method initialize(end_: nil, ocr: nil, should_parse: nil, start: nil)
|
|
1552
|
+
# Some parameter documentations has been truncated, see
|
|
1553
|
+
# {ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf}
|
|
1554
|
+
# for more details.
|
|
1555
|
+
#
|
|
1556
|
+
# PDF parsing controls. Use start/end to limit text extraction and embedded-image
|
|
1557
|
+
# detection/OCR to an inclusive 1-based page range.
|
|
1558
|
+
#
|
|
1559
|
+
# @param end_ [Integer] Last 1-based PDF page to parse. When omitted, parsing ends at the last page. Mus
|
|
1560
|
+
#
|
|
1561
|
+
# @param ocr [Boolean] When true, OCR the selected PDF pages that have no usable text layer (scans), re
|
|
1562
|
+
#
|
|
1563
|
+
# @param should_parse [Boolean] When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
|
|
1564
|
+
#
|
|
1565
|
+
# @param start [Integer] First 1-based PDF page to parse. When omitted, parsing starts at the first page.
|
|
1566
|
+
end
|
|
1567
|
+
end
|
|
1568
|
+
end
|
|
1569
|
+
|
|
1570
|
+
class HTML < ContextDev::Internal::Type::BaseModel
|
|
1571
|
+
# @!attribute format_
|
|
1572
|
+
# Return page content as HTML.
|
|
1573
|
+
#
|
|
1574
|
+
# @return [Symbol, :html]
|
|
1575
|
+
required :format_, const: :html, api_name: :format
|
|
1576
|
+
|
|
1577
|
+
# @!attribute source
|
|
1578
|
+
# How to find pages to crawl.
|
|
1579
|
+
#
|
|
1580
|
+
# @return [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Source::StartURL, ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Source::Sitemap]
|
|
1581
|
+
required :source, union: -> { ContextDev::BatchSubmitParams::Input::Crawl::Data::HTML::Source }
|
|
1582
|
+
|
|
1583
|
+
# @!attribute options
|
|
1584
|
+
# Options for HTML output.
|
|
1585
|
+
#
|
|
1586
|
+
# @return [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Options, nil]
|
|
1587
|
+
optional :options, -> { ContextDev::BatchSubmitParams::Input::Crawl::Data::HTML::Options }
|
|
1588
|
+
|
|
1589
|
+
# @!method initialize(source:, options: nil, format_: :html)
|
|
1590
|
+
# Crawl pages and return HTML.
|
|
1591
|
+
#
|
|
1592
|
+
# @param source [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Source::StartURL, ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Source::Sitemap] How to find pages to crawl.
|
|
1593
|
+
#
|
|
1594
|
+
# @param options [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Options] Options for HTML output.
|
|
1595
|
+
#
|
|
1596
|
+
# @param format_ [Symbol, :html] Return page content as HTML.
|
|
1597
|
+
|
|
1598
|
+
# How to find pages to crawl.
|
|
1599
|
+
#
|
|
1600
|
+
# @see ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML#source
|
|
1601
|
+
module Source
|
|
1602
|
+
extend ContextDev::Internal::Type::Union
|
|
1603
|
+
|
|
1604
|
+
discriminator :type
|
|
1605
|
+
|
|
1606
|
+
# Discover pages by following links from one URL.
|
|
1607
|
+
variant :start_url, -> { ContextDev::BatchSubmitParams::Input::Crawl::Data::HTML::Source::StartURL }
|
|
1608
|
+
|
|
1609
|
+
# Scrape the pages listed in a domain's sitemap. Links on those pages are not followed.
|
|
1610
|
+
variant :sitemap, -> { ContextDev::BatchSubmitParams::Input::Crawl::Data::HTML::Source::Sitemap }
|
|
1611
|
+
|
|
1612
|
+
class StartURL < ContextDev::Internal::Type::BaseModel
|
|
1613
|
+
# @!attribute type
|
|
1614
|
+
# Start from one page.
|
|
1615
|
+
#
|
|
1616
|
+
# @return [Symbol, :start_url]
|
|
1617
|
+
required :type, const: :start_url
|
|
1618
|
+
|
|
1619
|
+
# @!attribute url
|
|
1620
|
+
# Page where crawling begins. A URL without a scheme is read as https://.
|
|
1621
|
+
#
|
|
1622
|
+
# @return [String]
|
|
1623
|
+
required :url, String
|
|
1624
|
+
|
|
1625
|
+
# @!attribute controls
|
|
1626
|
+
# Limits and filters for page discovery.
|
|
1627
|
+
#
|
|
1628
|
+
# @return [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Source::StartURL::Controls, nil]
|
|
1629
|
+
optional :controls,
|
|
1630
|
+
-> { ContextDev::BatchSubmitParams::Input::Crawl::Data::HTML::Source::StartURL::Controls }
|
|
1631
|
+
|
|
1632
|
+
# @!method initialize(url:, controls: nil, type: :start_url)
|
|
1633
|
+
# Discover pages by following links from one URL.
|
|
1634
|
+
#
|
|
1635
|
+
# @param url [String] Page where crawling begins. A URL without a scheme is read as https://.
|
|
1636
|
+
#
|
|
1637
|
+
# @param controls [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Source::StartURL::Controls] Limits and filters for page discovery.
|
|
1638
|
+
#
|
|
1639
|
+
# @param type [Symbol, :start_url] Start from one page.
|
|
1640
|
+
|
|
1641
|
+
# @see ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Source::StartURL#controls
|
|
1642
|
+
class Controls < ContextDev::Internal::Type::BaseModel
|
|
1643
|
+
# @!attribute follow_subdomains
|
|
1644
|
+
# Follow links to subdomains.
|
|
1645
|
+
#
|
|
1646
|
+
# @return [Boolean, nil]
|
|
1647
|
+
optional :follow_subdomains,
|
|
1648
|
+
ContextDev::Internal::Type::Boolean,
|
|
1649
|
+
api_name: :followSubdomains
|
|
1650
|
+
|
|
1651
|
+
# @!attribute max_depth
|
|
1652
|
+
# Maximum link depth. Source pages are depth 0. No limit when omitted.
|
|
1653
|
+
#
|
|
1654
|
+
# @return [Integer, nil]
|
|
1655
|
+
optional :max_depth, Integer, api_name: :maxDepth
|
|
1656
|
+
|
|
1657
|
+
# @!attribute max_urls
|
|
1658
|
+
# Maximum pages to fetch. Unused reserved credits are refunded. Maximum 25000.
|
|
1659
|
+
#
|
|
1660
|
+
# @return [Integer, nil]
|
|
1661
|
+
optional :max_urls, Integer, api_name: :maxUrls
|
|
1662
|
+
|
|
1663
|
+
# @!attribute regex
|
|
1664
|
+
# RE2 pattern for URLs to include. The `start_url` itself is always included.
|
|
1665
|
+
#
|
|
1666
|
+
# @return [String, nil]
|
|
1667
|
+
optional :regex, String
|
|
1668
|
+
|
|
1669
|
+
# @!method initialize(follow_subdomains: nil, max_depth: nil, max_urls: nil, regex: nil)
|
|
1670
|
+
# Limits and filters for page discovery.
|
|
1671
|
+
#
|
|
1672
|
+
# @param follow_subdomains [Boolean] Follow links to subdomains.
|
|
1673
|
+
#
|
|
1674
|
+
# @param max_depth [Integer] Maximum link depth. Source pages are depth 0. No limit when omitted.
|
|
1675
|
+
#
|
|
1676
|
+
# @param max_urls [Integer] Maximum pages to fetch. Unused reserved credits are refunded. Maximum 25000.
|
|
1677
|
+
#
|
|
1678
|
+
# @param regex [String] RE2 pattern for URLs to include. The `start_url` itself is always included.
|
|
1679
|
+
end
|
|
1680
|
+
end
|
|
1681
|
+
|
|
1682
|
+
class Sitemap < ContextDev::Internal::Type::BaseModel
|
|
1683
|
+
# @!attribute domain
|
|
1684
|
+
# Domain whose sitemap lists the pages to scrape. A full URL is reduced to its
|
|
1685
|
+
# domain.
|
|
1686
|
+
#
|
|
1687
|
+
# @return [String]
|
|
1688
|
+
required :domain, String
|
|
1689
|
+
|
|
1690
|
+
# @!attribute type
|
|
1691
|
+
# Scrape the URLs in the domain's sitemap.
|
|
1692
|
+
#
|
|
1693
|
+
# @return [Symbol, :sitemap]
|
|
1694
|
+
required :type, const: :sitemap
|
|
1695
|
+
|
|
1696
|
+
# @!attribute controls
|
|
1697
|
+
# Limits and filters for the sitemap URLs. A sitemap batch scrapes exactly those
|
|
1698
|
+
# URLs and never follows links off them, so there is no crawl depth here.
|
|
1699
|
+
#
|
|
1700
|
+
# @return [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Source::Sitemap::Controls, nil]
|
|
1701
|
+
optional :controls,
|
|
1702
|
+
-> { ContextDev::BatchSubmitParams::Input::Crawl::Data::HTML::Source::Sitemap::Controls }
|
|
1703
|
+
|
|
1704
|
+
# @!method initialize(domain:, controls: nil, type: :sitemap)
|
|
1705
|
+
# Some parameter documentations has been truncated, see
|
|
1706
|
+
# {ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Source::Sitemap}
|
|
1707
|
+
# for more details.
|
|
1708
|
+
#
|
|
1709
|
+
# Scrape the pages listed in a domain's sitemap. Links on those pages are not
|
|
1710
|
+
# followed.
|
|
1711
|
+
#
|
|
1712
|
+
# @param domain [String] Domain whose sitemap lists the pages to scrape. A full URL is reduced to its dom
|
|
1713
|
+
#
|
|
1714
|
+
# @param controls [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Source::Sitemap::Controls] Limits and filters for the sitemap URLs. A sitemap batch scrapes exactly those U
|
|
1715
|
+
#
|
|
1716
|
+
# @param type [Symbol, :sitemap] Scrape the URLs in the domain's sitemap.
|
|
1717
|
+
|
|
1718
|
+
# @see ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Source::Sitemap#controls
|
|
1719
|
+
class Controls < ContextDev::Internal::Type::BaseModel
|
|
1720
|
+
# @!attribute max_urls
|
|
1721
|
+
# Maximum pages to fetch. Unused reserved credits are refunded. Maximum 25000.
|
|
1722
|
+
#
|
|
1723
|
+
# @return [Integer, nil]
|
|
1724
|
+
optional :max_urls, Integer, api_name: :maxUrls
|
|
1725
|
+
|
|
1726
|
+
# @!attribute regex
|
|
1727
|
+
# RE2 pattern; only sitemap URLs matching it are scraped.
|
|
1728
|
+
#
|
|
1729
|
+
# @return [String, nil]
|
|
1730
|
+
optional :regex, String
|
|
1731
|
+
|
|
1732
|
+
# @!method initialize(max_urls: nil, regex: nil)
|
|
1733
|
+
# Limits and filters for the sitemap URLs. A sitemap batch scrapes exactly those
|
|
1734
|
+
# URLs and never follows links off them, so there is no crawl depth here.
|
|
1735
|
+
#
|
|
1736
|
+
# @param max_urls [Integer] Maximum pages to fetch. Unused reserved credits are refunded. Maximum 25000.
|
|
1737
|
+
#
|
|
1738
|
+
# @param regex [String] RE2 pattern; only sitemap URLs matching it are scraped.
|
|
1739
|
+
end
|
|
1740
|
+
end
|
|
1741
|
+
|
|
1742
|
+
# @!method self.variants
|
|
1743
|
+
# @return [Array(ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Source::StartURL, ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Source::Sitemap)]
|
|
1744
|
+
end
|
|
1745
|
+
|
|
1746
|
+
# @see ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML#options
|
|
1747
|
+
class Options < ContextDev::Internal::Type::BaseModel
|
|
1748
|
+
# @!attribute country
|
|
1749
|
+
# Fetch the target page through a residential proxy in this country (ISO 3166-1
|
|
1750
|
+
# alpha-2).
|
|
1751
|
+
#
|
|
1752
|
+
# @return [Symbol, ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Country, nil]
|
|
1753
|
+
optional :country, enum: -> { ContextDev::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Country }
|
|
1754
|
+
|
|
1755
|
+
# @!attribute exclude_selectors
|
|
1756
|
+
# Remove elements matching these CSS selectors. Applied after `includeSelectors`,
|
|
1757
|
+
# so an element matching both is removed.
|
|
1758
|
+
#
|
|
1759
|
+
# @return [Array<String>, nil]
|
|
1760
|
+
optional :exclude_selectors,
|
|
1761
|
+
ContextDev::Internal::Type::ArrayOf[String],
|
|
1762
|
+
api_name: :excludeSelectors,
|
|
1763
|
+
nil?: true
|
|
1764
|
+
|
|
1765
|
+
# @!attribute include_selectors
|
|
1766
|
+
# Keep only the subtrees matching these CSS selectors. Filtered pages are always
|
|
1767
|
+
# fetched fresh, ignoring `maxAgeMs`.
|
|
1768
|
+
#
|
|
1769
|
+
# @return [Array<String>, nil]
|
|
1770
|
+
optional :include_selectors,
|
|
1771
|
+
ContextDev::Internal::Type::ArrayOf[String],
|
|
1772
|
+
api_name: :includeSelectors,
|
|
1773
|
+
nil?: true
|
|
1774
|
+
|
|
1775
|
+
# @!attribute max_age_ms
|
|
1776
|
+
# Return a cached result if a prior scrape for the same parameters exists and is
|
|
1777
|
+
# younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
|
|
1778
|
+
# omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
|
|
1779
|
+
#
|
|
1780
|
+
# @return [Integer, nil]
|
|
1781
|
+
optional :max_age_ms, Integer, api_name: :maxAgeMs, nil?: true
|
|
1782
|
+
|
|
1783
|
+
# @!attribute pdf
|
|
1784
|
+
# PDF parsing controls. Use start/end to limit text extraction and embedded-image
|
|
1785
|
+
# detection/OCR to an inclusive 1-based page range.
|
|
1786
|
+
#
|
|
1787
|
+
# @return [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf, nil]
|
|
1788
|
+
optional :pdf, -> { ContextDev::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf }
|
|
1789
|
+
|
|
1790
|
+
# @!attribute settle_animations
|
|
1791
|
+
# Wait briefly for CSS and transition animations to settle before extraction, on
|
|
1792
|
+
# pages that render in a browser.
|
|
1793
|
+
#
|
|
1794
|
+
# @return [Boolean, nil]
|
|
1795
|
+
optional :settle_animations, ContextDev::Internal::Type::Boolean, api_name: :settleAnimations
|
|
1796
|
+
|
|
1797
|
+
# @!attribute use_main_content_only
|
|
1798
|
+
# Return the main content without navigation or footers.
|
|
1799
|
+
#
|
|
1800
|
+
# @return [Boolean, nil]
|
|
1801
|
+
optional :use_main_content_only,
|
|
1802
|
+
ContextDev::Internal::Type::Boolean,
|
|
1803
|
+
api_name: :useMainContentOnly
|
|
1804
|
+
|
|
1805
|
+
# @!attribute wait_for_ms
|
|
1806
|
+
# How long to wait after initial page load, in milliseconds. `0` waits 500 ms.
|
|
1807
|
+
#
|
|
1808
|
+
# @return [Integer, nil]
|
|
1809
|
+
optional :wait_for_ms, Integer, api_name: :waitForMs
|
|
1810
|
+
|
|
1811
|
+
# @!method initialize(country: nil, exclude_selectors: nil, include_selectors: nil, max_age_ms: nil, pdf: nil, settle_animations: nil, use_main_content_only: nil, wait_for_ms: nil)
|
|
1812
|
+
# Some parameter documentations has been truncated, see
|
|
1813
|
+
# {ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Options} for
|
|
1814
|
+
# more details.
|
|
1815
|
+
#
|
|
1816
|
+
# Options for HTML output.
|
|
1817
|
+
#
|
|
1818
|
+
# @param country [Symbol, ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Country] Fetch the target page through a residential proxy in this country (ISO 3166-1 al
|
|
1819
|
+
#
|
|
1820
|
+
# @param exclude_selectors [Array<String>, nil] Remove elements matching these CSS selectors. Applied after `includeSelectors`,
|
|
1821
|
+
#
|
|
1822
|
+
# @param include_selectors [Array<String>, nil] Keep only the subtrees matching these CSS selectors. Filtered pages are always f
|
|
1823
|
+
#
|
|
1824
|
+
# @param max_age_ms [Integer, nil] Return a cached result if a prior scrape for the same parameters exists and is y
|
|
1825
|
+
#
|
|
1826
|
+
# @param pdf [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf] PDF parsing controls. Use start/end to limit text extraction and embedded-image
|
|
1827
|
+
#
|
|
1828
|
+
# @param settle_animations [Boolean] Wait briefly for CSS and transition animations to settle before extraction, on p
|
|
1829
|
+
#
|
|
1830
|
+
# @param use_main_content_only [Boolean] Return the main content without navigation or footers.
|
|
1831
|
+
#
|
|
1832
|
+
# @param wait_for_ms [Integer] How long to wait after initial page load, in milliseconds. `0` waits 500 ms.
|
|
1833
|
+
|
|
1834
|
+
# Fetch the target page through a residential proxy in this country (ISO 3166-1
|
|
1835
|
+
# alpha-2).
|
|
1836
|
+
#
|
|
1837
|
+
# @see ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Options#country
|
|
1838
|
+
module Country
|
|
1839
|
+
extend ContextDev::Internal::Type::Enum
|
|
1840
|
+
|
|
1841
|
+
AD = :ad
|
|
1842
|
+
AE = :ae
|
|
1843
|
+
AF = :af
|
|
1844
|
+
AG = :ag
|
|
1845
|
+
AI = :ai
|
|
1846
|
+
AL = :al
|
|
1847
|
+
AM = :am
|
|
1848
|
+
AO = :ao
|
|
1849
|
+
AR = :ar
|
|
1850
|
+
AT = :at
|
|
1851
|
+
AU = :au
|
|
1852
|
+
AW = :aw
|
|
1853
|
+
AZ = :az
|
|
1854
|
+
BA = :ba
|
|
1855
|
+
BB = :bb
|
|
1856
|
+
BD = :bd
|
|
1857
|
+
BE = :be
|
|
1858
|
+
BF = :bf
|
|
1859
|
+
BG = :bg
|
|
1860
|
+
BH = :bh
|
|
1861
|
+
BI = :bi
|
|
1862
|
+
BJ = :bj
|
|
1863
|
+
BM = :bm
|
|
1864
|
+
BN = :bn
|
|
1865
|
+
BO = :bo
|
|
1866
|
+
BQ = :bq
|
|
1867
|
+
BR = :br
|
|
1868
|
+
BS = :bs
|
|
1869
|
+
BW = :bw
|
|
1870
|
+
BY = :by
|
|
1871
|
+
BZ = :bz
|
|
1872
|
+
CA = :ca
|
|
1873
|
+
CD = :cd
|
|
1874
|
+
CF = :cf
|
|
1875
|
+
CG = :cg
|
|
1876
|
+
CH = :ch
|
|
1877
|
+
CI = :ci
|
|
1878
|
+
CL = :cl
|
|
1879
|
+
CM = :cm
|
|
1880
|
+
CN = :cn
|
|
1881
|
+
CO = :co
|
|
1882
|
+
CR = :cr
|
|
1883
|
+
CV = :cv
|
|
1884
|
+
CW = :cw
|
|
1885
|
+
CY = :cy
|
|
1886
|
+
CZ = :cz
|
|
1887
|
+
DE = :de
|
|
1888
|
+
DJ = :dj
|
|
1889
|
+
DK = :dk
|
|
1890
|
+
DM = :dm
|
|
1891
|
+
DO = :do
|
|
1892
|
+
DZ = :dz
|
|
1893
|
+
EC = :ec
|
|
1894
|
+
EE = :ee
|
|
1895
|
+
EG = :eg
|
|
1896
|
+
ES = :es
|
|
1897
|
+
ET = :et
|
|
1898
|
+
FI = :fi
|
|
1899
|
+
FJ = :fj
|
|
1900
|
+
FR = :fr
|
|
1901
|
+
GA = :ga
|
|
1902
|
+
GB = :gb
|
|
1903
|
+
GD = :gd
|
|
1904
|
+
GE = :ge
|
|
1905
|
+
GF = :gf
|
|
1906
|
+
GG = :gg
|
|
1907
|
+
GH = :gh
|
|
1908
|
+
GM = :gm
|
|
1909
|
+
GN = :gn
|
|
1910
|
+
GP = :gp
|
|
1911
|
+
GQ = :gq
|
|
1912
|
+
GR = :gr
|
|
1913
|
+
GT = :gt
|
|
1914
|
+
GU = :gu
|
|
1915
|
+
GW = :gw
|
|
1916
|
+
GY = :gy
|
|
1917
|
+
HK = :hk
|
|
1918
|
+
HN = :hn
|
|
1919
|
+
HR = :hr
|
|
1920
|
+
HT = :ht
|
|
1921
|
+
HU = :hu
|
|
1922
|
+
ID = :id
|
|
1923
|
+
IE = :ie
|
|
1924
|
+
IL = :il
|
|
1925
|
+
IM = :im
|
|
1926
|
+
IN = :in
|
|
1927
|
+
IQ = :iq
|
|
1928
|
+
IR = :ir
|
|
1929
|
+
IS = :is
|
|
1930
|
+
IT = :it
|
|
1931
|
+
JE = :je
|
|
1932
|
+
JM = :jm
|
|
1933
|
+
JO = :jo
|
|
1934
|
+
JP = :jp
|
|
1935
|
+
KE = :ke
|
|
1936
|
+
KG = :kg
|
|
1937
|
+
KH = :kh
|
|
1938
|
+
KN = :kn
|
|
1939
|
+
KR = :kr
|
|
1940
|
+
KW = :kw
|
|
1941
|
+
KY = :ky
|
|
1942
|
+
KZ = :kz
|
|
1943
|
+
LA = :la
|
|
1944
|
+
LB = :lb
|
|
1945
|
+
LC = :lc
|
|
1946
|
+
LK = :lk
|
|
1947
|
+
LR = :lr
|
|
1948
|
+
LS = :ls
|
|
1949
|
+
LT = :lt
|
|
1950
|
+
LU = :lu
|
|
1951
|
+
LV = :lv
|
|
1952
|
+
LY = :ly
|
|
1953
|
+
MA = :ma
|
|
1954
|
+
MC = :mc
|
|
1955
|
+
MD = :md
|
|
1956
|
+
ME = :me
|
|
1957
|
+
MF = :mf
|
|
1958
|
+
MG = :mg
|
|
1959
|
+
MK = :mk
|
|
1960
|
+
ML = :ml
|
|
1961
|
+
MM = :mm
|
|
1962
|
+
MN = :mn
|
|
1963
|
+
MO = :mo
|
|
1964
|
+
MQ = :mq
|
|
1965
|
+
MR = :mr
|
|
1966
|
+
MT = :mt
|
|
1967
|
+
MU = :mu
|
|
1968
|
+
MV = :mv
|
|
1969
|
+
MW = :mw
|
|
1970
|
+
MX = :mx
|
|
1971
|
+
MY = :my
|
|
1972
|
+
MZ = :mz
|
|
1973
|
+
NA = :na
|
|
1974
|
+
NC = :nc
|
|
1975
|
+
NE = :ne
|
|
1976
|
+
NG = :ng
|
|
1977
|
+
NI = :ni
|
|
1978
|
+
NL = :nl
|
|
1979
|
+
NO = :no
|
|
1980
|
+
NP = :np
|
|
1981
|
+
NZ = :nz
|
|
1982
|
+
OM = :om
|
|
1983
|
+
PA = :pa
|
|
1984
|
+
PE = :pe
|
|
1985
|
+
PF = :pf
|
|
1986
|
+
PG = :pg
|
|
1987
|
+
PH = :ph
|
|
1988
|
+
PK = :pk
|
|
1989
|
+
PL = :pl
|
|
1990
|
+
PR = :pr
|
|
1991
|
+
PS = :ps
|
|
1992
|
+
PT = :pt
|
|
1993
|
+
PY = :py
|
|
1994
|
+
QA = :qa
|
|
1995
|
+
RE = :re
|
|
1996
|
+
RO = :ro
|
|
1997
|
+
RS = :rs
|
|
1998
|
+
RU = :ru
|
|
1999
|
+
RW = :rw
|
|
2000
|
+
SA = :sa
|
|
2001
|
+
SC = :sc
|
|
2002
|
+
SD = :sd
|
|
2003
|
+
SE = :se
|
|
2004
|
+
SG = :sg
|
|
2005
|
+
SI = :si
|
|
2006
|
+
SK = :sk
|
|
2007
|
+
SL = :sl
|
|
2008
|
+
SM = :sm
|
|
2009
|
+
SN = :sn
|
|
2010
|
+
SO = :so
|
|
2011
|
+
SR = :sr
|
|
2012
|
+
SS = :ss
|
|
2013
|
+
ST = :st
|
|
2014
|
+
SV = :sv
|
|
2015
|
+
SX = :sx
|
|
2016
|
+
SY = :sy
|
|
2017
|
+
SZ = :sz
|
|
2018
|
+
TC = :tc
|
|
2019
|
+
TD = :td
|
|
2020
|
+
TG = :tg
|
|
2021
|
+
TH = :th
|
|
2022
|
+
TJ = :tj
|
|
2023
|
+
TL = :tl
|
|
2024
|
+
TM = :tm
|
|
2025
|
+
TN = :tn
|
|
2026
|
+
TR = :tr
|
|
2027
|
+
TT = :tt
|
|
2028
|
+
TW = :tw
|
|
2029
|
+
TZ = :tz
|
|
2030
|
+
UA = :ua
|
|
2031
|
+
UG = :ug
|
|
2032
|
+
US = :us
|
|
2033
|
+
UY = :uy
|
|
2034
|
+
UZ = :uz
|
|
2035
|
+
VC = :vc
|
|
2036
|
+
VE = :ve
|
|
2037
|
+
VG = :vg
|
|
2038
|
+
VI = :vi
|
|
2039
|
+
VN = :vn
|
|
2040
|
+
YE = :ye
|
|
2041
|
+
YT = :yt
|
|
2042
|
+
ZA = :za
|
|
2043
|
+
ZM = :zm
|
|
2044
|
+
ZW = :zw
|
|
2045
|
+
|
|
2046
|
+
# @!method self.values
|
|
2047
|
+
# @return [Array<Symbol>]
|
|
2048
|
+
end
|
|
2049
|
+
|
|
2050
|
+
# @see ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Options#pdf
|
|
2051
|
+
class Pdf < ContextDev::Internal::Type::BaseModel
|
|
2052
|
+
# @!attribute end_
|
|
2053
|
+
# Last 1-based PDF page to parse. When omitted, parsing ends at the last page.
|
|
2054
|
+
# Must be greater than or equal to start when both are provided.
|
|
2055
|
+
#
|
|
2056
|
+
# @return [Integer, nil]
|
|
2057
|
+
optional :end_, Integer, api_name: :end
|
|
2058
|
+
|
|
2059
|
+
# @!attribute ocr
|
|
2060
|
+
# When true, OCR the selected PDF pages that have no usable text layer (scans),
|
|
2061
|
+
# replacing each recovered page's text with the OCR result while pages with a real
|
|
2062
|
+
# text layer keep it. Billed at 1 credit per page OCR actually recovered, on top
|
|
2063
|
+
# of the base request cost. When false, no OCR runs.
|
|
2064
|
+
#
|
|
2065
|
+
# @return [Boolean, nil]
|
|
2066
|
+
optional :ocr, ContextDev::Internal::Type::Boolean
|
|
2067
|
+
|
|
2068
|
+
# @!attribute should_parse
|
|
2069
|
+
# When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
|
|
2070
|
+
# a 400 PDF_SKIPPED is returned.
|
|
2071
|
+
#
|
|
2072
|
+
# @return [Boolean, nil]
|
|
2073
|
+
optional :should_parse, ContextDev::Internal::Type::Boolean, api_name: :shouldParse
|
|
2074
|
+
|
|
2075
|
+
# @!attribute start
|
|
2076
|
+
# First 1-based PDF page to parse. When omitted, parsing starts at the first page.
|
|
2077
|
+
#
|
|
2078
|
+
# @return [Integer, nil]
|
|
2079
|
+
optional :start, Integer
|
|
2080
|
+
|
|
2081
|
+
# @!method initialize(end_: nil, ocr: nil, should_parse: nil, start: nil)
|
|
2082
|
+
# Some parameter documentations has been truncated, see
|
|
2083
|
+
# {ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf}
|
|
2084
|
+
# for more details.
|
|
2085
|
+
#
|
|
2086
|
+
# PDF parsing controls. Use start/end to limit text extraction and embedded-image
|
|
2087
|
+
# detection/OCR to an inclusive 1-based page range.
|
|
2088
|
+
#
|
|
2089
|
+
# @param end_ [Integer] Last 1-based PDF page to parse. When omitted, parsing ends at the last page. Mus
|
|
2090
|
+
#
|
|
2091
|
+
# @param ocr [Boolean] When true, OCR the selected PDF pages that have no usable text layer (scans), re
|
|
2092
|
+
#
|
|
2093
|
+
# @param should_parse [Boolean] When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
|
|
2094
|
+
#
|
|
2095
|
+
# @param start [Integer] First 1-based PDF page to parse. When omitted, parsing starts at the first page.
|
|
2096
|
+
end
|
|
2097
|
+
end
|
|
2098
|
+
end
|
|
2099
|
+
|
|
2100
|
+
# @!method self.variants
|
|
2101
|
+
# @return [Array(ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown, ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML)]
|
|
2102
|
+
end
|
|
2103
|
+
end
|
|
2104
|
+
|
|
2105
|
+
# @!method self.variants
|
|
2106
|
+
# @return [Array(ContextDev::Models::BatchSubmitParams::Input::Scrape, ContextDev::Models::BatchSubmitParams::Input::Crawl)]
|
|
53
2107
|
end
|
|
54
2108
|
end
|
|
55
2109
|
end
|