context.dev 2.8.0 → 2.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (116) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +28 -0
  3. data/README.md +1 -1
  4. data/lib/context_dev/client.rb +10 -0
  5. data/lib/context_dev/models/batch_delete_params.rb +22 -0
  6. data/lib/context_dev/models/batch_delete_response.rb +60 -0
  7. data/lib/context_dev/models/batch_get_results_response.rb +46 -4
  8. data/lib/context_dev/models/batch_list_response.rb +13 -4
  9. data/lib/context_dev/models/batch_retrieve_response.rb +13 -4
  10. data/lib/context_dev/models/batch_submit_params.rb +2080 -26
  11. data/lib/context_dev/models/batch_submit_response.rb +125 -531
  12. data/lib/context_dev/models/brand_retrieve_response.rb +50 -1
  13. data/lib/context_dev/models/brand_search_params.rb +41 -3
  14. data/lib/context_dev/models/brand_search_response.rb +3 -2
  15. data/lib/context_dev/models/crawl_controls.rb +21 -15
  16. data/lib/context_dev/models/news_search_params.rb +467 -0
  17. data/lib/context_dev/models/news_search_response.rb +238 -0
  18. data/lib/context_dev/models/parse_handle_params.rb +20 -147
  19. data/lib/context_dev/models/person_enrich_params.rb +176 -0
  20. data/lib/context_dev/models/person_enrich_response.rb +641 -0
  21. data/lib/context_dev/models/utility_prefetch_params.rb +19 -16
  22. data/lib/context_dev/models/utility_prefetch_response.rb +4 -5
  23. data/lib/context_dev/models/web_screenshot_params.rb +3 -30
  24. data/lib/context_dev/models/web_search_response.rb +1 -0
  25. data/lib/context_dev/models/web_web_crawl_md_params.rb +5 -4
  26. data/lib/context_dev/models/web_web_crawl_md_response.rb +30 -1
  27. data/lib/context_dev/models/web_web_scrape_html_params.rb +20 -156
  28. data/lib/context_dev/models/web_web_scrape_html_response.rb +30 -1
  29. data/lib/context_dev/models/web_web_scrape_images_params.rb +12 -124
  30. data/lib/context_dev/models/web_web_scrape_md_params.rb +40 -241
  31. data/lib/context_dev/models/web_web_scrape_md_response.rb +41 -2
  32. data/lib/context_dev/models/web_web_scrape_sitemap_params.rb +11 -1
  33. data/lib/context_dev/models/web_web_scrape_sitemap_response.rb +3 -2
  34. data/lib/context_dev/models.rb +6 -0
  35. data/lib/context_dev/resources/batch.rb +33 -7
  36. data/lib/context_dev/resources/brand.rb +10 -10
  37. data/lib/context_dev/resources/news.rb +51 -0
  38. data/lib/context_dev/resources/parse.rb +5 -5
  39. data/lib/context_dev/resources/people.rb +56 -0
  40. data/lib/context_dev/resources/utility.rb +7 -6
  41. data/lib/context_dev/resources/web.rb +46 -24
  42. data/lib/context_dev/version.rb +1 -1
  43. data/lib/context_dev.rb +8 -0
  44. data/rbi/context_dev/client.rbi +8 -0
  45. data/rbi/context_dev/models/batch_delete_params.rbi +40 -0
  46. data/rbi/context_dev/models/batch_delete_response.rbi +116 -0
  47. data/rbi/context_dev/models/batch_get_results_response.rbi +85 -3
  48. data/rbi/context_dev/models/batch_list_response.rbi +24 -8
  49. data/rbi/context_dev/models/batch_retrieve_response.rbi +24 -8
  50. data/rbi/context_dev/models/batch_submit_params.rbi +6421 -44
  51. data/rbi/context_dev/models/batch_submit_response.rbi +184 -1151
  52. data/rbi/context_dev/models/brand_retrieve_response.rbi +152 -0
  53. data/rbi/context_dev/models/brand_search_params.rbi +71 -2
  54. data/rbi/context_dev/models/brand_search_response.rbi +4 -2
  55. data/rbi/context_dev/models/crawl_controls.rbi +22 -28
  56. data/rbi/context_dev/models/news_search_params.rbi +1294 -0
  57. data/rbi/context_dev/models/news_search_response.rbi +423 -0
  58. data/rbi/context_dev/models/parse_handle_params.rbi +30 -323
  59. data/rbi/context_dev/models/person_enrich_params.rbi +332 -0
  60. data/rbi/context_dev/models/person_enrich_response.rbi +1295 -0
  61. data/rbi/context_dev/models/utility_prefetch_params.rbi +24 -16
  62. data/rbi/context_dev/models/utility_prefetch_response.rbi +8 -6
  63. data/rbi/context_dev/models/web_screenshot_params.rbi +4 -71
  64. data/rbi/context_dev/models/web_search_response.rbi +5 -0
  65. data/rbi/context_dev/models/web_web_crawl_md_params.rbi +8 -6
  66. data/rbi/context_dev/models/web_web_crawl_md_response.rbi +67 -0
  67. data/rbi/context_dev/models/web_web_scrape_html_params.rbi +30 -363
  68. data/rbi/context_dev/models/web_web_scrape_html_response.rbi +65 -0
  69. data/rbi/context_dev/models/web_web_scrape_images_params.rbi +16 -287
  70. data/rbi/context_dev/models/web_web_scrape_md_params.rbi +57 -558
  71. data/rbi/context_dev/models/web_web_scrape_md_response.rbi +80 -0
  72. data/rbi/context_dev/models/web_web_scrape_sitemap_params.rbi +15 -0
  73. data/rbi/context_dev/models/web_web_scrape_sitemap_response.rbi +4 -2
  74. data/rbi/context_dev/models.rbi +6 -0
  75. data/rbi/context_dev/resources/batch.rbi +32 -10
  76. data/rbi/context_dev/resources/brand.rbi +14 -8
  77. data/rbi/context_dev/resources/news.rbi +46 -0
  78. data/rbi/context_dev/resources/parse.rbi +10 -26
  79. data/rbi/context_dev/resources/people.rbi +47 -0
  80. data/rbi/context_dev/resources/utility.rbi +8 -6
  81. data/rbi/context_dev/resources/web.rbi +49 -66
  82. data/sig/context_dev/client.rbs +4 -0
  83. data/sig/context_dev/models/batch_delete_params.rbs +23 -0
  84. data/sig/context_dev/models/batch_delete_response.rbs +57 -0
  85. data/sig/context_dev/models/batch_get_results_response.rbs +30 -2
  86. data/sig/context_dev/models/batch_list_response.rbs +16 -2
  87. data/sig/context_dev/models/batch_retrieve_response.rbs +16 -2
  88. data/sig/context_dev/models/batch_submit_params.rbs +2666 -15
  89. data/sig/context_dev/models/batch_submit_response.rbs +78 -466
  90. data/sig/context_dev/models/brand_retrieve_response.rbs +62 -0
  91. data/sig/context_dev/models/brand_search_params.rbs +38 -1
  92. data/sig/context_dev/models/crawl_controls.rbs +16 -16
  93. data/sig/context_dev/models/news_search_params.rbs +532 -0
  94. data/sig/context_dev/models/news_search_response.rbs +206 -0
  95. data/sig/context_dev/models/parse_handle_params.rbs +25 -90
  96. data/sig/context_dev/models/person_enrich_params.rbs +199 -0
  97. data/sig/context_dev/models/person_enrich_response.rbs +638 -0
  98. data/sig/context_dev/models/utility_prefetch_params.rbs +2 -1
  99. data/sig/context_dev/models/utility_prefetch_response.rbs +2 -1
  100. data/sig/context_dev/models/web_screenshot_params.rbs +5 -18
  101. data/sig/context_dev/models/web_search_response.rbs +2 -0
  102. data/sig/context_dev/models/web_web_crawl_md_response.rbs +21 -0
  103. data/sig/context_dev/models/web_web_scrape_html_params.rbs +24 -94
  104. data/sig/context_dev/models/web_web_scrape_html_response.rbs +21 -0
  105. data/sig/context_dev/models/web_web_scrape_images_params.rbs +20 -72
  106. data/sig/context_dev/models/web_web_scrape_md_params.rbs +46 -148
  107. data/sig/context_dev/models/web_web_scrape_md_response.rbs +28 -0
  108. data/sig/context_dev/models/web_web_scrape_sitemap_params.rbs +7 -0
  109. data/sig/context_dev/models.rbs +6 -0
  110. data/sig/context_dev/resources/batch.rbs +8 -2
  111. data/sig/context_dev/resources/brand.rbs +3 -0
  112. data/sig/context_dev/resources/news.rbs +17 -0
  113. data/sig/context_dev/resources/parse.rbs +5 -5
  114. data/sig/context_dev/resources/people.rbs +19 -0
  115. data/sig/context_dev/resources/web.rbs +13 -11
  116. metadata +26 -2
@@ -7,49 +7,2103 @@ module ContextDev
7
7
  extend ContextDev::Internal::Type::RequestParameters::Converter
8
8
  include ContextDev::Internal::Type::RequestParameters
9
9
 
10
- # @!attribute identifiers
11
- # Known identifiers for the person. At least one identifier is required.
10
+ # @!attribute input
11
+ # Choose a URL list or a site crawl.
12
12
  #
13
- # @return [ContextDev::Models::BatchSubmitParams::Identifiers]
14
- required :identifiers, -> { ContextDev::BatchSubmitParams::Identifiers }
13
+ # @return [ContextDev::Models::BatchSubmitParams::Input::Scrape, ContextDev::Models::BatchSubmitParams::Input::Crawl]
14
+ required :input, union: -> { ContextDev::BatchSubmitParams::Input }
15
15
 
16
16
  # @!attribute tags
17
- # Optional tags for tracking usage. Up to 20 tags, each 1 to 50 characters.
17
+ # Tags stored on the batch. Filter the batch list by them later.
18
18
  #
19
19
  # @return [Array<String>, nil]
20
20
  optional :tags, ContextDev::Internal::Type::ArrayOf[String]
21
21
 
22
- # @!attribute timeout_ms
23
- # Optional timeout in milliseconds for the request. If the request takes longer
24
- # than this value, it will be aborted with a 408 status code. Maximum allowed
25
- # value is 300000ms (5 minutes).
22
+ # @!attribute webhook_url
23
+ # URL notified when the batch finishes.
26
24
  #
27
- # @return [Integer, nil]
28
- optional :timeout_ms, Integer, api_name: :timeoutMS
25
+ # @return [String, nil]
26
+ optional :webhook_url, String, api_name: :webhookUrl
29
27
 
30
- # @!method initialize(identifiers:, tags: nil, timeout_ms: nil, request_options: {})
28
+ # @!attribute idempotency_key
29
+ # Any string unique to this submission. Retries with the same key return the
30
+ # original batch.
31
+ #
32
+ # @return [String, nil]
33
+ optional :idempotency_key, String
34
+
35
+ # @!method initialize(input:, tags: nil, webhook_url: nil, idempotency_key: nil, request_options: {})
31
36
  # Some parameter documentations has been truncated, see
32
37
  # {ContextDev::Models::BatchSubmitParams} for more details.
33
38
  #
34
- # @param identifiers [ContextDev::Models::BatchSubmitParams::Identifiers] Known identifiers for the person. At least one identifier is required.
39
+ # @param input [ContextDev::Models::BatchSubmitParams::Input::Scrape, ContextDev::Models::BatchSubmitParams::Input::Crawl] Choose a URL list or a site crawl.
40
+ #
41
+ # @param tags [Array<String>] Tags stored on the batch. Filter the batch list by them later.
35
42
  #
36
- # @param tags [Array<String>] Optional tags for tracking usage. Up to 20 tags, each 1 to 50 characters.
43
+ # @param webhook_url [String] URL notified when the batch finishes.
37
44
  #
38
- # @param timeout_ms [Integer] Optional timeout in milliseconds for the request. If the request takes longer th
45
+ # @param idempotency_key [String] Any string unique to this submission. Retries with the same key return the origi
39
46
  #
40
47
  # @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}]
41
48
 
42
- class Identifiers < ContextDev::Internal::Type::BaseModel
43
- # @!attribute linkedin_url
44
- # LinkedIn profile URL, e.g. https://www.linkedin.com/in/yahia-bakour/.
45
- #
46
- # @return [String, nil]
47
- optional :linkedin_url, String, api_name: :linkedinUrl
48
-
49
- # @!method initialize(linkedin_url: nil)
50
- # Known identifiers for the person. At least one identifier is required.
51
- #
52
- # @param linkedin_url [String] LinkedIn profile URL, e.g. https://www.linkedin.com/in/yahia-bakour/.
49
+ # Choose a URL list or a site crawl.
50
+ module Input
51
+ extend ContextDev::Internal::Type::Union
52
+
53
+ discriminator :mode
54
+
55
+ # Scrape up to 25K URLs in one batch.
56
+ variant :scrape, -> { ContextDev::BatchSubmitParams::Input::Scrape }
57
+
58
+ # Crawl pages starting from a URL or from a domain's sitemap.
59
+ variant :crawl, -> { ContextDev::BatchSubmitParams::Input::Crawl }
60
+
61
+ class Scrape < ContextDev::Internal::Type::BaseModel
62
+ # @!attribute data
63
+ # Pages to scrape and their output format.
64
+ #
65
+ # @return [ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown, ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML]
66
+ required :data, union: -> { ContextDev::BatchSubmitParams::Input::Scrape::Data }
67
+
68
+ # @!attribute mode
69
+ # Scrape the pages in `data.urls`.
70
+ #
71
+ # @return [Symbol, :scrape]
72
+ required :mode, const: :scrape
73
+
74
+ # @!method initialize(data:, mode: :scrape)
75
+ # Scrape up to 25K URLs in one batch.
76
+ #
77
+ # @param data [ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown, ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML] Pages to scrape and their output format.
78
+ #
79
+ # @param mode [Symbol, :scrape] Scrape the pages in `data.urls`.
80
+
81
+ # Pages to scrape and their output format.
82
+ #
83
+ # @see ContextDev::Models::BatchSubmitParams::Input::Scrape#data
84
+ module Data
85
+ extend ContextDev::Internal::Type::Union
86
+
87
+ discriminator :format
88
+
89
+ # Scrape the listed pages as Markdown.
90
+ variant :markdown, -> { ContextDev::BatchSubmitParams::Input::Scrape::Data::Markdown }
91
+
92
+ # Scrape the listed pages as HTML.
93
+ variant :html, -> { ContextDev::BatchSubmitParams::Input::Scrape::Data::HTML }
94
+
95
+ class Markdown < ContextDev::Internal::Type::BaseModel
96
+ # @!attribute format_
97
+ # Return page content as Markdown.
98
+ #
99
+ # @return [Symbol, :markdown]
100
+ required :format_, const: :markdown, api_name: :format
101
+
102
+ # @!attribute urls
103
+ # Pages to scrape. Maximum 25000.
104
+ #
105
+ # @return [Array<ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::URL>]
106
+ required :urls,
107
+ -> { ContextDev::Internal::Type::ArrayOf[ContextDev::BatchSubmitParams::Input::Scrape::Data::Markdown::URL] }
108
+
109
+ # @!attribute options
110
+ # Options for Markdown output.
111
+ #
112
+ # @return [ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::Options, nil]
113
+ optional :options, -> { ContextDev::BatchSubmitParams::Input::Scrape::Data::Markdown::Options }
114
+
115
+ # @!method initialize(urls:, options: nil, format_: :markdown)
116
+ # Scrape the listed pages as Markdown.
117
+ #
118
+ # @param urls [Array<ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::URL>] Pages to scrape. Maximum 25000.
119
+ #
120
+ # @param options [ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::Options] Options for Markdown output.
121
+ #
122
+ # @param format_ [Symbol, :markdown] Return page content as Markdown.
123
+
124
+ class URL < ContextDev::Internal::Type::BaseModel
125
+ # @!attribute url
126
+ # Page URL to scrape.
127
+ #
128
+ # @return [String]
129
+ required :url, String
130
+
131
+ # @!attribute item_id
132
+ # Your ID for this page, returned with its result. The same URL can use different
133
+ # IDs.
134
+ #
135
+ # @return [String, nil]
136
+ optional :item_id, String, api_name: :itemId
137
+
138
+ # @!attribute meta
139
+ # Custom JSON returned unchanged with this page result.
140
+ #
141
+ # @return [Hash{Symbol=>Object}, nil]
142
+ optional :meta, ContextDev::Internal::Type::HashOf[ContextDev::Internal::Type::Unknown]
143
+
144
+ # @!method initialize(url:, item_id: nil, meta: nil)
145
+ # Some parameter documentations has been truncated, see
146
+ # {ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::URL} for
147
+ # more details.
148
+ #
149
+ # A page to scrape, with optional data for matching results.
150
+ #
151
+ # @param url [String] Page URL to scrape.
152
+ #
153
+ # @param item_id [String] Your ID for this page, returned with its result. The same URL can use different
154
+ #
155
+ # @param meta [Hash{Symbol=>Object}] Custom JSON returned unchanged with this page result.
156
+ end
157
+
158
+ # @see ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown#options
159
+ class Options < ContextDev::Internal::Type::BaseModel
160
+ # @!attribute country
161
+ # Fetch the target page through a residential proxy in this country (ISO 3166-1
162
+ # alpha-2).
163
+ #
164
+ # @return [Symbol, ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Country, nil]
165
+ optional :country,
166
+ enum: -> { ContextDev::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Country }
167
+
168
+ # @!attribute exclude_selectors
169
+ # Remove elements matching these CSS selectors. Applied after `includeSelectors`,
170
+ # so an element matching both is removed.
171
+ #
172
+ # @return [Array<String>, nil]
173
+ optional :exclude_selectors,
174
+ ContextDev::Internal::Type::ArrayOf[String],
175
+ api_name: :excludeSelectors,
176
+ nil?: true
177
+
178
+ # @!attribute include_html
179
+ # Also include each page's HTML in its result record, as an `html` field alongside
180
+ # the Markdown.
181
+ #
182
+ # @return [Boolean, nil]
183
+ optional :include_html, ContextDev::Internal::Type::Boolean, api_name: :includeHTML
184
+
185
+ # @!attribute include_images
186
+ # Include image references in the Markdown.
187
+ #
188
+ # @return [Boolean, nil]
189
+ optional :include_images, ContextDev::Internal::Type::Boolean, api_name: :includeImages
190
+
191
+ # @!attribute include_links
192
+ # Include links in the Markdown.
193
+ #
194
+ # @return [Boolean, nil]
195
+ optional :include_links, ContextDev::Internal::Type::Boolean, api_name: :includeLinks
196
+
197
+ # @!attribute include_selectors
198
+ # Keep only the subtrees matching these CSS selectors. Filtered pages are always
199
+ # fetched fresh, ignoring `maxAgeMs`.
200
+ #
201
+ # @return [Array<String>, nil]
202
+ optional :include_selectors,
203
+ ContextDev::Internal::Type::ArrayOf[String],
204
+ api_name: :includeSelectors,
205
+ nil?: true
206
+
207
+ # @!attribute max_age_ms
208
+ # Return a cached result if a prior scrape for the same parameters exists and is
209
+ # younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
210
+ # omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
211
+ #
212
+ # @return [Integer, nil]
213
+ optional :max_age_ms, Integer, api_name: :maxAgeMs, nil?: true
214
+
215
+ # @!attribute pdf
216
+ # PDF parsing controls. Use start/end to limit text extraction and embedded-image
217
+ # detection/OCR to an inclusive 1-based page range.
218
+ #
219
+ # @return [ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf, nil]
220
+ optional :pdf, -> { ContextDev::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf }
221
+
222
+ # @!attribute settle_animations
223
+ # Wait briefly for CSS and transition animations to settle before extraction, on
224
+ # pages that render in a browser.
225
+ #
226
+ # @return [Boolean, nil]
227
+ optional :settle_animations, ContextDev::Internal::Type::Boolean, api_name: :settleAnimations
228
+
229
+ # @!attribute shorten_base64_images
230
+ # Shorten inline base64 image data.
231
+ #
232
+ # @return [Boolean, nil]
233
+ optional :shorten_base64_images,
234
+ ContextDev::Internal::Type::Boolean,
235
+ api_name: :shortenBase64Images
236
+
237
+ # @!attribute use_main_content_only
238
+ # Return the main content without navigation or footers.
239
+ #
240
+ # @return [Boolean, nil]
241
+ optional :use_main_content_only,
242
+ ContextDev::Internal::Type::Boolean,
243
+ api_name: :useMainContentOnly
244
+
245
+ # @!attribute wait_for_ms
246
+ # How long to wait after initial page load, in milliseconds. `0` waits 500 ms.
247
+ #
248
+ # @return [Integer, nil]
249
+ optional :wait_for_ms, Integer, api_name: :waitForMs
250
+
251
+ # @!method initialize(country: nil, exclude_selectors: nil, include_html: nil, include_images: nil, include_links: nil, include_selectors: nil, max_age_ms: nil, pdf: nil, settle_animations: nil, shorten_base64_images: nil, use_main_content_only: nil, wait_for_ms: nil)
252
+ # Some parameter documentations has been truncated, see
253
+ # {ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::Options}
254
+ # for more details.
255
+ #
256
+ # Options for Markdown output.
257
+ #
258
+ # @param country [Symbol, ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Country] Fetch the target page through a residential proxy in this country (ISO 3166-1 al
259
+ #
260
+ # @param exclude_selectors [Array<String>, nil] Remove elements matching these CSS selectors. Applied after `includeSelectors`,
261
+ #
262
+ # @param include_html [Boolean] Also include each page's HTML in its result record, as an `html` field alongside
263
+ #
264
+ # @param include_images [Boolean] Include image references in the Markdown.
265
+ #
266
+ # @param include_links [Boolean] Include links in the Markdown.
267
+ #
268
+ # @param include_selectors [Array<String>, nil] Keep only the subtrees matching these CSS selectors. Filtered pages are always f
269
+ #
270
+ # @param max_age_ms [Integer, nil] Return a cached result if a prior scrape for the same parameters exists and is y
271
+ #
272
+ # @param pdf [ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf] PDF parsing controls. Use start/end to limit text extraction and embedded-image
273
+ #
274
+ # @param settle_animations [Boolean] Wait briefly for CSS and transition animations to settle before extraction, on p
275
+ #
276
+ # @param shorten_base64_images [Boolean] Shorten inline base64 image data.
277
+ #
278
+ # @param use_main_content_only [Boolean] Return the main content without navigation or footers.
279
+ #
280
+ # @param wait_for_ms [Integer] How long to wait after initial page load, in milliseconds. `0` waits 500 ms.
281
+
282
+ # Fetch the target page through a residential proxy in this country (ISO 3166-1
283
+ # alpha-2).
284
+ #
285
+ # @see ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::Options#country
286
+ module Country
287
+ extend ContextDev::Internal::Type::Enum
288
+
289
+ AD = :ad
290
+ AE = :ae
291
+ AF = :af
292
+ AG = :ag
293
+ AI = :ai
294
+ AL = :al
295
+ AM = :am
296
+ AO = :ao
297
+ AR = :ar
298
+ AT = :at
299
+ AU = :au
300
+ AW = :aw
301
+ AZ = :az
302
+ BA = :ba
303
+ BB = :bb
304
+ BD = :bd
305
+ BE = :be
306
+ BF = :bf
307
+ BG = :bg
308
+ BH = :bh
309
+ BI = :bi
310
+ BJ = :bj
311
+ BM = :bm
312
+ BN = :bn
313
+ BO = :bo
314
+ BQ = :bq
315
+ BR = :br
316
+ BS = :bs
317
+ BW = :bw
318
+ BY = :by
319
+ BZ = :bz
320
+ CA = :ca
321
+ CD = :cd
322
+ CF = :cf
323
+ CG = :cg
324
+ CH = :ch
325
+ CI = :ci
326
+ CL = :cl
327
+ CM = :cm
328
+ CN = :cn
329
+ CO = :co
330
+ CR = :cr
331
+ CV = :cv
332
+ CW = :cw
333
+ CY = :cy
334
+ CZ = :cz
335
+ DE = :de
336
+ DJ = :dj
337
+ DK = :dk
338
+ DM = :dm
339
+ DO = :do
340
+ DZ = :dz
341
+ EC = :ec
342
+ EE = :ee
343
+ EG = :eg
344
+ ES = :es
345
+ ET = :et
346
+ FI = :fi
347
+ FJ = :fj
348
+ FR = :fr
349
+ GA = :ga
350
+ GB = :gb
351
+ GD = :gd
352
+ GE = :ge
353
+ GF = :gf
354
+ GG = :gg
355
+ GH = :gh
356
+ GM = :gm
357
+ GN = :gn
358
+ GP = :gp
359
+ GQ = :gq
360
+ GR = :gr
361
+ GT = :gt
362
+ GU = :gu
363
+ GW = :gw
364
+ GY = :gy
365
+ HK = :hk
366
+ HN = :hn
367
+ HR = :hr
368
+ HT = :ht
369
+ HU = :hu
370
+ ID = :id
371
+ IE = :ie
372
+ IL = :il
373
+ IM = :im
374
+ IN = :in
375
+ IQ = :iq
376
+ IR = :ir
377
+ IS = :is
378
+ IT = :it
379
+ JE = :je
380
+ JM = :jm
381
+ JO = :jo
382
+ JP = :jp
383
+ KE = :ke
384
+ KG = :kg
385
+ KH = :kh
386
+ KN = :kn
387
+ KR = :kr
388
+ KW = :kw
389
+ KY = :ky
390
+ KZ = :kz
391
+ LA = :la
392
+ LB = :lb
393
+ LC = :lc
394
+ LK = :lk
395
+ LR = :lr
396
+ LS = :ls
397
+ LT = :lt
398
+ LU = :lu
399
+ LV = :lv
400
+ LY = :ly
401
+ MA = :ma
402
+ MC = :mc
403
+ MD = :md
404
+ ME = :me
405
+ MF = :mf
406
+ MG = :mg
407
+ MK = :mk
408
+ ML = :ml
409
+ MM = :mm
410
+ MN = :mn
411
+ MO = :mo
412
+ MQ = :mq
413
+ MR = :mr
414
+ MT = :mt
415
+ MU = :mu
416
+ MV = :mv
417
+ MW = :mw
418
+ MX = :mx
419
+ MY = :my
420
+ MZ = :mz
421
+ NA = :na
422
+ NC = :nc
423
+ NE = :ne
424
+ NG = :ng
425
+ NI = :ni
426
+ NL = :nl
427
+ NO = :no
428
+ NP = :np
429
+ NZ = :nz
430
+ OM = :om
431
+ PA = :pa
432
+ PE = :pe
433
+ PF = :pf
434
+ PG = :pg
435
+ PH = :ph
436
+ PK = :pk
437
+ PL = :pl
438
+ PR = :pr
439
+ PS = :ps
440
+ PT = :pt
441
+ PY = :py
442
+ QA = :qa
443
+ RE = :re
444
+ RO = :ro
445
+ RS = :rs
446
+ RU = :ru
447
+ RW = :rw
448
+ SA = :sa
449
+ SC = :sc
450
+ SD = :sd
451
+ SE = :se
452
+ SG = :sg
453
+ SI = :si
454
+ SK = :sk
455
+ SL = :sl
456
+ SM = :sm
457
+ SN = :sn
458
+ SO = :so
459
+ SR = :sr
460
+ SS = :ss
461
+ ST = :st
462
+ SV = :sv
463
+ SX = :sx
464
+ SY = :sy
465
+ SZ = :sz
466
+ TC = :tc
467
+ TD = :td
468
+ TG = :tg
469
+ TH = :th
470
+ TJ = :tj
471
+ TL = :tl
472
+ TM = :tm
473
+ TN = :tn
474
+ TR = :tr
475
+ TT = :tt
476
+ TW = :tw
477
+ TZ = :tz
478
+ UA = :ua
479
+ UG = :ug
480
+ US = :us
481
+ UY = :uy
482
+ UZ = :uz
483
+ VC = :vc
484
+ VE = :ve
485
+ VG = :vg
486
+ VI = :vi
487
+ VN = :vn
488
+ YE = :ye
489
+ YT = :yt
490
+ ZA = :za
491
+ ZM = :zm
492
+ ZW = :zw
493
+
494
+ # @!method self.values
495
+ # @return [Array<Symbol>]
496
+ end
497
+
498
+ # @see ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::Options#pdf
499
+ class Pdf < ContextDev::Internal::Type::BaseModel
500
+ # @!attribute end_
501
+ # Last 1-based PDF page to parse. When omitted, parsing ends at the last page.
502
+ # Must be greater than or equal to start when both are provided.
503
+ #
504
+ # @return [Integer, nil]
505
+ optional :end_, Integer, api_name: :end
506
+
507
+ # @!attribute ocr
508
+ # When true, OCR the selected PDF pages that have no usable text layer (scans),
509
+ # replacing each recovered page's text with the OCR result while pages with a real
510
+ # text layer keep it. Billed at 1 credit per page OCR actually recovered, on top
511
+ # of the base request cost. When false, no OCR runs.
512
+ #
513
+ # @return [Boolean, nil]
514
+ optional :ocr, ContextDev::Internal::Type::Boolean
515
+
516
+ # @!attribute should_parse
517
+ # When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
518
+ # a 400 PDF_SKIPPED is returned.
519
+ #
520
+ # @return [Boolean, nil]
521
+ optional :should_parse, ContextDev::Internal::Type::Boolean, api_name: :shouldParse
522
+
523
+ # @!attribute start
524
+ # First 1-based PDF page to parse. When omitted, parsing starts at the first page.
525
+ #
526
+ # @return [Integer, nil]
527
+ optional :start, Integer
528
+
529
+ # @!method initialize(end_: nil, ocr: nil, should_parse: nil, start: nil)
530
+ # Some parameter documentations has been truncated, see
531
+ # {ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf}
532
+ # for more details.
533
+ #
534
+ # PDF parsing controls. Use start/end to limit text extraction and embedded-image
535
+ # detection/OCR to an inclusive 1-based page range.
536
+ #
537
+ # @param end_ [Integer] Last 1-based PDF page to parse. When omitted, parsing ends at the last page. Mus
538
+ #
539
+ # @param ocr [Boolean] When true, OCR the selected PDF pages that have no usable text layer (scans), re
540
+ #
541
+ # @param should_parse [Boolean] When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
542
+ #
543
+ # @param start [Integer] First 1-based PDF page to parse. When omitted, parsing starts at the first page.
544
+ end
545
+ end
546
+ end
547
+
548
+ class HTML < ContextDev::Internal::Type::BaseModel
549
+ # @!attribute format_
550
+ # Return page content as HTML.
551
+ #
552
+ # @return [Symbol, :html]
553
+ required :format_, const: :html, api_name: :format
554
+
555
+ # @!attribute urls
556
+ # Pages to scrape. Maximum 25000.
557
+ #
558
+ # @return [Array<ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::URL>]
559
+ required :urls,
560
+ -> { ContextDev::Internal::Type::ArrayOf[ContextDev::BatchSubmitParams::Input::Scrape::Data::HTML::URL] }
561
+
562
+ # @!attribute options
563
+ # Options for HTML output.
564
+ #
565
+ # @return [ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::Options, nil]
566
+ optional :options, -> { ContextDev::BatchSubmitParams::Input::Scrape::Data::HTML::Options }
567
+
568
+ # @!method initialize(urls:, options: nil, format_: :html)
569
+ # Scrape the listed pages as HTML.
570
+ #
571
+ # @param urls [Array<ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::URL>] Pages to scrape. Maximum 25000.
572
+ #
573
+ # @param options [ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::Options] Options for HTML output.
574
+ #
575
+ # @param format_ [Symbol, :html] Return page content as HTML.
576
+
577
+ class URL < ContextDev::Internal::Type::BaseModel
578
+ # @!attribute url
579
+ # Page URL to scrape.
580
+ #
581
+ # @return [String]
582
+ required :url, String
583
+
584
+ # @!attribute item_id
585
+ # Your ID for this page, returned with its result. The same URL can use different
586
+ # IDs.
587
+ #
588
+ # @return [String, nil]
589
+ optional :item_id, String, api_name: :itemId
590
+
591
+ # @!attribute meta
592
+ # Custom JSON returned unchanged with this page result.
593
+ #
594
+ # @return [Hash{Symbol=>Object}, nil]
595
+ optional :meta, ContextDev::Internal::Type::HashOf[ContextDev::Internal::Type::Unknown]
596
+
597
+ # @!method initialize(url:, item_id: nil, meta: nil)
598
+ # Some parameter documentations has been truncated, see
599
+ # {ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::URL} for more
600
+ # details.
601
+ #
602
+ # A page to scrape, with optional data for matching results.
603
+ #
604
+ # @param url [String] Page URL to scrape.
605
+ #
606
+ # @param item_id [String] Your ID for this page, returned with its result. The same URL can use different
607
+ #
608
+ # @param meta [Hash{Symbol=>Object}] Custom JSON returned unchanged with this page result.
609
+ end
610
+
611
+ # @see ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML#options
612
+ class Options < ContextDev::Internal::Type::BaseModel
613
+ # @!attribute country
614
+ # Fetch the target page through a residential proxy in this country (ISO 3166-1
615
+ # alpha-2).
616
+ #
617
+ # @return [Symbol, ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Country, nil]
618
+ optional :country, enum: -> { ContextDev::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Country }
619
+
620
+ # @!attribute exclude_selectors
621
+ # Remove elements matching these CSS selectors. Applied after `includeSelectors`,
622
+ # so an element matching both is removed.
623
+ #
624
+ # @return [Array<String>, nil]
625
+ optional :exclude_selectors,
626
+ ContextDev::Internal::Type::ArrayOf[String],
627
+ api_name: :excludeSelectors,
628
+ nil?: true
629
+
630
+ # @!attribute include_selectors
631
+ # Keep only the subtrees matching these CSS selectors. Filtered pages are always
632
+ # fetched fresh, ignoring `maxAgeMs`.
633
+ #
634
+ # @return [Array<String>, nil]
635
+ optional :include_selectors,
636
+ ContextDev::Internal::Type::ArrayOf[String],
637
+ api_name: :includeSelectors,
638
+ nil?: true
639
+
640
+ # @!attribute max_age_ms
641
+ # Return a cached result if a prior scrape for the same parameters exists and is
642
+ # younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
643
+ # omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
644
+ #
645
+ # @return [Integer, nil]
646
+ optional :max_age_ms, Integer, api_name: :maxAgeMs, nil?: true
647
+
648
+ # @!attribute pdf
649
+ # PDF parsing controls. Use start/end to limit text extraction and embedded-image
650
+ # detection/OCR to an inclusive 1-based page range.
651
+ #
652
+ # @return [ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf, nil]
653
+ optional :pdf, -> { ContextDev::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf }
654
+
655
+ # @!attribute settle_animations
656
+ # Wait briefly for CSS and transition animations to settle before extraction, on
657
+ # pages that render in a browser.
658
+ #
659
+ # @return [Boolean, nil]
660
+ optional :settle_animations, ContextDev::Internal::Type::Boolean, api_name: :settleAnimations
661
+
662
+ # @!attribute use_main_content_only
663
+ # Return the main content without navigation or footers.
664
+ #
665
+ # @return [Boolean, nil]
666
+ optional :use_main_content_only,
667
+ ContextDev::Internal::Type::Boolean,
668
+ api_name: :useMainContentOnly
669
+
670
+ # @!attribute wait_for_ms
671
+ # How long to wait after initial page load, in milliseconds. `0` waits 500 ms.
672
+ #
673
+ # @return [Integer, nil]
674
+ optional :wait_for_ms, Integer, api_name: :waitForMs
675
+
676
+ # @!method initialize(country: nil, exclude_selectors: nil, include_selectors: nil, max_age_ms: nil, pdf: nil, settle_animations: nil, use_main_content_only: nil, wait_for_ms: nil)
677
+ # Some parameter documentations has been truncated, see
678
+ # {ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::Options} for
679
+ # more details.
680
+ #
681
+ # Options for HTML output.
682
+ #
683
+ # @param country [Symbol, ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Country] Fetch the target page through a residential proxy in this country (ISO 3166-1 al
684
+ #
685
+ # @param exclude_selectors [Array<String>, nil] Remove elements matching these CSS selectors. Applied after `includeSelectors`,
686
+ #
687
+ # @param include_selectors [Array<String>, nil] Keep only the subtrees matching these CSS selectors. Filtered pages are always f
688
+ #
689
+ # @param max_age_ms [Integer, nil] Return a cached result if a prior scrape for the same parameters exists and is y
690
+ #
691
+ # @param pdf [ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf] PDF parsing controls. Use start/end to limit text extraction and embedded-image
692
+ #
693
+ # @param settle_animations [Boolean] Wait briefly for CSS and transition animations to settle before extraction, on p
694
+ #
695
+ # @param use_main_content_only [Boolean] Return the main content without navigation or footers.
696
+ #
697
+ # @param wait_for_ms [Integer] How long to wait after initial page load, in milliseconds. `0` waits 500 ms.
698
+
699
+ # Fetch the target page through a residential proxy in this country (ISO 3166-1
700
+ # alpha-2).
701
+ #
702
+ # @see ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::Options#country
703
+ module Country
704
+ extend ContextDev::Internal::Type::Enum
705
+
706
+ AD = :ad
707
+ AE = :ae
708
+ AF = :af
709
+ AG = :ag
710
+ AI = :ai
711
+ AL = :al
712
+ AM = :am
713
+ AO = :ao
714
+ AR = :ar
715
+ AT = :at
716
+ AU = :au
717
+ AW = :aw
718
+ AZ = :az
719
+ BA = :ba
720
+ BB = :bb
721
+ BD = :bd
722
+ BE = :be
723
+ BF = :bf
724
+ BG = :bg
725
+ BH = :bh
726
+ BI = :bi
727
+ BJ = :bj
728
+ BM = :bm
729
+ BN = :bn
730
+ BO = :bo
731
+ BQ = :bq
732
+ BR = :br
733
+ BS = :bs
734
+ BW = :bw
735
+ BY = :by
736
+ BZ = :bz
737
+ CA = :ca
738
+ CD = :cd
739
+ CF = :cf
740
+ CG = :cg
741
+ CH = :ch
742
+ CI = :ci
743
+ CL = :cl
744
+ CM = :cm
745
+ CN = :cn
746
+ CO = :co
747
+ CR = :cr
748
+ CV = :cv
749
+ CW = :cw
750
+ CY = :cy
751
+ CZ = :cz
752
+ DE = :de
753
+ DJ = :dj
754
+ DK = :dk
755
+ DM = :dm
756
+ DO = :do
757
+ DZ = :dz
758
+ EC = :ec
759
+ EE = :ee
760
+ EG = :eg
761
+ ES = :es
762
+ ET = :et
763
+ FI = :fi
764
+ FJ = :fj
765
+ FR = :fr
766
+ GA = :ga
767
+ GB = :gb
768
+ GD = :gd
769
+ GE = :ge
770
+ GF = :gf
771
+ GG = :gg
772
+ GH = :gh
773
+ GM = :gm
774
+ GN = :gn
775
+ GP = :gp
776
+ GQ = :gq
777
+ GR = :gr
778
+ GT = :gt
779
+ GU = :gu
780
+ GW = :gw
781
+ GY = :gy
782
+ HK = :hk
783
+ HN = :hn
784
+ HR = :hr
785
+ HT = :ht
786
+ HU = :hu
787
+ ID = :id
788
+ IE = :ie
789
+ IL = :il
790
+ IM = :im
791
+ IN = :in
792
+ IQ = :iq
793
+ IR = :ir
794
+ IS = :is
795
+ IT = :it
796
+ JE = :je
797
+ JM = :jm
798
+ JO = :jo
799
+ JP = :jp
800
+ KE = :ke
801
+ KG = :kg
802
+ KH = :kh
803
+ KN = :kn
804
+ KR = :kr
805
+ KW = :kw
806
+ KY = :ky
807
+ KZ = :kz
808
+ LA = :la
809
+ LB = :lb
810
+ LC = :lc
811
+ LK = :lk
812
+ LR = :lr
813
+ LS = :ls
814
+ LT = :lt
815
+ LU = :lu
816
+ LV = :lv
817
+ LY = :ly
818
+ MA = :ma
819
+ MC = :mc
820
+ MD = :md
821
+ ME = :me
822
+ MF = :mf
823
+ MG = :mg
824
+ MK = :mk
825
+ ML = :ml
826
+ MM = :mm
827
+ MN = :mn
828
+ MO = :mo
829
+ MQ = :mq
830
+ MR = :mr
831
+ MT = :mt
832
+ MU = :mu
833
+ MV = :mv
834
+ MW = :mw
835
+ MX = :mx
836
+ MY = :my
837
+ MZ = :mz
838
+ NA = :na
839
+ NC = :nc
840
+ NE = :ne
841
+ NG = :ng
842
+ NI = :ni
843
+ NL = :nl
844
+ NO = :no
845
+ NP = :np
846
+ NZ = :nz
847
+ OM = :om
848
+ PA = :pa
849
+ PE = :pe
850
+ PF = :pf
851
+ PG = :pg
852
+ PH = :ph
853
+ PK = :pk
854
+ PL = :pl
855
+ PR = :pr
856
+ PS = :ps
857
+ PT = :pt
858
+ PY = :py
859
+ QA = :qa
860
+ RE = :re
861
+ RO = :ro
862
+ RS = :rs
863
+ RU = :ru
864
+ RW = :rw
865
+ SA = :sa
866
+ SC = :sc
867
+ SD = :sd
868
+ SE = :se
869
+ SG = :sg
870
+ SI = :si
871
+ SK = :sk
872
+ SL = :sl
873
+ SM = :sm
874
+ SN = :sn
875
+ SO = :so
876
+ SR = :sr
877
+ SS = :ss
878
+ ST = :st
879
+ SV = :sv
880
+ SX = :sx
881
+ SY = :sy
882
+ SZ = :sz
883
+ TC = :tc
884
+ TD = :td
885
+ TG = :tg
886
+ TH = :th
887
+ TJ = :tj
888
+ TL = :tl
889
+ TM = :tm
890
+ TN = :tn
891
+ TR = :tr
892
+ TT = :tt
893
+ TW = :tw
894
+ TZ = :tz
895
+ UA = :ua
896
+ UG = :ug
897
+ US = :us
898
+ UY = :uy
899
+ UZ = :uz
900
+ VC = :vc
901
+ VE = :ve
902
+ VG = :vg
903
+ VI = :vi
904
+ VN = :vn
905
+ YE = :ye
906
+ YT = :yt
907
+ ZA = :za
908
+ ZM = :zm
909
+ ZW = :zw
910
+
911
+ # @!method self.values
912
+ # @return [Array<Symbol>]
913
+ end
914
+
915
+ # @see ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::Options#pdf
916
+ class Pdf < ContextDev::Internal::Type::BaseModel
917
+ # @!attribute end_
918
+ # Last 1-based PDF page to parse. When omitted, parsing ends at the last page.
919
+ # Must be greater than or equal to start when both are provided.
920
+ #
921
+ # @return [Integer, nil]
922
+ optional :end_, Integer, api_name: :end
923
+
924
+ # @!attribute ocr
925
+ # When true, OCR the selected PDF pages that have no usable text layer (scans),
926
+ # replacing each recovered page's text with the OCR result while pages with a real
927
+ # text layer keep it. Billed at 1 credit per page OCR actually recovered, on top
928
+ # of the base request cost. When false, no OCR runs.
929
+ #
930
+ # @return [Boolean, nil]
931
+ optional :ocr, ContextDev::Internal::Type::Boolean
932
+
933
+ # @!attribute should_parse
934
+ # When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
935
+ # a 400 PDF_SKIPPED is returned.
936
+ #
937
+ # @return [Boolean, nil]
938
+ optional :should_parse, ContextDev::Internal::Type::Boolean, api_name: :shouldParse
939
+
940
+ # @!attribute start
941
+ # First 1-based PDF page to parse. When omitted, parsing starts at the first page.
942
+ #
943
+ # @return [Integer, nil]
944
+ optional :start, Integer
945
+
946
+ # @!method initialize(end_: nil, ocr: nil, should_parse: nil, start: nil)
947
+ # Some parameter documentations has been truncated, see
948
+ # {ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf}
949
+ # for more details.
950
+ #
951
+ # PDF parsing controls. Use start/end to limit text extraction and embedded-image
952
+ # detection/OCR to an inclusive 1-based page range.
953
+ #
954
+ # @param end_ [Integer] Last 1-based PDF page to parse. When omitted, parsing ends at the last page. Mus
955
+ #
956
+ # @param ocr [Boolean] When true, OCR the selected PDF pages that have no usable text layer (scans), re
957
+ #
958
+ # @param should_parse [Boolean] When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
959
+ #
960
+ # @param start [Integer] First 1-based PDF page to parse. When omitted, parsing starts at the first page.
961
+ end
962
+ end
963
+ end
964
+
965
+ # @!method self.variants
966
+ # @return [Array(ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown, ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML)]
967
+ end
968
+ end
969
+
970
+ class Crawl < ContextDev::Internal::Type::BaseModel
971
+ # @!attribute data
972
+ # Crawl source and output format.
973
+ #
974
+ # @return [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown, ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML]
975
+ required :data, union: -> { ContextDev::BatchSubmitParams::Input::Crawl::Data }
976
+
977
+ # @!attribute mode
978
+ # Discover and scrape pages from `data.source`.
979
+ #
980
+ # @return [Symbol, :crawl]
981
+ required :mode, const: :crawl
982
+
983
+ # @!method initialize(data:, mode: :crawl)
984
+ # Crawl pages starting from a URL or from a domain's sitemap.
985
+ #
986
+ # @param data [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown, ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML] Crawl source and output format.
987
+ #
988
+ # @param mode [Symbol, :crawl] Discover and scrape pages from `data.source`.
989
+
990
+ # Crawl source and output format.
991
+ #
992
+ # @see ContextDev::Models::BatchSubmitParams::Input::Crawl#data
993
+ module Data
994
+ extend ContextDev::Internal::Type::Union
995
+
996
+ discriminator :format
997
+
998
+ # Crawl pages and return Markdown.
999
+ variant :markdown, -> { ContextDev::BatchSubmitParams::Input::Crawl::Data::Markdown }
1000
+
1001
+ # Crawl pages and return HTML.
1002
+ variant :html, -> { ContextDev::BatchSubmitParams::Input::Crawl::Data::HTML }
1003
+
1004
+ class Markdown < ContextDev::Internal::Type::BaseModel
1005
+ # @!attribute format_
1006
+ # Return page content as Markdown.
1007
+ #
1008
+ # @return [Symbol, :markdown]
1009
+ required :format_, const: :markdown, api_name: :format
1010
+
1011
+ # @!attribute source
1012
+ # How to find pages to crawl.
1013
+ #
1014
+ # @return [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Source::StartURL, ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Source::Sitemap]
1015
+ required :source, union: -> { ContextDev::BatchSubmitParams::Input::Crawl::Data::Markdown::Source }
1016
+
1017
+ # @!attribute options
1018
+ # Options for Markdown output.
1019
+ #
1020
+ # @return [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Options, nil]
1021
+ optional :options, -> { ContextDev::BatchSubmitParams::Input::Crawl::Data::Markdown::Options }
1022
+
1023
+ # @!method initialize(source:, options: nil, format_: :markdown)
1024
+ # Crawl pages and return Markdown.
1025
+ #
1026
+ # @param source [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Source::StartURL, ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Source::Sitemap] How to find pages to crawl.
1027
+ #
1028
+ # @param options [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Options] Options for Markdown output.
1029
+ #
1030
+ # @param format_ [Symbol, :markdown] Return page content as Markdown.
1031
+
1032
+ # How to find pages to crawl.
1033
+ #
1034
+ # @see ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown#source
1035
+ module Source
1036
+ extend ContextDev::Internal::Type::Union
1037
+
1038
+ discriminator :type
1039
+
1040
+ # Discover pages by following links from one URL.
1041
+ variant :start_url, -> { ContextDev::BatchSubmitParams::Input::Crawl::Data::Markdown::Source::StartURL }
1042
+
1043
+ # Scrape the pages listed in a domain's sitemap. Links on those pages are not followed.
1044
+ variant :sitemap, -> { ContextDev::BatchSubmitParams::Input::Crawl::Data::Markdown::Source::Sitemap }
1045
+
1046
+ class StartURL < ContextDev::Internal::Type::BaseModel
1047
+ # @!attribute type
1048
+ # Start from one page.
1049
+ #
1050
+ # @return [Symbol, :start_url]
1051
+ required :type, const: :start_url
1052
+
1053
+ # @!attribute url
1054
+ # Page where crawling begins. A URL without a scheme is read as https://.
1055
+ #
1056
+ # @return [String]
1057
+ required :url, String
1058
+
1059
+ # @!attribute controls
1060
+ # Limits and filters for page discovery.
1061
+ #
1062
+ # @return [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Source::StartURL::Controls, nil]
1063
+ optional :controls,
1064
+ -> { ContextDev::BatchSubmitParams::Input::Crawl::Data::Markdown::Source::StartURL::Controls }
1065
+
1066
+ # @!method initialize(url:, controls: nil, type: :start_url)
1067
+ # Discover pages by following links from one URL.
1068
+ #
1069
+ # @param url [String] Page where crawling begins. A URL without a scheme is read as https://.
1070
+ #
1071
+ # @param controls [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Source::StartURL::Controls] Limits and filters for page discovery.
1072
+ #
1073
+ # @param type [Symbol, :start_url] Start from one page.
1074
+
1075
+ # @see ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Source::StartURL#controls
1076
+ class Controls < ContextDev::Internal::Type::BaseModel
1077
+ # @!attribute follow_subdomains
1078
+ # Follow links to subdomains.
1079
+ #
1080
+ # @return [Boolean, nil]
1081
+ optional :follow_subdomains,
1082
+ ContextDev::Internal::Type::Boolean,
1083
+ api_name: :followSubdomains
1084
+
1085
+ # @!attribute max_depth
1086
+ # Maximum link depth. Source pages are depth 0. No limit when omitted.
1087
+ #
1088
+ # @return [Integer, nil]
1089
+ optional :max_depth, Integer, api_name: :maxDepth
1090
+
1091
+ # @!attribute max_urls
1092
+ # Maximum pages to fetch. Unused reserved credits are refunded. Maximum 25000.
1093
+ #
1094
+ # @return [Integer, nil]
1095
+ optional :max_urls, Integer, api_name: :maxUrls
1096
+
1097
+ # @!attribute regex
1098
+ # RE2 pattern for URLs to include. The `start_url` itself is always included.
1099
+ #
1100
+ # @return [String, nil]
1101
+ optional :regex, String
1102
+
1103
+ # @!method initialize(follow_subdomains: nil, max_depth: nil, max_urls: nil, regex: nil)
1104
+ # Limits and filters for page discovery.
1105
+ #
1106
+ # @param follow_subdomains [Boolean] Follow links to subdomains.
1107
+ #
1108
+ # @param max_depth [Integer] Maximum link depth. Source pages are depth 0. No limit when omitted.
1109
+ #
1110
+ # @param max_urls [Integer] Maximum pages to fetch. Unused reserved credits are refunded. Maximum 25000.
1111
+ #
1112
+ # @param regex [String] RE2 pattern for URLs to include. The `start_url` itself is always included.
1113
+ end
1114
+ end
1115
+
1116
+ class Sitemap < ContextDev::Internal::Type::BaseModel
1117
+ # @!attribute domain
1118
+ # Domain whose sitemap lists the pages to scrape. A full URL is reduced to its
1119
+ # domain.
1120
+ #
1121
+ # @return [String]
1122
+ required :domain, String
1123
+
1124
+ # @!attribute type
1125
+ # Scrape the URLs in the domain's sitemap.
1126
+ #
1127
+ # @return [Symbol, :sitemap]
1128
+ required :type, const: :sitemap
1129
+
1130
+ # @!attribute controls
1131
+ # Limits and filters for the sitemap URLs. A sitemap batch scrapes exactly those
1132
+ # URLs and never follows links off them, so there is no crawl depth here.
1133
+ #
1134
+ # @return [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Source::Sitemap::Controls, nil]
1135
+ optional :controls,
1136
+ -> { ContextDev::BatchSubmitParams::Input::Crawl::Data::Markdown::Source::Sitemap::Controls }
1137
+
1138
+ # @!method initialize(domain:, controls: nil, type: :sitemap)
1139
+ # Some parameter documentations has been truncated, see
1140
+ # {ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Source::Sitemap}
1141
+ # for more details.
1142
+ #
1143
+ # Scrape the pages listed in a domain's sitemap. Links on those pages are not
1144
+ # followed.
1145
+ #
1146
+ # @param domain [String] Domain whose sitemap lists the pages to scrape. A full URL is reduced to its dom
1147
+ #
1148
+ # @param controls [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Source::Sitemap::Controls] Limits and filters for the sitemap URLs. A sitemap batch scrapes exactly those U
1149
+ #
1150
+ # @param type [Symbol, :sitemap] Scrape the URLs in the domain's sitemap.
1151
+
1152
+ # @see ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Source::Sitemap#controls
1153
+ class Controls < ContextDev::Internal::Type::BaseModel
1154
+ # @!attribute max_urls
1155
+ # Maximum pages to fetch. Unused reserved credits are refunded. Maximum 25000.
1156
+ #
1157
+ # @return [Integer, nil]
1158
+ optional :max_urls, Integer, api_name: :maxUrls
1159
+
1160
+ # @!attribute regex
1161
+ # RE2 pattern; only sitemap URLs matching it are scraped.
1162
+ #
1163
+ # @return [String, nil]
1164
+ optional :regex, String
1165
+
1166
+ # @!method initialize(max_urls: nil, regex: nil)
1167
+ # Limits and filters for the sitemap URLs. A sitemap batch scrapes exactly those
1168
+ # URLs and never follows links off them, so there is no crawl depth here.
1169
+ #
1170
+ # @param max_urls [Integer] Maximum pages to fetch. Unused reserved credits are refunded. Maximum 25000.
1171
+ #
1172
+ # @param regex [String] RE2 pattern; only sitemap URLs matching it are scraped.
1173
+ end
1174
+ end
1175
+
1176
+ # @!method self.variants
1177
+ # @return [Array(ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Source::StartURL, ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Source::Sitemap)]
1178
+ end
1179
+
1180
+ # @see ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown#options
1181
+ class Options < ContextDev::Internal::Type::BaseModel
1182
+ # @!attribute country
1183
+ # Fetch the target page through a residential proxy in this country (ISO 3166-1
1184
+ # alpha-2).
1185
+ #
1186
+ # @return [Symbol, ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Country, nil]
1187
+ optional :country,
1188
+ enum: -> { ContextDev::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Country }
1189
+
1190
+ # @!attribute exclude_selectors
1191
+ # Remove elements matching these CSS selectors. Applied after `includeSelectors`,
1192
+ # so an element matching both is removed.
1193
+ #
1194
+ # @return [Array<String>, nil]
1195
+ optional :exclude_selectors,
1196
+ ContextDev::Internal::Type::ArrayOf[String],
1197
+ api_name: :excludeSelectors,
1198
+ nil?: true
1199
+
1200
+ # @!attribute include_html
1201
+ # Also include each page's HTML in its result record, as an `html` field alongside
1202
+ # the Markdown.
1203
+ #
1204
+ # @return [Boolean, nil]
1205
+ optional :include_html, ContextDev::Internal::Type::Boolean, api_name: :includeHTML
1206
+
1207
+ # @!attribute include_images
1208
+ # Include image references in the Markdown.
1209
+ #
1210
+ # @return [Boolean, nil]
1211
+ optional :include_images, ContextDev::Internal::Type::Boolean, api_name: :includeImages
1212
+
1213
+ # @!attribute include_links
1214
+ # Include links in the Markdown.
1215
+ #
1216
+ # @return [Boolean, nil]
1217
+ optional :include_links, ContextDev::Internal::Type::Boolean, api_name: :includeLinks
1218
+
1219
+ # @!attribute include_selectors
1220
+ # Keep only the subtrees matching these CSS selectors. Filtered pages are always
1221
+ # fetched fresh, ignoring `maxAgeMs`.
1222
+ #
1223
+ # @return [Array<String>, nil]
1224
+ optional :include_selectors,
1225
+ ContextDev::Internal::Type::ArrayOf[String],
1226
+ api_name: :includeSelectors,
1227
+ nil?: true
1228
+
1229
+ # @!attribute max_age_ms
1230
+ # Return a cached result if a prior scrape for the same parameters exists and is
1231
+ # younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
1232
+ # omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
1233
+ #
1234
+ # @return [Integer, nil]
1235
+ optional :max_age_ms, Integer, api_name: :maxAgeMs, nil?: true
1236
+
1237
+ # @!attribute pdf
1238
+ # PDF parsing controls. Use start/end to limit text extraction and embedded-image
1239
+ # detection/OCR to an inclusive 1-based page range.
1240
+ #
1241
+ # @return [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf, nil]
1242
+ optional :pdf, -> { ContextDev::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf }
1243
+
1244
+ # @!attribute settle_animations
1245
+ # Wait briefly for CSS and transition animations to settle before extraction, on
1246
+ # pages that render in a browser.
1247
+ #
1248
+ # @return [Boolean, nil]
1249
+ optional :settle_animations, ContextDev::Internal::Type::Boolean, api_name: :settleAnimations
1250
+
1251
+ # @!attribute shorten_base64_images
1252
+ # Shorten inline base64 image data.
1253
+ #
1254
+ # @return [Boolean, nil]
1255
+ optional :shorten_base64_images,
1256
+ ContextDev::Internal::Type::Boolean,
1257
+ api_name: :shortenBase64Images
1258
+
1259
+ # @!attribute use_main_content_only
1260
+ # Return the main content without navigation or footers.
1261
+ #
1262
+ # @return [Boolean, nil]
1263
+ optional :use_main_content_only,
1264
+ ContextDev::Internal::Type::Boolean,
1265
+ api_name: :useMainContentOnly
1266
+
1267
+ # @!attribute wait_for_ms
1268
+ # How long to wait after initial page load, in milliseconds. `0` waits 500 ms.
1269
+ #
1270
+ # @return [Integer, nil]
1271
+ optional :wait_for_ms, Integer, api_name: :waitForMs
1272
+
1273
+ # @!method initialize(country: nil, exclude_selectors: nil, include_html: nil, include_images: nil, include_links: nil, include_selectors: nil, max_age_ms: nil, pdf: nil, settle_animations: nil, shorten_base64_images: nil, use_main_content_only: nil, wait_for_ms: nil)
1274
+ # Some parameter documentations has been truncated, see
1275
+ # {ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Options}
1276
+ # for more details.
1277
+ #
1278
+ # Options for Markdown output.
1279
+ #
1280
+ # @param country [Symbol, ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Country] Fetch the target page through a residential proxy in this country (ISO 3166-1 al
1281
+ #
1282
+ # @param exclude_selectors [Array<String>, nil] Remove elements matching these CSS selectors. Applied after `includeSelectors`,
1283
+ #
1284
+ # @param include_html [Boolean] Also include each page's HTML in its result record, as an `html` field alongside
1285
+ #
1286
+ # @param include_images [Boolean] Include image references in the Markdown.
1287
+ #
1288
+ # @param include_links [Boolean] Include links in the Markdown.
1289
+ #
1290
+ # @param include_selectors [Array<String>, nil] Keep only the subtrees matching these CSS selectors. Filtered pages are always f
1291
+ #
1292
+ # @param max_age_ms [Integer, nil] Return a cached result if a prior scrape for the same parameters exists and is y
1293
+ #
1294
+ # @param pdf [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf] PDF parsing controls. Use start/end to limit text extraction and embedded-image
1295
+ #
1296
+ # @param settle_animations [Boolean] Wait briefly for CSS and transition animations to settle before extraction, on p
1297
+ #
1298
+ # @param shorten_base64_images [Boolean] Shorten inline base64 image data.
1299
+ #
1300
+ # @param use_main_content_only [Boolean] Return the main content without navigation or footers.
1301
+ #
1302
+ # @param wait_for_ms [Integer] How long to wait after initial page load, in milliseconds. `0` waits 500 ms.
1303
+
1304
+ # Fetch the target page through a residential proxy in this country (ISO 3166-1
1305
+ # alpha-2).
1306
+ #
1307
+ # @see ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Options#country
1308
+ module Country
1309
+ extend ContextDev::Internal::Type::Enum
1310
+
1311
+ AD = :ad
1312
+ AE = :ae
1313
+ AF = :af
1314
+ AG = :ag
1315
+ AI = :ai
1316
+ AL = :al
1317
+ AM = :am
1318
+ AO = :ao
1319
+ AR = :ar
1320
+ AT = :at
1321
+ AU = :au
1322
+ AW = :aw
1323
+ AZ = :az
1324
+ BA = :ba
1325
+ BB = :bb
1326
+ BD = :bd
1327
+ BE = :be
1328
+ BF = :bf
1329
+ BG = :bg
1330
+ BH = :bh
1331
+ BI = :bi
1332
+ BJ = :bj
1333
+ BM = :bm
1334
+ BN = :bn
1335
+ BO = :bo
1336
+ BQ = :bq
1337
+ BR = :br
1338
+ BS = :bs
1339
+ BW = :bw
1340
+ BY = :by
1341
+ BZ = :bz
1342
+ CA = :ca
1343
+ CD = :cd
1344
+ CF = :cf
1345
+ CG = :cg
1346
+ CH = :ch
1347
+ CI = :ci
1348
+ CL = :cl
1349
+ CM = :cm
1350
+ CN = :cn
1351
+ CO = :co
1352
+ CR = :cr
1353
+ CV = :cv
1354
+ CW = :cw
1355
+ CY = :cy
1356
+ CZ = :cz
1357
+ DE = :de
1358
+ DJ = :dj
1359
+ DK = :dk
1360
+ DM = :dm
1361
+ DO = :do
1362
+ DZ = :dz
1363
+ EC = :ec
1364
+ EE = :ee
1365
+ EG = :eg
1366
+ ES = :es
1367
+ ET = :et
1368
+ FI = :fi
1369
+ FJ = :fj
1370
+ FR = :fr
1371
+ GA = :ga
1372
+ GB = :gb
1373
+ GD = :gd
1374
+ GE = :ge
1375
+ GF = :gf
1376
+ GG = :gg
1377
+ GH = :gh
1378
+ GM = :gm
1379
+ GN = :gn
1380
+ GP = :gp
1381
+ GQ = :gq
1382
+ GR = :gr
1383
+ GT = :gt
1384
+ GU = :gu
1385
+ GW = :gw
1386
+ GY = :gy
1387
+ HK = :hk
1388
+ HN = :hn
1389
+ HR = :hr
1390
+ HT = :ht
1391
+ HU = :hu
1392
+ ID = :id
1393
+ IE = :ie
1394
+ IL = :il
1395
+ IM = :im
1396
+ IN = :in
1397
+ IQ = :iq
1398
+ IR = :ir
1399
+ IS = :is
1400
+ IT = :it
1401
+ JE = :je
1402
+ JM = :jm
1403
+ JO = :jo
1404
+ JP = :jp
1405
+ KE = :ke
1406
+ KG = :kg
1407
+ KH = :kh
1408
+ KN = :kn
1409
+ KR = :kr
1410
+ KW = :kw
1411
+ KY = :ky
1412
+ KZ = :kz
1413
+ LA = :la
1414
+ LB = :lb
1415
+ LC = :lc
1416
+ LK = :lk
1417
+ LR = :lr
1418
+ LS = :ls
1419
+ LT = :lt
1420
+ LU = :lu
1421
+ LV = :lv
1422
+ LY = :ly
1423
+ MA = :ma
1424
+ MC = :mc
1425
+ MD = :md
1426
+ ME = :me
1427
+ MF = :mf
1428
+ MG = :mg
1429
+ MK = :mk
1430
+ ML = :ml
1431
+ MM = :mm
1432
+ MN = :mn
1433
+ MO = :mo
1434
+ MQ = :mq
1435
+ MR = :mr
1436
+ MT = :mt
1437
+ MU = :mu
1438
+ MV = :mv
1439
+ MW = :mw
1440
+ MX = :mx
1441
+ MY = :my
1442
+ MZ = :mz
1443
+ NA = :na
1444
+ NC = :nc
1445
+ NE = :ne
1446
+ NG = :ng
1447
+ NI = :ni
1448
+ NL = :nl
1449
+ NO = :no
1450
+ NP = :np
1451
+ NZ = :nz
1452
+ OM = :om
1453
+ PA = :pa
1454
+ PE = :pe
1455
+ PF = :pf
1456
+ PG = :pg
1457
+ PH = :ph
1458
+ PK = :pk
1459
+ PL = :pl
1460
+ PR = :pr
1461
+ PS = :ps
1462
+ PT = :pt
1463
+ PY = :py
1464
+ QA = :qa
1465
+ RE = :re
1466
+ RO = :ro
1467
+ RS = :rs
1468
+ RU = :ru
1469
+ RW = :rw
1470
+ SA = :sa
1471
+ SC = :sc
1472
+ SD = :sd
1473
+ SE = :se
1474
+ SG = :sg
1475
+ SI = :si
1476
+ SK = :sk
1477
+ SL = :sl
1478
+ SM = :sm
1479
+ SN = :sn
1480
+ SO = :so
1481
+ SR = :sr
1482
+ SS = :ss
1483
+ ST = :st
1484
+ SV = :sv
1485
+ SX = :sx
1486
+ SY = :sy
1487
+ SZ = :sz
1488
+ TC = :tc
1489
+ TD = :td
1490
+ TG = :tg
1491
+ TH = :th
1492
+ TJ = :tj
1493
+ TL = :tl
1494
+ TM = :tm
1495
+ TN = :tn
1496
+ TR = :tr
1497
+ TT = :tt
1498
+ TW = :tw
1499
+ TZ = :tz
1500
+ UA = :ua
1501
+ UG = :ug
1502
+ US = :us
1503
+ UY = :uy
1504
+ UZ = :uz
1505
+ VC = :vc
1506
+ VE = :ve
1507
+ VG = :vg
1508
+ VI = :vi
1509
+ VN = :vn
1510
+ YE = :ye
1511
+ YT = :yt
1512
+ ZA = :za
1513
+ ZM = :zm
1514
+ ZW = :zw
1515
+
1516
+ # @!method self.values
1517
+ # @return [Array<Symbol>]
1518
+ end
1519
+
1520
+ # @see ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Options#pdf
1521
+ class Pdf < ContextDev::Internal::Type::BaseModel
1522
+ # @!attribute end_
1523
+ # Last 1-based PDF page to parse. When omitted, parsing ends at the last page.
1524
+ # Must be greater than or equal to start when both are provided.
1525
+ #
1526
+ # @return [Integer, nil]
1527
+ optional :end_, Integer, api_name: :end
1528
+
1529
+ # @!attribute ocr
1530
+ # When true, OCR the selected PDF pages that have no usable text layer (scans),
1531
+ # replacing each recovered page's text with the OCR result while pages with a real
1532
+ # text layer keep it. Billed at 1 credit per page OCR actually recovered, on top
1533
+ # of the base request cost. When false, no OCR runs.
1534
+ #
1535
+ # @return [Boolean, nil]
1536
+ optional :ocr, ContextDev::Internal::Type::Boolean
1537
+
1538
+ # @!attribute should_parse
1539
+ # When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
1540
+ # a 400 PDF_SKIPPED is returned.
1541
+ #
1542
+ # @return [Boolean, nil]
1543
+ optional :should_parse, ContextDev::Internal::Type::Boolean, api_name: :shouldParse
1544
+
1545
+ # @!attribute start
1546
+ # First 1-based PDF page to parse. When omitted, parsing starts at the first page.
1547
+ #
1548
+ # @return [Integer, nil]
1549
+ optional :start, Integer
1550
+
1551
+ # @!method initialize(end_: nil, ocr: nil, should_parse: nil, start: nil)
1552
+ # Some parameter documentations has been truncated, see
1553
+ # {ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf}
1554
+ # for more details.
1555
+ #
1556
+ # PDF parsing controls. Use start/end to limit text extraction and embedded-image
1557
+ # detection/OCR to an inclusive 1-based page range.
1558
+ #
1559
+ # @param end_ [Integer] Last 1-based PDF page to parse. When omitted, parsing ends at the last page. Mus
1560
+ #
1561
+ # @param ocr [Boolean] When true, OCR the selected PDF pages that have no usable text layer (scans), re
1562
+ #
1563
+ # @param should_parse [Boolean] When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
1564
+ #
1565
+ # @param start [Integer] First 1-based PDF page to parse. When omitted, parsing starts at the first page.
1566
+ end
1567
+ end
1568
+ end
1569
+
1570
+ class HTML < ContextDev::Internal::Type::BaseModel
1571
+ # @!attribute format_
1572
+ # Return page content as HTML.
1573
+ #
1574
+ # @return [Symbol, :html]
1575
+ required :format_, const: :html, api_name: :format
1576
+
1577
+ # @!attribute source
1578
+ # How to find pages to crawl.
1579
+ #
1580
+ # @return [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Source::StartURL, ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Source::Sitemap]
1581
+ required :source, union: -> { ContextDev::BatchSubmitParams::Input::Crawl::Data::HTML::Source }
1582
+
1583
+ # @!attribute options
1584
+ # Options for HTML output.
1585
+ #
1586
+ # @return [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Options, nil]
1587
+ optional :options, -> { ContextDev::BatchSubmitParams::Input::Crawl::Data::HTML::Options }
1588
+
1589
+ # @!method initialize(source:, options: nil, format_: :html)
1590
+ # Crawl pages and return HTML.
1591
+ #
1592
+ # @param source [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Source::StartURL, ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Source::Sitemap] How to find pages to crawl.
1593
+ #
1594
+ # @param options [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Options] Options for HTML output.
1595
+ #
1596
+ # @param format_ [Symbol, :html] Return page content as HTML.
1597
+
1598
+ # How to find pages to crawl.
1599
+ #
1600
+ # @see ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML#source
1601
+ module Source
1602
+ extend ContextDev::Internal::Type::Union
1603
+
1604
+ discriminator :type
1605
+
1606
+ # Discover pages by following links from one URL.
1607
+ variant :start_url, -> { ContextDev::BatchSubmitParams::Input::Crawl::Data::HTML::Source::StartURL }
1608
+
1609
+ # Scrape the pages listed in a domain's sitemap. Links on those pages are not followed.
1610
+ variant :sitemap, -> { ContextDev::BatchSubmitParams::Input::Crawl::Data::HTML::Source::Sitemap }
1611
+
1612
+ class StartURL < ContextDev::Internal::Type::BaseModel
1613
+ # @!attribute type
1614
+ # Start from one page.
1615
+ #
1616
+ # @return [Symbol, :start_url]
1617
+ required :type, const: :start_url
1618
+
1619
+ # @!attribute url
1620
+ # Page where crawling begins. A URL without a scheme is read as https://.
1621
+ #
1622
+ # @return [String]
1623
+ required :url, String
1624
+
1625
+ # @!attribute controls
1626
+ # Limits and filters for page discovery.
1627
+ #
1628
+ # @return [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Source::StartURL::Controls, nil]
1629
+ optional :controls,
1630
+ -> { ContextDev::BatchSubmitParams::Input::Crawl::Data::HTML::Source::StartURL::Controls }
1631
+
1632
+ # @!method initialize(url:, controls: nil, type: :start_url)
1633
+ # Discover pages by following links from one URL.
1634
+ #
1635
+ # @param url [String] Page where crawling begins. A URL without a scheme is read as https://.
1636
+ #
1637
+ # @param controls [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Source::StartURL::Controls] Limits and filters for page discovery.
1638
+ #
1639
+ # @param type [Symbol, :start_url] Start from one page.
1640
+
1641
+ # @see ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Source::StartURL#controls
1642
+ class Controls < ContextDev::Internal::Type::BaseModel
1643
+ # @!attribute follow_subdomains
1644
+ # Follow links to subdomains.
1645
+ #
1646
+ # @return [Boolean, nil]
1647
+ optional :follow_subdomains,
1648
+ ContextDev::Internal::Type::Boolean,
1649
+ api_name: :followSubdomains
1650
+
1651
+ # @!attribute max_depth
1652
+ # Maximum link depth. Source pages are depth 0. No limit when omitted.
1653
+ #
1654
+ # @return [Integer, nil]
1655
+ optional :max_depth, Integer, api_name: :maxDepth
1656
+
1657
+ # @!attribute max_urls
1658
+ # Maximum pages to fetch. Unused reserved credits are refunded. Maximum 25000.
1659
+ #
1660
+ # @return [Integer, nil]
1661
+ optional :max_urls, Integer, api_name: :maxUrls
1662
+
1663
+ # @!attribute regex
1664
+ # RE2 pattern for URLs to include. The `start_url` itself is always included.
1665
+ #
1666
+ # @return [String, nil]
1667
+ optional :regex, String
1668
+
1669
+ # @!method initialize(follow_subdomains: nil, max_depth: nil, max_urls: nil, regex: nil)
1670
+ # Limits and filters for page discovery.
1671
+ #
1672
+ # @param follow_subdomains [Boolean] Follow links to subdomains.
1673
+ #
1674
+ # @param max_depth [Integer] Maximum link depth. Source pages are depth 0. No limit when omitted.
1675
+ #
1676
+ # @param max_urls [Integer] Maximum pages to fetch. Unused reserved credits are refunded. Maximum 25000.
1677
+ #
1678
+ # @param regex [String] RE2 pattern for URLs to include. The `start_url` itself is always included.
1679
+ end
1680
+ end
1681
+
1682
+ class Sitemap < ContextDev::Internal::Type::BaseModel
1683
+ # @!attribute domain
1684
+ # Domain whose sitemap lists the pages to scrape. A full URL is reduced to its
1685
+ # domain.
1686
+ #
1687
+ # @return [String]
1688
+ required :domain, String
1689
+
1690
+ # @!attribute type
1691
+ # Scrape the URLs in the domain's sitemap.
1692
+ #
1693
+ # @return [Symbol, :sitemap]
1694
+ required :type, const: :sitemap
1695
+
1696
+ # @!attribute controls
1697
+ # Limits and filters for the sitemap URLs. A sitemap batch scrapes exactly those
1698
+ # URLs and never follows links off them, so there is no crawl depth here.
1699
+ #
1700
+ # @return [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Source::Sitemap::Controls, nil]
1701
+ optional :controls,
1702
+ -> { ContextDev::BatchSubmitParams::Input::Crawl::Data::HTML::Source::Sitemap::Controls }
1703
+
1704
+ # @!method initialize(domain:, controls: nil, type: :sitemap)
1705
+ # Some parameter documentations has been truncated, see
1706
+ # {ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Source::Sitemap}
1707
+ # for more details.
1708
+ #
1709
+ # Scrape the pages listed in a domain's sitemap. Links on those pages are not
1710
+ # followed.
1711
+ #
1712
+ # @param domain [String] Domain whose sitemap lists the pages to scrape. A full URL is reduced to its dom
1713
+ #
1714
+ # @param controls [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Source::Sitemap::Controls] Limits and filters for the sitemap URLs. A sitemap batch scrapes exactly those U
1715
+ #
1716
+ # @param type [Symbol, :sitemap] Scrape the URLs in the domain's sitemap.
1717
+
1718
+ # @see ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Source::Sitemap#controls
1719
+ class Controls < ContextDev::Internal::Type::BaseModel
1720
+ # @!attribute max_urls
1721
+ # Maximum pages to fetch. Unused reserved credits are refunded. Maximum 25000.
1722
+ #
1723
+ # @return [Integer, nil]
1724
+ optional :max_urls, Integer, api_name: :maxUrls
1725
+
1726
+ # @!attribute regex
1727
+ # RE2 pattern; only sitemap URLs matching it are scraped.
1728
+ #
1729
+ # @return [String, nil]
1730
+ optional :regex, String
1731
+
1732
+ # @!method initialize(max_urls: nil, regex: nil)
1733
+ # Limits and filters for the sitemap URLs. A sitemap batch scrapes exactly those
1734
+ # URLs and never follows links off them, so there is no crawl depth here.
1735
+ #
1736
+ # @param max_urls [Integer] Maximum pages to fetch. Unused reserved credits are refunded. Maximum 25000.
1737
+ #
1738
+ # @param regex [String] RE2 pattern; only sitemap URLs matching it are scraped.
1739
+ end
1740
+ end
1741
+
1742
+ # @!method self.variants
1743
+ # @return [Array(ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Source::StartURL, ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Source::Sitemap)]
1744
+ end
1745
+
1746
+ # @see ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML#options
1747
+ class Options < ContextDev::Internal::Type::BaseModel
1748
+ # @!attribute country
1749
+ # Fetch the target page through a residential proxy in this country (ISO 3166-1
1750
+ # alpha-2).
1751
+ #
1752
+ # @return [Symbol, ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Country, nil]
1753
+ optional :country, enum: -> { ContextDev::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Country }
1754
+
1755
+ # @!attribute exclude_selectors
1756
+ # Remove elements matching these CSS selectors. Applied after `includeSelectors`,
1757
+ # so an element matching both is removed.
1758
+ #
1759
+ # @return [Array<String>, nil]
1760
+ optional :exclude_selectors,
1761
+ ContextDev::Internal::Type::ArrayOf[String],
1762
+ api_name: :excludeSelectors,
1763
+ nil?: true
1764
+
1765
+ # @!attribute include_selectors
1766
+ # Keep only the subtrees matching these CSS selectors. Filtered pages are always
1767
+ # fetched fresh, ignoring `maxAgeMs`.
1768
+ #
1769
+ # @return [Array<String>, nil]
1770
+ optional :include_selectors,
1771
+ ContextDev::Internal::Type::ArrayOf[String],
1772
+ api_name: :includeSelectors,
1773
+ nil?: true
1774
+
1775
+ # @!attribute max_age_ms
1776
+ # Return a cached result if a prior scrape for the same parameters exists and is
1777
+ # younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
1778
+ # omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
1779
+ #
1780
+ # @return [Integer, nil]
1781
+ optional :max_age_ms, Integer, api_name: :maxAgeMs, nil?: true
1782
+
1783
+ # @!attribute pdf
1784
+ # PDF parsing controls. Use start/end to limit text extraction and embedded-image
1785
+ # detection/OCR to an inclusive 1-based page range.
1786
+ #
1787
+ # @return [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf, nil]
1788
+ optional :pdf, -> { ContextDev::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf }
1789
+
1790
+ # @!attribute settle_animations
1791
+ # Wait briefly for CSS and transition animations to settle before extraction, on
1792
+ # pages that render in a browser.
1793
+ #
1794
+ # @return [Boolean, nil]
1795
+ optional :settle_animations, ContextDev::Internal::Type::Boolean, api_name: :settleAnimations
1796
+
1797
+ # @!attribute use_main_content_only
1798
+ # Return the main content without navigation or footers.
1799
+ #
1800
+ # @return [Boolean, nil]
1801
+ optional :use_main_content_only,
1802
+ ContextDev::Internal::Type::Boolean,
1803
+ api_name: :useMainContentOnly
1804
+
1805
+ # @!attribute wait_for_ms
1806
+ # How long to wait after initial page load, in milliseconds. `0` waits 500 ms.
1807
+ #
1808
+ # @return [Integer, nil]
1809
+ optional :wait_for_ms, Integer, api_name: :waitForMs
1810
+
1811
+ # @!method initialize(country: nil, exclude_selectors: nil, include_selectors: nil, max_age_ms: nil, pdf: nil, settle_animations: nil, use_main_content_only: nil, wait_for_ms: nil)
1812
+ # Some parameter documentations has been truncated, see
1813
+ # {ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Options} for
1814
+ # more details.
1815
+ #
1816
+ # Options for HTML output.
1817
+ #
1818
+ # @param country [Symbol, ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Country] Fetch the target page through a residential proxy in this country (ISO 3166-1 al
1819
+ #
1820
+ # @param exclude_selectors [Array<String>, nil] Remove elements matching these CSS selectors. Applied after `includeSelectors`,
1821
+ #
1822
+ # @param include_selectors [Array<String>, nil] Keep only the subtrees matching these CSS selectors. Filtered pages are always f
1823
+ #
1824
+ # @param max_age_ms [Integer, nil] Return a cached result if a prior scrape for the same parameters exists and is y
1825
+ #
1826
+ # @param pdf [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf] PDF parsing controls. Use start/end to limit text extraction and embedded-image
1827
+ #
1828
+ # @param settle_animations [Boolean] Wait briefly for CSS and transition animations to settle before extraction, on p
1829
+ #
1830
+ # @param use_main_content_only [Boolean] Return the main content without navigation or footers.
1831
+ #
1832
+ # @param wait_for_ms [Integer] How long to wait after initial page load, in milliseconds. `0` waits 500 ms.
1833
+
1834
+ # Fetch the target page through a residential proxy in this country (ISO 3166-1
1835
+ # alpha-2).
1836
+ #
1837
+ # @see ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Options#country
1838
+ module Country
1839
+ extend ContextDev::Internal::Type::Enum
1840
+
1841
+ AD = :ad
1842
+ AE = :ae
1843
+ AF = :af
1844
+ AG = :ag
1845
+ AI = :ai
1846
+ AL = :al
1847
+ AM = :am
1848
+ AO = :ao
1849
+ AR = :ar
1850
+ AT = :at
1851
+ AU = :au
1852
+ AW = :aw
1853
+ AZ = :az
1854
+ BA = :ba
1855
+ BB = :bb
1856
+ BD = :bd
1857
+ BE = :be
1858
+ BF = :bf
1859
+ BG = :bg
1860
+ BH = :bh
1861
+ BI = :bi
1862
+ BJ = :bj
1863
+ BM = :bm
1864
+ BN = :bn
1865
+ BO = :bo
1866
+ BQ = :bq
1867
+ BR = :br
1868
+ BS = :bs
1869
+ BW = :bw
1870
+ BY = :by
1871
+ BZ = :bz
1872
+ CA = :ca
1873
+ CD = :cd
1874
+ CF = :cf
1875
+ CG = :cg
1876
+ CH = :ch
1877
+ CI = :ci
1878
+ CL = :cl
1879
+ CM = :cm
1880
+ CN = :cn
1881
+ CO = :co
1882
+ CR = :cr
1883
+ CV = :cv
1884
+ CW = :cw
1885
+ CY = :cy
1886
+ CZ = :cz
1887
+ DE = :de
1888
+ DJ = :dj
1889
+ DK = :dk
1890
+ DM = :dm
1891
+ DO = :do
1892
+ DZ = :dz
1893
+ EC = :ec
1894
+ EE = :ee
1895
+ EG = :eg
1896
+ ES = :es
1897
+ ET = :et
1898
+ FI = :fi
1899
+ FJ = :fj
1900
+ FR = :fr
1901
+ GA = :ga
1902
+ GB = :gb
1903
+ GD = :gd
1904
+ GE = :ge
1905
+ GF = :gf
1906
+ GG = :gg
1907
+ GH = :gh
1908
+ GM = :gm
1909
+ GN = :gn
1910
+ GP = :gp
1911
+ GQ = :gq
1912
+ GR = :gr
1913
+ GT = :gt
1914
+ GU = :gu
1915
+ GW = :gw
1916
+ GY = :gy
1917
+ HK = :hk
1918
+ HN = :hn
1919
+ HR = :hr
1920
+ HT = :ht
1921
+ HU = :hu
1922
+ ID = :id
1923
+ IE = :ie
1924
+ IL = :il
1925
+ IM = :im
1926
+ IN = :in
1927
+ IQ = :iq
1928
+ IR = :ir
1929
+ IS = :is
1930
+ IT = :it
1931
+ JE = :je
1932
+ JM = :jm
1933
+ JO = :jo
1934
+ JP = :jp
1935
+ KE = :ke
1936
+ KG = :kg
1937
+ KH = :kh
1938
+ KN = :kn
1939
+ KR = :kr
1940
+ KW = :kw
1941
+ KY = :ky
1942
+ KZ = :kz
1943
+ LA = :la
1944
+ LB = :lb
1945
+ LC = :lc
1946
+ LK = :lk
1947
+ LR = :lr
1948
+ LS = :ls
1949
+ LT = :lt
1950
+ LU = :lu
1951
+ LV = :lv
1952
+ LY = :ly
1953
+ MA = :ma
1954
+ MC = :mc
1955
+ MD = :md
1956
+ ME = :me
1957
+ MF = :mf
1958
+ MG = :mg
1959
+ MK = :mk
1960
+ ML = :ml
1961
+ MM = :mm
1962
+ MN = :mn
1963
+ MO = :mo
1964
+ MQ = :mq
1965
+ MR = :mr
1966
+ MT = :mt
1967
+ MU = :mu
1968
+ MV = :mv
1969
+ MW = :mw
1970
+ MX = :mx
1971
+ MY = :my
1972
+ MZ = :mz
1973
+ NA = :na
1974
+ NC = :nc
1975
+ NE = :ne
1976
+ NG = :ng
1977
+ NI = :ni
1978
+ NL = :nl
1979
+ NO = :no
1980
+ NP = :np
1981
+ NZ = :nz
1982
+ OM = :om
1983
+ PA = :pa
1984
+ PE = :pe
1985
+ PF = :pf
1986
+ PG = :pg
1987
+ PH = :ph
1988
+ PK = :pk
1989
+ PL = :pl
1990
+ PR = :pr
1991
+ PS = :ps
1992
+ PT = :pt
1993
+ PY = :py
1994
+ QA = :qa
1995
+ RE = :re
1996
+ RO = :ro
1997
+ RS = :rs
1998
+ RU = :ru
1999
+ RW = :rw
2000
+ SA = :sa
2001
+ SC = :sc
2002
+ SD = :sd
2003
+ SE = :se
2004
+ SG = :sg
2005
+ SI = :si
2006
+ SK = :sk
2007
+ SL = :sl
2008
+ SM = :sm
2009
+ SN = :sn
2010
+ SO = :so
2011
+ SR = :sr
2012
+ SS = :ss
2013
+ ST = :st
2014
+ SV = :sv
2015
+ SX = :sx
2016
+ SY = :sy
2017
+ SZ = :sz
2018
+ TC = :tc
2019
+ TD = :td
2020
+ TG = :tg
2021
+ TH = :th
2022
+ TJ = :tj
2023
+ TL = :tl
2024
+ TM = :tm
2025
+ TN = :tn
2026
+ TR = :tr
2027
+ TT = :tt
2028
+ TW = :tw
2029
+ TZ = :tz
2030
+ UA = :ua
2031
+ UG = :ug
2032
+ US = :us
2033
+ UY = :uy
2034
+ UZ = :uz
2035
+ VC = :vc
2036
+ VE = :ve
2037
+ VG = :vg
2038
+ VI = :vi
2039
+ VN = :vn
2040
+ YE = :ye
2041
+ YT = :yt
2042
+ ZA = :za
2043
+ ZM = :zm
2044
+ ZW = :zw
2045
+
2046
+ # @!method self.values
2047
+ # @return [Array<Symbol>]
2048
+ end
2049
+
2050
+ # @see ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Options#pdf
2051
+ class Pdf < ContextDev::Internal::Type::BaseModel
2052
+ # @!attribute end_
2053
+ # Last 1-based PDF page to parse. When omitted, parsing ends at the last page.
2054
+ # Must be greater than or equal to start when both are provided.
2055
+ #
2056
+ # @return [Integer, nil]
2057
+ optional :end_, Integer, api_name: :end
2058
+
2059
+ # @!attribute ocr
2060
+ # When true, OCR the selected PDF pages that have no usable text layer (scans),
2061
+ # replacing each recovered page's text with the OCR result while pages with a real
2062
+ # text layer keep it. Billed at 1 credit per page OCR actually recovered, on top
2063
+ # of the base request cost. When false, no OCR runs.
2064
+ #
2065
+ # @return [Boolean, nil]
2066
+ optional :ocr, ContextDev::Internal::Type::Boolean
2067
+
2068
+ # @!attribute should_parse
2069
+ # When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
2070
+ # a 400 PDF_SKIPPED is returned.
2071
+ #
2072
+ # @return [Boolean, nil]
2073
+ optional :should_parse, ContextDev::Internal::Type::Boolean, api_name: :shouldParse
2074
+
2075
+ # @!attribute start
2076
+ # First 1-based PDF page to parse. When omitted, parsing starts at the first page.
2077
+ #
2078
+ # @return [Integer, nil]
2079
+ optional :start, Integer
2080
+
2081
+ # @!method initialize(end_: nil, ocr: nil, should_parse: nil, start: nil)
2082
+ # Some parameter documentations has been truncated, see
2083
+ # {ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf}
2084
+ # for more details.
2085
+ #
2086
+ # PDF parsing controls. Use start/end to limit text extraction and embedded-image
2087
+ # detection/OCR to an inclusive 1-based page range.
2088
+ #
2089
+ # @param end_ [Integer] Last 1-based PDF page to parse. When omitted, parsing ends at the last page. Mus
2090
+ #
2091
+ # @param ocr [Boolean] When true, OCR the selected PDF pages that have no usable text layer (scans), re
2092
+ #
2093
+ # @param should_parse [Boolean] When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
2094
+ #
2095
+ # @param start [Integer] First 1-based PDF page to parse. When omitted, parsing starts at the first page.
2096
+ end
2097
+ end
2098
+ end
2099
+
2100
+ # @!method self.variants
2101
+ # @return [Array(ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown, ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML)]
2102
+ end
2103
+ end
2104
+
2105
+ # @!method self.variants
2106
+ # @return [Array(ContextDev::Models::BatchSubmitParams::Input::Scrape, ContextDev::Models::BatchSubmitParams::Input::Crawl)]
53
2107
  end
54
2108
  end
55
2109
  end