context.dev 2.8.0 → 2.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (76) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +15 -0
  3. data/README.md +1 -1
  4. data/lib/context_dev/client.rb +5 -0
  5. data/lib/context_dev/models/batch_delete_params.rb +22 -0
  6. data/lib/context_dev/models/batch_delete_response.rb +60 -0
  7. data/lib/context_dev/models/batch_get_results_response.rb +13 -1
  8. data/lib/context_dev/models/batch_list_response.rb +13 -4
  9. data/lib/context_dev/models/batch_retrieve_response.rb +13 -4
  10. data/lib/context_dev/models/batch_submit_params.rb +2352 -26
  11. data/lib/context_dev/models/batch_submit_response.rb +125 -531
  12. data/lib/context_dev/models/brand_retrieve_response.rb +50 -1
  13. data/lib/context_dev/models/brand_search_response.rb +3 -2
  14. data/lib/context_dev/models/crawl_controls.rb +21 -15
  15. data/lib/context_dev/models/parse_handle_params.rb +11 -9
  16. data/lib/context_dev/models/person_enrich_params.rb +176 -0
  17. data/lib/context_dev/models/person_enrich_response.rb +581 -0
  18. data/lib/context_dev/models/web_search_response.rb +1 -0
  19. data/lib/context_dev/models/web_web_crawl_md_params.rb +5 -4
  20. data/lib/context_dev/models/web_web_scrape_html_params.rb +11 -9
  21. data/lib/context_dev/models/web_web_scrape_md_params.rb +11 -9
  22. data/lib/context_dev/models/web_web_scrape_sitemap_params.rb +11 -1
  23. data/lib/context_dev/models/web_web_scrape_sitemap_response.rb +3 -2
  24. data/lib/context_dev/models.rb +4 -0
  25. data/lib/context_dev/resources/batch.rb +33 -7
  26. data/lib/context_dev/resources/brand.rb +2 -1
  27. data/lib/context_dev/resources/parse.rb +1 -1
  28. data/lib/context_dev/resources/people.rb +56 -0
  29. data/lib/context_dev/resources/web.rb +21 -2
  30. data/lib/context_dev/version.rb +1 -1
  31. data/lib/context_dev.rb +5 -0
  32. data/rbi/context_dev/client.rbi +4 -0
  33. data/rbi/context_dev/models/batch_delete_params.rbi +40 -0
  34. data/rbi/context_dev/models/batch_delete_response.rbi +116 -0
  35. data/rbi/context_dev/models/batch_get_results_response.rbi +14 -1
  36. data/rbi/context_dev/models/batch_list_response.rbi +24 -8
  37. data/rbi/context_dev/models/batch_retrieve_response.rbi +24 -8
  38. data/rbi/context_dev/models/batch_submit_params.rbi +6955 -44
  39. data/rbi/context_dev/models/batch_submit_response.rbi +184 -1151
  40. data/rbi/context_dev/models/brand_retrieve_response.rbi +152 -0
  41. data/rbi/context_dev/models/brand_search_response.rbi +4 -2
  42. data/rbi/context_dev/models/crawl_controls.rbi +22 -28
  43. data/rbi/context_dev/models/parse_handle_params.rbi +15 -12
  44. data/rbi/context_dev/models/person_enrich_params.rbi +332 -0
  45. data/rbi/context_dev/models/person_enrich_response.rbi +1208 -0
  46. data/rbi/context_dev/models/web_search_response.rbi +5 -0
  47. data/rbi/context_dev/models/web_web_crawl_md_params.rbi +8 -6
  48. data/rbi/context_dev/models/web_web_scrape_html_params.rbi +15 -12
  49. data/rbi/context_dev/models/web_web_scrape_md_params.rbi +15 -12
  50. data/rbi/context_dev/models/web_web_scrape_sitemap_params.rbi +15 -0
  51. data/rbi/context_dev/models/web_web_scrape_sitemap_response.rbi +4 -2
  52. data/rbi/context_dev/models.rbi +4 -0
  53. data/rbi/context_dev/resources/batch.rbi +32 -10
  54. data/rbi/context_dev/resources/brand.rbi +2 -1
  55. data/rbi/context_dev/resources/parse.rbi +5 -5
  56. data/rbi/context_dev/resources/people.rbi +47 -0
  57. data/rbi/context_dev/resources/web.rbi +23 -1
  58. data/sig/context_dev/client.rbs +2 -0
  59. data/sig/context_dev/models/batch_delete_params.rbs +23 -0
  60. data/sig/context_dev/models/batch_delete_response.rbs +57 -0
  61. data/sig/context_dev/models/batch_get_results_response.rbs +9 -2
  62. data/sig/context_dev/models/batch_list_response.rbs +16 -2
  63. data/sig/context_dev/models/batch_retrieve_response.rbs +16 -2
  64. data/sig/context_dev/models/batch_submit_params.rbs +2756 -15
  65. data/sig/context_dev/models/batch_submit_response.rbs +78 -466
  66. data/sig/context_dev/models/brand_retrieve_response.rbs +62 -0
  67. data/sig/context_dev/models/crawl_controls.rbs +16 -16
  68. data/sig/context_dev/models/person_enrich_params.rbs +199 -0
  69. data/sig/context_dev/models/person_enrich_response.rbs +607 -0
  70. data/sig/context_dev/models/web_search_response.rbs +2 -0
  71. data/sig/context_dev/models/web_web_scrape_sitemap_params.rbs +7 -0
  72. data/sig/context_dev/models.rbs +4 -0
  73. data/sig/context_dev/resources/batch.rbs +8 -2
  74. data/sig/context_dev/resources/people.rbs +19 -0
  75. data/sig/context_dev/resources/web.rbs +1 -0
  76. metadata +17 -2
@@ -7,49 +7,2375 @@ module ContextDev
7
7
  extend ContextDev::Internal::Type::RequestParameters::Converter
8
8
  include ContextDev::Internal::Type::RequestParameters
9
9
 
10
- # @!attribute identifiers
11
- # Known identifiers for the person. At least one identifier is required.
10
+ # @!attribute input
11
+ # Choose a URL list or a site crawl.
12
12
  #
13
- # @return [ContextDev::Models::BatchSubmitParams::Identifiers]
14
- required :identifiers, -> { ContextDev::BatchSubmitParams::Identifiers }
13
+ # @return [ContextDev::Models::BatchSubmitParams::Input::Scrape, ContextDev::Models::BatchSubmitParams::Input::Crawl]
14
+ required :input, union: -> { ContextDev::BatchSubmitParams::Input }
15
15
 
16
16
  # @!attribute tags
17
- # Optional tags for tracking usage. Up to 20 tags, each 1 to 50 characters.
17
+ # Tags stored on the batch. Filter the batch list by them later.
18
18
  #
19
19
  # @return [Array<String>, nil]
20
20
  optional :tags, ContextDev::Internal::Type::ArrayOf[String]
21
21
 
22
- # @!attribute timeout_ms
23
- # Optional timeout in milliseconds for the request. If the request takes longer
24
- # than this value, it will be aborted with a 408 status code. Maximum allowed
25
- # value is 300000ms (5 minutes).
22
+ # @!attribute webhook_url
23
+ # URL notified when the batch finishes.
26
24
  #
27
- # @return [Integer, nil]
28
- optional :timeout_ms, Integer, api_name: :timeoutMS
25
+ # @return [String, nil]
26
+ optional :webhook_url, String, api_name: :webhookUrl
29
27
 
30
- # @!method initialize(identifiers:, tags: nil, timeout_ms: nil, request_options: {})
28
+ # @!attribute idempotency_key
29
+ # Any string unique to this submission. Retries with the same key return the
30
+ # original batch.
31
+ #
32
+ # @return [String, nil]
33
+ optional :idempotency_key, String
34
+
35
+ # @!method initialize(input:, tags: nil, webhook_url: nil, idempotency_key: nil, request_options: {})
31
36
  # Some parameter documentations has been truncated, see
32
37
  # {ContextDev::Models::BatchSubmitParams} for more details.
33
38
  #
34
- # @param identifiers [ContextDev::Models::BatchSubmitParams::Identifiers] Known identifiers for the person. At least one identifier is required.
39
+ # @param input [ContextDev::Models::BatchSubmitParams::Input::Scrape, ContextDev::Models::BatchSubmitParams::Input::Crawl] Choose a URL list or a site crawl.
35
40
  #
36
- # @param tags [Array<String>] Optional tags for tracking usage. Up to 20 tags, each 1 to 50 characters.
41
+ # @param tags [Array<String>] Tags stored on the batch. Filter the batch list by them later.
37
42
  #
38
- # @param timeout_ms [Integer] Optional timeout in milliseconds for the request. If the request takes longer th
43
+ # @param webhook_url [String] URL notified when the batch finishes.
44
+ #
45
+ # @param idempotency_key [String] Any string unique to this submission. Retries with the same key return the origi
39
46
  #
40
47
  # @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}]
41
48
 
42
- class Identifiers < ContextDev::Internal::Type::BaseModel
43
- # @!attribute linkedin_url
44
- # LinkedIn profile URL, e.g. https://www.linkedin.com/in/yahia-bakour/.
45
- #
46
- # @return [String, nil]
47
- optional :linkedin_url, String, api_name: :linkedinUrl
48
-
49
- # @!method initialize(linkedin_url: nil)
50
- # Known identifiers for the person. At least one identifier is required.
51
- #
52
- # @param linkedin_url [String] LinkedIn profile URL, e.g. https://www.linkedin.com/in/yahia-bakour/.
49
+ # Choose a URL list or a site crawl.
50
+ module Input
51
+ extend ContextDev::Internal::Type::Union
52
+
53
+ discriminator :mode
54
+
55
+ # Scrape up to 25K URLs in one batch.
56
+ variant :scrape, -> { ContextDev::BatchSubmitParams::Input::Scrape }
57
+
58
+ # Crawl pages starting from a URL or from a domain's sitemap.
59
+ variant :crawl, -> { ContextDev::BatchSubmitParams::Input::Crawl }
60
+
61
+ class Scrape < ContextDev::Internal::Type::BaseModel
62
+ # @!attribute data
63
+ # Pages to scrape and their output format.
64
+ #
65
+ # @return [ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown, ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML]
66
+ required :data, union: -> { ContextDev::BatchSubmitParams::Input::Scrape::Data }
67
+
68
+ # @!attribute mode
69
+ # Scrape the pages in `data.urls`.
70
+ #
71
+ # @return [Symbol, :scrape]
72
+ required :mode, const: :scrape
73
+
74
+ # @!method initialize(data:, mode: :scrape)
75
+ # Scrape up to 25K URLs in one batch.
76
+ #
77
+ # @param data [ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown, ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML] Pages to scrape and their output format.
78
+ #
79
+ # @param mode [Symbol, :scrape] Scrape the pages in `data.urls`.
80
+
81
+ # Pages to scrape and their output format.
82
+ #
83
+ # @see ContextDev::Models::BatchSubmitParams::Input::Scrape#data
84
+ module Data
85
+ extend ContextDev::Internal::Type::Union
86
+
87
+ discriminator :format
88
+
89
+ # Scrape the listed pages as Markdown.
90
+ variant :markdown, -> { ContextDev::BatchSubmitParams::Input::Scrape::Data::Markdown }
91
+
92
+ # Scrape the listed pages as HTML.
93
+ variant :html, -> { ContextDev::BatchSubmitParams::Input::Scrape::Data::HTML }
94
+
95
+ class Markdown < ContextDev::Internal::Type::BaseModel
96
+ # @!attribute format_
97
+ # Return page content as Markdown.
98
+ #
99
+ # @return [Symbol, :markdown]
100
+ required :format_, const: :markdown, api_name: :format
101
+
102
+ # @!attribute urls
103
+ # Pages to scrape. Maximum 25000.
104
+ #
105
+ # @return [Array<ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::URL>]
106
+ required :urls,
107
+ -> { ContextDev::Internal::Type::ArrayOf[ContextDev::BatchSubmitParams::Input::Scrape::Data::Markdown::URL] }
108
+
109
+ # @!attribute options
110
+ # Options for Markdown output.
111
+ #
112
+ # @return [ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::Options, nil]
113
+ optional :options, -> { ContextDev::BatchSubmitParams::Input::Scrape::Data::Markdown::Options }
114
+
115
+ # @!method initialize(urls:, options: nil, format_: :markdown)
116
+ # Scrape the listed pages as Markdown.
117
+ #
118
+ # @param urls [Array<ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::URL>] Pages to scrape. Maximum 25000.
119
+ #
120
+ # @param options [ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::Options] Options for Markdown output.
121
+ #
122
+ # @param format_ [Symbol, :markdown] Return page content as Markdown.
123
+
124
+ class URL < ContextDev::Internal::Type::BaseModel
125
+ # @!attribute url
126
+ # Page URL to scrape.
127
+ #
128
+ # @return [String]
129
+ required :url, String
130
+
131
+ # @!attribute item_id
132
+ # Your ID for this page, returned with its result. The same URL can use different
133
+ # IDs.
134
+ #
135
+ # @return [String, nil]
136
+ optional :item_id, String, api_name: :itemId
137
+
138
+ # @!attribute meta
139
+ # Custom JSON returned unchanged with this page result.
140
+ #
141
+ # @return [Hash{Symbol=>Object}, nil]
142
+ optional :meta, ContextDev::Internal::Type::HashOf[ContextDev::Internal::Type::Unknown]
143
+
144
+ # @!method initialize(url:, item_id: nil, meta: nil)
145
+ # Some parameter documentations has been truncated, see
146
+ # {ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::URL} for
147
+ # more details.
148
+ #
149
+ # A page to scrape, with optional data for matching results.
150
+ #
151
+ # @param url [String] Page URL to scrape.
152
+ #
153
+ # @param item_id [String] Your ID for this page, returned with its result. The same URL can use different
154
+ #
155
+ # @param meta [Hash{Symbol=>Object}] Custom JSON returned unchanged with this page result.
156
+ end
157
+
158
+ # @see ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown#options
159
+ class Options < ContextDev::Internal::Type::BaseModel
160
+ # @!attribute country
161
+ # Fetch the target page through a residential proxy in this country (ISO 3166-1
162
+ # alpha-2).
163
+ #
164
+ # @return [Symbol, ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Country, nil]
165
+ optional :country,
166
+ enum: -> { ContextDev::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Country }
167
+
168
+ # @!attribute exclude_selectors
169
+ # Remove elements matching these CSS selectors. Applied after `includeSelectors`,
170
+ # so an element matching both is removed.
171
+ #
172
+ # @return [Array<String>, nil]
173
+ optional :exclude_selectors,
174
+ ContextDev::Internal::Type::ArrayOf[String],
175
+ api_name: :excludeSelectors,
176
+ nil?: true
177
+
178
+ # @!attribute include_images
179
+ # Include image references in the Markdown.
180
+ #
181
+ # @return [Boolean, nil]
182
+ optional :include_images, ContextDev::Internal::Type::Boolean, api_name: :includeImages
183
+
184
+ # @!attribute include_links
185
+ # Include links in the Markdown.
186
+ #
187
+ # @return [Boolean, nil]
188
+ optional :include_links, ContextDev::Internal::Type::Boolean, api_name: :includeLinks
189
+
190
+ # @!attribute include_selectors
191
+ # Keep only the subtrees matching these CSS selectors. Filtered pages are always
192
+ # fetched fresh, ignoring `maxAgeMs`.
193
+ #
194
+ # @return [Array<String>, nil]
195
+ optional :include_selectors,
196
+ ContextDev::Internal::Type::ArrayOf[String],
197
+ api_name: :includeSelectors,
198
+ nil?: true
199
+
200
+ # @!attribute max_age_ms
201
+ # Return a cached result if a prior scrape for the same parameters exists and is
202
+ # younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
203
+ # omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
204
+ #
205
+ # @return [Integer, nil]
206
+ optional :max_age_ms, Integer, api_name: :maxAgeMs, nil?: true
207
+
208
+ # @!attribute pdf
209
+ # PDF parsing controls. Use start/end to limit text extraction and embedded-image
210
+ # detection/OCR to an inclusive 1-based page range.
211
+ #
212
+ # @return [ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf, nil]
213
+ optional :pdf, -> { ContextDev::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf }
214
+
215
+ # @!attribute settle_animations
216
+ # Wait briefly for CSS and transition animations to settle before extraction, on
217
+ # pages that render in a browser.
218
+ #
219
+ # @return [Boolean, nil]
220
+ optional :settle_animations, ContextDev::Internal::Type::Boolean, api_name: :settleAnimations
221
+
222
+ # @!attribute shorten_base64_images
223
+ # Shorten inline base64 image data.
224
+ #
225
+ # @return [Boolean, nil]
226
+ optional :shorten_base64_images,
227
+ ContextDev::Internal::Type::Boolean,
228
+ api_name: :shortenBase64Images
229
+
230
+ # @!attribute use_main_content_only
231
+ # Return the main content without navigation or footers.
232
+ #
233
+ # @return [Boolean, nil]
234
+ optional :use_main_content_only,
235
+ ContextDev::Internal::Type::Boolean,
236
+ api_name: :useMainContentOnly
237
+
238
+ # @!attribute wait_for_ms
239
+ # How long to wait after initial page load, in milliseconds. `0` waits 500 ms.
240
+ #
241
+ # @return [Integer, nil]
242
+ optional :wait_for_ms, Integer, api_name: :waitForMs
243
+
244
+ # @!method initialize(country: nil, exclude_selectors: nil, include_images: nil, include_links: nil, include_selectors: nil, max_age_ms: nil, pdf: nil, settle_animations: nil, shorten_base64_images: nil, use_main_content_only: nil, wait_for_ms: nil)
245
+ # Some parameter documentations has been truncated, see
246
+ # {ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::Options}
247
+ # for more details.
248
+ #
249
+ # Options for Markdown output.
250
+ #
251
+ # @param country [Symbol, ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Country] Fetch the target page through a residential proxy in this country (ISO 3166-1 al
252
+ #
253
+ # @param exclude_selectors [Array<String>, nil] Remove elements matching these CSS selectors. Applied after `includeSelectors`,
254
+ #
255
+ # @param include_images [Boolean] Include image references in the Markdown.
256
+ #
257
+ # @param include_links [Boolean] Include links in the Markdown.
258
+ #
259
+ # @param include_selectors [Array<String>, nil] Keep only the subtrees matching these CSS selectors. Filtered pages are always f
260
+ #
261
+ # @param max_age_ms [Integer, nil] Return a cached result if a prior scrape for the same parameters exists and is y
262
+ #
263
+ # @param pdf [ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf] PDF parsing controls. Use start/end to limit text extraction and embedded-image
264
+ #
265
+ # @param settle_animations [Boolean] Wait briefly for CSS and transition animations to settle before extraction, on p
266
+ #
267
+ # @param shorten_base64_images [Boolean] Shorten inline base64 image data.
268
+ #
269
+ # @param use_main_content_only [Boolean] Return the main content without navigation or footers.
270
+ #
271
+ # @param wait_for_ms [Integer] How long to wait after initial page load, in milliseconds. `0` waits 500 ms.
272
+
273
+ # Fetch the target page through a residential proxy in this country (ISO 3166-1
274
+ # alpha-2).
275
+ #
276
+ # @see ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::Options#country
277
+ module Country
278
+ extend ContextDev::Internal::Type::Enum
279
+
280
+ AD = :ad
281
+ AE = :ae
282
+ AF = :af
283
+ AG = :ag
284
+ AI = :ai
285
+ AL = :al
286
+ AM = :am
287
+ AO = :ao
288
+ AR = :ar
289
+ AT = :at
290
+ AU = :au
291
+ AW = :aw
292
+ AZ = :az
293
+ BA = :ba
294
+ BB = :bb
295
+ BD = :bd
296
+ BE = :be
297
+ BF = :bf
298
+ BG = :bg
299
+ BH = :bh
300
+ BI = :bi
301
+ BJ = :bj
302
+ BM = :bm
303
+ BN = :bn
304
+ BO = :bo
305
+ BQ = :bq
306
+ BR = :br
307
+ BS = :bs
308
+ BW = :bw
309
+ BY = :by
310
+ BZ = :bz
311
+ CA = :ca
312
+ CD = :cd
313
+ CF = :cf
314
+ CG = :cg
315
+ CH = :ch
316
+ CI = :ci
317
+ CL = :cl
318
+ CM = :cm
319
+ CN = :cn
320
+ CO = :co
321
+ CR = :cr
322
+ CV = :cv
323
+ CW = :cw
324
+ CY = :cy
325
+ CZ = :cz
326
+ DE = :de
327
+ DJ = :dj
328
+ DK = :dk
329
+ DM = :dm
330
+ DO = :do
331
+ DZ = :dz
332
+ EC = :ec
333
+ EE = :ee
334
+ EG = :eg
335
+ ES = :es
336
+ ET = :et
337
+ FI = :fi
338
+ FJ = :fj
339
+ FR = :fr
340
+ GA = :ga
341
+ GB = :gb
342
+ GD = :gd
343
+ GE = :ge
344
+ GF = :gf
345
+ GG = :gg
346
+ GH = :gh
347
+ GM = :gm
348
+ GN = :gn
349
+ GP = :gp
350
+ GQ = :gq
351
+ GR = :gr
352
+ GT = :gt
353
+ GU = :gu
354
+ GW = :gw
355
+ GY = :gy
356
+ HK = :hk
357
+ HN = :hn
358
+ HR = :hr
359
+ HT = :ht
360
+ HU = :hu
361
+ ID = :id
362
+ IE = :ie
363
+ IL = :il
364
+ IM = :im
365
+ IN = :in
366
+ IQ = :iq
367
+ IR = :ir
368
+ IS = :is
369
+ IT = :it
370
+ JE = :je
371
+ JM = :jm
372
+ JO = :jo
373
+ JP = :jp
374
+ KE = :ke
375
+ KG = :kg
376
+ KH = :kh
377
+ KN = :kn
378
+ KR = :kr
379
+ KW = :kw
380
+ KY = :ky
381
+ KZ = :kz
382
+ LA = :la
383
+ LB = :lb
384
+ LC = :lc
385
+ LK = :lk
386
+ LR = :lr
387
+ LS = :ls
388
+ LT = :lt
389
+ LU = :lu
390
+ LV = :lv
391
+ LY = :ly
392
+ MA = :ma
393
+ MC = :mc
394
+ MD = :md
395
+ ME = :me
396
+ MF = :mf
397
+ MG = :mg
398
+ MK = :mk
399
+ ML = :ml
400
+ MM = :mm
401
+ MN = :mn
402
+ MO = :mo
403
+ MQ = :mq
404
+ MR = :mr
405
+ MT = :mt
406
+ MU = :mu
407
+ MV = :mv
408
+ MW = :mw
409
+ MX = :mx
410
+ MY = :my
411
+ MZ = :mz
412
+ NA = :na
413
+ NC = :nc
414
+ NE = :ne
415
+ NG = :ng
416
+ NI = :ni
417
+ NL = :nl
418
+ NO = :no
419
+ NP = :np
420
+ NZ = :nz
421
+ OM = :om
422
+ PA = :pa
423
+ PE = :pe
424
+ PF = :pf
425
+ PG = :pg
426
+ PH = :ph
427
+ PK = :pk
428
+ PL = :pl
429
+ PR = :pr
430
+ PS = :ps
431
+ PT = :pt
432
+ PY = :py
433
+ QA = :qa
434
+ RE = :re
435
+ RO = :ro
436
+ RS = :rs
437
+ RU = :ru
438
+ RW = :rw
439
+ SA = :sa
440
+ SC = :sc
441
+ SD = :sd
442
+ SE = :se
443
+ SG = :sg
444
+ SI = :si
445
+ SK = :sk
446
+ SL = :sl
447
+ SM = :sm
448
+ SN = :sn
449
+ SO = :so
450
+ SR = :sr
451
+ SS = :ss
452
+ ST = :st
453
+ SV = :sv
454
+ SX = :sx
455
+ SY = :sy
456
+ SZ = :sz
457
+ TC = :tc
458
+ TD = :td
459
+ TG = :tg
460
+ TH = :th
461
+ TJ = :tj
462
+ TL = :tl
463
+ TM = :tm
464
+ TN = :tn
465
+ TR = :tr
466
+ TT = :tt
467
+ TW = :tw
468
+ TZ = :tz
469
+ UA = :ua
470
+ UG = :ug
471
+ US = :us
472
+ UY = :uy
473
+ UZ = :uz
474
+ VC = :vc
475
+ VE = :ve
476
+ VG = :vg
477
+ VI = :vi
478
+ VN = :vn
479
+ YE = :ye
480
+ YT = :yt
481
+ ZA = :za
482
+ ZM = :zm
483
+ ZW = :zw
484
+
485
+ # @!method self.values
486
+ # @return [Array<Symbol>]
487
+ end
488
+
489
+ # @see ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::Options#pdf
490
+ class Pdf < ContextDev::Internal::Type::BaseModel
491
+ # @!attribute end_
492
+ # Last 1-based PDF page to parse. When omitted, parsing ends at the last page.
493
+ # Must be greater than or equal to start when both are provided.
494
+ #
495
+ # @return [Integer, nil]
496
+ optional :end_, Integer, api_name: :end
497
+
498
+ # @!attribute ocr
499
+ # When true, OCR the selected PDF pages that have no usable text layer (scans),
500
+ # replacing each recovered page's text with the OCR result while pages with a real
501
+ # text layer keep it. Billed at 1 credit per page OCR actually recovered, on top
502
+ # of the base request cost. When false, no OCR runs.
503
+ #
504
+ # @return [Boolean, Symbol, ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf::Ocr, nil]
505
+ optional :ocr,
506
+ union: -> { ContextDev::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf::Ocr }
507
+
508
+ # @!attribute should_parse
509
+ # When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
510
+ # a 400 PDF_SKIPPED is returned.
511
+ #
512
+ # @return [Boolean, Symbol, ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf::ShouldParse, nil]
513
+ optional :should_parse,
514
+ union: -> {
515
+ ContextDev::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf::ShouldParse
516
+ },
517
+ api_name: :shouldParse
518
+
519
+ # @!attribute start
520
+ # First 1-based PDF page to parse. When omitted, parsing starts at the first page.
521
+ #
522
+ # @return [Integer, nil]
523
+ optional :start, Integer
524
+
525
+ # @!method initialize(end_: nil, ocr: nil, should_parse: nil, start: nil)
526
+ # Some parameter documentations has been truncated, see
527
+ # {ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf}
528
+ # for more details.
529
+ #
530
+ # PDF parsing controls. Use start/end to limit text extraction and embedded-image
531
+ # detection/OCR to an inclusive 1-based page range.
532
+ #
533
+ # @param end_ [Integer] Last 1-based PDF page to parse. When omitted, parsing ends at the last page. Mus
534
+ #
535
+ # @param ocr [Boolean, Symbol, ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf::Ocr] When true, OCR the selected PDF pages that have no usable text layer (scans), re
536
+ #
537
+ # @param should_parse [Boolean, Symbol, ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf::ShouldParse] When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
538
+ #
539
+ # @param start [Integer] First 1-based PDF page to parse. When omitted, parsing starts at the first page.
540
+
541
+ # When true, OCR the selected PDF pages that have no usable text layer (scans),
542
+ # replacing each recovered page's text with the OCR result while pages with a real
543
+ # text layer keep it. Billed at 1 credit per page OCR actually recovered, on top
544
+ # of the base request cost. When false, no OCR runs.
545
+ #
546
+ # @see ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf#ocr
547
+ module Ocr
548
+ extend ContextDev::Internal::Type::Union
549
+
550
+ variant ContextDev::Internal::Type::Boolean
551
+
552
+ variant const: -> { ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf::Ocr::TRUE }
553
+
554
+ variant const: -> { ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf::Ocr::FALSE }
555
+
556
+ # @!method self.variants
557
+ # @return [Array(Boolean, Symbol)]
558
+
559
+ define_sorbet_constant!(:Variants) do
560
+ T.type_alias do
561
+ T.any(
562
+ T::Boolean,
563
+ ContextDev::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf::Ocr::TaggedSymbol
564
+ )
565
+ end
566
+ end
567
+
568
+ # @!group
569
+
570
+ TRUE = :true
571
+ FALSE = :false
572
+
573
+ # @!endgroup
574
+ end
575
+
576
+ # When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
577
+ # a 400 PDF_SKIPPED is returned.
578
+ #
579
+ # @see ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf#should_parse
580
+ module ShouldParse
581
+ extend ContextDev::Internal::Type::Union
582
+
583
+ variant ContextDev::Internal::Type::Boolean
584
+
585
+ variant const: -> { ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf::ShouldParse::TRUE }
586
+
587
+ variant const: -> { ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf::ShouldParse::FALSE }
588
+
589
+ # @!method self.variants
590
+ # @return [Array(Boolean, Symbol)]
591
+
592
+ define_sorbet_constant!(:Variants) do
593
+ T.type_alias do
594
+ T.any(
595
+ T::Boolean,
596
+ ContextDev::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf::ShouldParse::TaggedSymbol
597
+ )
598
+ end
599
+ end
600
+
601
+ # @!group
602
+
603
+ TRUE = :true
604
+ FALSE = :false
605
+
606
+ # @!endgroup
607
+ end
608
+ end
609
+ end
610
+ end
611
+
612
+ class HTML < ContextDev::Internal::Type::BaseModel
613
+ # @!attribute format_
614
+ # Return page content as HTML.
615
+ #
616
+ # @return [Symbol, :html]
617
+ required :format_, const: :html, api_name: :format
618
+
619
+ # @!attribute urls
620
+ # Pages to scrape. Maximum 25000.
621
+ #
622
+ # @return [Array<ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::URL>]
623
+ required :urls,
624
+ -> { ContextDev::Internal::Type::ArrayOf[ContextDev::BatchSubmitParams::Input::Scrape::Data::HTML::URL] }
625
+
626
+ # @!attribute options
627
+ # Options for HTML output.
628
+ #
629
+ # @return [ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::Options, nil]
630
+ optional :options, -> { ContextDev::BatchSubmitParams::Input::Scrape::Data::HTML::Options }
631
+
632
+ # @!method initialize(urls:, options: nil, format_: :html)
633
+ # Scrape the listed pages as HTML.
634
+ #
635
+ # @param urls [Array<ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::URL>] Pages to scrape. Maximum 25000.
636
+ #
637
+ # @param options [ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::Options] Options for HTML output.
638
+ #
639
+ # @param format_ [Symbol, :html] Return page content as HTML.
640
+
641
+ class URL < ContextDev::Internal::Type::BaseModel
642
+ # @!attribute url
643
+ # Page URL to scrape.
644
+ #
645
+ # @return [String]
646
+ required :url, String
647
+
648
+ # @!attribute item_id
649
+ # Your ID for this page, returned with its result. The same URL can use different
650
+ # IDs.
651
+ #
652
+ # @return [String, nil]
653
+ optional :item_id, String, api_name: :itemId
654
+
655
+ # @!attribute meta
656
+ # Custom JSON returned unchanged with this page result.
657
+ #
658
+ # @return [Hash{Symbol=>Object}, nil]
659
+ optional :meta, ContextDev::Internal::Type::HashOf[ContextDev::Internal::Type::Unknown]
660
+
661
+ # @!method initialize(url:, item_id: nil, meta: nil)
662
+ # Some parameter documentations has been truncated, see
663
+ # {ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::URL} for more
664
+ # details.
665
+ #
666
+ # A page to scrape, with optional data for matching results.
667
+ #
668
+ # @param url [String] Page URL to scrape.
669
+ #
670
+ # @param item_id [String] Your ID for this page, returned with its result. The same URL can use different
671
+ #
672
+ # @param meta [Hash{Symbol=>Object}] Custom JSON returned unchanged with this page result.
673
+ end
674
+
675
+ # @see ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML#options
676
+ class Options < ContextDev::Internal::Type::BaseModel
677
+ # @!attribute country
678
+ # Fetch the target page through a residential proxy in this country (ISO 3166-1
679
+ # alpha-2).
680
+ #
681
+ # @return [Symbol, ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Country, nil]
682
+ optional :country, enum: -> { ContextDev::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Country }
683
+
684
+ # @!attribute exclude_selectors
685
+ # Remove elements matching these CSS selectors. Applied after `includeSelectors`,
686
+ # so an element matching both is removed.
687
+ #
688
+ # @return [Array<String>, nil]
689
+ optional :exclude_selectors,
690
+ ContextDev::Internal::Type::ArrayOf[String],
691
+ api_name: :excludeSelectors,
692
+ nil?: true
693
+
694
+ # @!attribute include_selectors
695
+ # Keep only the subtrees matching these CSS selectors. Filtered pages are always
696
+ # fetched fresh, ignoring `maxAgeMs`.
697
+ #
698
+ # @return [Array<String>, nil]
699
+ optional :include_selectors,
700
+ ContextDev::Internal::Type::ArrayOf[String],
701
+ api_name: :includeSelectors,
702
+ nil?: true
703
+
704
+ # @!attribute max_age_ms
705
+ # Return a cached result if a prior scrape for the same parameters exists and is
706
+ # younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
707
+ # omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
708
+ #
709
+ # @return [Integer, nil]
710
+ optional :max_age_ms, Integer, api_name: :maxAgeMs, nil?: true
711
+
712
+ # @!attribute pdf
713
+ # PDF parsing controls. Use start/end to limit text extraction and embedded-image
714
+ # detection/OCR to an inclusive 1-based page range.
715
+ #
716
+ # @return [ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf, nil]
717
+ optional :pdf, -> { ContextDev::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf }
718
+
719
+ # @!attribute settle_animations
720
+ # Wait briefly for CSS and transition animations to settle before extraction, on
721
+ # pages that render in a browser.
722
+ #
723
+ # @return [Boolean, nil]
724
+ optional :settle_animations, ContextDev::Internal::Type::Boolean, api_name: :settleAnimations
725
+
726
+ # @!attribute use_main_content_only
727
+ # Return the main content without navigation or footers.
728
+ #
729
+ # @return [Boolean, nil]
730
+ optional :use_main_content_only,
731
+ ContextDev::Internal::Type::Boolean,
732
+ api_name: :useMainContentOnly
733
+
734
+ # @!attribute wait_for_ms
735
+ # How long to wait after initial page load, in milliseconds. `0` waits 500 ms.
736
+ #
737
+ # @return [Integer, nil]
738
+ optional :wait_for_ms, Integer, api_name: :waitForMs
739
+
740
+ # @!method initialize(country: nil, exclude_selectors: nil, include_selectors: nil, max_age_ms: nil, pdf: nil, settle_animations: nil, use_main_content_only: nil, wait_for_ms: nil)
741
+ # Some parameter documentations has been truncated, see
742
+ # {ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::Options} for
743
+ # more details.
744
+ #
745
+ # Options for HTML output.
746
+ #
747
+ # @param country [Symbol, ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Country] Fetch the target page through a residential proxy in this country (ISO 3166-1 al
748
+ #
749
+ # @param exclude_selectors [Array<String>, nil] Remove elements matching these CSS selectors. Applied after `includeSelectors`,
750
+ #
751
+ # @param include_selectors [Array<String>, nil] Keep only the subtrees matching these CSS selectors. Filtered pages are always f
752
+ #
753
+ # @param max_age_ms [Integer, nil] Return a cached result if a prior scrape for the same parameters exists and is y
754
+ #
755
+ # @param pdf [ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf] PDF parsing controls. Use start/end to limit text extraction and embedded-image
756
+ #
757
+ # @param settle_animations [Boolean] Wait briefly for CSS and transition animations to settle before extraction, on p
758
+ #
759
+ # @param use_main_content_only [Boolean] Return the main content without navigation or footers.
760
+ #
761
+ # @param wait_for_ms [Integer] How long to wait after initial page load, in milliseconds. `0` waits 500 ms.
762
+
763
+ # Fetch the target page through a residential proxy in this country (ISO 3166-1
764
+ # alpha-2).
765
+ #
766
+ # @see ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::Options#country
767
+ module Country
768
+ extend ContextDev::Internal::Type::Enum
769
+
770
+ AD = :ad
771
+ AE = :ae
772
+ AF = :af
773
+ AG = :ag
774
+ AI = :ai
775
+ AL = :al
776
+ AM = :am
777
+ AO = :ao
778
+ AR = :ar
779
+ AT = :at
780
+ AU = :au
781
+ AW = :aw
782
+ AZ = :az
783
+ BA = :ba
784
+ BB = :bb
785
+ BD = :bd
786
+ BE = :be
787
+ BF = :bf
788
+ BG = :bg
789
+ BH = :bh
790
+ BI = :bi
791
+ BJ = :bj
792
+ BM = :bm
793
+ BN = :bn
794
+ BO = :bo
795
+ BQ = :bq
796
+ BR = :br
797
+ BS = :bs
798
+ BW = :bw
799
+ BY = :by
800
+ BZ = :bz
801
+ CA = :ca
802
+ CD = :cd
803
+ CF = :cf
804
+ CG = :cg
805
+ CH = :ch
806
+ CI = :ci
807
+ CL = :cl
808
+ CM = :cm
809
+ CN = :cn
810
+ CO = :co
811
+ CR = :cr
812
+ CV = :cv
813
+ CW = :cw
814
+ CY = :cy
815
+ CZ = :cz
816
+ DE = :de
817
+ DJ = :dj
818
+ DK = :dk
819
+ DM = :dm
820
+ DO = :do
821
+ DZ = :dz
822
+ EC = :ec
823
+ EE = :ee
824
+ EG = :eg
825
+ ES = :es
826
+ ET = :et
827
+ FI = :fi
828
+ FJ = :fj
829
+ FR = :fr
830
+ GA = :ga
831
+ GB = :gb
832
+ GD = :gd
833
+ GE = :ge
834
+ GF = :gf
835
+ GG = :gg
836
+ GH = :gh
837
+ GM = :gm
838
+ GN = :gn
839
+ GP = :gp
840
+ GQ = :gq
841
+ GR = :gr
842
+ GT = :gt
843
+ GU = :gu
844
+ GW = :gw
845
+ GY = :gy
846
+ HK = :hk
847
+ HN = :hn
848
+ HR = :hr
849
+ HT = :ht
850
+ HU = :hu
851
+ ID = :id
852
+ IE = :ie
853
+ IL = :il
854
+ IM = :im
855
+ IN = :in
856
+ IQ = :iq
857
+ IR = :ir
858
+ IS = :is
859
+ IT = :it
860
+ JE = :je
861
+ JM = :jm
862
+ JO = :jo
863
+ JP = :jp
864
+ KE = :ke
865
+ KG = :kg
866
+ KH = :kh
867
+ KN = :kn
868
+ KR = :kr
869
+ KW = :kw
870
+ KY = :ky
871
+ KZ = :kz
872
+ LA = :la
873
+ LB = :lb
874
+ LC = :lc
875
+ LK = :lk
876
+ LR = :lr
877
+ LS = :ls
878
+ LT = :lt
879
+ LU = :lu
880
+ LV = :lv
881
+ LY = :ly
882
+ MA = :ma
883
+ MC = :mc
884
+ MD = :md
885
+ ME = :me
886
+ MF = :mf
887
+ MG = :mg
888
+ MK = :mk
889
+ ML = :ml
890
+ MM = :mm
891
+ MN = :mn
892
+ MO = :mo
893
+ MQ = :mq
894
+ MR = :mr
895
+ MT = :mt
896
+ MU = :mu
897
+ MV = :mv
898
+ MW = :mw
899
+ MX = :mx
900
+ MY = :my
901
+ MZ = :mz
902
+ NA = :na
903
+ NC = :nc
904
+ NE = :ne
905
+ NG = :ng
906
+ NI = :ni
907
+ NL = :nl
908
+ NO = :no
909
+ NP = :np
910
+ NZ = :nz
911
+ OM = :om
912
+ PA = :pa
913
+ PE = :pe
914
+ PF = :pf
915
+ PG = :pg
916
+ PH = :ph
917
+ PK = :pk
918
+ PL = :pl
919
+ PR = :pr
920
+ PS = :ps
921
+ PT = :pt
922
+ PY = :py
923
+ QA = :qa
924
+ RE = :re
925
+ RO = :ro
926
+ RS = :rs
927
+ RU = :ru
928
+ RW = :rw
929
+ SA = :sa
930
+ SC = :sc
931
+ SD = :sd
932
+ SE = :se
933
+ SG = :sg
934
+ SI = :si
935
+ SK = :sk
936
+ SL = :sl
937
+ SM = :sm
938
+ SN = :sn
939
+ SO = :so
940
+ SR = :sr
941
+ SS = :ss
942
+ ST = :st
943
+ SV = :sv
944
+ SX = :sx
945
+ SY = :sy
946
+ SZ = :sz
947
+ TC = :tc
948
+ TD = :td
949
+ TG = :tg
950
+ TH = :th
951
+ TJ = :tj
952
+ TL = :tl
953
+ TM = :tm
954
+ TN = :tn
955
+ TR = :tr
956
+ TT = :tt
957
+ TW = :tw
958
+ TZ = :tz
959
+ UA = :ua
960
+ UG = :ug
961
+ US = :us
962
+ UY = :uy
963
+ UZ = :uz
964
+ VC = :vc
965
+ VE = :ve
966
+ VG = :vg
967
+ VI = :vi
968
+ VN = :vn
969
+ YE = :ye
970
+ YT = :yt
971
+ ZA = :za
972
+ ZM = :zm
973
+ ZW = :zw
974
+
975
+ # @!method self.values
976
+ # @return [Array<Symbol>]
977
+ end
978
+
979
+ # @see ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::Options#pdf
980
+ class Pdf < ContextDev::Internal::Type::BaseModel
981
+ # @!attribute end_
982
+ # Last 1-based PDF page to parse. When omitted, parsing ends at the last page.
983
+ # Must be greater than or equal to start when both are provided.
984
+ #
985
+ # @return [Integer, nil]
986
+ optional :end_, Integer, api_name: :end
987
+
988
+ # @!attribute ocr
989
+ # When true, OCR the selected PDF pages that have no usable text layer (scans),
990
+ # replacing each recovered page's text with the OCR result while pages with a real
991
+ # text layer keep it. Billed at 1 credit per page OCR actually recovered, on top
992
+ # of the base request cost. When false, no OCR runs.
993
+ #
994
+ # @return [Boolean, Symbol, ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf::Ocr, nil]
995
+ optional :ocr, union: -> { ContextDev::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf::Ocr }
996
+
997
+ # @!attribute should_parse
998
+ # When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
999
+ # a 400 PDF_SKIPPED is returned.
1000
+ #
1001
+ # @return [Boolean, Symbol, ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf::ShouldParse, nil]
1002
+ optional :should_parse,
1003
+ union: -> {
1004
+ ContextDev::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf::ShouldParse
1005
+ },
1006
+ api_name: :shouldParse
1007
+
1008
+ # @!attribute start
1009
+ # First 1-based PDF page to parse. When omitted, parsing starts at the first page.
1010
+ #
1011
+ # @return [Integer, nil]
1012
+ optional :start, Integer
1013
+
1014
+ # @!method initialize(end_: nil, ocr: nil, should_parse: nil, start: nil)
1015
+ # Some parameter documentations has been truncated, see
1016
+ # {ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf}
1017
+ # for more details.
1018
+ #
1019
+ # PDF parsing controls. Use start/end to limit text extraction and embedded-image
1020
+ # detection/OCR to an inclusive 1-based page range.
1021
+ #
1022
+ # @param end_ [Integer] Last 1-based PDF page to parse. When omitted, parsing ends at the last page. Mus
1023
+ #
1024
+ # @param ocr [Boolean, Symbol, ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf::Ocr] When true, OCR the selected PDF pages that have no usable text layer (scans), re
1025
+ #
1026
+ # @param should_parse [Boolean, Symbol, ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf::ShouldParse] When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
1027
+ #
1028
+ # @param start [Integer] First 1-based PDF page to parse. When omitted, parsing starts at the first page.
1029
+
1030
+ # When true, OCR the selected PDF pages that have no usable text layer (scans),
1031
+ # replacing each recovered page's text with the OCR result while pages with a real
1032
+ # text layer keep it. Billed at 1 credit per page OCR actually recovered, on top
1033
+ # of the base request cost. When false, no OCR runs.
1034
+ #
1035
+ # @see ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf#ocr
1036
+ module Ocr
1037
+ extend ContextDev::Internal::Type::Union
1038
+
1039
+ variant ContextDev::Internal::Type::Boolean
1040
+
1041
+ variant const: -> { ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf::Ocr::TRUE }
1042
+
1043
+ variant const: -> { ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf::Ocr::FALSE }
1044
+
1045
+ # @!method self.variants
1046
+ # @return [Array(Boolean, Symbol)]
1047
+
1048
+ define_sorbet_constant!(:Variants) do
1049
+ T.type_alias do
1050
+ T.any(
1051
+ T::Boolean,
1052
+ ContextDev::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf::Ocr::TaggedSymbol
1053
+ )
1054
+ end
1055
+ end
1056
+
1057
+ # @!group
1058
+
1059
+ TRUE = :true
1060
+ FALSE = :false
1061
+
1062
+ # @!endgroup
1063
+ end
1064
+
1065
+ # When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
1066
+ # a 400 PDF_SKIPPED is returned.
1067
+ #
1068
+ # @see ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf#should_parse
1069
+ module ShouldParse
1070
+ extend ContextDev::Internal::Type::Union
1071
+
1072
+ variant ContextDev::Internal::Type::Boolean
1073
+
1074
+ variant const: -> { ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf::ShouldParse::TRUE }
1075
+
1076
+ variant const: -> { ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf::ShouldParse::FALSE }
1077
+
1078
+ # @!method self.variants
1079
+ # @return [Array(Boolean, Symbol)]
1080
+
1081
+ define_sorbet_constant!(:Variants) do
1082
+ T.type_alias do
1083
+ T.any(
1084
+ T::Boolean,
1085
+ ContextDev::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf::ShouldParse::TaggedSymbol
1086
+ )
1087
+ end
1088
+ end
1089
+
1090
+ # @!group
1091
+
1092
+ TRUE = :true
1093
+ FALSE = :false
1094
+
1095
+ # @!endgroup
1096
+ end
1097
+ end
1098
+ end
1099
+ end
1100
+
1101
+ # @!method self.variants
1102
+ # @return [Array(ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown, ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML)]
1103
+ end
1104
+ end
1105
+
1106
+ class Crawl < ContextDev::Internal::Type::BaseModel
1107
+ # @!attribute data
1108
+ # Crawl source and output format.
1109
+ #
1110
+ # @return [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown, ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML]
1111
+ required :data, union: -> { ContextDev::BatchSubmitParams::Input::Crawl::Data }
1112
+
1113
+ # @!attribute mode
1114
+ # Discover and scrape pages from `data.source`.
1115
+ #
1116
+ # @return [Symbol, :crawl]
1117
+ required :mode, const: :crawl
1118
+
1119
+ # @!method initialize(data:, mode: :crawl)
1120
+ # Crawl pages starting from a URL or from a domain's sitemap.
1121
+ #
1122
+ # @param data [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown, ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML] Crawl source and output format.
1123
+ #
1124
+ # @param mode [Symbol, :crawl] Discover and scrape pages from `data.source`.
1125
+
1126
+ # Crawl source and output format.
1127
+ #
1128
+ # @see ContextDev::Models::BatchSubmitParams::Input::Crawl#data
1129
+ module Data
1130
+ extend ContextDev::Internal::Type::Union
1131
+
1132
+ discriminator :format
1133
+
1134
+ # Crawl pages and return Markdown.
1135
+ variant :markdown, -> { ContextDev::BatchSubmitParams::Input::Crawl::Data::Markdown }
1136
+
1137
+ # Crawl pages and return HTML.
1138
+ variant :html, -> { ContextDev::BatchSubmitParams::Input::Crawl::Data::HTML }
1139
+
1140
+ class Markdown < ContextDev::Internal::Type::BaseModel
1141
+ # @!attribute format_
1142
+ # Return page content as Markdown.
1143
+ #
1144
+ # @return [Symbol, :markdown]
1145
+ required :format_, const: :markdown, api_name: :format
1146
+
1147
+ # @!attribute source
1148
+ # How to find pages to crawl.
1149
+ #
1150
+ # @return [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Source::StartURL, ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Source::Sitemap]
1151
+ required :source, union: -> { ContextDev::BatchSubmitParams::Input::Crawl::Data::Markdown::Source }
1152
+
1153
+ # @!attribute options
1154
+ # Options for Markdown output.
1155
+ #
1156
+ # @return [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Options, nil]
1157
+ optional :options, -> { ContextDev::BatchSubmitParams::Input::Crawl::Data::Markdown::Options }
1158
+
1159
+ # @!method initialize(source:, options: nil, format_: :markdown)
1160
+ # Crawl pages and return Markdown.
1161
+ #
1162
+ # @param source [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Source::StartURL, ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Source::Sitemap] How to find pages to crawl.
1163
+ #
1164
+ # @param options [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Options] Options for Markdown output.
1165
+ #
1166
+ # @param format_ [Symbol, :markdown] Return page content as Markdown.
1167
+
1168
+ # How to find pages to crawl.
1169
+ #
1170
+ # @see ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown#source
1171
+ module Source
1172
+ extend ContextDev::Internal::Type::Union
1173
+
1174
+ discriminator :type
1175
+
1176
+ # Discover pages by following links from one URL.
1177
+ variant :start_url, -> { ContextDev::BatchSubmitParams::Input::Crawl::Data::Markdown::Source::StartURL }
1178
+
1179
+ # Scrape the pages listed in a domain's sitemap. Links on those pages are not followed.
1180
+ variant :sitemap, -> { ContextDev::BatchSubmitParams::Input::Crawl::Data::Markdown::Source::Sitemap }
1181
+
1182
+ class StartURL < ContextDev::Internal::Type::BaseModel
1183
+ # @!attribute type
1184
+ # Start from one page.
1185
+ #
1186
+ # @return [Symbol, :start_url]
1187
+ required :type, const: :start_url
1188
+
1189
+ # @!attribute url
1190
+ # Page where crawling begins. A URL without a scheme is read as https://.
1191
+ #
1192
+ # @return [String]
1193
+ required :url, String
1194
+
1195
+ # @!attribute controls
1196
+ # Limits and filters for page discovery.
1197
+ #
1198
+ # @return [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Source::StartURL::Controls, nil]
1199
+ optional :controls,
1200
+ -> { ContextDev::BatchSubmitParams::Input::Crawl::Data::Markdown::Source::StartURL::Controls }
1201
+
1202
+ # @!method initialize(url:, controls: nil, type: :start_url)
1203
+ # Discover pages by following links from one URL.
1204
+ #
1205
+ # @param url [String] Page where crawling begins. A URL without a scheme is read as https://.
1206
+ #
1207
+ # @param controls [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Source::StartURL::Controls] Limits and filters for page discovery.
1208
+ #
1209
+ # @param type [Symbol, :start_url] Start from one page.
1210
+
1211
+ # @see ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Source::StartURL#controls
1212
+ class Controls < ContextDev::Internal::Type::BaseModel
1213
+ # @!attribute follow_subdomains
1214
+ # Follow links to subdomains.
1215
+ #
1216
+ # @return [Boolean, nil]
1217
+ optional :follow_subdomains,
1218
+ ContextDev::Internal::Type::Boolean,
1219
+ api_name: :followSubdomains
1220
+
1221
+ # @!attribute max_depth
1222
+ # Maximum link depth. Source pages are depth 0. No limit when omitted.
1223
+ #
1224
+ # @return [Integer, nil]
1225
+ optional :max_depth, Integer, api_name: :maxDepth
1226
+
1227
+ # @!attribute max_urls
1228
+ # Maximum pages to fetch. Unused reserved credits are refunded. Maximum 25000.
1229
+ #
1230
+ # @return [Integer, nil]
1231
+ optional :max_urls, Integer, api_name: :maxUrls
1232
+
1233
+ # @!attribute regex
1234
+ # RE2 pattern for URLs to include. The `start_url` itself is always included.
1235
+ #
1236
+ # @return [String, nil]
1237
+ optional :regex, String
1238
+
1239
+ # @!method initialize(follow_subdomains: nil, max_depth: nil, max_urls: nil, regex: nil)
1240
+ # Limits and filters for page discovery.
1241
+ #
1242
+ # @param follow_subdomains [Boolean] Follow links to subdomains.
1243
+ #
1244
+ # @param max_depth [Integer] Maximum link depth. Source pages are depth 0. No limit when omitted.
1245
+ #
1246
+ # @param max_urls [Integer] Maximum pages to fetch. Unused reserved credits are refunded. Maximum 25000.
1247
+ #
1248
+ # @param regex [String] RE2 pattern for URLs to include. The `start_url` itself is always included.
1249
+ end
1250
+ end
1251
+
1252
+ class Sitemap < ContextDev::Internal::Type::BaseModel
1253
+ # @!attribute domain
1254
+ # Domain whose sitemap lists the pages to scrape. A full URL is reduced to its
1255
+ # domain.
1256
+ #
1257
+ # @return [String]
1258
+ required :domain, String
1259
+
1260
+ # @!attribute type
1261
+ # Scrape the URLs in the domain's sitemap.
1262
+ #
1263
+ # @return [Symbol, :sitemap]
1264
+ required :type, const: :sitemap
1265
+
1266
+ # @!attribute controls
1267
+ # Limits and filters for the sitemap URLs. A sitemap batch scrapes exactly those
1268
+ # URLs and never follows links off them, so there is no crawl depth here.
1269
+ #
1270
+ # @return [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Source::Sitemap::Controls, nil]
1271
+ optional :controls,
1272
+ -> { ContextDev::BatchSubmitParams::Input::Crawl::Data::Markdown::Source::Sitemap::Controls }
1273
+
1274
+ # @!method initialize(domain:, controls: nil, type: :sitemap)
1275
+ # Some parameter documentations has been truncated, see
1276
+ # {ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Source::Sitemap}
1277
+ # for more details.
1278
+ #
1279
+ # Scrape the pages listed in a domain's sitemap. Links on those pages are not
1280
+ # followed.
1281
+ #
1282
+ # @param domain [String] Domain whose sitemap lists the pages to scrape. A full URL is reduced to its dom
1283
+ #
1284
+ # @param controls [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Source::Sitemap::Controls] Limits and filters for the sitemap URLs. A sitemap batch scrapes exactly those U
1285
+ #
1286
+ # @param type [Symbol, :sitemap] Scrape the URLs in the domain's sitemap.
1287
+
1288
+ # @see ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Source::Sitemap#controls
1289
+ class Controls < ContextDev::Internal::Type::BaseModel
1290
+ # @!attribute max_urls
1291
+ # Maximum pages to fetch. Unused reserved credits are refunded. Maximum 25000.
1292
+ #
1293
+ # @return [Integer, nil]
1294
+ optional :max_urls, Integer, api_name: :maxUrls
1295
+
1296
+ # @!attribute regex
1297
+ # RE2 pattern; only sitemap URLs matching it are scraped.
1298
+ #
1299
+ # @return [String, nil]
1300
+ optional :regex, String
1301
+
1302
+ # @!method initialize(max_urls: nil, regex: nil)
1303
+ # Limits and filters for the sitemap URLs. A sitemap batch scrapes exactly those
1304
+ # URLs and never follows links off them, so there is no crawl depth here.
1305
+ #
1306
+ # @param max_urls [Integer] Maximum pages to fetch. Unused reserved credits are refunded. Maximum 25000.
1307
+ #
1308
+ # @param regex [String] RE2 pattern; only sitemap URLs matching it are scraped.
1309
+ end
1310
+ end
1311
+
1312
+ # @!method self.variants
1313
+ # @return [Array(ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Source::StartURL, ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Source::Sitemap)]
1314
+ end
1315
+
1316
+ # @see ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown#options
1317
+ class Options < ContextDev::Internal::Type::BaseModel
1318
+ # @!attribute country
1319
+ # Fetch the target page through a residential proxy in this country (ISO 3166-1
1320
+ # alpha-2).
1321
+ #
1322
+ # @return [Symbol, ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Country, nil]
1323
+ optional :country,
1324
+ enum: -> { ContextDev::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Country }
1325
+
1326
+ # @!attribute exclude_selectors
1327
+ # Remove elements matching these CSS selectors. Applied after `includeSelectors`,
1328
+ # so an element matching both is removed.
1329
+ #
1330
+ # @return [Array<String>, nil]
1331
+ optional :exclude_selectors,
1332
+ ContextDev::Internal::Type::ArrayOf[String],
1333
+ api_name: :excludeSelectors,
1334
+ nil?: true
1335
+
1336
+ # @!attribute include_images
1337
+ # Include image references in the Markdown.
1338
+ #
1339
+ # @return [Boolean, nil]
1340
+ optional :include_images, ContextDev::Internal::Type::Boolean, api_name: :includeImages
1341
+
1342
+ # @!attribute include_links
1343
+ # Include links in the Markdown.
1344
+ #
1345
+ # @return [Boolean, nil]
1346
+ optional :include_links, ContextDev::Internal::Type::Boolean, api_name: :includeLinks
1347
+
1348
+ # @!attribute include_selectors
1349
+ # Keep only the subtrees matching these CSS selectors. Filtered pages are always
1350
+ # fetched fresh, ignoring `maxAgeMs`.
1351
+ #
1352
+ # @return [Array<String>, nil]
1353
+ optional :include_selectors,
1354
+ ContextDev::Internal::Type::ArrayOf[String],
1355
+ api_name: :includeSelectors,
1356
+ nil?: true
1357
+
1358
+ # @!attribute max_age_ms
1359
+ # Return a cached result if a prior scrape for the same parameters exists and is
1360
+ # younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
1361
+ # omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
1362
+ #
1363
+ # @return [Integer, nil]
1364
+ optional :max_age_ms, Integer, api_name: :maxAgeMs, nil?: true
1365
+
1366
+ # @!attribute pdf
1367
+ # PDF parsing controls. Use start/end to limit text extraction and embedded-image
1368
+ # detection/OCR to an inclusive 1-based page range.
1369
+ #
1370
+ # @return [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf, nil]
1371
+ optional :pdf, -> { ContextDev::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf }
1372
+
1373
+ # @!attribute settle_animations
1374
+ # Wait briefly for CSS and transition animations to settle before extraction, on
1375
+ # pages that render in a browser.
1376
+ #
1377
+ # @return [Boolean, nil]
1378
+ optional :settle_animations, ContextDev::Internal::Type::Boolean, api_name: :settleAnimations
1379
+
1380
+ # @!attribute shorten_base64_images
1381
+ # Shorten inline base64 image data.
1382
+ #
1383
+ # @return [Boolean, nil]
1384
+ optional :shorten_base64_images,
1385
+ ContextDev::Internal::Type::Boolean,
1386
+ api_name: :shortenBase64Images
1387
+
1388
+ # @!attribute use_main_content_only
1389
+ # Return the main content without navigation or footers.
1390
+ #
1391
+ # @return [Boolean, nil]
1392
+ optional :use_main_content_only,
1393
+ ContextDev::Internal::Type::Boolean,
1394
+ api_name: :useMainContentOnly
1395
+
1396
+ # @!attribute wait_for_ms
1397
+ # How long to wait after initial page load, in milliseconds. `0` waits 500 ms.
1398
+ #
1399
+ # @return [Integer, nil]
1400
+ optional :wait_for_ms, Integer, api_name: :waitForMs
1401
+
1402
+ # @!method initialize(country: nil, exclude_selectors: nil, include_images: nil, include_links: nil, include_selectors: nil, max_age_ms: nil, pdf: nil, settle_animations: nil, shorten_base64_images: nil, use_main_content_only: nil, wait_for_ms: nil)
1403
+ # Some parameter documentations has been truncated, see
1404
+ # {ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Options}
1405
+ # for more details.
1406
+ #
1407
+ # Options for Markdown output.
1408
+ #
1409
+ # @param country [Symbol, ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Country] Fetch the target page through a residential proxy in this country (ISO 3166-1 al
1410
+ #
1411
+ # @param exclude_selectors [Array<String>, nil] Remove elements matching these CSS selectors. Applied after `includeSelectors`,
1412
+ #
1413
+ # @param include_images [Boolean] Include image references in the Markdown.
1414
+ #
1415
+ # @param include_links [Boolean] Include links in the Markdown.
1416
+ #
1417
+ # @param include_selectors [Array<String>, nil] Keep only the subtrees matching these CSS selectors. Filtered pages are always f
1418
+ #
1419
+ # @param max_age_ms [Integer, nil] Return a cached result if a prior scrape for the same parameters exists and is y
1420
+ #
1421
+ # @param pdf [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf] PDF parsing controls. Use start/end to limit text extraction and embedded-image
1422
+ #
1423
+ # @param settle_animations [Boolean] Wait briefly for CSS and transition animations to settle before extraction, on p
1424
+ #
1425
+ # @param shorten_base64_images [Boolean] Shorten inline base64 image data.
1426
+ #
1427
+ # @param use_main_content_only [Boolean] Return the main content without navigation or footers.
1428
+ #
1429
+ # @param wait_for_ms [Integer] How long to wait after initial page load, in milliseconds. `0` waits 500 ms.
1430
+
1431
+ # Fetch the target page through a residential proxy in this country (ISO 3166-1
1432
+ # alpha-2).
1433
+ #
1434
+ # @see ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Options#country
1435
+ module Country
1436
+ extend ContextDev::Internal::Type::Enum
1437
+
1438
+ AD = :ad
1439
+ AE = :ae
1440
+ AF = :af
1441
+ AG = :ag
1442
+ AI = :ai
1443
+ AL = :al
1444
+ AM = :am
1445
+ AO = :ao
1446
+ AR = :ar
1447
+ AT = :at
1448
+ AU = :au
1449
+ AW = :aw
1450
+ AZ = :az
1451
+ BA = :ba
1452
+ BB = :bb
1453
+ BD = :bd
1454
+ BE = :be
1455
+ BF = :bf
1456
+ BG = :bg
1457
+ BH = :bh
1458
+ BI = :bi
1459
+ BJ = :bj
1460
+ BM = :bm
1461
+ BN = :bn
1462
+ BO = :bo
1463
+ BQ = :bq
1464
+ BR = :br
1465
+ BS = :bs
1466
+ BW = :bw
1467
+ BY = :by
1468
+ BZ = :bz
1469
+ CA = :ca
1470
+ CD = :cd
1471
+ CF = :cf
1472
+ CG = :cg
1473
+ CH = :ch
1474
+ CI = :ci
1475
+ CL = :cl
1476
+ CM = :cm
1477
+ CN = :cn
1478
+ CO = :co
1479
+ CR = :cr
1480
+ CV = :cv
1481
+ CW = :cw
1482
+ CY = :cy
1483
+ CZ = :cz
1484
+ DE = :de
1485
+ DJ = :dj
1486
+ DK = :dk
1487
+ DM = :dm
1488
+ DO = :do
1489
+ DZ = :dz
1490
+ EC = :ec
1491
+ EE = :ee
1492
+ EG = :eg
1493
+ ES = :es
1494
+ ET = :et
1495
+ FI = :fi
1496
+ FJ = :fj
1497
+ FR = :fr
1498
+ GA = :ga
1499
+ GB = :gb
1500
+ GD = :gd
1501
+ GE = :ge
1502
+ GF = :gf
1503
+ GG = :gg
1504
+ GH = :gh
1505
+ GM = :gm
1506
+ GN = :gn
1507
+ GP = :gp
1508
+ GQ = :gq
1509
+ GR = :gr
1510
+ GT = :gt
1511
+ GU = :gu
1512
+ GW = :gw
1513
+ GY = :gy
1514
+ HK = :hk
1515
+ HN = :hn
1516
+ HR = :hr
1517
+ HT = :ht
1518
+ HU = :hu
1519
+ ID = :id
1520
+ IE = :ie
1521
+ IL = :il
1522
+ IM = :im
1523
+ IN = :in
1524
+ IQ = :iq
1525
+ IR = :ir
1526
+ IS = :is
1527
+ IT = :it
1528
+ JE = :je
1529
+ JM = :jm
1530
+ JO = :jo
1531
+ JP = :jp
1532
+ KE = :ke
1533
+ KG = :kg
1534
+ KH = :kh
1535
+ KN = :kn
1536
+ KR = :kr
1537
+ KW = :kw
1538
+ KY = :ky
1539
+ KZ = :kz
1540
+ LA = :la
1541
+ LB = :lb
1542
+ LC = :lc
1543
+ LK = :lk
1544
+ LR = :lr
1545
+ LS = :ls
1546
+ LT = :lt
1547
+ LU = :lu
1548
+ LV = :lv
1549
+ LY = :ly
1550
+ MA = :ma
1551
+ MC = :mc
1552
+ MD = :md
1553
+ ME = :me
1554
+ MF = :mf
1555
+ MG = :mg
1556
+ MK = :mk
1557
+ ML = :ml
1558
+ MM = :mm
1559
+ MN = :mn
1560
+ MO = :mo
1561
+ MQ = :mq
1562
+ MR = :mr
1563
+ MT = :mt
1564
+ MU = :mu
1565
+ MV = :mv
1566
+ MW = :mw
1567
+ MX = :mx
1568
+ MY = :my
1569
+ MZ = :mz
1570
+ NA = :na
1571
+ NC = :nc
1572
+ NE = :ne
1573
+ NG = :ng
1574
+ NI = :ni
1575
+ NL = :nl
1576
+ NO = :no
1577
+ NP = :np
1578
+ NZ = :nz
1579
+ OM = :om
1580
+ PA = :pa
1581
+ PE = :pe
1582
+ PF = :pf
1583
+ PG = :pg
1584
+ PH = :ph
1585
+ PK = :pk
1586
+ PL = :pl
1587
+ PR = :pr
1588
+ PS = :ps
1589
+ PT = :pt
1590
+ PY = :py
1591
+ QA = :qa
1592
+ RE = :re
1593
+ RO = :ro
1594
+ RS = :rs
1595
+ RU = :ru
1596
+ RW = :rw
1597
+ SA = :sa
1598
+ SC = :sc
1599
+ SD = :sd
1600
+ SE = :se
1601
+ SG = :sg
1602
+ SI = :si
1603
+ SK = :sk
1604
+ SL = :sl
1605
+ SM = :sm
1606
+ SN = :sn
1607
+ SO = :so
1608
+ SR = :sr
1609
+ SS = :ss
1610
+ ST = :st
1611
+ SV = :sv
1612
+ SX = :sx
1613
+ SY = :sy
1614
+ SZ = :sz
1615
+ TC = :tc
1616
+ TD = :td
1617
+ TG = :tg
1618
+ TH = :th
1619
+ TJ = :tj
1620
+ TL = :tl
1621
+ TM = :tm
1622
+ TN = :tn
1623
+ TR = :tr
1624
+ TT = :tt
1625
+ TW = :tw
1626
+ TZ = :tz
1627
+ UA = :ua
1628
+ UG = :ug
1629
+ US = :us
1630
+ UY = :uy
1631
+ UZ = :uz
1632
+ VC = :vc
1633
+ VE = :ve
1634
+ VG = :vg
1635
+ VI = :vi
1636
+ VN = :vn
1637
+ YE = :ye
1638
+ YT = :yt
1639
+ ZA = :za
1640
+ ZM = :zm
1641
+ ZW = :zw
1642
+
1643
+ # @!method self.values
1644
+ # @return [Array<Symbol>]
1645
+ end
1646
+
1647
+ # @see ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Options#pdf
1648
+ class Pdf < ContextDev::Internal::Type::BaseModel
1649
+ # @!attribute end_
1650
+ # Last 1-based PDF page to parse. When omitted, parsing ends at the last page.
1651
+ # Must be greater than or equal to start when both are provided.
1652
+ #
1653
+ # @return [Integer, nil]
1654
+ optional :end_, Integer, api_name: :end
1655
+
1656
+ # @!attribute ocr
1657
+ # When true, OCR the selected PDF pages that have no usable text layer (scans),
1658
+ # replacing each recovered page's text with the OCR result while pages with a real
1659
+ # text layer keep it. Billed at 1 credit per page OCR actually recovered, on top
1660
+ # of the base request cost. When false, no OCR runs.
1661
+ #
1662
+ # @return [Boolean, Symbol, ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf::Ocr, nil]
1663
+ optional :ocr,
1664
+ union: -> { ContextDev::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf::Ocr }
1665
+
1666
+ # @!attribute should_parse
1667
+ # When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
1668
+ # a 400 PDF_SKIPPED is returned.
1669
+ #
1670
+ # @return [Boolean, Symbol, ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf::ShouldParse, nil]
1671
+ optional :should_parse,
1672
+ union: -> {
1673
+ ContextDev::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf::ShouldParse
1674
+ },
1675
+ api_name: :shouldParse
1676
+
1677
+ # @!attribute start
1678
+ # First 1-based PDF page to parse. When omitted, parsing starts at the first page.
1679
+ #
1680
+ # @return [Integer, nil]
1681
+ optional :start, Integer
1682
+
1683
+ # @!method initialize(end_: nil, ocr: nil, should_parse: nil, start: nil)
1684
+ # Some parameter documentations has been truncated, see
1685
+ # {ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf}
1686
+ # for more details.
1687
+ #
1688
+ # PDF parsing controls. Use start/end to limit text extraction and embedded-image
1689
+ # detection/OCR to an inclusive 1-based page range.
1690
+ #
1691
+ # @param end_ [Integer] Last 1-based PDF page to parse. When omitted, parsing ends at the last page. Mus
1692
+ #
1693
+ # @param ocr [Boolean, Symbol, ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf::Ocr] When true, OCR the selected PDF pages that have no usable text layer (scans), re
1694
+ #
1695
+ # @param should_parse [Boolean, Symbol, ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf::ShouldParse] When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
1696
+ #
1697
+ # @param start [Integer] First 1-based PDF page to parse. When omitted, parsing starts at the first page.
1698
+
1699
+ # When true, OCR the selected PDF pages that have no usable text layer (scans),
1700
+ # replacing each recovered page's text with the OCR result while pages with a real
1701
+ # text layer keep it. Billed at 1 credit per page OCR actually recovered, on top
1702
+ # of the base request cost. When false, no OCR runs.
1703
+ #
1704
+ # @see ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf#ocr
1705
+ module Ocr
1706
+ extend ContextDev::Internal::Type::Union
1707
+
1708
+ variant ContextDev::Internal::Type::Boolean
1709
+
1710
+ variant const: -> { ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf::Ocr::TRUE }
1711
+
1712
+ variant const: -> { ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf::Ocr::FALSE }
1713
+
1714
+ # @!method self.variants
1715
+ # @return [Array(Boolean, Symbol)]
1716
+
1717
+ define_sorbet_constant!(:Variants) do
1718
+ T.type_alias do
1719
+ T.any(
1720
+ T::Boolean,
1721
+ ContextDev::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf::Ocr::TaggedSymbol
1722
+ )
1723
+ end
1724
+ end
1725
+
1726
+ # @!group
1727
+
1728
+ TRUE = :true
1729
+ FALSE = :false
1730
+
1731
+ # @!endgroup
1732
+ end
1733
+
1734
+ # When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
1735
+ # a 400 PDF_SKIPPED is returned.
1736
+ #
1737
+ # @see ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf#should_parse
1738
+ module ShouldParse
1739
+ extend ContextDev::Internal::Type::Union
1740
+
1741
+ variant ContextDev::Internal::Type::Boolean
1742
+
1743
+ variant const: -> { ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf::ShouldParse::TRUE }
1744
+
1745
+ variant const: -> { ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf::ShouldParse::FALSE }
1746
+
1747
+ # @!method self.variants
1748
+ # @return [Array(Boolean, Symbol)]
1749
+
1750
+ define_sorbet_constant!(:Variants) do
1751
+ T.type_alias do
1752
+ T.any(
1753
+ T::Boolean,
1754
+ ContextDev::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf::ShouldParse::TaggedSymbol
1755
+ )
1756
+ end
1757
+ end
1758
+
1759
+ # @!group
1760
+
1761
+ TRUE = :true
1762
+ FALSE = :false
1763
+
1764
+ # @!endgroup
1765
+ end
1766
+ end
1767
+ end
1768
+ end
1769
+
1770
+ class HTML < ContextDev::Internal::Type::BaseModel
1771
+ # @!attribute format_
1772
+ # Return page content as HTML.
1773
+ #
1774
+ # @return [Symbol, :html]
1775
+ required :format_, const: :html, api_name: :format
1776
+
1777
+ # @!attribute source
1778
+ # How to find pages to crawl.
1779
+ #
1780
+ # @return [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Source::StartURL, ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Source::Sitemap]
1781
+ required :source, union: -> { ContextDev::BatchSubmitParams::Input::Crawl::Data::HTML::Source }
1782
+
1783
+ # @!attribute options
1784
+ # Options for HTML output.
1785
+ #
1786
+ # @return [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Options, nil]
1787
+ optional :options, -> { ContextDev::BatchSubmitParams::Input::Crawl::Data::HTML::Options }
1788
+
1789
+ # @!method initialize(source:, options: nil, format_: :html)
1790
+ # Crawl pages and return HTML.
1791
+ #
1792
+ # @param source [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Source::StartURL, ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Source::Sitemap] How to find pages to crawl.
1793
+ #
1794
+ # @param options [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Options] Options for HTML output.
1795
+ #
1796
+ # @param format_ [Symbol, :html] Return page content as HTML.
1797
+
1798
+ # How to find pages to crawl.
1799
+ #
1800
+ # @see ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML#source
1801
+ module Source
1802
+ extend ContextDev::Internal::Type::Union
1803
+
1804
+ discriminator :type
1805
+
1806
+ # Discover pages by following links from one URL.
1807
+ variant :start_url, -> { ContextDev::BatchSubmitParams::Input::Crawl::Data::HTML::Source::StartURL }
1808
+
1809
+ # Scrape the pages listed in a domain's sitemap. Links on those pages are not followed.
1810
+ variant :sitemap, -> { ContextDev::BatchSubmitParams::Input::Crawl::Data::HTML::Source::Sitemap }
1811
+
1812
+ class StartURL < ContextDev::Internal::Type::BaseModel
1813
+ # @!attribute type
1814
+ # Start from one page.
1815
+ #
1816
+ # @return [Symbol, :start_url]
1817
+ required :type, const: :start_url
1818
+
1819
+ # @!attribute url
1820
+ # Page where crawling begins. A URL without a scheme is read as https://.
1821
+ #
1822
+ # @return [String]
1823
+ required :url, String
1824
+
1825
+ # @!attribute controls
1826
+ # Limits and filters for page discovery.
1827
+ #
1828
+ # @return [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Source::StartURL::Controls, nil]
1829
+ optional :controls,
1830
+ -> { ContextDev::BatchSubmitParams::Input::Crawl::Data::HTML::Source::StartURL::Controls }
1831
+
1832
+ # @!method initialize(url:, controls: nil, type: :start_url)
1833
+ # Discover pages by following links from one URL.
1834
+ #
1835
+ # @param url [String] Page where crawling begins. A URL without a scheme is read as https://.
1836
+ #
1837
+ # @param controls [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Source::StartURL::Controls] Limits and filters for page discovery.
1838
+ #
1839
+ # @param type [Symbol, :start_url] Start from one page.
1840
+
1841
+ # @see ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Source::StartURL#controls
1842
+ class Controls < ContextDev::Internal::Type::BaseModel
1843
+ # @!attribute follow_subdomains
1844
+ # Follow links to subdomains.
1845
+ #
1846
+ # @return [Boolean, nil]
1847
+ optional :follow_subdomains,
1848
+ ContextDev::Internal::Type::Boolean,
1849
+ api_name: :followSubdomains
1850
+
1851
+ # @!attribute max_depth
1852
+ # Maximum link depth. Source pages are depth 0. No limit when omitted.
1853
+ #
1854
+ # @return [Integer, nil]
1855
+ optional :max_depth, Integer, api_name: :maxDepth
1856
+
1857
+ # @!attribute max_urls
1858
+ # Maximum pages to fetch. Unused reserved credits are refunded. Maximum 25000.
1859
+ #
1860
+ # @return [Integer, nil]
1861
+ optional :max_urls, Integer, api_name: :maxUrls
1862
+
1863
+ # @!attribute regex
1864
+ # RE2 pattern for URLs to include. The `start_url` itself is always included.
1865
+ #
1866
+ # @return [String, nil]
1867
+ optional :regex, String
1868
+
1869
+ # @!method initialize(follow_subdomains: nil, max_depth: nil, max_urls: nil, regex: nil)
1870
+ # Limits and filters for page discovery.
1871
+ #
1872
+ # @param follow_subdomains [Boolean] Follow links to subdomains.
1873
+ #
1874
+ # @param max_depth [Integer] Maximum link depth. Source pages are depth 0. No limit when omitted.
1875
+ #
1876
+ # @param max_urls [Integer] Maximum pages to fetch. Unused reserved credits are refunded. Maximum 25000.
1877
+ #
1878
+ # @param regex [String] RE2 pattern for URLs to include. The `start_url` itself is always included.
1879
+ end
1880
+ end
1881
+
1882
+ class Sitemap < ContextDev::Internal::Type::BaseModel
1883
+ # @!attribute domain
1884
+ # Domain whose sitemap lists the pages to scrape. A full URL is reduced to its
1885
+ # domain.
1886
+ #
1887
+ # @return [String]
1888
+ required :domain, String
1889
+
1890
+ # @!attribute type
1891
+ # Scrape the URLs in the domain's sitemap.
1892
+ #
1893
+ # @return [Symbol, :sitemap]
1894
+ required :type, const: :sitemap
1895
+
1896
+ # @!attribute controls
1897
+ # Limits and filters for the sitemap URLs. A sitemap batch scrapes exactly those
1898
+ # URLs and never follows links off them, so there is no crawl depth here.
1899
+ #
1900
+ # @return [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Source::Sitemap::Controls, nil]
1901
+ optional :controls,
1902
+ -> { ContextDev::BatchSubmitParams::Input::Crawl::Data::HTML::Source::Sitemap::Controls }
1903
+
1904
+ # @!method initialize(domain:, controls: nil, type: :sitemap)
1905
+ # Some parameter documentations has been truncated, see
1906
+ # {ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Source::Sitemap}
1907
+ # for more details.
1908
+ #
1909
+ # Scrape the pages listed in a domain's sitemap. Links on those pages are not
1910
+ # followed.
1911
+ #
1912
+ # @param domain [String] Domain whose sitemap lists the pages to scrape. A full URL is reduced to its dom
1913
+ #
1914
+ # @param controls [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Source::Sitemap::Controls] Limits and filters for the sitemap URLs. A sitemap batch scrapes exactly those U
1915
+ #
1916
+ # @param type [Symbol, :sitemap] Scrape the URLs in the domain's sitemap.
1917
+
1918
+ # @see ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Source::Sitemap#controls
1919
+ class Controls < ContextDev::Internal::Type::BaseModel
1920
+ # @!attribute max_urls
1921
+ # Maximum pages to fetch. Unused reserved credits are refunded. Maximum 25000.
1922
+ #
1923
+ # @return [Integer, nil]
1924
+ optional :max_urls, Integer, api_name: :maxUrls
1925
+
1926
+ # @!attribute regex
1927
+ # RE2 pattern; only sitemap URLs matching it are scraped.
1928
+ #
1929
+ # @return [String, nil]
1930
+ optional :regex, String
1931
+
1932
+ # @!method initialize(max_urls: nil, regex: nil)
1933
+ # Limits and filters for the sitemap URLs. A sitemap batch scrapes exactly those
1934
+ # URLs and never follows links off them, so there is no crawl depth here.
1935
+ #
1936
+ # @param max_urls [Integer] Maximum pages to fetch. Unused reserved credits are refunded. Maximum 25000.
1937
+ #
1938
+ # @param regex [String] RE2 pattern; only sitemap URLs matching it are scraped.
1939
+ end
1940
+ end
1941
+
1942
+ # @!method self.variants
1943
+ # @return [Array(ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Source::StartURL, ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Source::Sitemap)]
1944
+ end
1945
+
1946
+ # @see ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML#options
1947
+ class Options < ContextDev::Internal::Type::BaseModel
1948
+ # @!attribute country
1949
+ # Fetch the target page through a residential proxy in this country (ISO 3166-1
1950
+ # alpha-2).
1951
+ #
1952
+ # @return [Symbol, ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Country, nil]
1953
+ optional :country, enum: -> { ContextDev::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Country }
1954
+
1955
+ # @!attribute exclude_selectors
1956
+ # Remove elements matching these CSS selectors. Applied after `includeSelectors`,
1957
+ # so an element matching both is removed.
1958
+ #
1959
+ # @return [Array<String>, nil]
1960
+ optional :exclude_selectors,
1961
+ ContextDev::Internal::Type::ArrayOf[String],
1962
+ api_name: :excludeSelectors,
1963
+ nil?: true
1964
+
1965
+ # @!attribute include_selectors
1966
+ # Keep only the subtrees matching these CSS selectors. Filtered pages are always
1967
+ # fetched fresh, ignoring `maxAgeMs`.
1968
+ #
1969
+ # @return [Array<String>, nil]
1970
+ optional :include_selectors,
1971
+ ContextDev::Internal::Type::ArrayOf[String],
1972
+ api_name: :includeSelectors,
1973
+ nil?: true
1974
+
1975
+ # @!attribute max_age_ms
1976
+ # Return a cached result if a prior scrape for the same parameters exists and is
1977
+ # younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
1978
+ # omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
1979
+ #
1980
+ # @return [Integer, nil]
1981
+ optional :max_age_ms, Integer, api_name: :maxAgeMs, nil?: true
1982
+
1983
+ # @!attribute pdf
1984
+ # PDF parsing controls. Use start/end to limit text extraction and embedded-image
1985
+ # detection/OCR to an inclusive 1-based page range.
1986
+ #
1987
+ # @return [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf, nil]
1988
+ optional :pdf, -> { ContextDev::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf }
1989
+
1990
+ # @!attribute settle_animations
1991
+ # Wait briefly for CSS and transition animations to settle before extraction, on
1992
+ # pages that render in a browser.
1993
+ #
1994
+ # @return [Boolean, nil]
1995
+ optional :settle_animations, ContextDev::Internal::Type::Boolean, api_name: :settleAnimations
1996
+
1997
+ # @!attribute use_main_content_only
1998
+ # Return the main content without navigation or footers.
1999
+ #
2000
+ # @return [Boolean, nil]
2001
+ optional :use_main_content_only,
2002
+ ContextDev::Internal::Type::Boolean,
2003
+ api_name: :useMainContentOnly
2004
+
2005
+ # @!attribute wait_for_ms
2006
+ # How long to wait after initial page load, in milliseconds. `0` waits 500 ms.
2007
+ #
2008
+ # @return [Integer, nil]
2009
+ optional :wait_for_ms, Integer, api_name: :waitForMs
2010
+
2011
+ # @!method initialize(country: nil, exclude_selectors: nil, include_selectors: nil, max_age_ms: nil, pdf: nil, settle_animations: nil, use_main_content_only: nil, wait_for_ms: nil)
2012
+ # Some parameter documentations has been truncated, see
2013
+ # {ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Options} for
2014
+ # more details.
2015
+ #
2016
+ # Options for HTML output.
2017
+ #
2018
+ # @param country [Symbol, ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Country] Fetch the target page through a residential proxy in this country (ISO 3166-1 al
2019
+ #
2020
+ # @param exclude_selectors [Array<String>, nil] Remove elements matching these CSS selectors. Applied after `includeSelectors`,
2021
+ #
2022
+ # @param include_selectors [Array<String>, nil] Keep only the subtrees matching these CSS selectors. Filtered pages are always f
2023
+ #
2024
+ # @param max_age_ms [Integer, nil] Return a cached result if a prior scrape for the same parameters exists and is y
2025
+ #
2026
+ # @param pdf [ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf] PDF parsing controls. Use start/end to limit text extraction and embedded-image
2027
+ #
2028
+ # @param settle_animations [Boolean] Wait briefly for CSS and transition animations to settle before extraction, on p
2029
+ #
2030
+ # @param use_main_content_only [Boolean] Return the main content without navigation or footers.
2031
+ #
2032
+ # @param wait_for_ms [Integer] How long to wait after initial page load, in milliseconds. `0` waits 500 ms.
2033
+
2034
+ # Fetch the target page through a residential proxy in this country (ISO 3166-1
2035
+ # alpha-2).
2036
+ #
2037
+ # @see ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Options#country
2038
+ module Country
2039
+ extend ContextDev::Internal::Type::Enum
2040
+
2041
+ AD = :ad
2042
+ AE = :ae
2043
+ AF = :af
2044
+ AG = :ag
2045
+ AI = :ai
2046
+ AL = :al
2047
+ AM = :am
2048
+ AO = :ao
2049
+ AR = :ar
2050
+ AT = :at
2051
+ AU = :au
2052
+ AW = :aw
2053
+ AZ = :az
2054
+ BA = :ba
2055
+ BB = :bb
2056
+ BD = :bd
2057
+ BE = :be
2058
+ BF = :bf
2059
+ BG = :bg
2060
+ BH = :bh
2061
+ BI = :bi
2062
+ BJ = :bj
2063
+ BM = :bm
2064
+ BN = :bn
2065
+ BO = :bo
2066
+ BQ = :bq
2067
+ BR = :br
2068
+ BS = :bs
2069
+ BW = :bw
2070
+ BY = :by
2071
+ BZ = :bz
2072
+ CA = :ca
2073
+ CD = :cd
2074
+ CF = :cf
2075
+ CG = :cg
2076
+ CH = :ch
2077
+ CI = :ci
2078
+ CL = :cl
2079
+ CM = :cm
2080
+ CN = :cn
2081
+ CO = :co
2082
+ CR = :cr
2083
+ CV = :cv
2084
+ CW = :cw
2085
+ CY = :cy
2086
+ CZ = :cz
2087
+ DE = :de
2088
+ DJ = :dj
2089
+ DK = :dk
2090
+ DM = :dm
2091
+ DO = :do
2092
+ DZ = :dz
2093
+ EC = :ec
2094
+ EE = :ee
2095
+ EG = :eg
2096
+ ES = :es
2097
+ ET = :et
2098
+ FI = :fi
2099
+ FJ = :fj
2100
+ FR = :fr
2101
+ GA = :ga
2102
+ GB = :gb
2103
+ GD = :gd
2104
+ GE = :ge
2105
+ GF = :gf
2106
+ GG = :gg
2107
+ GH = :gh
2108
+ GM = :gm
2109
+ GN = :gn
2110
+ GP = :gp
2111
+ GQ = :gq
2112
+ GR = :gr
2113
+ GT = :gt
2114
+ GU = :gu
2115
+ GW = :gw
2116
+ GY = :gy
2117
+ HK = :hk
2118
+ HN = :hn
2119
+ HR = :hr
2120
+ HT = :ht
2121
+ HU = :hu
2122
+ ID = :id
2123
+ IE = :ie
2124
+ IL = :il
2125
+ IM = :im
2126
+ IN = :in
2127
+ IQ = :iq
2128
+ IR = :ir
2129
+ IS = :is
2130
+ IT = :it
2131
+ JE = :je
2132
+ JM = :jm
2133
+ JO = :jo
2134
+ JP = :jp
2135
+ KE = :ke
2136
+ KG = :kg
2137
+ KH = :kh
2138
+ KN = :kn
2139
+ KR = :kr
2140
+ KW = :kw
2141
+ KY = :ky
2142
+ KZ = :kz
2143
+ LA = :la
2144
+ LB = :lb
2145
+ LC = :lc
2146
+ LK = :lk
2147
+ LR = :lr
2148
+ LS = :ls
2149
+ LT = :lt
2150
+ LU = :lu
2151
+ LV = :lv
2152
+ LY = :ly
2153
+ MA = :ma
2154
+ MC = :mc
2155
+ MD = :md
2156
+ ME = :me
2157
+ MF = :mf
2158
+ MG = :mg
2159
+ MK = :mk
2160
+ ML = :ml
2161
+ MM = :mm
2162
+ MN = :mn
2163
+ MO = :mo
2164
+ MQ = :mq
2165
+ MR = :mr
2166
+ MT = :mt
2167
+ MU = :mu
2168
+ MV = :mv
2169
+ MW = :mw
2170
+ MX = :mx
2171
+ MY = :my
2172
+ MZ = :mz
2173
+ NA = :na
2174
+ NC = :nc
2175
+ NE = :ne
2176
+ NG = :ng
2177
+ NI = :ni
2178
+ NL = :nl
2179
+ NO = :no
2180
+ NP = :np
2181
+ NZ = :nz
2182
+ OM = :om
2183
+ PA = :pa
2184
+ PE = :pe
2185
+ PF = :pf
2186
+ PG = :pg
2187
+ PH = :ph
2188
+ PK = :pk
2189
+ PL = :pl
2190
+ PR = :pr
2191
+ PS = :ps
2192
+ PT = :pt
2193
+ PY = :py
2194
+ QA = :qa
2195
+ RE = :re
2196
+ RO = :ro
2197
+ RS = :rs
2198
+ RU = :ru
2199
+ RW = :rw
2200
+ SA = :sa
2201
+ SC = :sc
2202
+ SD = :sd
2203
+ SE = :se
2204
+ SG = :sg
2205
+ SI = :si
2206
+ SK = :sk
2207
+ SL = :sl
2208
+ SM = :sm
2209
+ SN = :sn
2210
+ SO = :so
2211
+ SR = :sr
2212
+ SS = :ss
2213
+ ST = :st
2214
+ SV = :sv
2215
+ SX = :sx
2216
+ SY = :sy
2217
+ SZ = :sz
2218
+ TC = :tc
2219
+ TD = :td
2220
+ TG = :tg
2221
+ TH = :th
2222
+ TJ = :tj
2223
+ TL = :tl
2224
+ TM = :tm
2225
+ TN = :tn
2226
+ TR = :tr
2227
+ TT = :tt
2228
+ TW = :tw
2229
+ TZ = :tz
2230
+ UA = :ua
2231
+ UG = :ug
2232
+ US = :us
2233
+ UY = :uy
2234
+ UZ = :uz
2235
+ VC = :vc
2236
+ VE = :ve
2237
+ VG = :vg
2238
+ VI = :vi
2239
+ VN = :vn
2240
+ YE = :ye
2241
+ YT = :yt
2242
+ ZA = :za
2243
+ ZM = :zm
2244
+ ZW = :zw
2245
+
2246
+ # @!method self.values
2247
+ # @return [Array<Symbol>]
2248
+ end
2249
+
2250
+ # @see ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Options#pdf
2251
+ class Pdf < ContextDev::Internal::Type::BaseModel
2252
+ # @!attribute end_
2253
+ # Last 1-based PDF page to parse. When omitted, parsing ends at the last page.
2254
+ # Must be greater than or equal to start when both are provided.
2255
+ #
2256
+ # @return [Integer, nil]
2257
+ optional :end_, Integer, api_name: :end
2258
+
2259
+ # @!attribute ocr
2260
+ # When true, OCR the selected PDF pages that have no usable text layer (scans),
2261
+ # replacing each recovered page's text with the OCR result while pages with a real
2262
+ # text layer keep it. Billed at 1 credit per page OCR actually recovered, on top
2263
+ # of the base request cost. When false, no OCR runs.
2264
+ #
2265
+ # @return [Boolean, Symbol, ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf::Ocr, nil]
2266
+ optional :ocr, union: -> { ContextDev::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf::Ocr }
2267
+
2268
+ # @!attribute should_parse
2269
+ # When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
2270
+ # a 400 PDF_SKIPPED is returned.
2271
+ #
2272
+ # @return [Boolean, Symbol, ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf::ShouldParse, nil]
2273
+ optional :should_parse,
2274
+ union: -> {
2275
+ ContextDev::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf::ShouldParse
2276
+ },
2277
+ api_name: :shouldParse
2278
+
2279
+ # @!attribute start
2280
+ # First 1-based PDF page to parse. When omitted, parsing starts at the first page.
2281
+ #
2282
+ # @return [Integer, nil]
2283
+ optional :start, Integer
2284
+
2285
+ # @!method initialize(end_: nil, ocr: nil, should_parse: nil, start: nil)
2286
+ # Some parameter documentations has been truncated, see
2287
+ # {ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf}
2288
+ # for more details.
2289
+ #
2290
+ # PDF parsing controls. Use start/end to limit text extraction and embedded-image
2291
+ # detection/OCR to an inclusive 1-based page range.
2292
+ #
2293
+ # @param end_ [Integer] Last 1-based PDF page to parse. When omitted, parsing ends at the last page. Mus
2294
+ #
2295
+ # @param ocr [Boolean, Symbol, ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf::Ocr] When true, OCR the selected PDF pages that have no usable text layer (scans), re
2296
+ #
2297
+ # @param should_parse [Boolean, Symbol, ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf::ShouldParse] When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
2298
+ #
2299
+ # @param start [Integer] First 1-based PDF page to parse. When omitted, parsing starts at the first page.
2300
+
2301
+ # When true, OCR the selected PDF pages that have no usable text layer (scans),
2302
+ # replacing each recovered page's text with the OCR result while pages with a real
2303
+ # text layer keep it. Billed at 1 credit per page OCR actually recovered, on top
2304
+ # of the base request cost. When false, no OCR runs.
2305
+ #
2306
+ # @see ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf#ocr
2307
+ module Ocr
2308
+ extend ContextDev::Internal::Type::Union
2309
+
2310
+ variant ContextDev::Internal::Type::Boolean
2311
+
2312
+ variant const: -> { ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf::Ocr::TRUE }
2313
+
2314
+ variant const: -> { ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf::Ocr::FALSE }
2315
+
2316
+ # @!method self.variants
2317
+ # @return [Array(Boolean, Symbol)]
2318
+
2319
+ define_sorbet_constant!(:Variants) do
2320
+ T.type_alias do
2321
+ T.any(
2322
+ T::Boolean,
2323
+ ContextDev::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf::Ocr::TaggedSymbol
2324
+ )
2325
+ end
2326
+ end
2327
+
2328
+ # @!group
2329
+
2330
+ TRUE = :true
2331
+ FALSE = :false
2332
+
2333
+ # @!endgroup
2334
+ end
2335
+
2336
+ # When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
2337
+ # a 400 PDF_SKIPPED is returned.
2338
+ #
2339
+ # @see ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf#should_parse
2340
+ module ShouldParse
2341
+ extend ContextDev::Internal::Type::Union
2342
+
2343
+ variant ContextDev::Internal::Type::Boolean
2344
+
2345
+ variant const: -> { ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf::ShouldParse::TRUE }
2346
+
2347
+ variant const: -> { ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf::ShouldParse::FALSE }
2348
+
2349
+ # @!method self.variants
2350
+ # @return [Array(Boolean, Symbol)]
2351
+
2352
+ define_sorbet_constant!(:Variants) do
2353
+ T.type_alias do
2354
+ T.any(
2355
+ T::Boolean,
2356
+ ContextDev::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf::ShouldParse::TaggedSymbol
2357
+ )
2358
+ end
2359
+ end
2360
+
2361
+ # @!group
2362
+
2363
+ TRUE = :true
2364
+ FALSE = :false
2365
+
2366
+ # @!endgroup
2367
+ end
2368
+ end
2369
+ end
2370
+ end
2371
+
2372
+ # @!method self.variants
2373
+ # @return [Array(ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown, ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML)]
2374
+ end
2375
+ end
2376
+
2377
+ # @!method self.variants
2378
+ # @return [Array(ContextDev::Models::BatchSubmitParams::Input::Scrape, ContextDev::Models::BatchSubmitParams::Input::Crawl)]
53
2379
  end
54
2380
  end
55
2381
  end