context.dev 2.9.0 → 2.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (80) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +23 -0
  3. data/README.md +1 -1
  4. data/lib/context_dev/client.rb +5 -0
  5. data/lib/context_dev/models/batch_get_results_response.rb +33 -3
  6. data/lib/context_dev/models/batch_submit_params.rb +44 -316
  7. data/lib/context_dev/models/brand_retrieve_response.rb +29 -1
  8. data/lib/context_dev/models/brand_retrieve_simplified_response.rb +30 -1
  9. data/lib/context_dev/models/brand_search_params.rb +41 -3
  10. data/lib/context_dev/models/news_search_params.rb +467 -0
  11. data/lib/context_dev/models/news_search_response.rb +284 -0
  12. data/lib/context_dev/models/parse_handle_params.rb +15 -144
  13. data/lib/context_dev/models/person_enrich_response.rb +61 -1
  14. data/lib/context_dev/models/utility_prefetch_params.rb +19 -16
  15. data/lib/context_dev/models/utility_prefetch_response.rb +4 -5
  16. data/lib/context_dev/models/web_screenshot_params.rb +16 -31
  17. data/lib/context_dev/models/web_web_crawl_md_response.rb +30 -1
  18. data/lib/context_dev/models/web_web_scrape_html_params.rb +15 -153
  19. data/lib/context_dev/models/web_web_scrape_html_response.rb +30 -1
  20. data/lib/context_dev/models/web_web_scrape_images_params.rb +12 -124
  21. data/lib/context_dev/models/web_web_scrape_md_params.rb +35 -238
  22. data/lib/context_dev/models/web_web_scrape_md_response.rb +41 -2
  23. data/lib/context_dev/models.rb +2 -0
  24. data/lib/context_dev/resources/brand.rb +10 -11
  25. data/lib/context_dev/resources/news.rb +51 -0
  26. data/lib/context_dev/resources/parse.rb +5 -5
  27. data/lib/context_dev/resources/utility.rb +7 -6
  28. data/lib/context_dev/resources/web.rb +30 -24
  29. data/lib/context_dev/version.rb +1 -1
  30. data/lib/context_dev.rb +3 -0
  31. data/rbi/context_dev/client.rbi +4 -0
  32. data/rbi/context_dev/models/batch_get_results_response.rbi +71 -2
  33. data/rbi/context_dev/models/batch_submit_params.rbi +58 -592
  34. data/rbi/context_dev/models/brand_retrieve_response.rbi +80 -3
  35. data/rbi/context_dev/models/brand_retrieve_simplified_response.rbi +80 -3
  36. data/rbi/context_dev/models/brand_search_params.rbi +71 -2
  37. data/rbi/context_dev/models/news_search_params.rbi +1294 -0
  38. data/rbi/context_dev/models/news_search_response.rbi +489 -0
  39. data/rbi/context_dev/models/parse_handle_params.rbi +20 -316
  40. data/rbi/context_dev/models/person_enrich_response.rbi +87 -0
  41. data/rbi/context_dev/models/utility_prefetch_params.rbi +24 -16
  42. data/rbi/context_dev/models/utility_prefetch_response.rbi +8 -6
  43. data/rbi/context_dev/models/web_screenshot_params.rbi +23 -71
  44. data/rbi/context_dev/models/web_web_crawl_md_response.rbi +67 -0
  45. data/rbi/context_dev/models/web_web_scrape_html_params.rbi +20 -356
  46. data/rbi/context_dev/models/web_web_scrape_html_response.rbi +65 -0
  47. data/rbi/context_dev/models/web_web_scrape_images_params.rbi +16 -287
  48. data/rbi/context_dev/models/web_web_scrape_md_params.rbi +47 -551
  49. data/rbi/context_dev/models/web_web_scrape_md_response.rbi +80 -0
  50. data/rbi/context_dev/models.rbi +2 -0
  51. data/rbi/context_dev/resources/brand.rbi +14 -9
  52. data/rbi/context_dev/resources/news.rbi +46 -0
  53. data/rbi/context_dev/resources/parse.rbi +5 -21
  54. data/rbi/context_dev/resources/utility.rbi +8 -6
  55. data/rbi/context_dev/resources/web.rbi +34 -66
  56. data/sig/context_dev/client.rbs +2 -0
  57. data/sig/context_dev/models/batch_get_results_response.rbs +21 -0
  58. data/sig/context_dev/models/batch_submit_params.rbs +54 -144
  59. data/sig/context_dev/models/brand_retrieve_response.rbs +33 -3
  60. data/sig/context_dev/models/brand_retrieve_simplified_response.rbs +33 -3
  61. data/sig/context_dev/models/brand_search_params.rbs +38 -1
  62. data/sig/context_dev/models/news_search_params.rbs +532 -0
  63. data/sig/context_dev/models/news_search_response.rbs +206 -0
  64. data/sig/context_dev/models/parse_handle_params.rbs +25 -90
  65. data/sig/context_dev/models/person_enrich_response.rbs +31 -0
  66. data/sig/context_dev/models/utility_prefetch_params.rbs +2 -1
  67. data/sig/context_dev/models/utility_prefetch_response.rbs +2 -1
  68. data/sig/context_dev/models/web_screenshot_params.rbs +12 -18
  69. data/sig/context_dev/models/web_web_crawl_md_response.rbs +21 -0
  70. data/sig/context_dev/models/web_web_scrape_html_params.rbs +24 -94
  71. data/sig/context_dev/models/web_web_scrape_html_response.rbs +21 -0
  72. data/sig/context_dev/models/web_web_scrape_images_params.rbs +20 -72
  73. data/sig/context_dev/models/web_web_scrape_md_params.rbs +46 -148
  74. data/sig/context_dev/models/web_web_scrape_md_response.rbs +28 -0
  75. data/sig/context_dev/models.rbs +2 -0
  76. data/sig/context_dev/resources/brand.rbs +3 -0
  77. data/sig/context_dev/resources/news.rbs +17 -0
  78. data/sig/context_dev/resources/parse.rbs +5 -5
  79. data/sig/context_dev/resources/web.rbs +13 -11
  80. metadata +11 -2
checksums.yaml CHANGED
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  SHA256:
3
- metadata.gz: 69a2da1329540a2c1c0c290e8a62fbf719dfd285d7bafa68f0f1bae6a691d46d
4
- data.tar.gz: 6bb32ce7431dbc1d7e831d6401058a52e6fec88bce87f782c12bc77a2f94d945
3
+ metadata.gz: 366622603821e809f6f9eedf42ab688325b500d9dcae8e9cea5ff059edf08d53
4
+ data.tar.gz: deb733e2b4bb0a4def45544a438a143f799bb74e1cd51412fd4881d7f4cd83fc
5
5
  SHA512:
6
- metadata.gz: 05f9612ffb040bf683cf176f4c29054037587a7896fce546afc631b0c412075a5292bdcb60ef8d7cf7aa9b665ba8b555d2a4e8f6697989e9effb1eed243597b5
7
- data.tar.gz: a98751043b31c39b8005fe357584cbddbd0685b8af8e400c6ca3f227f49cee25a07e6c4346d773be85bac7a4480d5e3ac0d32c9496c5b7b775c7fc41dfc5a01b
6
+ metadata.gz: fc24b4ae331e3ac3097c98f525783b79062620de5c4bd761569358ffe8a33b092da8a102baf2e777e0912d56fa981d11aa066daae83024e50d8684c5d07225d5
7
+ data.tar.gz: 2ccea91ca85c5a826930ceaae09cd8eb04cfef0dc8cdac14c7b616007fa0a1fff6d0079993d192781afdcdd23f8558030faf5b287a2952fe5438b611f65ae476
data/CHANGELOG.md CHANGED
@@ -1,5 +1,28 @@
1
1
  # Changelog
2
2
 
3
+ ## 2.11.0 (2026-08-18)
4
+
5
+ Full Changelog: [v2.10.0...v2.11.0](https://github.com/context-dot-dev/context-ruby-sdk/compare/v2.10.0...v2.11.0)
6
+
7
+ ### Features
8
+
9
+ * **api:** api update ([9bff49f](https://github.com/context-dot-dev/context-ruby-sdk/commit/9bff49f9a966fd40691f2cd7a41ba96ec7679243))
10
+ * **api:** api update ([5b19e0c](https://github.com/context-dot-dev/context-ruby-sdk/commit/5b19e0ce1f7ad615a92e28440bec6ab3703ffe02))
11
+ * **api:** api update ([f4ccf22](https://github.com/context-dot-dev/context-ruby-sdk/commit/f4ccf220f130672950d3231a6b9952c166168ef9))
12
+
13
+ ## 2.10.0 (2026-08-17)
14
+
15
+ Full Changelog: [v2.9.0...v2.10.0](https://github.com/context-dot-dev/context-ruby-sdk/compare/v2.9.0...v2.10.0)
16
+
17
+ ### Features
18
+
19
+ * **api:** api update ([6f90ebf](https://github.com/context-dot-dev/context-ruby-sdk/commit/6f90ebff6164a8daf2f6c4bf8d1492ade1ad2779))
20
+ * **api:** api update ([95cbdde](https://github.com/context-dot-dev/context-ruby-sdk/commit/95cbddeb3a3351a03c394709dfeaf3fb5c98be92))
21
+ * **api:** api update ([97520df](https://github.com/context-dot-dev/context-ruby-sdk/commit/97520df52a798122460e2f219d7b582d2f6830da))
22
+ * **api:** api update ([2b73772](https://github.com/context-dot-dev/context-ruby-sdk/commit/2b737722e962cf7a5aba20a2b596d379f7f2bce0))
23
+ * **api:** api update ([630b415](https://github.com/context-dot-dev/context-ruby-sdk/commit/630b4156f12004d2da154e26003271cd143ccbb8))
24
+ * **api:** manual updates ([54ad450](https://github.com/context-dot-dev/context-ruby-sdk/commit/54ad450f892542378cec0ce5430d1436e1437a4d))
25
+
3
26
  ## 2.9.0 (2026-08-07)
4
27
 
5
28
  Full Changelog: [v2.8.0...v2.9.0](https://github.com/context-dot-dev/context-ruby-sdk/compare/v2.8.0...v2.9.0)
data/README.md CHANGED
@@ -26,7 +26,7 @@ To use this gem, install via Bundler by adding the following to your application
26
26
  <!-- x-release-please-start-version -->
27
27
 
28
28
  ```ruby
29
- gem "context.dev", "~> 2.9.0"
29
+ gem "context.dev", "~> 2.11.0"
30
30
  ```
31
31
 
32
32
  <!-- x-release-please-end -->
@@ -50,6 +50,10 @@ module ContextDev
50
50
  # @return [ContextDev::Resources::People]
51
51
  attr_reader :people
52
52
 
53
+ # Search live first-party RSS and free historical news data by company identity.
54
+ # @return [ContextDev::Resources::News]
55
+ attr_reader :news
56
+
53
57
  # @api private
54
58
  #
55
59
  # @return [Hash{String=>String}]
@@ -120,6 +124,7 @@ module ContextDev
120
124
  @monitors = ContextDev::Resources::Monitors.new(client: self)
121
125
  @batch = ContextDev::Resources::Batch.new(client: self)
122
126
  @people = ContextDev::Resources::People.new(client: self)
127
+ @news = ContextDev::Resources::News.new(client: self)
123
128
  end
124
129
  end
125
130
  end
@@ -86,7 +86,8 @@ module ContextDev
86
86
  required :url, String
87
87
 
88
88
  # @!attribute html
89
- # Raw page HTML. Present on html batches.
89
+ # Page HTML. Present on html batches, and on markdown batches submitted with
90
+ # `options.includeHTML`.
90
91
  #
91
92
  # @return [String, nil]
92
93
  optional :html, String
@@ -130,7 +131,7 @@ module ContextDev
130
131
  #
131
132
  # @param url [String] URL as submitted, or as discovered by the crawl.
132
133
  #
133
- # @param html [String] Raw page HTML. Present on html batches.
134
+ # @param html [String] Page HTML. Present on html batches, and on markdown batches submitted with `opti
134
135
  #
135
136
  # @param item_id [String] Caller-supplied identifier echoed from submission.
136
137
  #
@@ -196,6 +197,14 @@ module ContextDev
196
197
  # @return [String, nil]
197
198
  optional :favicon, String
198
199
 
200
+ # @!attribute headings
201
+ # Page headings (h1–h6) in document order, extracted from the unfiltered document.
202
+ # Capped at the first 500 headings. Omitted when the page has none.
203
+ #
204
+ # @return [Array<ContextDev::Models::BatchGetResultsResponse::Data::Ok::Metadata::Heading>, nil]
205
+ optional :headings,
206
+ -> { ContextDev::Internal::Type::ArrayOf[ContextDev::Models::BatchGetResultsResponse::Data::Ok::Metadata::Heading] }
207
+
199
208
  # @!attribute image
200
209
  # Primary resolved preview image from Open Graph, Twitter, or image metadata.
201
210
  #
@@ -267,7 +276,7 @@ module ContextDev
267
276
  optional :twitter,
268
277
  -> { ContextDev::Internal::Type::HashOf[union: ContextDev::Models::BatchGetResultsResponse::Data::Ok::Metadata::Twitter] }
269
278
 
270
- # @!method initialize(final_url:, source_url:, additional_meta: nil, alternates: nil, author: nil, canonical_url: nil, description: nil, favicon: nil, image: nil, json_ld: nil, keywords: nil, language: nil, modified_time: nil, open_graph: nil, published_time: nil, robots: nil, site_name: nil, title: nil, twitter: nil)
279
+ # @!method initialize(final_url:, source_url:, additional_meta: nil, alternates: nil, author: nil, canonical_url: nil, description: nil, favicon: nil, headings: nil, image: nil, json_ld: nil, keywords: nil, language: nil, modified_time: nil, open_graph: nil, published_time: nil, robots: nil, site_name: nil, title: nil, twitter: nil)
271
280
  # Some parameter documentations has been truncated, see
272
281
  # {ContextDev::Models::BatchGetResultsResponse::Data::Ok::Metadata} for more
273
282
  # details.
@@ -290,6 +299,8 @@ module ContextDev
290
299
  #
291
300
  # @param favicon [String] Resolved favicon URL, when present.
292
301
  #
302
+ # @param headings [Array<ContextDev::Models::BatchGetResultsResponse::Data::Ok::Metadata::Heading>] Page headings (h1–h6) in document order, extracted from the unfiltered document.
303
+ #
293
304
  # @param image [String] Primary resolved preview image from Open Graph, Twitter, or image metadata.
294
305
  #
295
306
  # @param json_ld [Array<Hash{Symbol=>Object}>] JSON-LD structured data blocks parsed from the page.
@@ -361,6 +372,25 @@ module ContextDev
361
372
  # @param type [String] Alternate resource MIME type, when present.
362
373
  end
363
374
 
375
+ class Heading < ContextDev::Internal::Type::BaseModel
376
+ # @!attribute level
377
+ # Heading level, 1–6 (from h1–h6).
378
+ #
379
+ # @return [Integer]
380
+ required :level, Integer
381
+
382
+ # @!attribute text
383
+ # Heading text with whitespace collapsed, truncated to 1000 characters.
384
+ #
385
+ # @return [String]
386
+ required :text, String
387
+
388
+ # @!method initialize(level:, text:)
389
+ # @param level [Integer] Heading level, 1–6 (from h1–h6).
390
+ #
391
+ # @param text [String] Heading text with whitespace collapsed, truncated to 1000 characters.
392
+ end
393
+
364
394
  module OpenGraph
365
395
  extend ContextDev::Internal::Type::Union
366
396
 
@@ -175,6 +175,13 @@ module ContextDev
175
175
  api_name: :excludeSelectors,
176
176
  nil?: true
177
177
 
178
+ # @!attribute include_html
179
+ # Also include each page's HTML in its result record, as an `html` field alongside
180
+ # the Markdown.
181
+ #
182
+ # @return [Boolean, nil]
183
+ optional :include_html, ContextDev::Internal::Type::Boolean, api_name: :includeHTML
184
+
178
185
  # @!attribute include_images
179
186
  # Include image references in the Markdown.
180
187
  #
@@ -241,7 +248,7 @@ module ContextDev
241
248
  # @return [Integer, nil]
242
249
  optional :wait_for_ms, Integer, api_name: :waitForMs
243
250
 
244
- # @!method initialize(country: nil, exclude_selectors: nil, include_images: nil, include_links: nil, include_selectors: nil, max_age_ms: nil, pdf: nil, settle_animations: nil, shorten_base64_images: nil, use_main_content_only: nil, wait_for_ms: nil)
251
+ # @!method initialize(country: nil, exclude_selectors: nil, include_html: nil, include_images: nil, include_links: nil, include_selectors: nil, max_age_ms: nil, pdf: nil, settle_animations: nil, shorten_base64_images: nil, use_main_content_only: nil, wait_for_ms: nil)
245
252
  # Some parameter documentations has been truncated, see
246
253
  # {ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::Options}
247
254
  # for more details.
@@ -252,6 +259,8 @@ module ContextDev
252
259
  #
253
260
  # @param exclude_selectors [Array<String>, nil] Remove elements matching these CSS selectors. Applied after `includeSelectors`,
254
261
  #
262
+ # @param include_html [Boolean] Also include each page's HTML in its result record, as an `html` field alongside
263
+ #
255
264
  # @param include_images [Boolean] Include image references in the Markdown.
256
265
  #
257
266
  # @param include_links [Boolean] Include links in the Markdown.
@@ -501,20 +510,15 @@ module ContextDev
501
510
  # text layer keep it. Billed at 1 credit per page OCR actually recovered, on top
502
511
  # of the base request cost. When false, no OCR runs.
503
512
  #
504
- # @return [Boolean, Symbol, ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf::Ocr, nil]
505
- optional :ocr,
506
- union: -> { ContextDev::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf::Ocr }
513
+ # @return [Boolean, nil]
514
+ optional :ocr, ContextDev::Internal::Type::Boolean
507
515
 
508
516
  # @!attribute should_parse
509
517
  # When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
510
518
  # a 400 PDF_SKIPPED is returned.
511
519
  #
512
- # @return [Boolean, Symbol, ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf::ShouldParse, nil]
513
- optional :should_parse,
514
- union: -> {
515
- ContextDev::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf::ShouldParse
516
- },
517
- api_name: :shouldParse
520
+ # @return [Boolean, nil]
521
+ optional :should_parse, ContextDev::Internal::Type::Boolean, api_name: :shouldParse
518
522
 
519
523
  # @!attribute start
520
524
  # First 1-based PDF page to parse. When omitted, parsing starts at the first page.
@@ -532,79 +536,11 @@ module ContextDev
532
536
  #
533
537
  # @param end_ [Integer] Last 1-based PDF page to parse. When omitted, parsing ends at the last page. Mus
534
538
  #
535
- # @param ocr [Boolean, Symbol, ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf::Ocr] When true, OCR the selected PDF pages that have no usable text layer (scans), re
539
+ # @param ocr [Boolean] When true, OCR the selected PDF pages that have no usable text layer (scans), re
536
540
  #
537
- # @param should_parse [Boolean, Symbol, ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf::ShouldParse] When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
541
+ # @param should_parse [Boolean] When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
538
542
  #
539
543
  # @param start [Integer] First 1-based PDF page to parse. When omitted, parsing starts at the first page.
540
-
541
- # When true, OCR the selected PDF pages that have no usable text layer (scans),
542
- # replacing each recovered page's text with the OCR result while pages with a real
543
- # text layer keep it. Billed at 1 credit per page OCR actually recovered, on top
544
- # of the base request cost. When false, no OCR runs.
545
- #
546
- # @see ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf#ocr
547
- module Ocr
548
- extend ContextDev::Internal::Type::Union
549
-
550
- variant ContextDev::Internal::Type::Boolean
551
-
552
- variant const: -> { ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf::Ocr::TRUE }
553
-
554
- variant const: -> { ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf::Ocr::FALSE }
555
-
556
- # @!method self.variants
557
- # @return [Array(Boolean, Symbol)]
558
-
559
- define_sorbet_constant!(:Variants) do
560
- T.type_alias do
561
- T.any(
562
- T::Boolean,
563
- ContextDev::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf::Ocr::TaggedSymbol
564
- )
565
- end
566
- end
567
-
568
- # @!group
569
-
570
- TRUE = :true
571
- FALSE = :false
572
-
573
- # @!endgroup
574
- end
575
-
576
- # When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
577
- # a 400 PDF_SKIPPED is returned.
578
- #
579
- # @see ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf#should_parse
580
- module ShouldParse
581
- extend ContextDev::Internal::Type::Union
582
-
583
- variant ContextDev::Internal::Type::Boolean
584
-
585
- variant const: -> { ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf::ShouldParse::TRUE }
586
-
587
- variant const: -> { ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf::ShouldParse::FALSE }
588
-
589
- # @!method self.variants
590
- # @return [Array(Boolean, Symbol)]
591
-
592
- define_sorbet_constant!(:Variants) do
593
- T.type_alias do
594
- T.any(
595
- T::Boolean,
596
- ContextDev::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf::ShouldParse::TaggedSymbol
597
- )
598
- end
599
- end
600
-
601
- # @!group
602
-
603
- TRUE = :true
604
- FALSE = :false
605
-
606
- # @!endgroup
607
- end
608
544
  end
609
545
  end
610
546
  end
@@ -991,19 +927,15 @@ module ContextDev
991
927
  # text layer keep it. Billed at 1 credit per page OCR actually recovered, on top
992
928
  # of the base request cost. When false, no OCR runs.
993
929
  #
994
- # @return [Boolean, Symbol, ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf::Ocr, nil]
995
- optional :ocr, union: -> { ContextDev::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf::Ocr }
930
+ # @return [Boolean, nil]
931
+ optional :ocr, ContextDev::Internal::Type::Boolean
996
932
 
997
933
  # @!attribute should_parse
998
934
  # When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
999
935
  # a 400 PDF_SKIPPED is returned.
1000
936
  #
1001
- # @return [Boolean, Symbol, ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf::ShouldParse, nil]
1002
- optional :should_parse,
1003
- union: -> {
1004
- ContextDev::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf::ShouldParse
1005
- },
1006
- api_name: :shouldParse
937
+ # @return [Boolean, nil]
938
+ optional :should_parse, ContextDev::Internal::Type::Boolean, api_name: :shouldParse
1007
939
 
1008
940
  # @!attribute start
1009
941
  # First 1-based PDF page to parse. When omitted, parsing starts at the first page.
@@ -1021,79 +953,11 @@ module ContextDev
1021
953
  #
1022
954
  # @param end_ [Integer] Last 1-based PDF page to parse. When omitted, parsing ends at the last page. Mus
1023
955
  #
1024
- # @param ocr [Boolean, Symbol, ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf::Ocr] When true, OCR the selected PDF pages that have no usable text layer (scans), re
956
+ # @param ocr [Boolean] When true, OCR the selected PDF pages that have no usable text layer (scans), re
1025
957
  #
1026
- # @param should_parse [Boolean, Symbol, ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf::ShouldParse] When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
958
+ # @param should_parse [Boolean] When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
1027
959
  #
1028
960
  # @param start [Integer] First 1-based PDF page to parse. When omitted, parsing starts at the first page.
1029
-
1030
- # When true, OCR the selected PDF pages that have no usable text layer (scans),
1031
- # replacing each recovered page's text with the OCR result while pages with a real
1032
- # text layer keep it. Billed at 1 credit per page OCR actually recovered, on top
1033
- # of the base request cost. When false, no OCR runs.
1034
- #
1035
- # @see ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf#ocr
1036
- module Ocr
1037
- extend ContextDev::Internal::Type::Union
1038
-
1039
- variant ContextDev::Internal::Type::Boolean
1040
-
1041
- variant const: -> { ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf::Ocr::TRUE }
1042
-
1043
- variant const: -> { ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf::Ocr::FALSE }
1044
-
1045
- # @!method self.variants
1046
- # @return [Array(Boolean, Symbol)]
1047
-
1048
- define_sorbet_constant!(:Variants) do
1049
- T.type_alias do
1050
- T.any(
1051
- T::Boolean,
1052
- ContextDev::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf::Ocr::TaggedSymbol
1053
- )
1054
- end
1055
- end
1056
-
1057
- # @!group
1058
-
1059
- TRUE = :true
1060
- FALSE = :false
1061
-
1062
- # @!endgroup
1063
- end
1064
-
1065
- # When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
1066
- # a 400 PDF_SKIPPED is returned.
1067
- #
1068
- # @see ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf#should_parse
1069
- module ShouldParse
1070
- extend ContextDev::Internal::Type::Union
1071
-
1072
- variant ContextDev::Internal::Type::Boolean
1073
-
1074
- variant const: -> { ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf::ShouldParse::TRUE }
1075
-
1076
- variant const: -> { ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf::ShouldParse::FALSE }
1077
-
1078
- # @!method self.variants
1079
- # @return [Array(Boolean, Symbol)]
1080
-
1081
- define_sorbet_constant!(:Variants) do
1082
- T.type_alias do
1083
- T.any(
1084
- T::Boolean,
1085
- ContextDev::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf::ShouldParse::TaggedSymbol
1086
- )
1087
- end
1088
- end
1089
-
1090
- # @!group
1091
-
1092
- TRUE = :true
1093
- FALSE = :false
1094
-
1095
- # @!endgroup
1096
- end
1097
961
  end
1098
962
  end
1099
963
  end
@@ -1333,6 +1197,13 @@ module ContextDev
1333
1197
  api_name: :excludeSelectors,
1334
1198
  nil?: true
1335
1199
 
1200
+ # @!attribute include_html
1201
+ # Also include each page's HTML in its result record, as an `html` field alongside
1202
+ # the Markdown.
1203
+ #
1204
+ # @return [Boolean, nil]
1205
+ optional :include_html, ContextDev::Internal::Type::Boolean, api_name: :includeHTML
1206
+
1336
1207
  # @!attribute include_images
1337
1208
  # Include image references in the Markdown.
1338
1209
  #
@@ -1399,7 +1270,7 @@ module ContextDev
1399
1270
  # @return [Integer, nil]
1400
1271
  optional :wait_for_ms, Integer, api_name: :waitForMs
1401
1272
 
1402
- # @!method initialize(country: nil, exclude_selectors: nil, include_images: nil, include_links: nil, include_selectors: nil, max_age_ms: nil, pdf: nil, settle_animations: nil, shorten_base64_images: nil, use_main_content_only: nil, wait_for_ms: nil)
1273
+ # @!method initialize(country: nil, exclude_selectors: nil, include_html: nil, include_images: nil, include_links: nil, include_selectors: nil, max_age_ms: nil, pdf: nil, settle_animations: nil, shorten_base64_images: nil, use_main_content_only: nil, wait_for_ms: nil)
1403
1274
  # Some parameter documentations has been truncated, see
1404
1275
  # {ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Options}
1405
1276
  # for more details.
@@ -1410,6 +1281,8 @@ module ContextDev
1410
1281
  #
1411
1282
  # @param exclude_selectors [Array<String>, nil] Remove elements matching these CSS selectors. Applied after `includeSelectors`,
1412
1283
  #
1284
+ # @param include_html [Boolean] Also include each page's HTML in its result record, as an `html` field alongside
1285
+ #
1413
1286
  # @param include_images [Boolean] Include image references in the Markdown.
1414
1287
  #
1415
1288
  # @param include_links [Boolean] Include links in the Markdown.
@@ -1659,20 +1532,15 @@ module ContextDev
1659
1532
  # text layer keep it. Billed at 1 credit per page OCR actually recovered, on top
1660
1533
  # of the base request cost. When false, no OCR runs.
1661
1534
  #
1662
- # @return [Boolean, Symbol, ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf::Ocr, nil]
1663
- optional :ocr,
1664
- union: -> { ContextDev::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf::Ocr }
1535
+ # @return [Boolean, nil]
1536
+ optional :ocr, ContextDev::Internal::Type::Boolean
1665
1537
 
1666
1538
  # @!attribute should_parse
1667
1539
  # When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
1668
1540
  # a 400 PDF_SKIPPED is returned.
1669
1541
  #
1670
- # @return [Boolean, Symbol, ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf::ShouldParse, nil]
1671
- optional :should_parse,
1672
- union: -> {
1673
- ContextDev::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf::ShouldParse
1674
- },
1675
- api_name: :shouldParse
1542
+ # @return [Boolean, nil]
1543
+ optional :should_parse, ContextDev::Internal::Type::Boolean, api_name: :shouldParse
1676
1544
 
1677
1545
  # @!attribute start
1678
1546
  # First 1-based PDF page to parse. When omitted, parsing starts at the first page.
@@ -1690,79 +1558,11 @@ module ContextDev
1690
1558
  #
1691
1559
  # @param end_ [Integer] Last 1-based PDF page to parse. When omitted, parsing ends at the last page. Mus
1692
1560
  #
1693
- # @param ocr [Boolean, Symbol, ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf::Ocr] When true, OCR the selected PDF pages that have no usable text layer (scans), re
1561
+ # @param ocr [Boolean] When true, OCR the selected PDF pages that have no usable text layer (scans), re
1694
1562
  #
1695
- # @param should_parse [Boolean, Symbol, ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf::ShouldParse] When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
1563
+ # @param should_parse [Boolean] When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
1696
1564
  #
1697
1565
  # @param start [Integer] First 1-based PDF page to parse. When omitted, parsing starts at the first page.
1698
-
1699
- # When true, OCR the selected PDF pages that have no usable text layer (scans),
1700
- # replacing each recovered page's text with the OCR result while pages with a real
1701
- # text layer keep it. Billed at 1 credit per page OCR actually recovered, on top
1702
- # of the base request cost. When false, no OCR runs.
1703
- #
1704
- # @see ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf#ocr
1705
- module Ocr
1706
- extend ContextDev::Internal::Type::Union
1707
-
1708
- variant ContextDev::Internal::Type::Boolean
1709
-
1710
- variant const: -> { ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf::Ocr::TRUE }
1711
-
1712
- variant const: -> { ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf::Ocr::FALSE }
1713
-
1714
- # @!method self.variants
1715
- # @return [Array(Boolean, Symbol)]
1716
-
1717
- define_sorbet_constant!(:Variants) do
1718
- T.type_alias do
1719
- T.any(
1720
- T::Boolean,
1721
- ContextDev::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf::Ocr::TaggedSymbol
1722
- )
1723
- end
1724
- end
1725
-
1726
- # @!group
1727
-
1728
- TRUE = :true
1729
- FALSE = :false
1730
-
1731
- # @!endgroup
1732
- end
1733
-
1734
- # When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
1735
- # a 400 PDF_SKIPPED is returned.
1736
- #
1737
- # @see ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf#should_parse
1738
- module ShouldParse
1739
- extend ContextDev::Internal::Type::Union
1740
-
1741
- variant ContextDev::Internal::Type::Boolean
1742
-
1743
- variant const: -> { ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf::ShouldParse::TRUE }
1744
-
1745
- variant const: -> { ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf::ShouldParse::FALSE }
1746
-
1747
- # @!method self.variants
1748
- # @return [Array(Boolean, Symbol)]
1749
-
1750
- define_sorbet_constant!(:Variants) do
1751
- T.type_alias do
1752
- T.any(
1753
- T::Boolean,
1754
- ContextDev::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf::ShouldParse::TaggedSymbol
1755
- )
1756
- end
1757
- end
1758
-
1759
- # @!group
1760
-
1761
- TRUE = :true
1762
- FALSE = :false
1763
-
1764
- # @!endgroup
1765
- end
1766
1566
  end
1767
1567
  end
1768
1568
  end
@@ -2262,19 +2062,15 @@ module ContextDev
2262
2062
  # text layer keep it. Billed at 1 credit per page OCR actually recovered, on top
2263
2063
  # of the base request cost. When false, no OCR runs.
2264
2064
  #
2265
- # @return [Boolean, Symbol, ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf::Ocr, nil]
2266
- optional :ocr, union: -> { ContextDev::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf::Ocr }
2065
+ # @return [Boolean, nil]
2066
+ optional :ocr, ContextDev::Internal::Type::Boolean
2267
2067
 
2268
2068
  # @!attribute should_parse
2269
2069
  # When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
2270
2070
  # a 400 PDF_SKIPPED is returned.
2271
2071
  #
2272
- # @return [Boolean, Symbol, ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf::ShouldParse, nil]
2273
- optional :should_parse,
2274
- union: -> {
2275
- ContextDev::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf::ShouldParse
2276
- },
2277
- api_name: :shouldParse
2072
+ # @return [Boolean, nil]
2073
+ optional :should_parse, ContextDev::Internal::Type::Boolean, api_name: :shouldParse
2278
2074
 
2279
2075
  # @!attribute start
2280
2076
  # First 1-based PDF page to parse. When omitted, parsing starts at the first page.
@@ -2292,79 +2088,11 @@ module ContextDev
2292
2088
  #
2293
2089
  # @param end_ [Integer] Last 1-based PDF page to parse. When omitted, parsing ends at the last page. Mus
2294
2090
  #
2295
- # @param ocr [Boolean, Symbol, ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf::Ocr] When true, OCR the selected PDF pages that have no usable text layer (scans), re
2091
+ # @param ocr [Boolean] When true, OCR the selected PDF pages that have no usable text layer (scans), re
2296
2092
  #
2297
- # @param should_parse [Boolean, Symbol, ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf::ShouldParse] When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
2093
+ # @param should_parse [Boolean] When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
2298
2094
  #
2299
2095
  # @param start [Integer] First 1-based PDF page to parse. When omitted, parsing starts at the first page.
2300
-
2301
- # When true, OCR the selected PDF pages that have no usable text layer (scans),
2302
- # replacing each recovered page's text with the OCR result while pages with a real
2303
- # text layer keep it. Billed at 1 credit per page OCR actually recovered, on top
2304
- # of the base request cost. When false, no OCR runs.
2305
- #
2306
- # @see ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf#ocr
2307
- module Ocr
2308
- extend ContextDev::Internal::Type::Union
2309
-
2310
- variant ContextDev::Internal::Type::Boolean
2311
-
2312
- variant const: -> { ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf::Ocr::TRUE }
2313
-
2314
- variant const: -> { ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf::Ocr::FALSE }
2315
-
2316
- # @!method self.variants
2317
- # @return [Array(Boolean, Symbol)]
2318
-
2319
- define_sorbet_constant!(:Variants) do
2320
- T.type_alias do
2321
- T.any(
2322
- T::Boolean,
2323
- ContextDev::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf::Ocr::TaggedSymbol
2324
- )
2325
- end
2326
- end
2327
-
2328
- # @!group
2329
-
2330
- TRUE = :true
2331
- FALSE = :false
2332
-
2333
- # @!endgroup
2334
- end
2335
-
2336
- # When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
2337
- # a 400 PDF_SKIPPED is returned.
2338
- #
2339
- # @see ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf#should_parse
2340
- module ShouldParse
2341
- extend ContextDev::Internal::Type::Union
2342
-
2343
- variant ContextDev::Internal::Type::Boolean
2344
-
2345
- variant const: -> { ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf::ShouldParse::TRUE }
2346
-
2347
- variant const: -> { ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf::ShouldParse::FALSE }
2348
-
2349
- # @!method self.variants
2350
- # @return [Array(Boolean, Symbol)]
2351
-
2352
- define_sorbet_constant!(:Variants) do
2353
- T.type_alias do
2354
- T.any(
2355
- T::Boolean,
2356
- ContextDev::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf::ShouldParse::TaggedSymbol
2357
- )
2358
- end
2359
- end
2360
-
2361
- # @!group
2362
-
2363
- TRUE = :true
2364
- FALSE = :false
2365
-
2366
- # @!endgroup
2367
- end
2368
2096
  end
2369
2097
  end
2370
2098
  end