context.dev 2.9.0 → 2.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (80) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +23 -0
  3. data/README.md +1 -1
  4. data/lib/context_dev/client.rb +5 -0
  5. data/lib/context_dev/models/batch_get_results_response.rb +33 -3
  6. data/lib/context_dev/models/batch_submit_params.rb +44 -316
  7. data/lib/context_dev/models/brand_retrieve_response.rb +29 -1
  8. data/lib/context_dev/models/brand_retrieve_simplified_response.rb +30 -1
  9. data/lib/context_dev/models/brand_search_params.rb +41 -3
  10. data/lib/context_dev/models/news_search_params.rb +467 -0
  11. data/lib/context_dev/models/news_search_response.rb +284 -0
  12. data/lib/context_dev/models/parse_handle_params.rb +15 -144
  13. data/lib/context_dev/models/person_enrich_response.rb +61 -1
  14. data/lib/context_dev/models/utility_prefetch_params.rb +19 -16
  15. data/lib/context_dev/models/utility_prefetch_response.rb +4 -5
  16. data/lib/context_dev/models/web_screenshot_params.rb +16 -31
  17. data/lib/context_dev/models/web_web_crawl_md_response.rb +30 -1
  18. data/lib/context_dev/models/web_web_scrape_html_params.rb +15 -153
  19. data/lib/context_dev/models/web_web_scrape_html_response.rb +30 -1
  20. data/lib/context_dev/models/web_web_scrape_images_params.rb +12 -124
  21. data/lib/context_dev/models/web_web_scrape_md_params.rb +35 -238
  22. data/lib/context_dev/models/web_web_scrape_md_response.rb +41 -2
  23. data/lib/context_dev/models.rb +2 -0
  24. data/lib/context_dev/resources/brand.rb +10 -11
  25. data/lib/context_dev/resources/news.rb +51 -0
  26. data/lib/context_dev/resources/parse.rb +5 -5
  27. data/lib/context_dev/resources/utility.rb +7 -6
  28. data/lib/context_dev/resources/web.rb +30 -24
  29. data/lib/context_dev/version.rb +1 -1
  30. data/lib/context_dev.rb +3 -0
  31. data/rbi/context_dev/client.rbi +4 -0
  32. data/rbi/context_dev/models/batch_get_results_response.rbi +71 -2
  33. data/rbi/context_dev/models/batch_submit_params.rbi +58 -592
  34. data/rbi/context_dev/models/brand_retrieve_response.rbi +80 -3
  35. data/rbi/context_dev/models/brand_retrieve_simplified_response.rbi +80 -3
  36. data/rbi/context_dev/models/brand_search_params.rbi +71 -2
  37. data/rbi/context_dev/models/news_search_params.rbi +1294 -0
  38. data/rbi/context_dev/models/news_search_response.rbi +489 -0
  39. data/rbi/context_dev/models/parse_handle_params.rbi +20 -316
  40. data/rbi/context_dev/models/person_enrich_response.rbi +87 -0
  41. data/rbi/context_dev/models/utility_prefetch_params.rbi +24 -16
  42. data/rbi/context_dev/models/utility_prefetch_response.rbi +8 -6
  43. data/rbi/context_dev/models/web_screenshot_params.rbi +23 -71
  44. data/rbi/context_dev/models/web_web_crawl_md_response.rbi +67 -0
  45. data/rbi/context_dev/models/web_web_scrape_html_params.rbi +20 -356
  46. data/rbi/context_dev/models/web_web_scrape_html_response.rbi +65 -0
  47. data/rbi/context_dev/models/web_web_scrape_images_params.rbi +16 -287
  48. data/rbi/context_dev/models/web_web_scrape_md_params.rbi +47 -551
  49. data/rbi/context_dev/models/web_web_scrape_md_response.rbi +80 -0
  50. data/rbi/context_dev/models.rbi +2 -0
  51. data/rbi/context_dev/resources/brand.rbi +14 -9
  52. data/rbi/context_dev/resources/news.rbi +46 -0
  53. data/rbi/context_dev/resources/parse.rbi +5 -21
  54. data/rbi/context_dev/resources/utility.rbi +8 -6
  55. data/rbi/context_dev/resources/web.rbi +34 -66
  56. data/sig/context_dev/client.rbs +2 -0
  57. data/sig/context_dev/models/batch_get_results_response.rbs +21 -0
  58. data/sig/context_dev/models/batch_submit_params.rbs +54 -144
  59. data/sig/context_dev/models/brand_retrieve_response.rbs +33 -3
  60. data/sig/context_dev/models/brand_retrieve_simplified_response.rbs +33 -3
  61. data/sig/context_dev/models/brand_search_params.rbs +38 -1
  62. data/sig/context_dev/models/news_search_params.rbs +532 -0
  63. data/sig/context_dev/models/news_search_response.rbs +206 -0
  64. data/sig/context_dev/models/parse_handle_params.rbs +25 -90
  65. data/sig/context_dev/models/person_enrich_response.rbs +31 -0
  66. data/sig/context_dev/models/utility_prefetch_params.rbs +2 -1
  67. data/sig/context_dev/models/utility_prefetch_response.rbs +2 -1
  68. data/sig/context_dev/models/web_screenshot_params.rbs +12 -18
  69. data/sig/context_dev/models/web_web_crawl_md_response.rbs +21 -0
  70. data/sig/context_dev/models/web_web_scrape_html_params.rbs +24 -94
  71. data/sig/context_dev/models/web_web_scrape_html_response.rbs +21 -0
  72. data/sig/context_dev/models/web_web_scrape_images_params.rbs +20 -72
  73. data/sig/context_dev/models/web_web_scrape_md_params.rbs +46 -148
  74. data/sig/context_dev/models/web_web_scrape_md_response.rbs +28 -0
  75. data/sig/context_dev/models.rbs +2 -0
  76. data/sig/context_dev/resources/brand.rbs +3 -0
  77. data/sig/context_dev/resources/news.rbs +17 -0
  78. data/sig/context_dev/resources/parse.rbs +5 -5
  79. data/sig/context_dev/resources/web.rbs +13 -11
  80. metadata +11 -2
@@ -343,6 +343,14 @@ module ContextDev
343
343
  sig { returns(T.nilable(T::Array[String])) }
344
344
  attr_accessor :exclude_selectors
345
345
 
346
+ # Also include each page's HTML in its result record, as an `html` field alongside
347
+ # the Markdown.
348
+ sig { returns(T.nilable(T::Boolean)) }
349
+ attr_reader :include_html
350
+
351
+ sig { params(include_html: T::Boolean).void }
352
+ attr_writer :include_html
353
+
346
354
  # Include image references in the Markdown.
347
355
  sig { returns(T.nilable(T::Boolean)) }
348
356
  attr_reader :include_images
@@ -422,6 +430,7 @@ module ContextDev
422
430
  country:
423
431
  ContextDev::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Country::OrSymbol,
424
432
  exclude_selectors: T.nilable(T::Array[String]),
433
+ include_html: T::Boolean,
425
434
  include_images: T::Boolean,
426
435
  include_links: T::Boolean,
427
436
  include_selectors: T.nilable(T::Array[String]),
@@ -441,6 +450,9 @@ module ContextDev
441
450
  # Remove elements matching these CSS selectors. Applied after `includeSelectors`,
442
451
  # so an element matching both is removed.
443
452
  exclude_selectors: nil,
453
+ # Also include each page's HTML in its result record, as an `html` field alongside
454
+ # the Markdown.
455
+ include_html: nil,
444
456
  # Include image references in the Markdown.
445
457
  include_images: nil,
446
458
  # Include links in the Markdown.
@@ -473,6 +485,7 @@ module ContextDev
473
485
  country:
474
486
  ContextDev::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Country::OrSymbol,
475
487
  exclude_selectors: T.nilable(T::Array[String]),
488
+ include_html: T::Boolean,
476
489
  include_images: T::Boolean,
477
490
  include_links: T::Boolean,
478
491
  include_selectors: T.nilable(T::Array[String]),
@@ -1556,52 +1569,18 @@ module ContextDev
1556
1569
  # replacing each recovered page's text with the OCR result while pages with a real
1557
1570
  # text layer keep it. Billed at 1 credit per page OCR actually recovered, on top
1558
1571
  # of the base request cost. When false, no OCR runs.
1559
- sig do
1560
- returns(
1561
- T.nilable(
1562
- T.any(
1563
- T::Boolean,
1564
- ContextDev::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf::Ocr::OrSymbol
1565
- )
1566
- )
1567
- )
1568
- end
1572
+ sig { returns(T.nilable(T::Boolean)) }
1569
1573
  attr_reader :ocr
1570
1574
 
1571
- sig do
1572
- params(
1573
- ocr:
1574
- T.any(
1575
- T::Boolean,
1576
- ContextDev::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf::Ocr::OrSymbol
1577
- )
1578
- ).void
1579
- end
1575
+ sig { params(ocr: T::Boolean).void }
1580
1576
  attr_writer :ocr
1581
1577
 
1582
1578
  # When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
1583
1579
  # a 400 PDF_SKIPPED is returned.
1584
- sig do
1585
- returns(
1586
- T.nilable(
1587
- T.any(
1588
- T::Boolean,
1589
- ContextDev::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf::ShouldParse::OrSymbol
1590
- )
1591
- )
1592
- )
1593
- end
1580
+ sig { returns(T.nilable(T::Boolean)) }
1594
1581
  attr_reader :should_parse
1595
1582
 
1596
- sig do
1597
- params(
1598
- should_parse:
1599
- T.any(
1600
- T::Boolean,
1601
- ContextDev::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf::ShouldParse::OrSymbol
1602
- )
1603
- ).void
1604
- end
1583
+ sig { params(should_parse: T::Boolean).void }
1605
1584
  attr_writer :should_parse
1606
1585
 
1607
1586
  # First 1-based PDF page to parse. When omitted, parsing starts at the first page.
@@ -1616,16 +1595,8 @@ module ContextDev
1616
1595
  sig do
1617
1596
  params(
1618
1597
  end_: Integer,
1619
- ocr:
1620
- T.any(
1621
- T::Boolean,
1622
- ContextDev::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf::Ocr::OrSymbol
1623
- ),
1624
- should_parse:
1625
- T.any(
1626
- T::Boolean,
1627
- ContextDev::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf::ShouldParse::OrSymbol
1628
- ),
1598
+ ocr: T::Boolean,
1599
+ should_parse: T::Boolean,
1629
1600
  start: Integer
1630
1601
  ).returns(T.attached_class)
1631
1602
  end
@@ -1650,112 +1621,14 @@ module ContextDev
1650
1621
  override.returns(
1651
1622
  {
1652
1623
  end_: Integer,
1653
- ocr:
1654
- T.any(
1655
- T::Boolean,
1656
- ContextDev::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf::Ocr::OrSymbol
1657
- ),
1658
- should_parse:
1659
- T.any(
1660
- T::Boolean,
1661
- ContextDev::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf::ShouldParse::OrSymbol
1662
- ),
1624
+ ocr: T::Boolean,
1625
+ should_parse: T::Boolean,
1663
1626
  start: Integer
1664
1627
  }
1665
1628
  )
1666
1629
  end
1667
1630
  def to_hash
1668
1631
  end
1669
-
1670
- # When true, OCR the selected PDF pages that have no usable text layer (scans),
1671
- # replacing each recovered page's text with the OCR result while pages with a real
1672
- # text layer keep it. Billed at 1 credit per page OCR actually recovered, on top
1673
- # of the base request cost. When false, no OCR runs.
1674
- module Ocr
1675
- extend ContextDev::Internal::Type::Union
1676
-
1677
- Variants =
1678
- T.type_alias do
1679
- T.any(
1680
- T::Boolean,
1681
- ContextDev::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf::Ocr::TaggedSymbol
1682
- )
1683
- end
1684
-
1685
- sig do
1686
- override.returns(
1687
- T::Array[
1688
- ContextDev::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf::Ocr::Variants
1689
- ]
1690
- )
1691
- end
1692
- def self.variants
1693
- end
1694
-
1695
- TaggedSymbol =
1696
- T.type_alias do
1697
- T.all(
1698
- Symbol,
1699
- ContextDev::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf::Ocr
1700
- )
1701
- end
1702
- OrSymbol = T.type_alias { T.any(Symbol, String) }
1703
-
1704
- TRUE =
1705
- T.let(
1706
- :true,
1707
- ContextDev::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf::Ocr::TaggedSymbol
1708
- )
1709
- FALSE =
1710
- T.let(
1711
- :false,
1712
- ContextDev::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf::Ocr::TaggedSymbol
1713
- )
1714
- end
1715
-
1716
- # When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
1717
- # a 400 PDF_SKIPPED is returned.
1718
- module ShouldParse
1719
- extend ContextDev::Internal::Type::Union
1720
-
1721
- Variants =
1722
- T.type_alias do
1723
- T.any(
1724
- T::Boolean,
1725
- ContextDev::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf::ShouldParse::TaggedSymbol
1726
- )
1727
- end
1728
-
1729
- sig do
1730
- override.returns(
1731
- T::Array[
1732
- ContextDev::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf::ShouldParse::Variants
1733
- ]
1734
- )
1735
- end
1736
- def self.variants
1737
- end
1738
-
1739
- TaggedSymbol =
1740
- T.type_alias do
1741
- T.all(
1742
- Symbol,
1743
- ContextDev::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf::ShouldParse
1744
- )
1745
- end
1746
- OrSymbol = T.type_alias { T.any(Symbol, String) }
1747
-
1748
- TRUE =
1749
- T.let(
1750
- :true,
1751
- ContextDev::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf::ShouldParse::TaggedSymbol
1752
- )
1753
- FALSE =
1754
- T.let(
1755
- :false,
1756
- ContextDev::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf::ShouldParse::TaggedSymbol
1757
- )
1758
- end
1759
1632
  end
1760
1633
  end
1761
1634
  end
@@ -3112,52 +2985,18 @@ module ContextDev
3112
2985
  # replacing each recovered page's text with the OCR result while pages with a real
3113
2986
  # text layer keep it. Billed at 1 credit per page OCR actually recovered, on top
3114
2987
  # of the base request cost. When false, no OCR runs.
3115
- sig do
3116
- returns(
3117
- T.nilable(
3118
- T.any(
3119
- T::Boolean,
3120
- ContextDev::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf::Ocr::OrSymbol
3121
- )
3122
- )
3123
- )
3124
- end
2988
+ sig { returns(T.nilable(T::Boolean)) }
3125
2989
  attr_reader :ocr
3126
2990
 
3127
- sig do
3128
- params(
3129
- ocr:
3130
- T.any(
3131
- T::Boolean,
3132
- ContextDev::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf::Ocr::OrSymbol
3133
- )
3134
- ).void
3135
- end
2991
+ sig { params(ocr: T::Boolean).void }
3136
2992
  attr_writer :ocr
3137
2993
 
3138
2994
  # When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
3139
2995
  # a 400 PDF_SKIPPED is returned.
3140
- sig do
3141
- returns(
3142
- T.nilable(
3143
- T.any(
3144
- T::Boolean,
3145
- ContextDev::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf::ShouldParse::OrSymbol
3146
- )
3147
- )
3148
- )
3149
- end
2996
+ sig { returns(T.nilable(T::Boolean)) }
3150
2997
  attr_reader :should_parse
3151
2998
 
3152
- sig do
3153
- params(
3154
- should_parse:
3155
- T.any(
3156
- T::Boolean,
3157
- ContextDev::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf::ShouldParse::OrSymbol
3158
- )
3159
- ).void
3160
- end
2999
+ sig { params(should_parse: T::Boolean).void }
3161
3000
  attr_writer :should_parse
3162
3001
 
3163
3002
  # First 1-based PDF page to parse. When omitted, parsing starts at the first page.
@@ -3172,16 +3011,8 @@ module ContextDev
3172
3011
  sig do
3173
3012
  params(
3174
3013
  end_: Integer,
3175
- ocr:
3176
- T.any(
3177
- T::Boolean,
3178
- ContextDev::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf::Ocr::OrSymbol
3179
- ),
3180
- should_parse:
3181
- T.any(
3182
- T::Boolean,
3183
- ContextDev::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf::ShouldParse::OrSymbol
3184
- ),
3014
+ ocr: T::Boolean,
3015
+ should_parse: T::Boolean,
3185
3016
  start: Integer
3186
3017
  ).returns(T.attached_class)
3187
3018
  end
@@ -3206,112 +3037,14 @@ module ContextDev
3206
3037
  override.returns(
3207
3038
  {
3208
3039
  end_: Integer,
3209
- ocr:
3210
- T.any(
3211
- T::Boolean,
3212
- ContextDev::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf::Ocr::OrSymbol
3213
- ),
3214
- should_parse:
3215
- T.any(
3216
- T::Boolean,
3217
- ContextDev::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf::ShouldParse::OrSymbol
3218
- ),
3040
+ ocr: T::Boolean,
3041
+ should_parse: T::Boolean,
3219
3042
  start: Integer
3220
3043
  }
3221
3044
  )
3222
3045
  end
3223
3046
  def to_hash
3224
3047
  end
3225
-
3226
- # When true, OCR the selected PDF pages that have no usable text layer (scans),
3227
- # replacing each recovered page's text with the OCR result while pages with a real
3228
- # text layer keep it. Billed at 1 credit per page OCR actually recovered, on top
3229
- # of the base request cost. When false, no OCR runs.
3230
- module Ocr
3231
- extend ContextDev::Internal::Type::Union
3232
-
3233
- Variants =
3234
- T.type_alias do
3235
- T.any(
3236
- T::Boolean,
3237
- ContextDev::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf::Ocr::TaggedSymbol
3238
- )
3239
- end
3240
-
3241
- sig do
3242
- override.returns(
3243
- T::Array[
3244
- ContextDev::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf::Ocr::Variants
3245
- ]
3246
- )
3247
- end
3248
- def self.variants
3249
- end
3250
-
3251
- TaggedSymbol =
3252
- T.type_alias do
3253
- T.all(
3254
- Symbol,
3255
- ContextDev::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf::Ocr
3256
- )
3257
- end
3258
- OrSymbol = T.type_alias { T.any(Symbol, String) }
3259
-
3260
- TRUE =
3261
- T.let(
3262
- :true,
3263
- ContextDev::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf::Ocr::TaggedSymbol
3264
- )
3265
- FALSE =
3266
- T.let(
3267
- :false,
3268
- ContextDev::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf::Ocr::TaggedSymbol
3269
- )
3270
- end
3271
-
3272
- # When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
3273
- # a 400 PDF_SKIPPED is returned.
3274
- module ShouldParse
3275
- extend ContextDev::Internal::Type::Union
3276
-
3277
- Variants =
3278
- T.type_alias do
3279
- T.any(
3280
- T::Boolean,
3281
- ContextDev::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf::ShouldParse::TaggedSymbol
3282
- )
3283
- end
3284
-
3285
- sig do
3286
- override.returns(
3287
- T::Array[
3288
- ContextDev::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf::ShouldParse::Variants
3289
- ]
3290
- )
3291
- end
3292
- def self.variants
3293
- end
3294
-
3295
- TaggedSymbol =
3296
- T.type_alias do
3297
- T.all(
3298
- Symbol,
3299
- ContextDev::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf::ShouldParse
3300
- )
3301
- end
3302
- OrSymbol = T.type_alias { T.any(Symbol, String) }
3303
-
3304
- TRUE =
3305
- T.let(
3306
- :true,
3307
- ContextDev::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf::ShouldParse::TaggedSymbol
3308
- )
3309
- FALSE =
3310
- T.let(
3311
- :false,
3312
- ContextDev::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf::ShouldParse::TaggedSymbol
3313
- )
3314
- end
3315
3048
  end
3316
3049
  end
3317
3050
  end
@@ -3794,6 +3527,14 @@ module ContextDev
3794
3527
  sig { returns(T.nilable(T::Array[String])) }
3795
3528
  attr_accessor :exclude_selectors
3796
3529
 
3530
+ # Also include each page's HTML in its result record, as an `html` field alongside
3531
+ # the Markdown.
3532
+ sig { returns(T.nilable(T::Boolean)) }
3533
+ attr_reader :include_html
3534
+
3535
+ sig { params(include_html: T::Boolean).void }
3536
+ attr_writer :include_html
3537
+
3797
3538
  # Include image references in the Markdown.
3798
3539
  sig { returns(T.nilable(T::Boolean)) }
3799
3540
  attr_reader :include_images
@@ -3873,6 +3614,7 @@ module ContextDev
3873
3614
  country:
3874
3615
  ContextDev::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Country::OrSymbol,
3875
3616
  exclude_selectors: T.nilable(T::Array[String]),
3617
+ include_html: T::Boolean,
3876
3618
  include_images: T::Boolean,
3877
3619
  include_links: T::Boolean,
3878
3620
  include_selectors: T.nilable(T::Array[String]),
@@ -3892,6 +3634,9 @@ module ContextDev
3892
3634
  # Remove elements matching these CSS selectors. Applied after `includeSelectors`,
3893
3635
  # so an element matching both is removed.
3894
3636
  exclude_selectors: nil,
3637
+ # Also include each page's HTML in its result record, as an `html` field alongside
3638
+ # the Markdown.
3639
+ include_html: nil,
3895
3640
  # Include image references in the Markdown.
3896
3641
  include_images: nil,
3897
3642
  # Include links in the Markdown.
@@ -3924,6 +3669,7 @@ module ContextDev
3924
3669
  country:
3925
3670
  ContextDev::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Country::OrSymbol,
3926
3671
  exclude_selectors: T.nilable(T::Array[String]),
3672
+ include_html: T::Boolean,
3927
3673
  include_images: T::Boolean,
3928
3674
  include_links: T::Boolean,
3929
3675
  include_selectors: T.nilable(T::Array[String]),
@@ -5007,52 +4753,18 @@ module ContextDev
5007
4753
  # replacing each recovered page's text with the OCR result while pages with a real
5008
4754
  # text layer keep it. Billed at 1 credit per page OCR actually recovered, on top
5009
4755
  # of the base request cost. When false, no OCR runs.
5010
- sig do
5011
- returns(
5012
- T.nilable(
5013
- T.any(
5014
- T::Boolean,
5015
- ContextDev::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf::Ocr::OrSymbol
5016
- )
5017
- )
5018
- )
5019
- end
4756
+ sig { returns(T.nilable(T::Boolean)) }
5020
4757
  attr_reader :ocr
5021
4758
 
5022
- sig do
5023
- params(
5024
- ocr:
5025
- T.any(
5026
- T::Boolean,
5027
- ContextDev::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf::Ocr::OrSymbol
5028
- )
5029
- ).void
5030
- end
4759
+ sig { params(ocr: T::Boolean).void }
5031
4760
  attr_writer :ocr
5032
4761
 
5033
4762
  # When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
5034
4763
  # a 400 PDF_SKIPPED is returned.
5035
- sig do
5036
- returns(
5037
- T.nilable(
5038
- T.any(
5039
- T::Boolean,
5040
- ContextDev::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf::ShouldParse::OrSymbol
5041
- )
5042
- )
5043
- )
5044
- end
4764
+ sig { returns(T.nilable(T::Boolean)) }
5045
4765
  attr_reader :should_parse
5046
4766
 
5047
- sig do
5048
- params(
5049
- should_parse:
5050
- T.any(
5051
- T::Boolean,
5052
- ContextDev::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf::ShouldParse::OrSymbol
5053
- )
5054
- ).void
5055
- end
4767
+ sig { params(should_parse: T::Boolean).void }
5056
4768
  attr_writer :should_parse
5057
4769
 
5058
4770
  # First 1-based PDF page to parse. When omitted, parsing starts at the first page.
@@ -5067,16 +4779,8 @@ module ContextDev
5067
4779
  sig do
5068
4780
  params(
5069
4781
  end_: Integer,
5070
- ocr:
5071
- T.any(
5072
- T::Boolean,
5073
- ContextDev::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf::Ocr::OrSymbol
5074
- ),
5075
- should_parse:
5076
- T.any(
5077
- T::Boolean,
5078
- ContextDev::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf::ShouldParse::OrSymbol
5079
- ),
4782
+ ocr: T::Boolean,
4783
+ should_parse: T::Boolean,
5080
4784
  start: Integer
5081
4785
  ).returns(T.attached_class)
5082
4786
  end
@@ -5101,112 +4805,14 @@ module ContextDev
5101
4805
  override.returns(
5102
4806
  {
5103
4807
  end_: Integer,
5104
- ocr:
5105
- T.any(
5106
- T::Boolean,
5107
- ContextDev::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf::Ocr::OrSymbol
5108
- ),
5109
- should_parse:
5110
- T.any(
5111
- T::Boolean,
5112
- ContextDev::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf::ShouldParse::OrSymbol
5113
- ),
4808
+ ocr: T::Boolean,
4809
+ should_parse: T::Boolean,
5114
4810
  start: Integer
5115
4811
  }
5116
4812
  )
5117
4813
  end
5118
4814
  def to_hash
5119
4815
  end
5120
-
5121
- # When true, OCR the selected PDF pages that have no usable text layer (scans),
5122
- # replacing each recovered page's text with the OCR result while pages with a real
5123
- # text layer keep it. Billed at 1 credit per page OCR actually recovered, on top
5124
- # of the base request cost. When false, no OCR runs.
5125
- module Ocr
5126
- extend ContextDev::Internal::Type::Union
5127
-
5128
- Variants =
5129
- T.type_alias do
5130
- T.any(
5131
- T::Boolean,
5132
- ContextDev::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf::Ocr::TaggedSymbol
5133
- )
5134
- end
5135
-
5136
- sig do
5137
- override.returns(
5138
- T::Array[
5139
- ContextDev::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf::Ocr::Variants
5140
- ]
5141
- )
5142
- end
5143
- def self.variants
5144
- end
5145
-
5146
- TaggedSymbol =
5147
- T.type_alias do
5148
- T.all(
5149
- Symbol,
5150
- ContextDev::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf::Ocr
5151
- )
5152
- end
5153
- OrSymbol = T.type_alias { T.any(Symbol, String) }
5154
-
5155
- TRUE =
5156
- T.let(
5157
- :true,
5158
- ContextDev::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf::Ocr::TaggedSymbol
5159
- )
5160
- FALSE =
5161
- T.let(
5162
- :false,
5163
- ContextDev::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf::Ocr::TaggedSymbol
5164
- )
5165
- end
5166
-
5167
- # When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
5168
- # a 400 PDF_SKIPPED is returned.
5169
- module ShouldParse
5170
- extend ContextDev::Internal::Type::Union
5171
-
5172
- Variants =
5173
- T.type_alias do
5174
- T.any(
5175
- T::Boolean,
5176
- ContextDev::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf::ShouldParse::TaggedSymbol
5177
- )
5178
- end
5179
-
5180
- sig do
5181
- override.returns(
5182
- T::Array[
5183
- ContextDev::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf::ShouldParse::Variants
5184
- ]
5185
- )
5186
- end
5187
- def self.variants
5188
- end
5189
-
5190
- TaggedSymbol =
5191
- T.type_alias do
5192
- T.all(
5193
- Symbol,
5194
- ContextDev::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf::ShouldParse
5195
- )
5196
- end
5197
- OrSymbol = T.type_alias { T.any(Symbol, String) }
5198
-
5199
- TRUE =
5200
- T.let(
5201
- :true,
5202
- ContextDev::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf::ShouldParse::TaggedSymbol
5203
- )
5204
- FALSE =
5205
- T.let(
5206
- :false,
5207
- ContextDev::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf::ShouldParse::TaggedSymbol
5208
- )
5209
- end
5210
4816
  end
5211
4817
  end
5212
4818
  end
@@ -6787,52 +6393,18 @@ module ContextDev
6787
6393
  # replacing each recovered page's text with the OCR result while pages with a real
6788
6394
  # text layer keep it. Billed at 1 credit per page OCR actually recovered, on top
6789
6395
  # of the base request cost. When false, no OCR runs.
6790
- sig do
6791
- returns(
6792
- T.nilable(
6793
- T.any(
6794
- T::Boolean,
6795
- ContextDev::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf::Ocr::OrSymbol
6796
- )
6797
- )
6798
- )
6799
- end
6396
+ sig { returns(T.nilable(T::Boolean)) }
6800
6397
  attr_reader :ocr
6801
6398
 
6802
- sig do
6803
- params(
6804
- ocr:
6805
- T.any(
6806
- T::Boolean,
6807
- ContextDev::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf::Ocr::OrSymbol
6808
- )
6809
- ).void
6810
- end
6399
+ sig { params(ocr: T::Boolean).void }
6811
6400
  attr_writer :ocr
6812
6401
 
6813
6402
  # When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
6814
6403
  # a 400 PDF_SKIPPED is returned.
6815
- sig do
6816
- returns(
6817
- T.nilable(
6818
- T.any(
6819
- T::Boolean,
6820
- ContextDev::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf::ShouldParse::OrSymbol
6821
- )
6822
- )
6823
- )
6824
- end
6404
+ sig { returns(T.nilable(T::Boolean)) }
6825
6405
  attr_reader :should_parse
6826
6406
 
6827
- sig do
6828
- params(
6829
- should_parse:
6830
- T.any(
6831
- T::Boolean,
6832
- ContextDev::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf::ShouldParse::OrSymbol
6833
- )
6834
- ).void
6835
- end
6407
+ sig { params(should_parse: T::Boolean).void }
6836
6408
  attr_writer :should_parse
6837
6409
 
6838
6410
  # First 1-based PDF page to parse. When omitted, parsing starts at the first page.
@@ -6847,16 +6419,8 @@ module ContextDev
6847
6419
  sig do
6848
6420
  params(
6849
6421
  end_: Integer,
6850
- ocr:
6851
- T.any(
6852
- T::Boolean,
6853
- ContextDev::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf::Ocr::OrSymbol
6854
- ),
6855
- should_parse:
6856
- T.any(
6857
- T::Boolean,
6858
- ContextDev::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf::ShouldParse::OrSymbol
6859
- ),
6422
+ ocr: T::Boolean,
6423
+ should_parse: T::Boolean,
6860
6424
  start: Integer
6861
6425
  ).returns(T.attached_class)
6862
6426
  end
@@ -6881,112 +6445,14 @@ module ContextDev
6881
6445
  override.returns(
6882
6446
  {
6883
6447
  end_: Integer,
6884
- ocr:
6885
- T.any(
6886
- T::Boolean,
6887
- ContextDev::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf::Ocr::OrSymbol
6888
- ),
6889
- should_parse:
6890
- T.any(
6891
- T::Boolean,
6892
- ContextDev::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf::ShouldParse::OrSymbol
6893
- ),
6448
+ ocr: T::Boolean,
6449
+ should_parse: T::Boolean,
6894
6450
  start: Integer
6895
6451
  }
6896
6452
  )
6897
6453
  end
6898
6454
  def to_hash
6899
6455
  end
6900
-
6901
- # When true, OCR the selected PDF pages that have no usable text layer (scans),
6902
- # replacing each recovered page's text with the OCR result while pages with a real
6903
- # text layer keep it. Billed at 1 credit per page OCR actually recovered, on top
6904
- # of the base request cost. When false, no OCR runs.
6905
- module Ocr
6906
- extend ContextDev::Internal::Type::Union
6907
-
6908
- Variants =
6909
- T.type_alias do
6910
- T.any(
6911
- T::Boolean,
6912
- ContextDev::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf::Ocr::TaggedSymbol
6913
- )
6914
- end
6915
-
6916
- sig do
6917
- override.returns(
6918
- T::Array[
6919
- ContextDev::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf::Ocr::Variants
6920
- ]
6921
- )
6922
- end
6923
- def self.variants
6924
- end
6925
-
6926
- TaggedSymbol =
6927
- T.type_alias do
6928
- T.all(
6929
- Symbol,
6930
- ContextDev::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf::Ocr
6931
- )
6932
- end
6933
- OrSymbol = T.type_alias { T.any(Symbol, String) }
6934
-
6935
- TRUE =
6936
- T.let(
6937
- :true,
6938
- ContextDev::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf::Ocr::TaggedSymbol
6939
- )
6940
- FALSE =
6941
- T.let(
6942
- :false,
6943
- ContextDev::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf::Ocr::TaggedSymbol
6944
- )
6945
- end
6946
-
6947
- # When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
6948
- # a 400 PDF_SKIPPED is returned.
6949
- module ShouldParse
6950
- extend ContextDev::Internal::Type::Union
6951
-
6952
- Variants =
6953
- T.type_alias do
6954
- T.any(
6955
- T::Boolean,
6956
- ContextDev::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf::ShouldParse::TaggedSymbol
6957
- )
6958
- end
6959
-
6960
- sig do
6961
- override.returns(
6962
- T::Array[
6963
- ContextDev::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf::ShouldParse::Variants
6964
- ]
6965
- )
6966
- end
6967
- def self.variants
6968
- end
6969
-
6970
- TaggedSymbol =
6971
- T.type_alias do
6972
- T.all(
6973
- Symbol,
6974
- ContextDev::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf::ShouldParse
6975
- )
6976
- end
6977
- OrSymbol = T.type_alias { T.any(Symbol, String) }
6978
-
6979
- TRUE =
6980
- T.let(
6981
- :true,
6982
- ContextDev::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf::ShouldParse::TaggedSymbol
6983
- )
6984
- FALSE =
6985
- T.let(
6986
- :false,
6987
- ContextDev::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf::ShouldParse::TaggedSymbol
6988
- )
6989
- end
6990
6456
  end
6991
6457
  end
6992
6458
  end