context.dev 2.8.0 → 2.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (76) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +15 -0
  3. data/README.md +1 -1
  4. data/lib/context_dev/client.rb +5 -0
  5. data/lib/context_dev/models/batch_delete_params.rb +22 -0
  6. data/lib/context_dev/models/batch_delete_response.rb +60 -0
  7. data/lib/context_dev/models/batch_get_results_response.rb +13 -1
  8. data/lib/context_dev/models/batch_list_response.rb +13 -4
  9. data/lib/context_dev/models/batch_retrieve_response.rb +13 -4
  10. data/lib/context_dev/models/batch_submit_params.rb +2352 -26
  11. data/lib/context_dev/models/batch_submit_response.rb +125 -531
  12. data/lib/context_dev/models/brand_retrieve_response.rb +50 -1
  13. data/lib/context_dev/models/brand_search_response.rb +3 -2
  14. data/lib/context_dev/models/crawl_controls.rb +21 -15
  15. data/lib/context_dev/models/parse_handle_params.rb +11 -9
  16. data/lib/context_dev/models/person_enrich_params.rb +176 -0
  17. data/lib/context_dev/models/person_enrich_response.rb +581 -0
  18. data/lib/context_dev/models/web_search_response.rb +1 -0
  19. data/lib/context_dev/models/web_web_crawl_md_params.rb +5 -4
  20. data/lib/context_dev/models/web_web_scrape_html_params.rb +11 -9
  21. data/lib/context_dev/models/web_web_scrape_md_params.rb +11 -9
  22. data/lib/context_dev/models/web_web_scrape_sitemap_params.rb +11 -1
  23. data/lib/context_dev/models/web_web_scrape_sitemap_response.rb +3 -2
  24. data/lib/context_dev/models.rb +4 -0
  25. data/lib/context_dev/resources/batch.rb +33 -7
  26. data/lib/context_dev/resources/brand.rb +2 -1
  27. data/lib/context_dev/resources/parse.rb +1 -1
  28. data/lib/context_dev/resources/people.rb +56 -0
  29. data/lib/context_dev/resources/web.rb +21 -2
  30. data/lib/context_dev/version.rb +1 -1
  31. data/lib/context_dev.rb +5 -0
  32. data/rbi/context_dev/client.rbi +4 -0
  33. data/rbi/context_dev/models/batch_delete_params.rbi +40 -0
  34. data/rbi/context_dev/models/batch_delete_response.rbi +116 -0
  35. data/rbi/context_dev/models/batch_get_results_response.rbi +14 -1
  36. data/rbi/context_dev/models/batch_list_response.rbi +24 -8
  37. data/rbi/context_dev/models/batch_retrieve_response.rbi +24 -8
  38. data/rbi/context_dev/models/batch_submit_params.rbi +6955 -44
  39. data/rbi/context_dev/models/batch_submit_response.rbi +184 -1151
  40. data/rbi/context_dev/models/brand_retrieve_response.rbi +152 -0
  41. data/rbi/context_dev/models/brand_search_response.rbi +4 -2
  42. data/rbi/context_dev/models/crawl_controls.rbi +22 -28
  43. data/rbi/context_dev/models/parse_handle_params.rbi +15 -12
  44. data/rbi/context_dev/models/person_enrich_params.rbi +332 -0
  45. data/rbi/context_dev/models/person_enrich_response.rbi +1208 -0
  46. data/rbi/context_dev/models/web_search_response.rbi +5 -0
  47. data/rbi/context_dev/models/web_web_crawl_md_params.rbi +8 -6
  48. data/rbi/context_dev/models/web_web_scrape_html_params.rbi +15 -12
  49. data/rbi/context_dev/models/web_web_scrape_md_params.rbi +15 -12
  50. data/rbi/context_dev/models/web_web_scrape_sitemap_params.rbi +15 -0
  51. data/rbi/context_dev/models/web_web_scrape_sitemap_response.rbi +4 -2
  52. data/rbi/context_dev/models.rbi +4 -0
  53. data/rbi/context_dev/resources/batch.rbi +32 -10
  54. data/rbi/context_dev/resources/brand.rbi +2 -1
  55. data/rbi/context_dev/resources/parse.rbi +5 -5
  56. data/rbi/context_dev/resources/people.rbi +47 -0
  57. data/rbi/context_dev/resources/web.rbi +23 -1
  58. data/sig/context_dev/client.rbs +2 -0
  59. data/sig/context_dev/models/batch_delete_params.rbs +23 -0
  60. data/sig/context_dev/models/batch_delete_response.rbs +57 -0
  61. data/sig/context_dev/models/batch_get_results_response.rbs +9 -2
  62. data/sig/context_dev/models/batch_list_response.rbs +16 -2
  63. data/sig/context_dev/models/batch_retrieve_response.rbs +16 -2
  64. data/sig/context_dev/models/batch_submit_params.rbs +2756 -15
  65. data/sig/context_dev/models/batch_submit_response.rbs +78 -466
  66. data/sig/context_dev/models/brand_retrieve_response.rbs +62 -0
  67. data/sig/context_dev/models/crawl_controls.rbs +16 -16
  68. data/sig/context_dev/models/person_enrich_params.rbs +199 -0
  69. data/sig/context_dev/models/person_enrich_response.rbs +607 -0
  70. data/sig/context_dev/models/web_search_response.rbs +2 -0
  71. data/sig/context_dev/models/web_web_scrape_sitemap_params.rbs +7 -0
  72. data/sig/context_dev/models.rbs +4 -0
  73. data/sig/context_dev/resources/batch.rbs +8 -2
  74. data/sig/context_dev/resources/people.rbs +19 -0
  75. data/sig/context_dev/resources/web.rbs +1 -0
  76. metadata +17 -2
@@ -224,6 +224,11 @@ module ContextDev
224
224
  :TIMEOUT,
225
225
  ContextDev::Models::WebSearchResponse::Result::Markdown::Code::TaggedSymbol
226
226
  )
227
+ CONTENT_TOO_LARGE =
228
+ T.let(
229
+ :CONTENT_TOO_LARGE,
230
+ ContextDev::Models::WebSearchResponse::Result::Markdown::Code::TaggedSymbol
231
+ )
227
232
  WEBSITE_ACCESS_ERROR =
228
233
  T.let(
229
234
  :WEBSITE_ACCESS_ERROR,
@@ -553,9 +553,10 @@ module ContextDev
553
553
  sig { params(end_: Integer).void }
554
554
  attr_writer :end_
555
555
 
556
- # When true, detect and OCR images embedded in the selected PDF pages, inserting
557
- # recognized text at each image's position in page reading order while preserving
558
- # the PDF text layer. This is separate from automatic scanned-PDF OCR fallback.
556
+ # When true, OCR the selected PDF pages that have no usable text layer (scans),
557
+ # replacing each recovered page's text with the OCR result while pages with a real
558
+ # text layer keep it. Billed at 1 credit per page OCR actually recovered, on top
559
+ # of the base request cost.
559
560
  sig { returns(T.nilable(T::Boolean)) }
560
561
  attr_reader :ocr
561
562
 
@@ -591,9 +592,10 @@ module ContextDev
591
592
  # Last 1-based PDF page to parse. When omitted, parsing ends at the last page.
592
593
  # Must be greater than or equal to start when both are provided.
593
594
  end_: nil,
594
- # When true, detect and OCR images embedded in the selected PDF pages, inserting
595
- # recognized text at each image's position in page reading order while preserving
596
- # the PDF text layer. This is separate from automatic scanned-PDF OCR fallback.
595
+ # When true, OCR the selected PDF pages that have no usable text layer (scans),
596
+ # replacing each recovered page's text with the OCR result while pages with a real
597
+ # text layer keep it. Billed at 1 credit per page OCR actually recovered, on top
598
+ # of the base request cost.
597
599
  ocr: nil,
598
600
  # When true, PDF pages are fetched and parsed. When false, PDF pages are skipped
599
601
  # entirely (not included in results and not counted as failures).
@@ -901,9 +901,10 @@ module ContextDev
901
901
  sig { params(end_: Integer).void }
902
902
  attr_writer :end_
903
903
 
904
- # When true, detect and OCR images embedded in the selected PDF pages, inserting
905
- # recognized text at each image's position in page reading order while preserving
906
- # the PDF text layer. This is separate from automatic scanned-PDF OCR fallback.
904
+ # When true, OCR the selected PDF pages that have no usable text layer (scans),
905
+ # replacing each recovered page's text with the OCR result while pages with a real
906
+ # text layer keep it. Billed at 1 credit per page OCR actually recovered, on top
907
+ # of the base request cost. When false, no OCR runs.
907
908
  sig do
908
909
  returns(
909
910
  T.nilable(
@@ -928,7 +929,7 @@ module ContextDev
928
929
  attr_writer :ocr
929
930
 
930
931
  # When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
931
- # a 400 WEBSITE_ACCESS_ERROR is returned.
932
+ # a 400 PDF_SKIPPED is returned.
932
933
  sig do
933
934
  returns(
934
935
  T.nilable(
@@ -981,12 +982,13 @@ module ContextDev
981
982
  # Last 1-based PDF page to parse. When omitted, parsing ends at the last page.
982
983
  # Must be greater than or equal to start when both are provided.
983
984
  end_: nil,
984
- # When true, detect and OCR images embedded in the selected PDF pages, inserting
985
- # recognized text at each image's position in page reading order while preserving
986
- # the PDF text layer. This is separate from automatic scanned-PDF OCR fallback.
985
+ # When true, OCR the selected PDF pages that have no usable text layer (scans),
986
+ # replacing each recovered page's text with the OCR result while pages with a real
987
+ # text layer keep it. Billed at 1 credit per page OCR actually recovered, on top
988
+ # of the base request cost. When false, no OCR runs.
987
989
  ocr: nil,
988
990
  # When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
989
- # a 400 WEBSITE_ACCESS_ERROR is returned.
991
+ # a 400 PDF_SKIPPED is returned.
990
992
  should_parse: nil,
991
993
  # First 1-based PDF page to parse. When omitted, parsing starts at the first page.
992
994
  start: nil
@@ -1014,9 +1016,10 @@ module ContextDev
1014
1016
  def to_hash
1015
1017
  end
1016
1018
 
1017
- # When true, detect and OCR images embedded in the selected PDF pages, inserting
1018
- # recognized text at each image's position in page reading order while preserving
1019
- # the PDF text layer. This is separate from automatic scanned-PDF OCR fallback.
1019
+ # When true, OCR the selected PDF pages that have no usable text layer (scans),
1020
+ # replacing each recovered page's text with the OCR result while pages with a real
1021
+ # text layer keep it. Billed at 1 credit per page OCR actually recovered, on top
1022
+ # of the base request cost. When false, no OCR runs.
1020
1023
  module Ocr
1021
1024
  extend ContextDev::Internal::Type::Union
1022
1025
 
@@ -1055,7 +1058,7 @@ module ContextDev
1055
1058
  end
1056
1059
 
1057
1060
  # When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
1058
- # a 400 WEBSITE_ACCESS_ERROR is returned.
1061
+ # a 400 PDF_SKIPPED is returned.
1059
1062
  module ShouldParse
1060
1063
  extend ContextDev::Internal::Type::Union
1061
1064
 
@@ -874,9 +874,10 @@ module ContextDev
874
874
  sig { params(end_: Integer).void }
875
875
  attr_writer :end_
876
876
 
877
- # When true, detect and OCR images embedded in the selected PDF pages, inserting
878
- # recognized text at each image's position in page reading order while preserving
879
- # the PDF text layer. This is separate from automatic scanned-PDF OCR fallback.
877
+ # When true, OCR the selected PDF pages that have no usable text layer (scans),
878
+ # replacing each recovered page's text with the OCR result while pages with a real
879
+ # text layer keep it. Billed at 1 credit per page OCR actually recovered, on top
880
+ # of the base request cost. When false, no OCR runs.
880
881
  sig do
881
882
  returns(
882
883
  T.nilable(
@@ -901,7 +902,7 @@ module ContextDev
901
902
  attr_writer :ocr
902
903
 
903
904
  # When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
904
- # a 400 WEBSITE_ACCESS_ERROR is returned.
905
+ # a 400 PDF_SKIPPED is returned.
905
906
  sig do
906
907
  returns(
907
908
  T.nilable(
@@ -954,12 +955,13 @@ module ContextDev
954
955
  # Last 1-based PDF page to parse. When omitted, parsing ends at the last page.
955
956
  # Must be greater than or equal to start when both are provided.
956
957
  end_: nil,
957
- # When true, detect and OCR images embedded in the selected PDF pages, inserting
958
- # recognized text at each image's position in page reading order while preserving
959
- # the PDF text layer. This is separate from automatic scanned-PDF OCR fallback.
958
+ # When true, OCR the selected PDF pages that have no usable text layer (scans),
959
+ # replacing each recovered page's text with the OCR result while pages with a real
960
+ # text layer keep it. Billed at 1 credit per page OCR actually recovered, on top
961
+ # of the base request cost. When false, no OCR runs.
960
962
  ocr: nil,
961
963
  # When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
962
- # a 400 WEBSITE_ACCESS_ERROR is returned.
964
+ # a 400 PDF_SKIPPED is returned.
963
965
  should_parse: nil,
964
966
  # First 1-based PDF page to parse. When omitted, parsing starts at the first page.
965
967
  start: nil
@@ -987,9 +989,10 @@ module ContextDev
987
989
  def to_hash
988
990
  end
989
991
 
990
- # When true, detect and OCR images embedded in the selected PDF pages, inserting
991
- # recognized text at each image's position in page reading order while preserving
992
- # the PDF text layer. This is separate from automatic scanned-PDF OCR fallback.
992
+ # When true, OCR the selected PDF pages that have no usable text layer (scans),
993
+ # replacing each recovered page's text with the OCR result while pages with a real
994
+ # text layer keep it. Billed at 1 credit per page OCR actually recovered, on top
995
+ # of the base request cost. When false, no OCR runs.
993
996
  module Ocr
994
997
  extend ContextDev::Internal::Type::Union
995
998
 
@@ -1028,7 +1031,7 @@ module ContextDev
1028
1031
  end
1029
1032
 
1030
1033
  # When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
1031
- # a 400 WEBSITE_ACCESS_ERROR is returned.
1034
+ # a 400 PDF_SKIPPED is returned.
1032
1035
  module ShouldParse
1033
1036
  extend ContextDev::Internal::Type::Union
1034
1037
 
@@ -35,6 +35,15 @@ module ContextDev
35
35
  sig { params(max_links: Integer).void }
36
36
  attr_writer :max_links
37
37
 
38
+ # Optional search phrase. When provided, the crawled sitemap is filtered to the
39
+ # pages whose URLs are about that phrase, most relevant first, and the request
40
+ # costs 2 credits instead of 1.
41
+ sig { returns(T.nilable(String)) }
42
+ attr_reader :search
43
+
44
+ sig { params(search: String).void }
45
+ attr_writer :search
46
+
38
47
  # Optional explicit sitemap URL. When provided, exactly this sitemap is crawled
39
48
  # instead of discovering the domain's sitemaps.
40
49
  sig { returns(T.nilable(String)) }
@@ -88,6 +97,7 @@ module ContextDev
88
97
  domain: String,
89
98
  headers: T::Hash[Symbol, String],
90
99
  max_links: Integer,
100
+ search: String,
91
101
  sitemap_url: String,
92
102
  tags: T::Array[String],
93
103
  timeout_ms: Integer,
@@ -106,6 +116,10 @@ module ContextDev
106
116
  # Maximum number of links to return from the sitemap crawl. Defaults to 10,000.
107
117
  # Minimum is 1, maximum is 100,000.
108
118
  max_links: nil,
119
+ # Optional search phrase. When provided, the crawled sitemap is filtered to the
120
+ # pages whose URLs are about that phrase, most relevant first, and the request
121
+ # costs 2 credits instead of 1.
122
+ search: nil,
109
123
  # Optional explicit sitemap URL. When provided, exactly this sitemap is crawled
110
124
  # instead of discovering the domain's sitemaps.
111
125
  sitemap_url: nil,
@@ -135,6 +149,7 @@ module ContextDev
135
149
  domain: String,
136
150
  headers: T::Hash[Symbol, String],
137
151
  max_links: Integer,
152
+ search: String,
138
153
  sitemap_url: String,
139
154
  tags: T::Array[String],
140
155
  timeout_ms: Integer,
@@ -34,7 +34,8 @@ module ContextDev
34
34
  end
35
35
  attr_accessor :success
36
36
 
37
- # Array of discovered page URLs from the sitemap (max 500)
37
+ # Discovered page URLs from the sitemap, up to `maxLinks`. When `search` is set
38
+ # these are only the matching pages, most relevant first.
38
39
  sig { returns(T::Array[String]) }
39
40
  attr_accessor :urls
40
41
 
@@ -75,7 +76,8 @@ module ContextDev
75
76
  meta:,
76
77
  # Indicates success
77
78
  success:,
78
- # Array of discovered page URLs from the sitemap (max 500)
79
+ # Discovered page URLs from the sitemap, up to `maxLinks`. When `search` is set
80
+ # these are only the matching pages, most relevant first.
79
81
  urls:,
80
82
  # Metadata about the API key used for the request. Included in every response
81
83
  # whenever a valid API key is provided, even when the response status is not 200.
@@ -7,6 +7,8 @@ module ContextDev
7
7
 
8
8
  BatchCancelParams = ContextDev::Models::BatchCancelParams
9
9
 
10
+ BatchDeleteParams = ContextDev::Models::BatchDeleteParams
11
+
10
12
  BatchGetResultsParams = ContextDev::Models::BatchGetResultsParams
11
13
 
12
14
  BatchListParams = ContextDev::Models::BatchListParams
@@ -64,6 +66,8 @@ module ContextDev
64
66
 
65
67
  ParseHandleParams = ContextDev::Models::ParseHandleParams
66
68
 
69
+ PersonEnrichParams = ContextDev::Models::PersonEnrichParams
70
+
67
71
  UtilityPrefetchParams = ContextDev::Models::UtilityPrefetchParams
68
72
 
69
73
  WebExtractCompetitorsParams = ContextDev::Models::WebExtractCompetitorsParams
@@ -2,6 +2,7 @@
2
2
 
3
3
  module ContextDev
4
4
  module Resources
5
+ # Scrape many pages or crawl a site asynchronously.
5
6
  class Batch
6
7
  # Check progress, and get download links once the batch finishes.
7
8
  sig do
@@ -49,6 +50,21 @@ module ContextDev
49
50
  )
50
51
  end
51
52
 
53
+ # Permanently delete a finished batch and its stored results. Active batches must
54
+ # settle first.
55
+ sig do
56
+ params(
57
+ batch_id: String,
58
+ request_options: ContextDev::RequestOptions::OrHash
59
+ ).returns(ContextDev::Models::BatchDeleteResponse)
60
+ end
61
+ def delete(
62
+ # ID of the batch to retrieve or cancel.
63
+ batch_id,
64
+ request_options: {}
65
+ )
66
+ end
67
+
52
68
  # Stop a batch from starting new pages. In-progress pages finish, and unused
53
69
  # credits are refunded.
54
70
  sig do
@@ -86,24 +102,30 @@ module ContextDev
86
102
  )
87
103
  end
88
104
 
89
- # Retrieve and normalize a person profile from identifiers.
105
+ # Scrape 25K URLs or crawl large websites asynchronously.
90
106
  sig do
91
107
  params(
92
- identifiers: ContextDev::BatchSubmitParams::Identifiers::OrHash,
108
+ input:
109
+ T.any(
110
+ ContextDev::BatchSubmitParams::Input::Scrape::OrHash,
111
+ ContextDev::BatchSubmitParams::Input::Crawl::OrHash
112
+ ),
93
113
  tags: T::Array[String],
94
- timeout_ms: Integer,
114
+ webhook_url: String,
115
+ idempotency_key: String,
95
116
  request_options: ContextDev::RequestOptions::OrHash
96
117
  ).returns(ContextDev::Models::BatchSubmitResponse)
97
118
  end
98
119
  def submit(
99
- # Known identifiers for the person. At least one identifier is required.
100
- identifiers:,
101
- # Optional tags for tracking usage. Up to 20 tags, each 1 to 50 characters.
120
+ # Body param: Choose a URL list or a site crawl.
121
+ input:,
122
+ # Body param: Tags stored on the batch. Filter the batch list by them later.
102
123
  tags: nil,
103
- # Optional timeout in milliseconds for the request. If the request takes longer
104
- # than this value, it will be aborted with a 408 status code. Maximum allowed
105
- # value is 300000ms (5 minutes).
106
- timeout_ms: nil,
124
+ # Body param: URL notified when the batch finishes.
125
+ webhook_url: nil,
126
+ # Header param: Any string unique to this submission. Retries with the same key
127
+ # return the original batch.
128
+ idempotency_key: nil,
107
129
  request_options: {}
108
130
  )
109
131
  end
@@ -65,7 +65,8 @@ module ContextDev
65
65
  end
66
66
 
67
67
  # Search brands by name or domain and get back up to 10 lightweight matches
68
- # (domain, name, logo), most popular first: by Tranco rank, then market cap for
68
+ # (domain, name, logo). Name matches rank ahead of domain matches; within each
69
+ # group the most popular brands come first: by Tranco rank, then market cap for
69
70
  # brands outside the Tranco list, with text relevance breaking ties. Matching is
70
71
  # prefix-based with no typo tolerance, so it is suited to autocomplete. Only
71
72
  # brands already in the Context.dev index are returned — use /brand/retrieve to
@@ -49,11 +49,11 @@ module ContextDev
49
49
  include_images: nil,
50
50
  # Query param: Preserve hyperlinks in Markdown output
51
51
  include_links: nil,
52
- # Query param: When true for PDF inputs, detect and OCR images embedded in the
53
- # selected pages, inserting recognized text at each image's position in page
54
- # reading order while preserving the PDF text layer. pdf.start/pdf.end limit the
55
- # inclusive page range. When false, all OCR is disabled, including the automatic
56
- # scanned-PDF fallback.
52
+ # Query param: When true for PDF inputs, OCR the selected pages that have no
53
+ # usable text layer (scans), replacing each recovered page's text with the OCR
54
+ # result while pages with a real text layer keep it. pdf.start/pdf.end limit the
55
+ # inclusive page range. Billed at 1 credit per page OCR actually recovered, on top
56
+ # of the base request cost. When false, no OCR runs.
57
57
  ocr: nil,
58
58
  # Query param: PDF page-range options as a JSON object, e.g. {"start": 2, "end":
59
59
  # 5}.
@@ -0,0 +1,47 @@
1
+ # typed: strong
2
+
3
+ module ContextDev
4
+ module Resources
5
+ class People
6
+ # Finds and normalizes the best available person candidate from additive identity
7
+ # clues, then assigns an identity match score from 0 to 100. Available on all paid
8
+ # plans. Successful requests cost 20 credits. Disposable and free email addresses
9
+ # (like gmail.com, yahoo.com) will throw a 422 error.
10
+ sig do
11
+ params(
12
+ company: ContextDev::PersonEnrichParams::Company::OrHash,
13
+ education:
14
+ T::Array[ContextDev::PersonEnrichParams::Education::OrHash],
15
+ email: String,
16
+ location: ContextDev::PersonEnrichParams::Location::OrHash,
17
+ name: ContextDev::PersonEnrichParams::Name::OrHash,
18
+ social_urls: T::Array[String],
19
+ tags: T::Array[String],
20
+ timeout_ms: Integer,
21
+ request_options: ContextDev::RequestOptions::OrHash
22
+ ).returns(ContextDev::Models::PersonEnrichResponse)
23
+ end
24
+ def enrich(
25
+ company: nil,
26
+ education: nil,
27
+ email: nil,
28
+ location: nil,
29
+ name: nil,
30
+ social_urls: nil,
31
+ # Optional tags for tracking usage. Up to 20 tags, each 1 to 50 characters.
32
+ tags: nil,
33
+ # Optional timeout in milliseconds for the request. If the request takes longer
34
+ # than this value, it will be aborted with a 408 status code. Maximum allowed
35
+ # value is 300000ms (5 minutes).
36
+ timeout_ms: nil,
37
+ request_options: {}
38
+ )
39
+ end
40
+
41
+ # @api private
42
+ sig { params(client: ContextDev::Client).returns(T.attached_class) }
43
+ def self.new(client:)
44
+ end
45
+ end
46
+ end
47
+ end
@@ -595,6 +595,18 @@ module ContextDev
595
595
  # responses from a recognized API key; use error_code to distinguish stable
596
596
  # failure categories.
597
597
  #
598
+ # ### YouTube
599
+ #
600
+ # YouTube URLs return the video or channel itself rather than the surrounding
601
+ # player and navigation chrome. A URL addressing a single video (`/watch`,
602
+ # `youtu.be`, `/shorts`, `/embed`, `/live`) returns its title, channel, duration,
603
+ # view count, keywords, full description, and the transcript when the video has
604
+ # captions that can be retrieved; videos without captions return everything except
605
+ # the transcript. A channel URL (`/channel/UC…`, `/@handle`, `/c/…`, `/user/…`)
606
+ # returns its name, handle, subscriber count, video count, and full description.
607
+ # When `includeImages=true`, video responses also include the thumbnail and
608
+ # channel responses include the avatar. Costs the same as any other scrape.
609
+ #
598
610
  # ### Billing & errors
599
611
  #
600
612
  # | HTTP status | Billed? | Meaning |
@@ -604,6 +616,7 @@ module ContextDev
604
616
  # | 401 / 403 | No | Invalid/disabled key, insufficient permissions, or credits exhausted; inspect error_code |
605
617
  # | 404 | No | Target page returned or fingerprinted as not found |
606
618
  # | 408 | No | Request timed out |
619
+ # | 413 | No | Target content exceeds the maximum supported size (20 MB) |
607
620
  # | 415 | No | Unsupported content type |
608
621
  # | 429 | No | Per-minute rate limit exceeded; honor Retry-After |
609
622
  # | 500 | No | Internal error |
@@ -727,12 +740,17 @@ module ContextDev
727
740
  )
728
741
  end
729
742
 
730
- # Crawl an entire website's sitemap and return all discovered page URLs.
743
+ # Crawl an entire website's sitemap and return all discovered page URLs. Pass
744
+ # `search` to have the crawled sitemap filtered down to the pages about a phrase
745
+ # (for example `pricing and plans` or `api authentication docs`), most relevant
746
+ # first — a searched crawl scans the whole sitemap and costs 2 credits instead
747
+ # of 1.
731
748
  sig do
732
749
  params(
733
750
  domain: String,
734
751
  headers: T::Hash[Symbol, String],
735
752
  max_links: Integer,
753
+ search: String,
736
754
  sitemap_url: String,
737
755
  tags: T::Array[String],
738
756
  timeout_ms: Integer,
@@ -751,6 +769,10 @@ module ContextDev
751
769
  # Maximum number of links to return from the sitemap crawl. Defaults to 10,000.
752
770
  # Minimum is 1, maximum is 100,000.
753
771
  max_links: nil,
772
+ # Optional search phrase. When provided, the crawled sitemap is filtered to the
773
+ # pages whose URLs are about that phrase, most relevant first, and the request
774
+ # costs 2 credits instead of 1.
775
+ search: nil,
754
776
  # Optional explicit sitemap URL. When provided, exactly this sitemap is crawled
755
777
  # instead of discovering the domain's sitemaps.
756
778
  sitemap_url: nil,
@@ -26,6 +26,8 @@ module ContextDev
26
26
 
27
27
  attr_reader batch: ContextDev::Resources::Batch
28
28
 
29
+ attr_reader people: ContextDev::Resources::People
30
+
29
31
  private def auth_headers: -> ::Hash[String, String]
30
32
 
31
33
  def initialize: (
@@ -0,0 +1,23 @@
1
+ module ContextDev
2
+ module Models
3
+ type batch_delete_params =
4
+ { batch_id: String } & ContextDev::Internal::Type::request_parameters
5
+
6
+ class BatchDeleteParams < ContextDev::Internal::Type::BaseModel
7
+ extend ContextDev::Internal::Type::RequestParameters::Converter
8
+ include ContextDev::Internal::Type::RequestParameters
9
+
10
+ attr_accessor batch_id: String
11
+
12
+ def initialize: (
13
+ batch_id: String,
14
+ ?request_options: ContextDev::request_opts
15
+ ) -> void
16
+
17
+ def to_hash: -> {
18
+ batch_id: String,
19
+ request_options: ContextDev::RequestOptions
20
+ }
21
+ end
22
+ end
23
+ end
@@ -0,0 +1,57 @@
1
+ module ContextDev
2
+ module Models
3
+ type batch_delete_response =
4
+ {
5
+ id: String,
6
+ deleted: bool,
7
+ key_metadata: ContextDev::Models::BatchDeleteResponse::KeyMetadata
8
+ }
9
+
10
+ class BatchDeleteResponse < ContextDev::Internal::Type::BaseModel
11
+ attr_reader id: String?
12
+
13
+ def id=: (String) -> String
14
+
15
+ attr_reader deleted: bool?
16
+
17
+ def deleted=: (bool) -> bool
18
+
19
+ attr_reader key_metadata: ContextDev::Models::BatchDeleteResponse::KeyMetadata?
20
+
21
+ def key_metadata=: (
22
+ ContextDev::Models::BatchDeleteResponse::KeyMetadata
23
+ ) -> ContextDev::Models::BatchDeleteResponse::KeyMetadata
24
+
25
+ def initialize: (
26
+ ?id: String,
27
+ ?deleted: bool,
28
+ ?key_metadata: ContextDev::Models::BatchDeleteResponse::KeyMetadata
29
+ ) -> void
30
+
31
+ def to_hash: -> {
32
+ id: String,
33
+ deleted: bool,
34
+ key_metadata: ContextDev::Models::BatchDeleteResponse::KeyMetadata
35
+ }
36
+
37
+ type key_metadata =
38
+ { credits_consumed: Integer, credits_remaining: Integer }
39
+
40
+ class KeyMetadata < ContextDev::Internal::Type::BaseModel
41
+ attr_accessor credits_consumed: Integer
42
+
43
+ attr_accessor credits_remaining: Integer
44
+
45
+ def initialize: (
46
+ credits_consumed: Integer,
47
+ credits_remaining: Integer
48
+ ) -> void
49
+
50
+ def to_hash: -> {
51
+ credits_consumed: Integer,
52
+ credits_remaining: Integer
53
+ }
54
+ end
55
+ end
56
+ end
57
+ end
@@ -58,7 +58,8 @@ module ContextDev
58
58
  html: String,
59
59
  item_id: String,
60
60
  markdown: String,
61
- meta: ::Hash[Symbol, top]
61
+ meta: ::Hash[Symbol, top],
62
+ ocr_pages: Integer
62
63
  }
63
64
 
64
65
  class Ok < ContextDev::Internal::Type::BaseModel
@@ -88,6 +89,10 @@ module ContextDev
88
89
 
89
90
  def meta=: (::Hash[Symbol, top]) -> ::Hash[Symbol, top]
90
91
 
92
+ attr_reader ocr_pages: Integer?
93
+
94
+ def ocr_pages=: (Integer) -> Integer
95
+
91
96
  def initialize: (
92
97
  final_url: String,
93
98
  http_status: Integer?,
@@ -97,6 +102,7 @@ module ContextDev
97
102
  ?item_id: String,
98
103
  ?markdown: String,
99
104
  ?meta: ::Hash[Symbol, top],
105
+ ?ocr_pages: Integer,
100
106
  ?status: :ok
101
107
  ) -> void
102
108
 
@@ -109,7 +115,8 @@ module ContextDev
109
115
  html: String,
110
116
  item_id: String,
111
117
  markdown: String,
112
- meta: ::Hash[Symbol, top]
118
+ meta: ::Hash[Symbol, top],
119
+ ocr_pages: Integer
113
120
  }
114
121
 
115
122
  type metadata =
@@ -117,22 +117,36 @@ module ContextDev
117
117
  timing: ContextDev::Models::BatchListResponse::Data::Timing
118
118
  }
119
119
 
120
- type credits = { net: Integer, refunded: Integer, reserved: Integer }
120
+ type credits =
121
+ {
122
+ net: Integer,
123
+ ocr_charged: Integer,
124
+ refunded: Integer,
125
+ reserved: Integer
126
+ }
121
127
 
122
128
  class Credits < ContextDev::Internal::Type::BaseModel
123
129
  attr_accessor net: Integer
124
130
 
131
+ attr_accessor ocr_charged: Integer
132
+
125
133
  attr_accessor refunded: Integer
126
134
 
127
135
  attr_accessor reserved: Integer
128
136
 
129
137
  def initialize: (
130
138
  net: Integer,
139
+ ocr_charged: Integer,
131
140
  refunded: Integer,
132
141
  reserved: Integer
133
142
  ) -> void
134
143
 
135
- def to_hash: -> { net: Integer, refunded: Integer, reserved: Integer }
144
+ def to_hash: -> {
145
+ net: Integer,
146
+ ocr_charged: Integer,
147
+ refunded: Integer,
148
+ reserved: Integer
149
+ }
136
150
  end
137
151
 
138
152
  type format_ = :markdown | :html