context.dev 2.8.0 → 2.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (116) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +28 -0
  3. data/README.md +1 -1
  4. data/lib/context_dev/client.rb +10 -0
  5. data/lib/context_dev/models/batch_delete_params.rb +22 -0
  6. data/lib/context_dev/models/batch_delete_response.rb +60 -0
  7. data/lib/context_dev/models/batch_get_results_response.rb +46 -4
  8. data/lib/context_dev/models/batch_list_response.rb +13 -4
  9. data/lib/context_dev/models/batch_retrieve_response.rb +13 -4
  10. data/lib/context_dev/models/batch_submit_params.rb +2080 -26
  11. data/lib/context_dev/models/batch_submit_response.rb +125 -531
  12. data/lib/context_dev/models/brand_retrieve_response.rb +50 -1
  13. data/lib/context_dev/models/brand_search_params.rb +41 -3
  14. data/lib/context_dev/models/brand_search_response.rb +3 -2
  15. data/lib/context_dev/models/crawl_controls.rb +21 -15
  16. data/lib/context_dev/models/news_search_params.rb +467 -0
  17. data/lib/context_dev/models/news_search_response.rb +238 -0
  18. data/lib/context_dev/models/parse_handle_params.rb +20 -147
  19. data/lib/context_dev/models/person_enrich_params.rb +176 -0
  20. data/lib/context_dev/models/person_enrich_response.rb +641 -0
  21. data/lib/context_dev/models/utility_prefetch_params.rb +19 -16
  22. data/lib/context_dev/models/utility_prefetch_response.rb +4 -5
  23. data/lib/context_dev/models/web_screenshot_params.rb +3 -30
  24. data/lib/context_dev/models/web_search_response.rb +1 -0
  25. data/lib/context_dev/models/web_web_crawl_md_params.rb +5 -4
  26. data/lib/context_dev/models/web_web_crawl_md_response.rb +30 -1
  27. data/lib/context_dev/models/web_web_scrape_html_params.rb +20 -156
  28. data/lib/context_dev/models/web_web_scrape_html_response.rb +30 -1
  29. data/lib/context_dev/models/web_web_scrape_images_params.rb +12 -124
  30. data/lib/context_dev/models/web_web_scrape_md_params.rb +40 -241
  31. data/lib/context_dev/models/web_web_scrape_md_response.rb +41 -2
  32. data/lib/context_dev/models/web_web_scrape_sitemap_params.rb +11 -1
  33. data/lib/context_dev/models/web_web_scrape_sitemap_response.rb +3 -2
  34. data/lib/context_dev/models.rb +6 -0
  35. data/lib/context_dev/resources/batch.rb +33 -7
  36. data/lib/context_dev/resources/brand.rb +10 -10
  37. data/lib/context_dev/resources/news.rb +51 -0
  38. data/lib/context_dev/resources/parse.rb +5 -5
  39. data/lib/context_dev/resources/people.rb +56 -0
  40. data/lib/context_dev/resources/utility.rb +7 -6
  41. data/lib/context_dev/resources/web.rb +46 -24
  42. data/lib/context_dev/version.rb +1 -1
  43. data/lib/context_dev.rb +8 -0
  44. data/rbi/context_dev/client.rbi +8 -0
  45. data/rbi/context_dev/models/batch_delete_params.rbi +40 -0
  46. data/rbi/context_dev/models/batch_delete_response.rbi +116 -0
  47. data/rbi/context_dev/models/batch_get_results_response.rbi +85 -3
  48. data/rbi/context_dev/models/batch_list_response.rbi +24 -8
  49. data/rbi/context_dev/models/batch_retrieve_response.rbi +24 -8
  50. data/rbi/context_dev/models/batch_submit_params.rbi +6421 -44
  51. data/rbi/context_dev/models/batch_submit_response.rbi +184 -1151
  52. data/rbi/context_dev/models/brand_retrieve_response.rbi +152 -0
  53. data/rbi/context_dev/models/brand_search_params.rbi +71 -2
  54. data/rbi/context_dev/models/brand_search_response.rbi +4 -2
  55. data/rbi/context_dev/models/crawl_controls.rbi +22 -28
  56. data/rbi/context_dev/models/news_search_params.rbi +1294 -0
  57. data/rbi/context_dev/models/news_search_response.rbi +423 -0
  58. data/rbi/context_dev/models/parse_handle_params.rbi +30 -323
  59. data/rbi/context_dev/models/person_enrich_params.rbi +332 -0
  60. data/rbi/context_dev/models/person_enrich_response.rbi +1295 -0
  61. data/rbi/context_dev/models/utility_prefetch_params.rbi +24 -16
  62. data/rbi/context_dev/models/utility_prefetch_response.rbi +8 -6
  63. data/rbi/context_dev/models/web_screenshot_params.rbi +4 -71
  64. data/rbi/context_dev/models/web_search_response.rbi +5 -0
  65. data/rbi/context_dev/models/web_web_crawl_md_params.rbi +8 -6
  66. data/rbi/context_dev/models/web_web_crawl_md_response.rbi +67 -0
  67. data/rbi/context_dev/models/web_web_scrape_html_params.rbi +30 -363
  68. data/rbi/context_dev/models/web_web_scrape_html_response.rbi +65 -0
  69. data/rbi/context_dev/models/web_web_scrape_images_params.rbi +16 -287
  70. data/rbi/context_dev/models/web_web_scrape_md_params.rbi +57 -558
  71. data/rbi/context_dev/models/web_web_scrape_md_response.rbi +80 -0
  72. data/rbi/context_dev/models/web_web_scrape_sitemap_params.rbi +15 -0
  73. data/rbi/context_dev/models/web_web_scrape_sitemap_response.rbi +4 -2
  74. data/rbi/context_dev/models.rbi +6 -0
  75. data/rbi/context_dev/resources/batch.rbi +32 -10
  76. data/rbi/context_dev/resources/brand.rbi +14 -8
  77. data/rbi/context_dev/resources/news.rbi +46 -0
  78. data/rbi/context_dev/resources/parse.rbi +10 -26
  79. data/rbi/context_dev/resources/people.rbi +47 -0
  80. data/rbi/context_dev/resources/utility.rbi +8 -6
  81. data/rbi/context_dev/resources/web.rbi +49 -66
  82. data/sig/context_dev/client.rbs +4 -0
  83. data/sig/context_dev/models/batch_delete_params.rbs +23 -0
  84. data/sig/context_dev/models/batch_delete_response.rbs +57 -0
  85. data/sig/context_dev/models/batch_get_results_response.rbs +30 -2
  86. data/sig/context_dev/models/batch_list_response.rbs +16 -2
  87. data/sig/context_dev/models/batch_retrieve_response.rbs +16 -2
  88. data/sig/context_dev/models/batch_submit_params.rbs +2666 -15
  89. data/sig/context_dev/models/batch_submit_response.rbs +78 -466
  90. data/sig/context_dev/models/brand_retrieve_response.rbs +62 -0
  91. data/sig/context_dev/models/brand_search_params.rbs +38 -1
  92. data/sig/context_dev/models/crawl_controls.rbs +16 -16
  93. data/sig/context_dev/models/news_search_params.rbs +532 -0
  94. data/sig/context_dev/models/news_search_response.rbs +206 -0
  95. data/sig/context_dev/models/parse_handle_params.rbs +25 -90
  96. data/sig/context_dev/models/person_enrich_params.rbs +199 -0
  97. data/sig/context_dev/models/person_enrich_response.rbs +638 -0
  98. data/sig/context_dev/models/utility_prefetch_params.rbs +2 -1
  99. data/sig/context_dev/models/utility_prefetch_response.rbs +2 -1
  100. data/sig/context_dev/models/web_screenshot_params.rbs +5 -18
  101. data/sig/context_dev/models/web_search_response.rbs +2 -0
  102. data/sig/context_dev/models/web_web_crawl_md_response.rbs +21 -0
  103. data/sig/context_dev/models/web_web_scrape_html_params.rbs +24 -94
  104. data/sig/context_dev/models/web_web_scrape_html_response.rbs +21 -0
  105. data/sig/context_dev/models/web_web_scrape_images_params.rbs +20 -72
  106. data/sig/context_dev/models/web_web_scrape_md_params.rbs +46 -148
  107. data/sig/context_dev/models/web_web_scrape_md_response.rbs +28 -0
  108. data/sig/context_dev/models/web_web_scrape_sitemap_params.rbs +7 -0
  109. data/sig/context_dev/models.rbs +6 -0
  110. data/sig/context_dev/resources/batch.rbs +8 -2
  111. data/sig/context_dev/resources/brand.rbs +3 -0
  112. data/sig/context_dev/resources/news.rbs +17 -0
  113. data/sig/context_dev/resources/parse.rbs +5 -5
  114. data/sig/context_dev/resources/people.rbs +19 -0
  115. data/sig/context_dev/resources/web.rbs +13 -11
  116. metadata +26 -2
@@ -23,7 +23,8 @@ module ContextDev
23
23
  required :success, enum: -> { ContextDev::Models::WebWebScrapeSitemapResponse::Success }
24
24
 
25
25
  # @!attribute urls
26
- # Array of discovered page URLs from the sitemap (max 500)
26
+ # Discovered page URLs from the sitemap, up to `maxLinks`. When `search` is set
27
+ # these are only the matching pages, most relevant first.
27
28
  #
28
29
  # @return [Array<String>]
29
30
  required :urls, ContextDev::Internal::Type::ArrayOf[String]
@@ -45,7 +46,7 @@ module ContextDev
45
46
  #
46
47
  # @param success [Boolean, ContextDev::Models::WebWebScrapeSitemapResponse::Success] Indicates success
47
48
  #
48
- # @param urls [Array<String>] Array of discovered page URLs from the sitemap (max 500)
49
+ # @param urls [Array<String>] Discovered page URLs from the sitemap, up to `maxLinks`. When `search` is set th
49
50
  #
50
51
  # @param key_metadata [ContextDev::Models::WebWebScrapeSitemapResponse::KeyMetadata] Metadata about the API key used for the request. Included in every response when
51
52
 
@@ -45,6 +45,8 @@ module ContextDev
45
45
 
46
46
  BatchCancelParams = ContextDev::Models::BatchCancelParams
47
47
 
48
+ BatchDeleteParams = ContextDev::Models::BatchDeleteParams
49
+
48
50
  BatchGetResultsParams = ContextDev::Models::BatchGetResultsParams
49
51
 
50
52
  BatchListParams = ContextDev::Models::BatchListParams
@@ -95,10 +97,14 @@ module ContextDev
95
97
 
96
98
  MonitorUpdateParams = ContextDev::Models::MonitorUpdateParams
97
99
 
100
+ NewsSearchParams = ContextDev::Models::NewsSearchParams
101
+
98
102
  PageErrorCount = ContextDev::Models::PageErrorCount
99
103
 
100
104
  ParseHandleParams = ContextDev::Models::ParseHandleParams
101
105
 
106
+ PersonEnrichParams = ContextDev::Models::PersonEnrichParams
107
+
102
108
  UtilityPrefetchParams = ContextDev::Models::UtilityPrefetchParams
103
109
 
104
110
  WebExtractCompetitorsParams = ContextDev::Models::WebExtractCompetitorsParams
@@ -2,6 +2,7 @@
2
2
 
3
3
  module ContextDev
4
4
  module Resources
5
+ # Scrape many pages or crawl a site asynchronously.
5
6
  class Batch
6
7
  # Check progress, and get download links once the batch finishes.
7
8
  #
@@ -60,6 +61,27 @@ module ContextDev
60
61
  )
61
62
  end
62
63
 
64
+ # Permanently delete a finished batch and its stored results. Active batches must
65
+ # settle first.
66
+ #
67
+ # @overload delete(batch_id, request_options: {})
68
+ #
69
+ # @param batch_id [String] ID of the batch to retrieve or cancel.
70
+ #
71
+ # @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
72
+ #
73
+ # @return [ContextDev::Models::BatchDeleteResponse]
74
+ #
75
+ # @see ContextDev::Models::BatchDeleteParams
76
+ def delete(batch_id, params = {})
77
+ @client.request(
78
+ method: :delete,
79
+ path: ["batch/%1$s", batch_id],
80
+ model: ContextDev::Models::BatchDeleteResponse,
81
+ options: params[:request_options]
82
+ )
83
+ end
84
+
63
85
  # Stop a batch from starting new pages. In-progress pages finish, and unused
64
86
  # credits are refunded.
65
87
  #
@@ -115,15 +137,17 @@ module ContextDev
115
137
  # Some parameter documentations has been truncated, see
116
138
  # {ContextDev::Models::BatchSubmitParams} for more details.
117
139
  #
118
- # Retrieve and normalize a person profile from identifiers.
140
+ # Scrape 25K URLs or crawl large websites asynchronously.
141
+ #
142
+ # @overload submit(input:, tags: nil, webhook_url: nil, idempotency_key: nil, request_options: {})
119
143
  #
120
- # @overload submit(identifiers:, tags: nil, timeout_ms: nil, request_options: {})
144
+ # @param input [ContextDev::Models::BatchSubmitParams::Input::Scrape, ContextDev::Models::BatchSubmitParams::Input::Crawl] Body param: Choose a URL list or a site crawl.
121
145
  #
122
- # @param identifiers [ContextDev::Models::BatchSubmitParams::Identifiers] Known identifiers for the person. At least one identifier is required.
146
+ # @param tags [Array<String>] Body param: Tags stored on the batch. Filter the batch list by them later.
123
147
  #
124
- # @param tags [Array<String>] Optional tags for tracking usage. Up to 20 tags, each 1 to 50 characters.
148
+ # @param webhook_url [String] Body param: URL notified when the batch finishes.
125
149
  #
126
- # @param timeout_ms [Integer] Optional timeout in milliseconds for the request. If the request takes longer th
150
+ # @param idempotency_key [String] Header param: Any string unique to this submission. Retries with the same key re
127
151
  #
128
152
  # @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
129
153
  #
@@ -132,10 +156,12 @@ module ContextDev
132
156
  # @see ContextDev::Models::BatchSubmitParams
133
157
  def submit(params)
134
158
  parsed, options = ContextDev::BatchSubmitParams.dump_request(params)
159
+ header_params = {idempotency_key: "idempotency-key"}
135
160
  @client.request(
136
161
  method: :post,
137
- path: "people/retrieve",
138
- body: parsed,
162
+ path: "batch/submit",
163
+ headers: parsed.slice(*header_params.keys).transform_keys(header_params),
164
+ body: parsed.except(*header_params.keys),
139
165
  model: ContextDev::Models::BatchSubmitResponse,
140
166
  options: options
141
167
  )
@@ -68,20 +68,20 @@ module ContextDev
68
68
  # Some parameter documentations has been truncated, see
69
69
  # {ContextDev::Models::BrandSearchParams} for more details.
70
70
  #
71
- # Search brands by name or domain and get back up to 10 lightweight matches
72
- # (domain, name, logo), most popular first: by Tranco rank, then market cap for
73
- # brands outside the Tranco list, with text relevance breaking ties. Matching is
74
- # prefix-based with no typo tolerance, so it is suited to autocomplete. Only
75
- # brands already in the Context.dev index are returned — use /brand/retrieve to
76
- # fetch (and index) a specific domain. Free on Pro and Scale plans; costs 1 credit
77
- # per request on the Free and Starter plans.
71
+ # Search indexed brands by name or domain
78
72
  #
79
- # @overload search(query:, tags: nil, request_options: {})
73
+ # @overload search(query:, autocomplete: nil, query_by: nil, tags: nil, typo_tolerance: nil, request_options: {})
80
74
  #
81
- # @param query [String] Search term, matched against brand names and domains by prefix (e.g. 'nike', 'ni
75
+ # @param query [String] Search term, matched against the fields selected by queryBy (e.g. 'nike', 'nike.
76
+ #
77
+ # @param autocomplete [Boolean] Whether the search term matches by prefix, so partial words match as they are ty
78
+ #
79
+ # @param query_by [Array<Symbol, ContextDev::Models::BrandSearchParams::QueryBy>] Fields to match the search term against, as a comma-separated list or repeated p
82
80
  #
83
81
  # @param tags [Array<String>] Optional comma-separated caller-defined tags for tracking this request. Tags are
84
82
  #
83
+ # @param typo_tolerance [Integer] Maximum number of typos tolerated when matching, from 0 to 2. Defaults to 0 (no
84
+ #
85
85
  # @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
86
86
  #
87
87
  # @return [ContextDev::Models::BrandSearchResponse]
@@ -93,7 +93,7 @@ module ContextDev
93
93
  @client.request(
94
94
  method: :get,
95
95
  path: "brand/search",
96
- query: query,
96
+ query: query.transform_keys(query_by: "queryBy", typo_tolerance: "typoTolerance"),
97
97
  model: ContextDev::Models::BrandSearchResponse,
98
98
  options: options
99
99
  )
@@ -0,0 +1,51 @@
1
+ # frozen_string_literal: true
2
+
3
+ module ContextDev
4
+ module Resources
5
+ # Search live first-party RSS and free historical news data by company identity.
6
+ class News
7
+ # Searches live and historical company news for one company, identified in
8
+ # searchBy by name, domain, ticker (optionally disambiguated by exchange), or
9
+ # ISIN. Results can be filtered by publisher domain, publisher country, article
10
+ # language, article type, and published-at date, and include stable story IDs,
11
+ # source metadata, verified entity relevance, and cursor pagination.
12
+ #
13
+ # @overload search(search_by:, cursor: nil, filter_by: nil, limit: nil, sort_by: nil, tags: nil, request_options: {})
14
+ #
15
+ # @param search_by [ContextDev::Models::NewsSearchParams::SearchBy] What to search for.
16
+ #
17
+ # @param cursor [String, nil] Opaque next_cursor from the previous response, or null for the first page.
18
+ #
19
+ # @param filter_by [ContextDev::Models::NewsSearchParams::FilterBy] Optional result filters.
20
+ #
21
+ # @param limit [Integer] Maximum results to return. Defaults to 10.
22
+ #
23
+ # @param sort_by [ContextDev::Models::NewsSearchParams::SortBy] Result ordering. Defaults to newest.
24
+ #
25
+ # @param tags [Array<String>] Optional tags for tracking usage. Up to 20 tags, each 1 to 50 characters.
26
+ #
27
+ # @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
28
+ #
29
+ # @return [ContextDev::Models::NewsSearchResponse]
30
+ #
31
+ # @see ContextDev::Models::NewsSearchParams
32
+ def search(params)
33
+ parsed, options = ContextDev::NewsSearchParams.dump_request(params)
34
+ @client.request(
35
+ method: :post,
36
+ path: "news/search",
37
+ body: parsed,
38
+ model: ContextDev::Models::NewsSearchResponse,
39
+ options: options
40
+ )
41
+ end
42
+
43
+ # @api private
44
+ #
45
+ # @param client [ContextDev::Client]
46
+ def initialize(client:)
47
+ @client = client
48
+ end
49
+ end
50
+ end
51
+ end
@@ -17,19 +17,19 @@ module ContextDev
17
17
  #
18
18
  # @param extension [Symbol, ContextDev::Models::ParseHandleParams::Extension] Query param: Optional file extension hint, such as pdf, docx, xlsx, pptx, html,
19
19
  #
20
- # @param include_images [Boolean, Symbol, ContextDev::Models::ParseHandleParams::IncludeImages] Query param: Include image references in Markdown output
20
+ # @param include_images [Boolean] Query param: Include image references in Markdown output
21
21
  #
22
- # @param include_links [Boolean, Symbol, ContextDev::Models::ParseHandleParams::IncludeLinks] Query param: Preserve hyperlinks in Markdown output
22
+ # @param include_links [Boolean] Query param: Preserve hyperlinks in Markdown output
23
23
  #
24
- # @param ocr [Boolean, Symbol, ContextDev::Models::ParseHandleParams::Ocr] Query param: When true for PDF inputs, detect and OCR images embedded in the sel
24
+ # @param ocr [Boolean] Query param: When true for PDF inputs, OCR the selected pages that have no usabl
25
25
  #
26
26
  # @param pdf [ContextDev::Models::ParseHandleParams::Pdf] Query param: PDF page-range options as a JSON object, e.g. {"start": 2, "end": 5
27
27
  #
28
- # @param shorten_base64_images [Boolean, Symbol, ContextDev::Models::ParseHandleParams::ShortenBase64Images] Query param: Shorten base64-encoded image data in the Markdown output
28
+ # @param shorten_base64_images [Boolean] Query param: Shorten base64-encoded image data in the Markdown output
29
29
  #
30
30
  # @param tags [Array<String>] Query param: Optional comma-separated caller-defined tags for tracking this requ
31
31
  #
32
- # @param use_main_content_only [Boolean, Symbol, ContextDev::Models::ParseHandleParams::UseMainContentOnly] Query param: Extract only the main content from HTML-like inputs
32
+ # @param use_main_content_only [Boolean] Query param: Extract only the main content from HTML-like inputs
33
33
  #
34
34
  # @param zdr [Symbol, ContextDev::Models::ParseHandleParams::Zdr] Query param: Set to enabled to bypass shared caches and omit request and respons
35
35
  #
@@ -0,0 +1,56 @@
1
+ # frozen_string_literal: true
2
+
3
+ module ContextDev
4
+ module Resources
5
+ class People
6
+ # Some parameter documentations has been truncated, see
7
+ # {ContextDev::Models::PersonEnrichParams} for more details.
8
+ #
9
+ # Finds and normalizes the best available person candidate from additive identity
10
+ # clues, then assigns an identity match score from 0 to 100. Available on all paid
11
+ # plans. Successful requests cost 20 credits. Disposable and free email addresses
12
+ # (like gmail.com, yahoo.com) will throw a 422 error.
13
+ #
14
+ # @overload enrich(company: nil, education: nil, email: nil, location: nil, name: nil, social_urls: nil, tags: nil, timeout_ms: nil, request_options: {})
15
+ #
16
+ # @param company [ContextDev::Models::PersonEnrichParams::Company]
17
+ #
18
+ # @param education [Array<ContextDev::Models::PersonEnrichParams::Education>]
19
+ #
20
+ # @param email [String]
21
+ #
22
+ # @param location [ContextDev::Models::PersonEnrichParams::Location]
23
+ #
24
+ # @param name [ContextDev::Models::PersonEnrichParams::Name]
25
+ #
26
+ # @param social_urls [Array<String>]
27
+ #
28
+ # @param tags [Array<String>] Optional tags for tracking usage. Up to 20 tags, each 1 to 50 characters.
29
+ #
30
+ # @param timeout_ms [Integer] Optional timeout in milliseconds for the request. If the request takes longer th
31
+ #
32
+ # @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
33
+ #
34
+ # @return [ContextDev::Models::PersonEnrichResponse]
35
+ #
36
+ # @see ContextDev::Models::PersonEnrichParams
37
+ def enrich(params = {})
38
+ parsed, options = ContextDev::PersonEnrichParams.dump_request(params)
39
+ @client.request(
40
+ method: :post,
41
+ path: "people/enrich",
42
+ body: parsed,
43
+ model: ContextDev::Models::PersonEnrichResponse,
44
+ options: options
45
+ )
46
+ end
47
+
48
+ # @api private
49
+ #
50
+ # @param client [ContextDev::Client]
51
+ def initialize(client:)
52
+ @client = client
53
+ end
54
+ end
55
+ end
56
+ end
@@ -6,16 +6,17 @@ module ContextDev
6
6
  # Some parameter documentations has been truncated, see
7
7
  # {ContextDev::Models::UtilityPrefetchParams} for more details.
8
8
  #
9
- # Signal that you may fetch brand data soon to improve latency. The type field
10
- # selects what to prefetch (currently only 'brand') and identifier carries exactly
11
- # one lookup key: a domain, or an email whose domain is extracted and validated
12
- # (free email providers and disposable email addresses are not allowed).
9
+ # Signal that you may fetch data soon to improve latency. The type field selects
10
+ # what to prefetch ('brand' queues a brand data fetch, 'styleguide' queues a
11
+ # styleguide extraction) and identifier carries exactly one lookup key: a domain,
12
+ # or an email whose domain is extracted and validated (free email providers and
13
+ # disposable email addresses are not allowed).
13
14
  #
14
15
  # @overload prefetch(identifier:, type:, tags: nil, timeout_ms: nil, request_options: {})
15
16
  #
16
- # @param identifier [ContextDev::Models::UtilityPrefetchParams::Identifier::UtilityPrefetchDomainIdentifier, ContextDev::Models::UtilityPrefetchParams::Identifier::UtilityPrefetchEmailIdentifier] Identifier of the brand to prefetch. Provide exactly one of domain or email.
17
+ # @param identifier [ContextDev::Models::UtilityPrefetchParams::Identifier::UtilityPrefetchDomainIdentifier, ContextDev::Models::UtilityPrefetchParams::Identifier::UtilityPrefetchEmailIdentifier] Identifier of the target to prefetch. Provide exactly one of domain or email.
17
18
  #
18
- # @param type [Symbol, ContextDev::Models::UtilityPrefetchParams::Type] What to prefetch. Currently only 'brand' is supported.
19
+ # @param type [Symbol, ContextDev::Models::UtilityPrefetchParams::Type] What to prefetch: 'brand' warms the brand data cache, 'styleguide' warms the sty
19
20
  #
20
21
  # @param tags [Array<String>] Optional tags for tracking usage. Up to 20 tags, each 1 to 50 characters.
21
22
  #
@@ -188,7 +188,7 @@ module ContextDev
188
188
  #
189
189
  # @param full_screenshot [Symbol, ContextDev::Models::WebScreenshotParams::FullScreenshot] Optional parameter to determine screenshot type. If 'true', takes a full page sc
190
190
  #
191
- # @param handle_cookie_popup [Boolean, Symbol, ContextDev::Models::WebScreenshotParams::HandleCookiePopup] Optional parameter to control cookie/consent popup handling. If 'true', we dismi
191
+ # @param handle_cookie_popup [Boolean] Optional parameter to control cookie/consent popup handling. If 'true', we dismi
192
192
  #
193
193
  # @param max_age_ms [Integer, nil] Return a cached screenshot if a prior screenshot for the same parameters exists
194
194
  #
@@ -359,7 +359,7 @@ module ContextDev
359
359
  #
360
360
  # @param headers [Hash{Symbol=>String}] Optional outbound HTTP headers forwarded only to the target URL, sent as deep-ob
361
361
  #
362
- # @param include_frames [Boolean, Symbol, ContextDev::Models::WebWebScrapeHTMLParams::IncludeFrames] When true, iframes are rendered inline into the returned HTML.
362
+ # @param include_frames [Boolean] When true, iframes are rendered inline into the returned HTML.
363
363
  #
364
364
  # @param include_selectors [Array<String>, nil] CSS selectors. When provided, only matching subtrees (and their descendants) are
365
365
  #
@@ -367,13 +367,13 @@ module ContextDev
367
367
  #
368
368
  # @param pdf [ContextDev::Models::WebWebScrapeHTMLParams::Pdf] PDF parsing controls. Use start/end to limit text extraction and embedded-image
369
369
  #
370
- # @param settle_animations [Boolean, Symbol, ContextDev::Models::WebWebScrapeHTMLParams::SettleAnimations] When true, waits briefly for CSS and transition animations to settle before extr
370
+ # @param settle_animations [Boolean] When true, waits briefly for CSS and transition animations to settle before extr
371
371
  #
372
372
  # @param tags [Array<String>] Optional comma-separated caller-defined tags for tracking this request. Tags are
373
373
  #
374
374
  # @param timeout_ms [Integer] Optional timeout in milliseconds for the request. If the request takes longer th
375
375
  #
376
- # @param use_main_content_only [Boolean, Symbol, ContextDev::Models::WebWebScrapeHTMLParams::UseMainContentOnly] When true, return only the page's main content in the HTML response, excluding h
376
+ # @param use_main_content_only [Boolean] When true, return only the page's main content in the HTML response, excluding h
377
377
  #
378
378
  # @param wait_for_ms [Integer, nil] Optional browser wait time in milliseconds after initial page load. Min: 0. Max:
379
379
  #
@@ -420,7 +420,7 @@ module ContextDev
420
420
  #
421
421
  # @param actions [Array<ContextDev::Models::WebWebScrapeImagesParams::Action::Wait, ContextDev::Models::WebWebScrapeImagesParams::Action::Perform>, nil] Optional browser actions executed in array order after the page loads and before
422
422
  #
423
- # @param dedupe [Boolean, Symbol, ContextDev::Models::WebWebScrapeImagesParams::Dedupe] When true, visually duplicate images are removed: every image is loaded and perc
423
+ # @param dedupe [Boolean] When true, visually duplicate images are removed: every image is loaded and perc
424
424
  #
425
425
  # @param enrichment [ContextDev::Models::WebWebScrapeImagesParams::Enrichment, nil] Optional per-image processing, sent as deep-object query params such as enrichme
426
426
  #
@@ -462,20 +462,33 @@ module ContextDev
462
462
  # responses from a recognized API key; use error_code to distinguish stable
463
463
  # failure categories.
464
464
  #
465
+ # ### YouTube
466
+ #
467
+ # YouTube URLs return the video or channel itself rather than the surrounding
468
+ # player and navigation chrome. A URL addressing a single video (`/watch`,
469
+ # `youtu.be`, `/shorts`, `/embed`, `/live`) returns its title, channel, duration,
470
+ # view count, keywords, full description, and the transcript when the video has
471
+ # captions that can be retrieved; videos without captions return everything except
472
+ # the transcript. A channel URL (`/channel/UC…`, `/@handle`, `/c/…`, `/user/…`)
473
+ # returns its name, handle, subscriber count, video count, and full description.
474
+ # When `includeImages=true`, video responses also include the thumbnail and
475
+ # channel responses include the avatar. Costs the same as any other scrape.
476
+ #
465
477
  # ### Billing & errors
466
478
  #
467
- # | HTTP status | Billed? | Meaning |
468
- # | ----------- | ----------------------------------------- | ---------------------------------------------------------------------------------------- |
469
- # | 200 | Yes — 1 credit, or 2 credits with actions | Successful scrape, including a zero-length result when includeSelectors matched nothing |
470
- # | 400 | No | Invalid input, skipped PDF, or the page could not be scraped |
471
- # | 401 / 403 | No | Invalid/disabled key, insufficient permissions, or credits exhausted; inspect error_code |
472
- # | 404 | No | Target page returned or fingerprinted as not found |
473
- # | 408 | No | Request timed out |
474
- # | 415 | No | Unsupported content type |
475
- # | 429 | No | Per-minute rate limit exceeded; honor Retry-After |
476
- # | 500 | No | Internal error |
479
+ # | HTTP status | Billed? | Meaning |
480
+ # | ----------- | ----------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
481
+ # | 200 | Yes — 1 credit, or 2 credits with actions | Successful scrape, including a zero-length result when includeSelectors matched nothing |
482
+ # | 400 | No | Invalid input, skipped PDF, or the page could not be scraped. error_code WEBSITE_BLOCKED specifically means the site answered with an anti-bot challenge, CAPTCHA wall, or login shell instead of the page (even when the site returned HTTP 200) — retrying later or from another country sometimes succeeds |
483
+ # | 401 / 403 | No | Invalid/disabled key, insufficient permissions, or credits exhausted; inspect error_code |
484
+ # | 404 | No | Target page returned or fingerprinted as not found |
485
+ # | 408 | No | Request timed out |
486
+ # | 413 | No | Target content exceeds the maximum supported size (20 MB) |
487
+ # | 415 | No | Unsupported content type |
488
+ # | 429 | No | Per-minute rate limit exceeded; honor Retry-After |
489
+ # | 500 | No | Internal error |
477
490
  #
478
- # @overload web_scrape_md(url:, actions: nil, country: nil, exclude_selectors: nil, headers: nil, include_frames: nil, include_images: nil, include_links: nil, include_selectors: nil, max_age_ms: nil, pdf: nil, settle_animations: nil, shorten_base64_images: nil, tags: nil, timeout_ms: nil, use_main_content_only: nil, wait_for_ms: nil, zdr: nil, request_options: {})
491
+ # @overload web_scrape_md(url:, actions: nil, country: nil, exclude_selectors: nil, headers: nil, include_frames: nil, include_html: nil, include_images: nil, include_links: nil, include_selectors: nil, max_age_ms: nil, pdf: nil, settle_animations: nil, shorten_base64_images: nil, tags: nil, timeout_ms: nil, use_main_content_only: nil, wait_for_ms: nil, zdr: nil, request_options: {})
479
492
  #
480
493
  # @param url [String] Full URL to scrape into LLM usable Markdown (must include http:// or https:// pr
481
494
  #
@@ -487,11 +500,13 @@ module ContextDev
487
500
  #
488
501
  # @param headers [Hash{Symbol=>String}] Optional outbound HTTP headers forwarded only to the target URL, sent as deep-ob
489
502
  #
490
- # @param include_frames [Boolean, Symbol, ContextDev::Models::WebWebScrapeMdParams::IncludeFrames] When true, the contents of iframes are rendered to Markdown.
503
+ # @param include_frames [Boolean] When true, the contents of iframes are rendered to Markdown.
491
504
  #
492
- # @param include_images [Boolean, Symbol, ContextDev::Models::WebWebScrapeMdParams::IncludeImages] Include image references in Markdown output
505
+ # @param include_html [Boolean] When true, the response also includes an `html` field with the page HTML the Mar
493
506
  #
494
- # @param include_links [Boolean, Symbol, ContextDev::Models::WebWebScrapeMdParams::IncludeLinks] Preserve hyperlinks in Markdown output
507
+ # @param include_images [Boolean] Include image references in Markdown output
508
+ #
509
+ # @param include_links [Boolean] Preserve hyperlinks in Markdown output
495
510
  #
496
511
  # @param include_selectors [Array<String>, nil] CSS selectors. When provided, only matching HTML subtrees (and their descendants
497
512
  #
@@ -499,15 +514,15 @@ module ContextDev
499
514
  #
500
515
  # @param pdf [ContextDev::Models::WebWebScrapeMdParams::Pdf] PDF parsing controls. Use start/end to limit text extraction and embedded-image
501
516
  #
502
- # @param settle_animations [Boolean, Symbol, ContextDev::Models::WebWebScrapeMdParams::SettleAnimations] When true, waits briefly for CSS and transition animations to settle before conv
517
+ # @param settle_animations [Boolean] When true, waits briefly for CSS and transition animations to settle before conv
503
518
  #
504
- # @param shorten_base64_images [Boolean, Symbol, ContextDev::Models::WebWebScrapeMdParams::ShortenBase64Images] Shorten base64-encoded image data in the Markdown output
519
+ # @param shorten_base64_images [Boolean] Shorten base64-encoded image data in the Markdown output
505
520
  #
506
521
  # @param tags [Array<String>] Optional comma-separated caller-defined tags for tracking this request. Tags are
507
522
  #
508
523
  # @param timeout_ms [Integer] Optional timeout in milliseconds for the request. If the request takes longer th
509
524
  #
510
- # @param use_main_content_only [Boolean, Symbol, ContextDev::Models::WebWebScrapeMdParams::UseMainContentOnly] Extract only the main content of the page, excluding headers, footers, sidebars,
525
+ # @param use_main_content_only [Boolean] Extract only the main content of the page, excluding headers, footers, sidebars,
511
526
  #
512
527
  # @param wait_for_ms [Integer, nil] Optional browser wait time in milliseconds after initial page load before conver
513
528
  #
@@ -527,6 +542,7 @@ module ContextDev
527
542
  query: query.transform_keys(
528
543
  exclude_selectors: "excludeSelectors",
529
544
  include_frames: "includeFrames",
545
+ include_html: "includeHTML",
530
546
  include_images: "includeImages",
531
547
  include_links: "includeLinks",
532
548
  include_selectors: "includeSelectors",
@@ -545,9 +561,13 @@ module ContextDev
545
561
  # Some parameter documentations has been truncated, see
546
562
  # {ContextDev::Models::WebWebScrapeSitemapParams} for more details.
547
563
  #
548
- # Crawl an entire website's sitemap and return all discovered page URLs.
564
+ # Crawl an entire website's sitemap and return all discovered page URLs. Pass
565
+ # `search` to have the crawled sitemap filtered down to the pages about a phrase
566
+ # (for example `pricing and plans` or `api authentication docs`), most relevant
567
+ # first — a searched crawl scans the whole sitemap and costs 2 credits instead
568
+ # of 1.
549
569
  #
550
- # @overload web_scrape_sitemap(domain:, headers: nil, max_links: nil, sitemap_url: nil, tags: nil, timeout_ms: nil, url_regex: nil, zdr: nil, request_options: {})
570
+ # @overload web_scrape_sitemap(domain:, headers: nil, max_links: nil, search: nil, sitemap_url: nil, tags: nil, timeout_ms: nil, url_regex: nil, zdr: nil, request_options: {})
551
571
  #
552
572
  # @param domain [String] Domain to build a sitemap for
553
573
  #
@@ -555,6 +575,8 @@ module ContextDev
555
575
  #
556
576
  # @param max_links [Integer] Maximum number of links to return from the sitemap crawl. Defaults to 10,000. Mi
557
577
  #
578
+ # @param search [String] Optional search phrase. When provided, the crawled sitemap is filtered to the pa
579
+ #
558
580
  # @param sitemap_url [String] Optional explicit sitemap URL. When provided, exactly this sitemap is crawled in
559
581
  #
560
582
  # @param tags [Array<String>] Optional comma-separated caller-defined tags for tracking this request. Tags are
@@ -1,5 +1,5 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  module ContextDev
4
- VERSION = "2.8.0"
4
+ VERSION = "2.10.0"
5
5
  end
data/lib/context_dev.rb CHANGED
@@ -58,6 +58,8 @@ require_relative "context_dev/models/ai_extract_products_params"
58
58
  require_relative "context_dev/models/ai_extract_products_response"
59
59
  require_relative "context_dev/models/batch_cancel_params"
60
60
  require_relative "context_dev/models/batch_cancel_response"
61
+ require_relative "context_dev/models/batch_delete_params"
62
+ require_relative "context_dev/models/batch_delete_response"
61
63
  require_relative "context_dev/models/batch_get_results_params"
62
64
  require_relative "context_dev/models/batch_get_results_response"
63
65
  require_relative "context_dev/models/batch_list_params"
@@ -105,9 +107,13 @@ require_relative "context_dev/models/monitor_run_params"
105
107
  require_relative "context_dev/models/monitor_run_response"
106
108
  require_relative "context_dev/models/monitor_update_params"
107
109
  require_relative "context_dev/models/monitor_update_response"
110
+ require_relative "context_dev/models/news_search_params"
111
+ require_relative "context_dev/models/news_search_response"
108
112
  require_relative "context_dev/models/page_error_count"
109
113
  require_relative "context_dev/models/parse_handle_params"
110
114
  require_relative "context_dev/models/parse_handle_response"
115
+ require_relative "context_dev/models/person_enrich_params"
116
+ require_relative "context_dev/models/person_enrich_response"
111
117
  require_relative "context_dev/models/utility_prefetch_params"
112
118
  require_relative "context_dev/models/utility_prefetch_response"
113
119
  require_relative "context_dev/models/web_extract_competitors_params"
@@ -139,6 +145,8 @@ require_relative "context_dev/resources/batch"
139
145
  require_relative "context_dev/resources/brand"
140
146
  require_relative "context_dev/resources/industry"
141
147
  require_relative "context_dev/resources/monitors"
148
+ require_relative "context_dev/resources/news"
142
149
  require_relative "context_dev/resources/parse"
150
+ require_relative "context_dev/resources/people"
143
151
  require_relative "context_dev/resources/utility"
144
152
  require_relative "context_dev/resources/web"
@@ -38,9 +38,17 @@ module ContextDev
38
38
  sig { returns(ContextDev::Resources::Monitors) }
39
39
  attr_reader :monitors
40
40
 
41
+ # Scrape many pages or crawl a site asynchronously.
41
42
  sig { returns(ContextDev::Resources::Batch) }
42
43
  attr_reader :batch
43
44
 
45
+ sig { returns(ContextDev::Resources::People) }
46
+ attr_reader :people
47
+
48
+ # Search live first-party RSS and free historical news data by company identity.
49
+ sig { returns(ContextDev::Resources::News) }
50
+ attr_reader :news
51
+
44
52
  # @api private
45
53
  sig { override.returns(T::Hash[String, String]) }
46
54
  private def auth_headers
@@ -0,0 +1,40 @@
1
+ # typed: strong
2
+
3
+ module ContextDev
4
+ module Models
5
+ class BatchDeleteParams < ContextDev::Internal::Type::BaseModel
6
+ extend ContextDev::Internal::Type::RequestParameters::Converter
7
+ include ContextDev::Internal::Type::RequestParameters
8
+
9
+ OrHash =
10
+ T.type_alias do
11
+ T.any(ContextDev::BatchDeleteParams, ContextDev::Internal::AnyHash)
12
+ end
13
+
14
+ # ID of the batch to retrieve or cancel.
15
+ sig { returns(String) }
16
+ attr_accessor :batch_id
17
+
18
+ sig do
19
+ params(
20
+ batch_id: String,
21
+ request_options: ContextDev::RequestOptions::OrHash
22
+ ).returns(T.attached_class)
23
+ end
24
+ def self.new(
25
+ # ID of the batch to retrieve or cancel.
26
+ batch_id:,
27
+ request_options: {}
28
+ )
29
+ end
30
+
31
+ sig do
32
+ override.returns(
33
+ { batch_id: String, request_options: ContextDev::RequestOptions }
34
+ )
35
+ end
36
+ def to_hash
37
+ end
38
+ end
39
+ end
40
+ end