context.dev 2.8.0 → 2.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +28 -0
- data/README.md +1 -1
- data/lib/context_dev/client.rb +10 -0
- data/lib/context_dev/models/batch_delete_params.rb +22 -0
- data/lib/context_dev/models/batch_delete_response.rb +60 -0
- data/lib/context_dev/models/batch_get_results_response.rb +46 -4
- data/lib/context_dev/models/batch_list_response.rb +13 -4
- data/lib/context_dev/models/batch_retrieve_response.rb +13 -4
- data/lib/context_dev/models/batch_submit_params.rb +2080 -26
- data/lib/context_dev/models/batch_submit_response.rb +125 -531
- data/lib/context_dev/models/brand_retrieve_response.rb +50 -1
- data/lib/context_dev/models/brand_search_params.rb +41 -3
- data/lib/context_dev/models/brand_search_response.rb +3 -2
- data/lib/context_dev/models/crawl_controls.rb +21 -15
- data/lib/context_dev/models/news_search_params.rb +467 -0
- data/lib/context_dev/models/news_search_response.rb +238 -0
- data/lib/context_dev/models/parse_handle_params.rb +20 -147
- data/lib/context_dev/models/person_enrich_params.rb +176 -0
- data/lib/context_dev/models/person_enrich_response.rb +641 -0
- data/lib/context_dev/models/utility_prefetch_params.rb +19 -16
- data/lib/context_dev/models/utility_prefetch_response.rb +4 -5
- data/lib/context_dev/models/web_screenshot_params.rb +3 -30
- data/lib/context_dev/models/web_search_response.rb +1 -0
- data/lib/context_dev/models/web_web_crawl_md_params.rb +5 -4
- data/lib/context_dev/models/web_web_crawl_md_response.rb +30 -1
- data/lib/context_dev/models/web_web_scrape_html_params.rb +20 -156
- data/lib/context_dev/models/web_web_scrape_html_response.rb +30 -1
- data/lib/context_dev/models/web_web_scrape_images_params.rb +12 -124
- data/lib/context_dev/models/web_web_scrape_md_params.rb +40 -241
- data/lib/context_dev/models/web_web_scrape_md_response.rb +41 -2
- data/lib/context_dev/models/web_web_scrape_sitemap_params.rb +11 -1
- data/lib/context_dev/models/web_web_scrape_sitemap_response.rb +3 -2
- data/lib/context_dev/models.rb +6 -0
- data/lib/context_dev/resources/batch.rb +33 -7
- data/lib/context_dev/resources/brand.rb +10 -10
- data/lib/context_dev/resources/news.rb +51 -0
- data/lib/context_dev/resources/parse.rb +5 -5
- data/lib/context_dev/resources/people.rb +56 -0
- data/lib/context_dev/resources/utility.rb +7 -6
- data/lib/context_dev/resources/web.rb +46 -24
- data/lib/context_dev/version.rb +1 -1
- data/lib/context_dev.rb +8 -0
- data/rbi/context_dev/client.rbi +8 -0
- data/rbi/context_dev/models/batch_delete_params.rbi +40 -0
- data/rbi/context_dev/models/batch_delete_response.rbi +116 -0
- data/rbi/context_dev/models/batch_get_results_response.rbi +85 -3
- data/rbi/context_dev/models/batch_list_response.rbi +24 -8
- data/rbi/context_dev/models/batch_retrieve_response.rbi +24 -8
- data/rbi/context_dev/models/batch_submit_params.rbi +6421 -44
- data/rbi/context_dev/models/batch_submit_response.rbi +184 -1151
- data/rbi/context_dev/models/brand_retrieve_response.rbi +152 -0
- data/rbi/context_dev/models/brand_search_params.rbi +71 -2
- data/rbi/context_dev/models/brand_search_response.rbi +4 -2
- data/rbi/context_dev/models/crawl_controls.rbi +22 -28
- data/rbi/context_dev/models/news_search_params.rbi +1294 -0
- data/rbi/context_dev/models/news_search_response.rbi +423 -0
- data/rbi/context_dev/models/parse_handle_params.rbi +30 -323
- data/rbi/context_dev/models/person_enrich_params.rbi +332 -0
- data/rbi/context_dev/models/person_enrich_response.rbi +1295 -0
- data/rbi/context_dev/models/utility_prefetch_params.rbi +24 -16
- data/rbi/context_dev/models/utility_prefetch_response.rbi +8 -6
- data/rbi/context_dev/models/web_screenshot_params.rbi +4 -71
- data/rbi/context_dev/models/web_search_response.rbi +5 -0
- data/rbi/context_dev/models/web_web_crawl_md_params.rbi +8 -6
- data/rbi/context_dev/models/web_web_crawl_md_response.rbi +67 -0
- data/rbi/context_dev/models/web_web_scrape_html_params.rbi +30 -363
- data/rbi/context_dev/models/web_web_scrape_html_response.rbi +65 -0
- data/rbi/context_dev/models/web_web_scrape_images_params.rbi +16 -287
- data/rbi/context_dev/models/web_web_scrape_md_params.rbi +57 -558
- data/rbi/context_dev/models/web_web_scrape_md_response.rbi +80 -0
- data/rbi/context_dev/models/web_web_scrape_sitemap_params.rbi +15 -0
- data/rbi/context_dev/models/web_web_scrape_sitemap_response.rbi +4 -2
- data/rbi/context_dev/models.rbi +6 -0
- data/rbi/context_dev/resources/batch.rbi +32 -10
- data/rbi/context_dev/resources/brand.rbi +14 -8
- data/rbi/context_dev/resources/news.rbi +46 -0
- data/rbi/context_dev/resources/parse.rbi +10 -26
- data/rbi/context_dev/resources/people.rbi +47 -0
- data/rbi/context_dev/resources/utility.rbi +8 -6
- data/rbi/context_dev/resources/web.rbi +49 -66
- data/sig/context_dev/client.rbs +4 -0
- data/sig/context_dev/models/batch_delete_params.rbs +23 -0
- data/sig/context_dev/models/batch_delete_response.rbs +57 -0
- data/sig/context_dev/models/batch_get_results_response.rbs +30 -2
- data/sig/context_dev/models/batch_list_response.rbs +16 -2
- data/sig/context_dev/models/batch_retrieve_response.rbs +16 -2
- data/sig/context_dev/models/batch_submit_params.rbs +2666 -15
- data/sig/context_dev/models/batch_submit_response.rbs +78 -466
- data/sig/context_dev/models/brand_retrieve_response.rbs +62 -0
- data/sig/context_dev/models/brand_search_params.rbs +38 -1
- data/sig/context_dev/models/crawl_controls.rbs +16 -16
- data/sig/context_dev/models/news_search_params.rbs +532 -0
- data/sig/context_dev/models/news_search_response.rbs +206 -0
- data/sig/context_dev/models/parse_handle_params.rbs +25 -90
- data/sig/context_dev/models/person_enrich_params.rbs +199 -0
- data/sig/context_dev/models/person_enrich_response.rbs +638 -0
- data/sig/context_dev/models/utility_prefetch_params.rbs +2 -1
- data/sig/context_dev/models/utility_prefetch_response.rbs +2 -1
- data/sig/context_dev/models/web_screenshot_params.rbs +5 -18
- data/sig/context_dev/models/web_search_response.rbs +2 -0
- data/sig/context_dev/models/web_web_crawl_md_response.rbs +21 -0
- data/sig/context_dev/models/web_web_scrape_html_params.rbs +24 -94
- data/sig/context_dev/models/web_web_scrape_html_response.rbs +21 -0
- data/sig/context_dev/models/web_web_scrape_images_params.rbs +20 -72
- data/sig/context_dev/models/web_web_scrape_md_params.rbs +46 -148
- data/sig/context_dev/models/web_web_scrape_md_response.rbs +28 -0
- data/sig/context_dev/models/web_web_scrape_sitemap_params.rbs +7 -0
- data/sig/context_dev/models.rbs +6 -0
- data/sig/context_dev/resources/batch.rbs +8 -2
- data/sig/context_dev/resources/brand.rbs +3 -0
- data/sig/context_dev/resources/news.rbs +17 -0
- data/sig/context_dev/resources/parse.rbs +5 -5
- data/sig/context_dev/resources/people.rbs +19 -0
- data/sig/context_dev/resources/web.rbs +13 -11
- metadata +26 -2
|
@@ -23,7 +23,8 @@ module ContextDev
|
|
|
23
23
|
required :success, enum: -> { ContextDev::Models::WebWebScrapeSitemapResponse::Success }
|
|
24
24
|
|
|
25
25
|
# @!attribute urls
|
|
26
|
-
#
|
|
26
|
+
# Discovered page URLs from the sitemap, up to `maxLinks`. When `search` is set
|
|
27
|
+
# these are only the matching pages, most relevant first.
|
|
27
28
|
#
|
|
28
29
|
# @return [Array<String>]
|
|
29
30
|
required :urls, ContextDev::Internal::Type::ArrayOf[String]
|
|
@@ -45,7 +46,7 @@ module ContextDev
|
|
|
45
46
|
#
|
|
46
47
|
# @param success [Boolean, ContextDev::Models::WebWebScrapeSitemapResponse::Success] Indicates success
|
|
47
48
|
#
|
|
48
|
-
# @param urls [Array<String>]
|
|
49
|
+
# @param urls [Array<String>] Discovered page URLs from the sitemap, up to `maxLinks`. When `search` is set th
|
|
49
50
|
#
|
|
50
51
|
# @param key_metadata [ContextDev::Models::WebWebScrapeSitemapResponse::KeyMetadata] Metadata about the API key used for the request. Included in every response when
|
|
51
52
|
|
data/lib/context_dev/models.rb
CHANGED
|
@@ -45,6 +45,8 @@ module ContextDev
|
|
|
45
45
|
|
|
46
46
|
BatchCancelParams = ContextDev::Models::BatchCancelParams
|
|
47
47
|
|
|
48
|
+
BatchDeleteParams = ContextDev::Models::BatchDeleteParams
|
|
49
|
+
|
|
48
50
|
BatchGetResultsParams = ContextDev::Models::BatchGetResultsParams
|
|
49
51
|
|
|
50
52
|
BatchListParams = ContextDev::Models::BatchListParams
|
|
@@ -95,10 +97,14 @@ module ContextDev
|
|
|
95
97
|
|
|
96
98
|
MonitorUpdateParams = ContextDev::Models::MonitorUpdateParams
|
|
97
99
|
|
|
100
|
+
NewsSearchParams = ContextDev::Models::NewsSearchParams
|
|
101
|
+
|
|
98
102
|
PageErrorCount = ContextDev::Models::PageErrorCount
|
|
99
103
|
|
|
100
104
|
ParseHandleParams = ContextDev::Models::ParseHandleParams
|
|
101
105
|
|
|
106
|
+
PersonEnrichParams = ContextDev::Models::PersonEnrichParams
|
|
107
|
+
|
|
102
108
|
UtilityPrefetchParams = ContextDev::Models::UtilityPrefetchParams
|
|
103
109
|
|
|
104
110
|
WebExtractCompetitorsParams = ContextDev::Models::WebExtractCompetitorsParams
|
|
@@ -2,6 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
module ContextDev
|
|
4
4
|
module Resources
|
|
5
|
+
# Scrape many pages or crawl a site asynchronously.
|
|
5
6
|
class Batch
|
|
6
7
|
# Check progress, and get download links once the batch finishes.
|
|
7
8
|
#
|
|
@@ -60,6 +61,27 @@ module ContextDev
|
|
|
60
61
|
)
|
|
61
62
|
end
|
|
62
63
|
|
|
64
|
+
# Permanently delete a finished batch and its stored results. Active batches must
|
|
65
|
+
# settle first.
|
|
66
|
+
#
|
|
67
|
+
# @overload delete(batch_id, request_options: {})
|
|
68
|
+
#
|
|
69
|
+
# @param batch_id [String] ID of the batch to retrieve or cancel.
|
|
70
|
+
#
|
|
71
|
+
# @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
|
|
72
|
+
#
|
|
73
|
+
# @return [ContextDev::Models::BatchDeleteResponse]
|
|
74
|
+
#
|
|
75
|
+
# @see ContextDev::Models::BatchDeleteParams
|
|
76
|
+
def delete(batch_id, params = {})
|
|
77
|
+
@client.request(
|
|
78
|
+
method: :delete,
|
|
79
|
+
path: ["batch/%1$s", batch_id],
|
|
80
|
+
model: ContextDev::Models::BatchDeleteResponse,
|
|
81
|
+
options: params[:request_options]
|
|
82
|
+
)
|
|
83
|
+
end
|
|
84
|
+
|
|
63
85
|
# Stop a batch from starting new pages. In-progress pages finish, and unused
|
|
64
86
|
# credits are refunded.
|
|
65
87
|
#
|
|
@@ -115,15 +137,17 @@ module ContextDev
|
|
|
115
137
|
# Some parameter documentations has been truncated, see
|
|
116
138
|
# {ContextDev::Models::BatchSubmitParams} for more details.
|
|
117
139
|
#
|
|
118
|
-
#
|
|
140
|
+
# Scrape 25K URLs or crawl large websites asynchronously.
|
|
141
|
+
#
|
|
142
|
+
# @overload submit(input:, tags: nil, webhook_url: nil, idempotency_key: nil, request_options: {})
|
|
119
143
|
#
|
|
120
|
-
# @
|
|
144
|
+
# @param input [ContextDev::Models::BatchSubmitParams::Input::Scrape, ContextDev::Models::BatchSubmitParams::Input::Crawl] Body param: Choose a URL list or a site crawl.
|
|
121
145
|
#
|
|
122
|
-
# @param
|
|
146
|
+
# @param tags [Array<String>] Body param: Tags stored on the batch. Filter the batch list by them later.
|
|
123
147
|
#
|
|
124
|
-
# @param
|
|
148
|
+
# @param webhook_url [String] Body param: URL notified when the batch finishes.
|
|
125
149
|
#
|
|
126
|
-
# @param
|
|
150
|
+
# @param idempotency_key [String] Header param: Any string unique to this submission. Retries with the same key re
|
|
127
151
|
#
|
|
128
152
|
# @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
|
|
129
153
|
#
|
|
@@ -132,10 +156,12 @@ module ContextDev
|
|
|
132
156
|
# @see ContextDev::Models::BatchSubmitParams
|
|
133
157
|
def submit(params)
|
|
134
158
|
parsed, options = ContextDev::BatchSubmitParams.dump_request(params)
|
|
159
|
+
header_params = {idempotency_key: "idempotency-key"}
|
|
135
160
|
@client.request(
|
|
136
161
|
method: :post,
|
|
137
|
-
path: "
|
|
138
|
-
|
|
162
|
+
path: "batch/submit",
|
|
163
|
+
headers: parsed.slice(*header_params.keys).transform_keys(header_params),
|
|
164
|
+
body: parsed.except(*header_params.keys),
|
|
139
165
|
model: ContextDev::Models::BatchSubmitResponse,
|
|
140
166
|
options: options
|
|
141
167
|
)
|
|
@@ -68,20 +68,20 @@ module ContextDev
|
|
|
68
68
|
# Some parameter documentations has been truncated, see
|
|
69
69
|
# {ContextDev::Models::BrandSearchParams} for more details.
|
|
70
70
|
#
|
|
71
|
-
# Search brands by name or domain
|
|
72
|
-
# (domain, name, logo), most popular first: by Tranco rank, then market cap for
|
|
73
|
-
# brands outside the Tranco list, with text relevance breaking ties. Matching is
|
|
74
|
-
# prefix-based with no typo tolerance, so it is suited to autocomplete. Only
|
|
75
|
-
# brands already in the Context.dev index are returned — use /brand/retrieve to
|
|
76
|
-
# fetch (and index) a specific domain. Free on Pro and Scale plans; costs 1 credit
|
|
77
|
-
# per request on the Free and Starter plans.
|
|
71
|
+
# Search indexed brands by name or domain
|
|
78
72
|
#
|
|
79
|
-
# @overload search(query:, tags: nil, request_options: {})
|
|
73
|
+
# @overload search(query:, autocomplete: nil, query_by: nil, tags: nil, typo_tolerance: nil, request_options: {})
|
|
80
74
|
#
|
|
81
|
-
# @param query [String] Search term, matched against
|
|
75
|
+
# @param query [String] Search term, matched against the fields selected by queryBy (e.g. 'nike', 'nike.
|
|
76
|
+
#
|
|
77
|
+
# @param autocomplete [Boolean] Whether the search term matches by prefix, so partial words match as they are ty
|
|
78
|
+
#
|
|
79
|
+
# @param query_by [Array<Symbol, ContextDev::Models::BrandSearchParams::QueryBy>] Fields to match the search term against, as a comma-separated list or repeated p
|
|
82
80
|
#
|
|
83
81
|
# @param tags [Array<String>] Optional comma-separated caller-defined tags for tracking this request. Tags are
|
|
84
82
|
#
|
|
83
|
+
# @param typo_tolerance [Integer] Maximum number of typos tolerated when matching, from 0 to 2. Defaults to 0 (no
|
|
84
|
+
#
|
|
85
85
|
# @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
|
|
86
86
|
#
|
|
87
87
|
# @return [ContextDev::Models::BrandSearchResponse]
|
|
@@ -93,7 +93,7 @@ module ContextDev
|
|
|
93
93
|
@client.request(
|
|
94
94
|
method: :get,
|
|
95
95
|
path: "brand/search",
|
|
96
|
-
query: query,
|
|
96
|
+
query: query.transform_keys(query_by: "queryBy", typo_tolerance: "typoTolerance"),
|
|
97
97
|
model: ContextDev::Models::BrandSearchResponse,
|
|
98
98
|
options: options
|
|
99
99
|
)
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module ContextDev
|
|
4
|
+
module Resources
|
|
5
|
+
# Search live first-party RSS and free historical news data by company identity.
|
|
6
|
+
class News
|
|
7
|
+
# Searches live and historical company news for one company, identified in
|
|
8
|
+
# searchBy by name, domain, ticker (optionally disambiguated by exchange), or
|
|
9
|
+
# ISIN. Results can be filtered by publisher domain, publisher country, article
|
|
10
|
+
# language, article type, and published-at date, and include stable story IDs,
|
|
11
|
+
# source metadata, verified entity relevance, and cursor pagination.
|
|
12
|
+
#
|
|
13
|
+
# @overload search(search_by:, cursor: nil, filter_by: nil, limit: nil, sort_by: nil, tags: nil, request_options: {})
|
|
14
|
+
#
|
|
15
|
+
# @param search_by [ContextDev::Models::NewsSearchParams::SearchBy] What to search for.
|
|
16
|
+
#
|
|
17
|
+
# @param cursor [String, nil] Opaque next_cursor from the previous response, or null for the first page.
|
|
18
|
+
#
|
|
19
|
+
# @param filter_by [ContextDev::Models::NewsSearchParams::FilterBy] Optional result filters.
|
|
20
|
+
#
|
|
21
|
+
# @param limit [Integer] Maximum results to return. Defaults to 10.
|
|
22
|
+
#
|
|
23
|
+
# @param sort_by [ContextDev::Models::NewsSearchParams::SortBy] Result ordering. Defaults to newest.
|
|
24
|
+
#
|
|
25
|
+
# @param tags [Array<String>] Optional tags for tracking usage. Up to 20 tags, each 1 to 50 characters.
|
|
26
|
+
#
|
|
27
|
+
# @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
|
|
28
|
+
#
|
|
29
|
+
# @return [ContextDev::Models::NewsSearchResponse]
|
|
30
|
+
#
|
|
31
|
+
# @see ContextDev::Models::NewsSearchParams
|
|
32
|
+
def search(params)
|
|
33
|
+
parsed, options = ContextDev::NewsSearchParams.dump_request(params)
|
|
34
|
+
@client.request(
|
|
35
|
+
method: :post,
|
|
36
|
+
path: "news/search",
|
|
37
|
+
body: parsed,
|
|
38
|
+
model: ContextDev::Models::NewsSearchResponse,
|
|
39
|
+
options: options
|
|
40
|
+
)
|
|
41
|
+
end
|
|
42
|
+
|
|
43
|
+
# @api private
|
|
44
|
+
#
|
|
45
|
+
# @param client [ContextDev::Client]
|
|
46
|
+
def initialize(client:)
|
|
47
|
+
@client = client
|
|
48
|
+
end
|
|
49
|
+
end
|
|
50
|
+
end
|
|
51
|
+
end
|
|
@@ -17,19 +17,19 @@ module ContextDev
|
|
|
17
17
|
#
|
|
18
18
|
# @param extension [Symbol, ContextDev::Models::ParseHandleParams::Extension] Query param: Optional file extension hint, such as pdf, docx, xlsx, pptx, html,
|
|
19
19
|
#
|
|
20
|
-
# @param include_images [Boolean
|
|
20
|
+
# @param include_images [Boolean] Query param: Include image references in Markdown output
|
|
21
21
|
#
|
|
22
|
-
# @param include_links [Boolean
|
|
22
|
+
# @param include_links [Boolean] Query param: Preserve hyperlinks in Markdown output
|
|
23
23
|
#
|
|
24
|
-
# @param ocr [Boolean
|
|
24
|
+
# @param ocr [Boolean] Query param: When true for PDF inputs, OCR the selected pages that have no usabl
|
|
25
25
|
#
|
|
26
26
|
# @param pdf [ContextDev::Models::ParseHandleParams::Pdf] Query param: PDF page-range options as a JSON object, e.g. {"start": 2, "end": 5
|
|
27
27
|
#
|
|
28
|
-
# @param shorten_base64_images [Boolean
|
|
28
|
+
# @param shorten_base64_images [Boolean] Query param: Shorten base64-encoded image data in the Markdown output
|
|
29
29
|
#
|
|
30
30
|
# @param tags [Array<String>] Query param: Optional comma-separated caller-defined tags for tracking this requ
|
|
31
31
|
#
|
|
32
|
-
# @param use_main_content_only [Boolean
|
|
32
|
+
# @param use_main_content_only [Boolean] Query param: Extract only the main content from HTML-like inputs
|
|
33
33
|
#
|
|
34
34
|
# @param zdr [Symbol, ContextDev::Models::ParseHandleParams::Zdr] Query param: Set to enabled to bypass shared caches and omit request and respons
|
|
35
35
|
#
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module ContextDev
|
|
4
|
+
module Resources
|
|
5
|
+
class People
|
|
6
|
+
# Some parameter documentations has been truncated, see
|
|
7
|
+
# {ContextDev::Models::PersonEnrichParams} for more details.
|
|
8
|
+
#
|
|
9
|
+
# Finds and normalizes the best available person candidate from additive identity
|
|
10
|
+
# clues, then assigns an identity match score from 0 to 100. Available on all paid
|
|
11
|
+
# plans. Successful requests cost 20 credits. Disposable and free email addresses
|
|
12
|
+
# (like gmail.com, yahoo.com) will throw a 422 error.
|
|
13
|
+
#
|
|
14
|
+
# @overload enrich(company: nil, education: nil, email: nil, location: nil, name: nil, social_urls: nil, tags: nil, timeout_ms: nil, request_options: {})
|
|
15
|
+
#
|
|
16
|
+
# @param company [ContextDev::Models::PersonEnrichParams::Company]
|
|
17
|
+
#
|
|
18
|
+
# @param education [Array<ContextDev::Models::PersonEnrichParams::Education>]
|
|
19
|
+
#
|
|
20
|
+
# @param email [String]
|
|
21
|
+
#
|
|
22
|
+
# @param location [ContextDev::Models::PersonEnrichParams::Location]
|
|
23
|
+
#
|
|
24
|
+
# @param name [ContextDev::Models::PersonEnrichParams::Name]
|
|
25
|
+
#
|
|
26
|
+
# @param social_urls [Array<String>]
|
|
27
|
+
#
|
|
28
|
+
# @param tags [Array<String>] Optional tags for tracking usage. Up to 20 tags, each 1 to 50 characters.
|
|
29
|
+
#
|
|
30
|
+
# @param timeout_ms [Integer] Optional timeout in milliseconds for the request. If the request takes longer th
|
|
31
|
+
#
|
|
32
|
+
# @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
|
|
33
|
+
#
|
|
34
|
+
# @return [ContextDev::Models::PersonEnrichResponse]
|
|
35
|
+
#
|
|
36
|
+
# @see ContextDev::Models::PersonEnrichParams
|
|
37
|
+
def enrich(params = {})
|
|
38
|
+
parsed, options = ContextDev::PersonEnrichParams.dump_request(params)
|
|
39
|
+
@client.request(
|
|
40
|
+
method: :post,
|
|
41
|
+
path: "people/enrich",
|
|
42
|
+
body: parsed,
|
|
43
|
+
model: ContextDev::Models::PersonEnrichResponse,
|
|
44
|
+
options: options
|
|
45
|
+
)
|
|
46
|
+
end
|
|
47
|
+
|
|
48
|
+
# @api private
|
|
49
|
+
#
|
|
50
|
+
# @param client [ContextDev::Client]
|
|
51
|
+
def initialize(client:)
|
|
52
|
+
@client = client
|
|
53
|
+
end
|
|
54
|
+
end
|
|
55
|
+
end
|
|
56
|
+
end
|
|
@@ -6,16 +6,17 @@ module ContextDev
|
|
|
6
6
|
# Some parameter documentations has been truncated, see
|
|
7
7
|
# {ContextDev::Models::UtilityPrefetchParams} for more details.
|
|
8
8
|
#
|
|
9
|
-
# Signal that you may fetch
|
|
10
|
-
#
|
|
11
|
-
# one lookup key: a domain,
|
|
12
|
-
#
|
|
9
|
+
# Signal that you may fetch data soon to improve latency. The type field selects
|
|
10
|
+
# what to prefetch ('brand' queues a brand data fetch, 'styleguide' queues a
|
|
11
|
+
# styleguide extraction) and identifier carries exactly one lookup key: a domain,
|
|
12
|
+
# or an email whose domain is extracted and validated (free email providers and
|
|
13
|
+
# disposable email addresses are not allowed).
|
|
13
14
|
#
|
|
14
15
|
# @overload prefetch(identifier:, type:, tags: nil, timeout_ms: nil, request_options: {})
|
|
15
16
|
#
|
|
16
|
-
# @param identifier [ContextDev::Models::UtilityPrefetchParams::Identifier::UtilityPrefetchDomainIdentifier, ContextDev::Models::UtilityPrefetchParams::Identifier::UtilityPrefetchEmailIdentifier] Identifier of the
|
|
17
|
+
# @param identifier [ContextDev::Models::UtilityPrefetchParams::Identifier::UtilityPrefetchDomainIdentifier, ContextDev::Models::UtilityPrefetchParams::Identifier::UtilityPrefetchEmailIdentifier] Identifier of the target to prefetch. Provide exactly one of domain or email.
|
|
17
18
|
#
|
|
18
|
-
# @param type [Symbol, ContextDev::Models::UtilityPrefetchParams::Type] What to prefetch
|
|
19
|
+
# @param type [Symbol, ContextDev::Models::UtilityPrefetchParams::Type] What to prefetch: 'brand' warms the brand data cache, 'styleguide' warms the sty
|
|
19
20
|
#
|
|
20
21
|
# @param tags [Array<String>] Optional tags for tracking usage. Up to 20 tags, each 1 to 50 characters.
|
|
21
22
|
#
|
|
@@ -188,7 +188,7 @@ module ContextDev
|
|
|
188
188
|
#
|
|
189
189
|
# @param full_screenshot [Symbol, ContextDev::Models::WebScreenshotParams::FullScreenshot] Optional parameter to determine screenshot type. If 'true', takes a full page sc
|
|
190
190
|
#
|
|
191
|
-
# @param handle_cookie_popup [Boolean
|
|
191
|
+
# @param handle_cookie_popup [Boolean] Optional parameter to control cookie/consent popup handling. If 'true', we dismi
|
|
192
192
|
#
|
|
193
193
|
# @param max_age_ms [Integer, nil] Return a cached screenshot if a prior screenshot for the same parameters exists
|
|
194
194
|
#
|
|
@@ -359,7 +359,7 @@ module ContextDev
|
|
|
359
359
|
#
|
|
360
360
|
# @param headers [Hash{Symbol=>String}] Optional outbound HTTP headers forwarded only to the target URL, sent as deep-ob
|
|
361
361
|
#
|
|
362
|
-
# @param include_frames [Boolean
|
|
362
|
+
# @param include_frames [Boolean] When true, iframes are rendered inline into the returned HTML.
|
|
363
363
|
#
|
|
364
364
|
# @param include_selectors [Array<String>, nil] CSS selectors. When provided, only matching subtrees (and their descendants) are
|
|
365
365
|
#
|
|
@@ -367,13 +367,13 @@ module ContextDev
|
|
|
367
367
|
#
|
|
368
368
|
# @param pdf [ContextDev::Models::WebWebScrapeHTMLParams::Pdf] PDF parsing controls. Use start/end to limit text extraction and embedded-image
|
|
369
369
|
#
|
|
370
|
-
# @param settle_animations [Boolean
|
|
370
|
+
# @param settle_animations [Boolean] When true, waits briefly for CSS and transition animations to settle before extr
|
|
371
371
|
#
|
|
372
372
|
# @param tags [Array<String>] Optional comma-separated caller-defined tags for tracking this request. Tags are
|
|
373
373
|
#
|
|
374
374
|
# @param timeout_ms [Integer] Optional timeout in milliseconds for the request. If the request takes longer th
|
|
375
375
|
#
|
|
376
|
-
# @param use_main_content_only [Boolean
|
|
376
|
+
# @param use_main_content_only [Boolean] When true, return only the page's main content in the HTML response, excluding h
|
|
377
377
|
#
|
|
378
378
|
# @param wait_for_ms [Integer, nil] Optional browser wait time in milliseconds after initial page load. Min: 0. Max:
|
|
379
379
|
#
|
|
@@ -420,7 +420,7 @@ module ContextDev
|
|
|
420
420
|
#
|
|
421
421
|
# @param actions [Array<ContextDev::Models::WebWebScrapeImagesParams::Action::Wait, ContextDev::Models::WebWebScrapeImagesParams::Action::Perform>, nil] Optional browser actions executed in array order after the page loads and before
|
|
422
422
|
#
|
|
423
|
-
# @param dedupe [Boolean
|
|
423
|
+
# @param dedupe [Boolean] When true, visually duplicate images are removed: every image is loaded and perc
|
|
424
424
|
#
|
|
425
425
|
# @param enrichment [ContextDev::Models::WebWebScrapeImagesParams::Enrichment, nil] Optional per-image processing, sent as deep-object query params such as enrichme
|
|
426
426
|
#
|
|
@@ -462,20 +462,33 @@ module ContextDev
|
|
|
462
462
|
# responses from a recognized API key; use error_code to distinguish stable
|
|
463
463
|
# failure categories.
|
|
464
464
|
#
|
|
465
|
+
# ### YouTube
|
|
466
|
+
#
|
|
467
|
+
# YouTube URLs return the video or channel itself rather than the surrounding
|
|
468
|
+
# player and navigation chrome. A URL addressing a single video (`/watch`,
|
|
469
|
+
# `youtu.be`, `/shorts`, `/embed`, `/live`) returns its title, channel, duration,
|
|
470
|
+
# view count, keywords, full description, and the transcript when the video has
|
|
471
|
+
# captions that can be retrieved; videos without captions return everything except
|
|
472
|
+
# the transcript. A channel URL (`/channel/UC…`, `/@handle`, `/c/…`, `/user/…`)
|
|
473
|
+
# returns its name, handle, subscriber count, video count, and full description.
|
|
474
|
+
# When `includeImages=true`, video responses also include the thumbnail and
|
|
475
|
+
# channel responses include the avatar. Costs the same as any other scrape.
|
|
476
|
+
#
|
|
465
477
|
# ### Billing & errors
|
|
466
478
|
#
|
|
467
|
-
# | HTTP status | Billed? | Meaning
|
|
468
|
-
# | ----------- | ----------------------------------------- |
|
|
469
|
-
# | 200 | Yes — 1 credit, or 2 credits with actions | Successful scrape, including a zero-length result when includeSelectors matched nothing
|
|
470
|
-
# | 400 | No | Invalid input, skipped PDF, or the page could not be scraped
|
|
471
|
-
# | 401 / 403 | No | Invalid/disabled key, insufficient permissions, or credits exhausted; inspect error_code
|
|
472
|
-
# | 404 | No | Target page returned or fingerprinted as not found
|
|
473
|
-
# | 408 | No | Request timed out
|
|
474
|
-
# |
|
|
475
|
-
# |
|
|
476
|
-
# |
|
|
479
|
+
# | HTTP status | Billed? | Meaning |
|
|
480
|
+
# | ----------- | ----------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
|
481
|
+
# | 200 | Yes — 1 credit, or 2 credits with actions | Successful scrape, including a zero-length result when includeSelectors matched nothing |
|
|
482
|
+
# | 400 | No | Invalid input, skipped PDF, or the page could not be scraped. error_code WEBSITE_BLOCKED specifically means the site answered with an anti-bot challenge, CAPTCHA wall, or login shell instead of the page (even when the site returned HTTP 200) — retrying later or from another country sometimes succeeds |
|
|
483
|
+
# | 401 / 403 | No | Invalid/disabled key, insufficient permissions, or credits exhausted; inspect error_code |
|
|
484
|
+
# | 404 | No | Target page returned or fingerprinted as not found |
|
|
485
|
+
# | 408 | No | Request timed out |
|
|
486
|
+
# | 413 | No | Target content exceeds the maximum supported size (20 MB) |
|
|
487
|
+
# | 415 | No | Unsupported content type |
|
|
488
|
+
# | 429 | No | Per-minute rate limit exceeded; honor Retry-After |
|
|
489
|
+
# | 500 | No | Internal error |
|
|
477
490
|
#
|
|
478
|
-
# @overload web_scrape_md(url:, actions: nil, country: nil, exclude_selectors: nil, headers: nil, include_frames: nil, include_images: nil, include_links: nil, include_selectors: nil, max_age_ms: nil, pdf: nil, settle_animations: nil, shorten_base64_images: nil, tags: nil, timeout_ms: nil, use_main_content_only: nil, wait_for_ms: nil, zdr: nil, request_options: {})
|
|
491
|
+
# @overload web_scrape_md(url:, actions: nil, country: nil, exclude_selectors: nil, headers: nil, include_frames: nil, include_html: nil, include_images: nil, include_links: nil, include_selectors: nil, max_age_ms: nil, pdf: nil, settle_animations: nil, shorten_base64_images: nil, tags: nil, timeout_ms: nil, use_main_content_only: nil, wait_for_ms: nil, zdr: nil, request_options: {})
|
|
479
492
|
#
|
|
480
493
|
# @param url [String] Full URL to scrape into LLM usable Markdown (must include http:// or https:// pr
|
|
481
494
|
#
|
|
@@ -487,11 +500,13 @@ module ContextDev
|
|
|
487
500
|
#
|
|
488
501
|
# @param headers [Hash{Symbol=>String}] Optional outbound HTTP headers forwarded only to the target URL, sent as deep-ob
|
|
489
502
|
#
|
|
490
|
-
# @param include_frames [Boolean
|
|
503
|
+
# @param include_frames [Boolean] When true, the contents of iframes are rendered to Markdown.
|
|
491
504
|
#
|
|
492
|
-
# @param
|
|
505
|
+
# @param include_html [Boolean] When true, the response also includes an `html` field with the page HTML the Mar
|
|
493
506
|
#
|
|
494
|
-
# @param
|
|
507
|
+
# @param include_images [Boolean] Include image references in Markdown output
|
|
508
|
+
#
|
|
509
|
+
# @param include_links [Boolean] Preserve hyperlinks in Markdown output
|
|
495
510
|
#
|
|
496
511
|
# @param include_selectors [Array<String>, nil] CSS selectors. When provided, only matching HTML subtrees (and their descendants
|
|
497
512
|
#
|
|
@@ -499,15 +514,15 @@ module ContextDev
|
|
|
499
514
|
#
|
|
500
515
|
# @param pdf [ContextDev::Models::WebWebScrapeMdParams::Pdf] PDF parsing controls. Use start/end to limit text extraction and embedded-image
|
|
501
516
|
#
|
|
502
|
-
# @param settle_animations [Boolean
|
|
517
|
+
# @param settle_animations [Boolean] When true, waits briefly for CSS and transition animations to settle before conv
|
|
503
518
|
#
|
|
504
|
-
# @param shorten_base64_images [Boolean
|
|
519
|
+
# @param shorten_base64_images [Boolean] Shorten base64-encoded image data in the Markdown output
|
|
505
520
|
#
|
|
506
521
|
# @param tags [Array<String>] Optional comma-separated caller-defined tags for tracking this request. Tags are
|
|
507
522
|
#
|
|
508
523
|
# @param timeout_ms [Integer] Optional timeout in milliseconds for the request. If the request takes longer th
|
|
509
524
|
#
|
|
510
|
-
# @param use_main_content_only [Boolean
|
|
525
|
+
# @param use_main_content_only [Boolean] Extract only the main content of the page, excluding headers, footers, sidebars,
|
|
511
526
|
#
|
|
512
527
|
# @param wait_for_ms [Integer, nil] Optional browser wait time in milliseconds after initial page load before conver
|
|
513
528
|
#
|
|
@@ -527,6 +542,7 @@ module ContextDev
|
|
|
527
542
|
query: query.transform_keys(
|
|
528
543
|
exclude_selectors: "excludeSelectors",
|
|
529
544
|
include_frames: "includeFrames",
|
|
545
|
+
include_html: "includeHTML",
|
|
530
546
|
include_images: "includeImages",
|
|
531
547
|
include_links: "includeLinks",
|
|
532
548
|
include_selectors: "includeSelectors",
|
|
@@ -545,9 +561,13 @@ module ContextDev
|
|
|
545
561
|
# Some parameter documentations has been truncated, see
|
|
546
562
|
# {ContextDev::Models::WebWebScrapeSitemapParams} for more details.
|
|
547
563
|
#
|
|
548
|
-
# Crawl an entire website's sitemap and return all discovered page URLs.
|
|
564
|
+
# Crawl an entire website's sitemap and return all discovered page URLs. Pass
|
|
565
|
+
# `search` to have the crawled sitemap filtered down to the pages about a phrase
|
|
566
|
+
# (for example `pricing and plans` or `api authentication docs`), most relevant
|
|
567
|
+
# first — a searched crawl scans the whole sitemap and costs 2 credits instead
|
|
568
|
+
# of 1.
|
|
549
569
|
#
|
|
550
|
-
# @overload web_scrape_sitemap(domain:, headers: nil, max_links: nil, sitemap_url: nil, tags: nil, timeout_ms: nil, url_regex: nil, zdr: nil, request_options: {})
|
|
570
|
+
# @overload web_scrape_sitemap(domain:, headers: nil, max_links: nil, search: nil, sitemap_url: nil, tags: nil, timeout_ms: nil, url_regex: nil, zdr: nil, request_options: {})
|
|
551
571
|
#
|
|
552
572
|
# @param domain [String] Domain to build a sitemap for
|
|
553
573
|
#
|
|
@@ -555,6 +575,8 @@ module ContextDev
|
|
|
555
575
|
#
|
|
556
576
|
# @param max_links [Integer] Maximum number of links to return from the sitemap crawl. Defaults to 10,000. Mi
|
|
557
577
|
#
|
|
578
|
+
# @param search [String] Optional search phrase. When provided, the crawled sitemap is filtered to the pa
|
|
579
|
+
#
|
|
558
580
|
# @param sitemap_url [String] Optional explicit sitemap URL. When provided, exactly this sitemap is crawled in
|
|
559
581
|
#
|
|
560
582
|
# @param tags [Array<String>] Optional comma-separated caller-defined tags for tracking this request. Tags are
|
data/lib/context_dev/version.rb
CHANGED
data/lib/context_dev.rb
CHANGED
|
@@ -58,6 +58,8 @@ require_relative "context_dev/models/ai_extract_products_params"
|
|
|
58
58
|
require_relative "context_dev/models/ai_extract_products_response"
|
|
59
59
|
require_relative "context_dev/models/batch_cancel_params"
|
|
60
60
|
require_relative "context_dev/models/batch_cancel_response"
|
|
61
|
+
require_relative "context_dev/models/batch_delete_params"
|
|
62
|
+
require_relative "context_dev/models/batch_delete_response"
|
|
61
63
|
require_relative "context_dev/models/batch_get_results_params"
|
|
62
64
|
require_relative "context_dev/models/batch_get_results_response"
|
|
63
65
|
require_relative "context_dev/models/batch_list_params"
|
|
@@ -105,9 +107,13 @@ require_relative "context_dev/models/monitor_run_params"
|
|
|
105
107
|
require_relative "context_dev/models/monitor_run_response"
|
|
106
108
|
require_relative "context_dev/models/monitor_update_params"
|
|
107
109
|
require_relative "context_dev/models/monitor_update_response"
|
|
110
|
+
require_relative "context_dev/models/news_search_params"
|
|
111
|
+
require_relative "context_dev/models/news_search_response"
|
|
108
112
|
require_relative "context_dev/models/page_error_count"
|
|
109
113
|
require_relative "context_dev/models/parse_handle_params"
|
|
110
114
|
require_relative "context_dev/models/parse_handle_response"
|
|
115
|
+
require_relative "context_dev/models/person_enrich_params"
|
|
116
|
+
require_relative "context_dev/models/person_enrich_response"
|
|
111
117
|
require_relative "context_dev/models/utility_prefetch_params"
|
|
112
118
|
require_relative "context_dev/models/utility_prefetch_response"
|
|
113
119
|
require_relative "context_dev/models/web_extract_competitors_params"
|
|
@@ -139,6 +145,8 @@ require_relative "context_dev/resources/batch"
|
|
|
139
145
|
require_relative "context_dev/resources/brand"
|
|
140
146
|
require_relative "context_dev/resources/industry"
|
|
141
147
|
require_relative "context_dev/resources/monitors"
|
|
148
|
+
require_relative "context_dev/resources/news"
|
|
142
149
|
require_relative "context_dev/resources/parse"
|
|
150
|
+
require_relative "context_dev/resources/people"
|
|
143
151
|
require_relative "context_dev/resources/utility"
|
|
144
152
|
require_relative "context_dev/resources/web"
|
data/rbi/context_dev/client.rbi
CHANGED
|
@@ -38,9 +38,17 @@ module ContextDev
|
|
|
38
38
|
sig { returns(ContextDev::Resources::Monitors) }
|
|
39
39
|
attr_reader :monitors
|
|
40
40
|
|
|
41
|
+
# Scrape many pages or crawl a site asynchronously.
|
|
41
42
|
sig { returns(ContextDev::Resources::Batch) }
|
|
42
43
|
attr_reader :batch
|
|
43
44
|
|
|
45
|
+
sig { returns(ContextDev::Resources::People) }
|
|
46
|
+
attr_reader :people
|
|
47
|
+
|
|
48
|
+
# Search live first-party RSS and free historical news data by company identity.
|
|
49
|
+
sig { returns(ContextDev::Resources::News) }
|
|
50
|
+
attr_reader :news
|
|
51
|
+
|
|
44
52
|
# @api private
|
|
45
53
|
sig { override.returns(T::Hash[String, String]) }
|
|
46
54
|
private def auth_headers
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
# typed: strong
|
|
2
|
+
|
|
3
|
+
module ContextDev
|
|
4
|
+
module Models
|
|
5
|
+
class BatchDeleteParams < ContextDev::Internal::Type::BaseModel
|
|
6
|
+
extend ContextDev::Internal::Type::RequestParameters::Converter
|
|
7
|
+
include ContextDev::Internal::Type::RequestParameters
|
|
8
|
+
|
|
9
|
+
OrHash =
|
|
10
|
+
T.type_alias do
|
|
11
|
+
T.any(ContextDev::BatchDeleteParams, ContextDev::Internal::AnyHash)
|
|
12
|
+
end
|
|
13
|
+
|
|
14
|
+
# ID of the batch to retrieve or cancel.
|
|
15
|
+
sig { returns(String) }
|
|
16
|
+
attr_accessor :batch_id
|
|
17
|
+
|
|
18
|
+
sig do
|
|
19
|
+
params(
|
|
20
|
+
batch_id: String,
|
|
21
|
+
request_options: ContextDev::RequestOptions::OrHash
|
|
22
|
+
).returns(T.attached_class)
|
|
23
|
+
end
|
|
24
|
+
def self.new(
|
|
25
|
+
# ID of the batch to retrieve or cancel.
|
|
26
|
+
batch_id:,
|
|
27
|
+
request_options: {}
|
|
28
|
+
)
|
|
29
|
+
end
|
|
30
|
+
|
|
31
|
+
sig do
|
|
32
|
+
override.returns(
|
|
33
|
+
{ batch_id: String, request_options: ContextDev::RequestOptions }
|
|
34
|
+
)
|
|
35
|
+
end
|
|
36
|
+
def to_hash
|
|
37
|
+
end
|
|
38
|
+
end
|
|
39
|
+
end
|
|
40
|
+
end
|