context.dev 2.9.0 → 2.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +23 -0
- data/README.md +1 -1
- data/lib/context_dev/client.rb +5 -0
- data/lib/context_dev/models/batch_get_results_response.rb +33 -3
- data/lib/context_dev/models/batch_submit_params.rb +44 -316
- data/lib/context_dev/models/brand_retrieve_response.rb +29 -1
- data/lib/context_dev/models/brand_retrieve_simplified_response.rb +30 -1
- data/lib/context_dev/models/brand_search_params.rb +41 -3
- data/lib/context_dev/models/news_search_params.rb +467 -0
- data/lib/context_dev/models/news_search_response.rb +284 -0
- data/lib/context_dev/models/parse_handle_params.rb +15 -144
- data/lib/context_dev/models/person_enrich_response.rb +61 -1
- data/lib/context_dev/models/utility_prefetch_params.rb +19 -16
- data/lib/context_dev/models/utility_prefetch_response.rb +4 -5
- data/lib/context_dev/models/web_screenshot_params.rb +16 -31
- data/lib/context_dev/models/web_web_crawl_md_response.rb +30 -1
- data/lib/context_dev/models/web_web_scrape_html_params.rb +15 -153
- data/lib/context_dev/models/web_web_scrape_html_response.rb +30 -1
- data/lib/context_dev/models/web_web_scrape_images_params.rb +12 -124
- data/lib/context_dev/models/web_web_scrape_md_params.rb +35 -238
- data/lib/context_dev/models/web_web_scrape_md_response.rb +41 -2
- data/lib/context_dev/models.rb +2 -0
- data/lib/context_dev/resources/brand.rb +10 -11
- data/lib/context_dev/resources/news.rb +51 -0
- data/lib/context_dev/resources/parse.rb +5 -5
- data/lib/context_dev/resources/utility.rb +7 -6
- data/lib/context_dev/resources/web.rb +30 -24
- data/lib/context_dev/version.rb +1 -1
- data/lib/context_dev.rb +3 -0
- data/rbi/context_dev/client.rbi +4 -0
- data/rbi/context_dev/models/batch_get_results_response.rbi +71 -2
- data/rbi/context_dev/models/batch_submit_params.rbi +58 -592
- data/rbi/context_dev/models/brand_retrieve_response.rbi +80 -3
- data/rbi/context_dev/models/brand_retrieve_simplified_response.rbi +80 -3
- data/rbi/context_dev/models/brand_search_params.rbi +71 -2
- data/rbi/context_dev/models/news_search_params.rbi +1294 -0
- data/rbi/context_dev/models/news_search_response.rbi +489 -0
- data/rbi/context_dev/models/parse_handle_params.rbi +20 -316
- data/rbi/context_dev/models/person_enrich_response.rbi +87 -0
- data/rbi/context_dev/models/utility_prefetch_params.rbi +24 -16
- data/rbi/context_dev/models/utility_prefetch_response.rbi +8 -6
- data/rbi/context_dev/models/web_screenshot_params.rbi +23 -71
- data/rbi/context_dev/models/web_web_crawl_md_response.rbi +67 -0
- data/rbi/context_dev/models/web_web_scrape_html_params.rbi +20 -356
- data/rbi/context_dev/models/web_web_scrape_html_response.rbi +65 -0
- data/rbi/context_dev/models/web_web_scrape_images_params.rbi +16 -287
- data/rbi/context_dev/models/web_web_scrape_md_params.rbi +47 -551
- data/rbi/context_dev/models/web_web_scrape_md_response.rbi +80 -0
- data/rbi/context_dev/models.rbi +2 -0
- data/rbi/context_dev/resources/brand.rbi +14 -9
- data/rbi/context_dev/resources/news.rbi +46 -0
- data/rbi/context_dev/resources/parse.rbi +5 -21
- data/rbi/context_dev/resources/utility.rbi +8 -6
- data/rbi/context_dev/resources/web.rbi +34 -66
- data/sig/context_dev/client.rbs +2 -0
- data/sig/context_dev/models/batch_get_results_response.rbs +21 -0
- data/sig/context_dev/models/batch_submit_params.rbs +54 -144
- data/sig/context_dev/models/brand_retrieve_response.rbs +33 -3
- data/sig/context_dev/models/brand_retrieve_simplified_response.rbs +33 -3
- data/sig/context_dev/models/brand_search_params.rbs +38 -1
- data/sig/context_dev/models/news_search_params.rbs +532 -0
- data/sig/context_dev/models/news_search_response.rbs +206 -0
- data/sig/context_dev/models/parse_handle_params.rbs +25 -90
- data/sig/context_dev/models/person_enrich_response.rbs +31 -0
- data/sig/context_dev/models/utility_prefetch_params.rbs +2 -1
- data/sig/context_dev/models/utility_prefetch_response.rbs +2 -1
- data/sig/context_dev/models/web_screenshot_params.rbs +12 -18
- data/sig/context_dev/models/web_web_crawl_md_response.rbs +21 -0
- data/sig/context_dev/models/web_web_scrape_html_params.rbs +24 -94
- data/sig/context_dev/models/web_web_scrape_html_response.rbs +21 -0
- data/sig/context_dev/models/web_web_scrape_images_params.rbs +20 -72
- data/sig/context_dev/models/web_web_scrape_md_params.rbs +46 -148
- data/sig/context_dev/models/web_web_scrape_md_response.rbs +28 -0
- data/sig/context_dev/models.rbs +2 -0
- data/sig/context_dev/resources/brand.rbs +3 -0
- data/sig/context_dev/resources/news.rbs +17 -0
- data/sig/context_dev/resources/parse.rbs +5 -5
- data/sig/context_dev/resources/web.rbs +13 -11
- metadata +11 -2
|
@@ -68,21 +68,20 @@ module ContextDev
|
|
|
68
68
|
# Some parameter documentations has been truncated, see
|
|
69
69
|
# {ContextDev::Models::BrandSearchParams} for more details.
|
|
70
70
|
#
|
|
71
|
-
# Search brands by name or domain
|
|
72
|
-
# (domain, name, logo). Name matches rank ahead of domain matches; within each
|
|
73
|
-
# group the most popular brands come first: by Tranco rank, then market cap for
|
|
74
|
-
# brands outside the Tranco list, with text relevance breaking ties. Matching is
|
|
75
|
-
# prefix-based with no typo tolerance, so it is suited to autocomplete. Only
|
|
76
|
-
# brands already in the Context.dev index are returned — use /brand/retrieve to
|
|
77
|
-
# fetch (and index) a specific domain. Free on Pro and Scale plans; costs 1 credit
|
|
78
|
-
# per request on the Free and Starter plans.
|
|
71
|
+
# Search indexed brands by name or domain
|
|
79
72
|
#
|
|
80
|
-
# @overload search(query:, tags: nil, request_options: {})
|
|
73
|
+
# @overload search(query:, autocomplete: nil, query_by: nil, tags: nil, typo_tolerance: nil, request_options: {})
|
|
81
74
|
#
|
|
82
|
-
# @param query [String] Search term, matched against
|
|
75
|
+
# @param query [String] Search term, matched against the fields selected by queryBy (e.g. 'nike', 'nike.
|
|
76
|
+
#
|
|
77
|
+
# @param autocomplete [Boolean] Whether the search term matches by prefix, so partial words match as they are ty
|
|
78
|
+
#
|
|
79
|
+
# @param query_by [Array<Symbol, ContextDev::Models::BrandSearchParams::QueryBy>] Fields to match the search term against, as a comma-separated list or repeated p
|
|
83
80
|
#
|
|
84
81
|
# @param tags [Array<String>] Optional comma-separated caller-defined tags for tracking this request. Tags are
|
|
85
82
|
#
|
|
83
|
+
# @param typo_tolerance [Integer] Maximum number of typos tolerated when matching, from 0 to 2. Defaults to 0 (no
|
|
84
|
+
#
|
|
86
85
|
# @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
|
|
87
86
|
#
|
|
88
87
|
# @return [ContextDev::Models::BrandSearchResponse]
|
|
@@ -94,7 +93,7 @@ module ContextDev
|
|
|
94
93
|
@client.request(
|
|
95
94
|
method: :get,
|
|
96
95
|
path: "brand/search",
|
|
97
|
-
query: query,
|
|
96
|
+
query: query.transform_keys(query_by: "queryBy", typo_tolerance: "typoTolerance"),
|
|
98
97
|
model: ContextDev::Models::BrandSearchResponse,
|
|
99
98
|
options: options
|
|
100
99
|
)
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module ContextDev
|
|
4
|
+
module Resources
|
|
5
|
+
# Search live first-party RSS and free historical news data by company identity.
|
|
6
|
+
class News
|
|
7
|
+
# Searches live and historical company news for one company, identified in
|
|
8
|
+
# searchBy by name, domain, ticker (optionally disambiguated by exchange), or
|
|
9
|
+
# ISIN. Results can be filtered by publisher domain, publisher country, article
|
|
10
|
+
# language, article type, and published-at date, and include stable story IDs,
|
|
11
|
+
# source metadata, verified entity relevance, and cursor pagination.
|
|
12
|
+
#
|
|
13
|
+
# @overload search(search_by:, cursor: nil, filter_by: nil, limit: nil, sort_by: nil, tags: nil, request_options: {})
|
|
14
|
+
#
|
|
15
|
+
# @param search_by [ContextDev::Models::NewsSearchParams::SearchBy] What to search for.
|
|
16
|
+
#
|
|
17
|
+
# @param cursor [String, nil] Opaque next_cursor from the previous response, or null for the first page.
|
|
18
|
+
#
|
|
19
|
+
# @param filter_by [ContextDev::Models::NewsSearchParams::FilterBy] Optional result filters.
|
|
20
|
+
#
|
|
21
|
+
# @param limit [Integer] Maximum results to return. Defaults to 10.
|
|
22
|
+
#
|
|
23
|
+
# @param sort_by [ContextDev::Models::NewsSearchParams::SortBy] Result ordering. Defaults to newest.
|
|
24
|
+
#
|
|
25
|
+
# @param tags [Array<String>] Optional tags for tracking usage. Up to 20 tags, each 1 to 50 characters.
|
|
26
|
+
#
|
|
27
|
+
# @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
|
|
28
|
+
#
|
|
29
|
+
# @return [ContextDev::Models::NewsSearchResponse]
|
|
30
|
+
#
|
|
31
|
+
# @see ContextDev::Models::NewsSearchParams
|
|
32
|
+
def search(params)
|
|
33
|
+
parsed, options = ContextDev::NewsSearchParams.dump_request(params)
|
|
34
|
+
@client.request(
|
|
35
|
+
method: :post,
|
|
36
|
+
path: "news/search",
|
|
37
|
+
body: parsed,
|
|
38
|
+
model: ContextDev::Models::NewsSearchResponse,
|
|
39
|
+
options: options
|
|
40
|
+
)
|
|
41
|
+
end
|
|
42
|
+
|
|
43
|
+
# @api private
|
|
44
|
+
#
|
|
45
|
+
# @param client [ContextDev::Client]
|
|
46
|
+
def initialize(client:)
|
|
47
|
+
@client = client
|
|
48
|
+
end
|
|
49
|
+
end
|
|
50
|
+
end
|
|
51
|
+
end
|
|
@@ -17,19 +17,19 @@ module ContextDev
|
|
|
17
17
|
#
|
|
18
18
|
# @param extension [Symbol, ContextDev::Models::ParseHandleParams::Extension] Query param: Optional file extension hint, such as pdf, docx, xlsx, pptx, html,
|
|
19
19
|
#
|
|
20
|
-
# @param include_images [Boolean
|
|
20
|
+
# @param include_images [Boolean] Query param: Include image references in Markdown output
|
|
21
21
|
#
|
|
22
|
-
# @param include_links [Boolean
|
|
22
|
+
# @param include_links [Boolean] Query param: Preserve hyperlinks in Markdown output
|
|
23
23
|
#
|
|
24
|
-
# @param ocr [Boolean
|
|
24
|
+
# @param ocr [Boolean] Query param: When true for PDF inputs, OCR the selected pages that have no usabl
|
|
25
25
|
#
|
|
26
26
|
# @param pdf [ContextDev::Models::ParseHandleParams::Pdf] Query param: PDF page-range options as a JSON object, e.g. {"start": 2, "end": 5
|
|
27
27
|
#
|
|
28
|
-
# @param shorten_base64_images [Boolean
|
|
28
|
+
# @param shorten_base64_images [Boolean] Query param: Shorten base64-encoded image data in the Markdown output
|
|
29
29
|
#
|
|
30
30
|
# @param tags [Array<String>] Query param: Optional comma-separated caller-defined tags for tracking this requ
|
|
31
31
|
#
|
|
32
|
-
# @param use_main_content_only [Boolean
|
|
32
|
+
# @param use_main_content_only [Boolean] Query param: Extract only the main content from HTML-like inputs
|
|
33
33
|
#
|
|
34
34
|
# @param zdr [Symbol, ContextDev::Models::ParseHandleParams::Zdr] Query param: Set to enabled to bypass shared caches and omit request and respons
|
|
35
35
|
#
|
|
@@ -6,16 +6,17 @@ module ContextDev
|
|
|
6
6
|
# Some parameter documentations has been truncated, see
|
|
7
7
|
# {ContextDev::Models::UtilityPrefetchParams} for more details.
|
|
8
8
|
#
|
|
9
|
-
# Signal that you may fetch
|
|
10
|
-
#
|
|
11
|
-
# one lookup key: a domain,
|
|
12
|
-
#
|
|
9
|
+
# Signal that you may fetch data soon to improve latency. The type field selects
|
|
10
|
+
# what to prefetch ('brand' queues a brand data fetch, 'styleguide' queues a
|
|
11
|
+
# styleguide extraction) and identifier carries exactly one lookup key: a domain,
|
|
12
|
+
# or an email whose domain is extracted and validated (free email providers and
|
|
13
|
+
# disposable email addresses are not allowed).
|
|
13
14
|
#
|
|
14
15
|
# @overload prefetch(identifier:, type:, tags: nil, timeout_ms: nil, request_options: {})
|
|
15
16
|
#
|
|
16
|
-
# @param identifier [ContextDev::Models::UtilityPrefetchParams::Identifier::UtilityPrefetchDomainIdentifier, ContextDev::Models::UtilityPrefetchParams::Identifier::UtilityPrefetchEmailIdentifier] Identifier of the
|
|
17
|
+
# @param identifier [ContextDev::Models::UtilityPrefetchParams::Identifier::UtilityPrefetchDomainIdentifier, ContextDev::Models::UtilityPrefetchParams::Identifier::UtilityPrefetchEmailIdentifier] Identifier of the target to prefetch. Provide exactly one of domain or email.
|
|
17
18
|
#
|
|
18
|
-
# @param type [Symbol, ContextDev::Models::UtilityPrefetchParams::Type] What to prefetch
|
|
19
|
+
# @param type [Symbol, ContextDev::Models::UtilityPrefetchParams::Type] What to prefetch: 'brand' warms the brand data cache, 'styleguide' warms the sty
|
|
19
20
|
#
|
|
20
21
|
# @param tags [Array<String>] Optional tags for tracking usage. Up to 20 tags, each 1 to 50 characters.
|
|
21
22
|
#
|
|
@@ -176,7 +176,9 @@ module ContextDev
|
|
|
176
176
|
#
|
|
177
177
|
# Capture a screenshot of a website.
|
|
178
178
|
#
|
|
179
|
-
# @overload screenshot(color_scheme: nil, country: nil, direct_url: nil, domain: nil, full_screenshot: nil, handle_cookie_popup: nil, max_age_ms: nil, page: nil, scroll_offset: nil, tags: nil, timeout_ms: nil, viewport: nil, wait_for_ms: nil, zdr: nil, request_options: {})
|
|
179
|
+
# @overload screenshot(clear_popups: nil, color_scheme: nil, country: nil, direct_url: nil, domain: nil, full_screenshot: nil, handle_cookie_popup: nil, max_age_ms: nil, page: nil, scroll_offset: nil, tags: nil, timeout_ms: nil, viewport: nil, wait_for_ms: nil, zdr: nil, request_options: {})
|
|
180
|
+
#
|
|
181
|
+
# @param clear_popups [Boolean] Optional parameter for comprehensive popup cleanup. If 'true', the browser dismi
|
|
180
182
|
#
|
|
181
183
|
# @param color_scheme [Symbol, ContextDev::Models::WebScreenshotParams::ColorScheme] Optional parameter to choose the site's visual theme in the screenshot. Use 'lig
|
|
182
184
|
#
|
|
@@ -188,7 +190,7 @@ module ContextDev
|
|
|
188
190
|
#
|
|
189
191
|
# @param full_screenshot [Symbol, ContextDev::Models::WebScreenshotParams::FullScreenshot] Optional parameter to determine screenshot type. If 'true', takes a full page sc
|
|
190
192
|
#
|
|
191
|
-
# @param handle_cookie_popup [Boolean
|
|
193
|
+
# @param handle_cookie_popup [Boolean] Optional parameter to control cookie/consent popup handling. If 'true', we dismi
|
|
192
194
|
#
|
|
193
195
|
# @param max_age_ms [Integer, nil] Return a cached screenshot if a prior screenshot for the same parameters exists
|
|
194
196
|
#
|
|
@@ -218,6 +220,7 @@ module ContextDev
|
|
|
218
220
|
method: :get,
|
|
219
221
|
path: "web/screenshot",
|
|
220
222
|
query: query.transform_keys(
|
|
223
|
+
clear_popups: "clearPopups",
|
|
221
224
|
color_scheme: "colorScheme",
|
|
222
225
|
direct_url: "directUrl",
|
|
223
226
|
full_screenshot: "fullScreenshot",
|
|
@@ -359,7 +362,7 @@ module ContextDev
|
|
|
359
362
|
#
|
|
360
363
|
# @param headers [Hash{Symbol=>String}] Optional outbound HTTP headers forwarded only to the target URL, sent as deep-ob
|
|
361
364
|
#
|
|
362
|
-
# @param include_frames [Boolean
|
|
365
|
+
# @param include_frames [Boolean] When true, iframes are rendered inline into the returned HTML.
|
|
363
366
|
#
|
|
364
367
|
# @param include_selectors [Array<String>, nil] CSS selectors. When provided, only matching subtrees (and their descendants) are
|
|
365
368
|
#
|
|
@@ -367,13 +370,13 @@ module ContextDev
|
|
|
367
370
|
#
|
|
368
371
|
# @param pdf [ContextDev::Models::WebWebScrapeHTMLParams::Pdf] PDF parsing controls. Use start/end to limit text extraction and embedded-image
|
|
369
372
|
#
|
|
370
|
-
# @param settle_animations [Boolean
|
|
373
|
+
# @param settle_animations [Boolean] When true, waits briefly for CSS and transition animations to settle before extr
|
|
371
374
|
#
|
|
372
375
|
# @param tags [Array<String>] Optional comma-separated caller-defined tags for tracking this request. Tags are
|
|
373
376
|
#
|
|
374
377
|
# @param timeout_ms [Integer] Optional timeout in milliseconds for the request. If the request takes longer th
|
|
375
378
|
#
|
|
376
|
-
# @param use_main_content_only [Boolean
|
|
379
|
+
# @param use_main_content_only [Boolean] When true, return only the page's main content in the HTML response, excluding h
|
|
377
380
|
#
|
|
378
381
|
# @param wait_for_ms [Integer, nil] Optional browser wait time in milliseconds after initial page load. Min: 0. Max:
|
|
379
382
|
#
|
|
@@ -420,7 +423,7 @@ module ContextDev
|
|
|
420
423
|
#
|
|
421
424
|
# @param actions [Array<ContextDev::Models::WebWebScrapeImagesParams::Action::Wait, ContextDev::Models::WebWebScrapeImagesParams::Action::Perform>, nil] Optional browser actions executed in array order after the page loads and before
|
|
422
425
|
#
|
|
423
|
-
# @param dedupe [Boolean
|
|
426
|
+
# @param dedupe [Boolean] When true, visually duplicate images are removed: every image is loaded and perc
|
|
424
427
|
#
|
|
425
428
|
# @param enrichment [ContextDev::Models::WebWebScrapeImagesParams::Enrichment, nil] Optional per-image processing, sent as deep-object query params such as enrichme
|
|
426
429
|
#
|
|
@@ -476,19 +479,19 @@ module ContextDev
|
|
|
476
479
|
#
|
|
477
480
|
# ### Billing & errors
|
|
478
481
|
#
|
|
479
|
-
# | HTTP status | Billed? | Meaning
|
|
480
|
-
# | ----------- | ----------------------------------------- |
|
|
481
|
-
# | 200 | Yes — 1 credit, or 2 credits with actions | Successful scrape, including a zero-length result when includeSelectors matched nothing
|
|
482
|
-
# | 400 | No | Invalid input, skipped PDF, or the page could not be scraped
|
|
483
|
-
# | 401 / 403 | No | Invalid/disabled key, insufficient permissions, or credits exhausted; inspect error_code
|
|
484
|
-
# | 404 | No | Target page returned or fingerprinted as not found
|
|
485
|
-
# | 408 | No | Request timed out
|
|
486
|
-
# | 413 | No | Target content exceeds the maximum supported size (20 MB)
|
|
487
|
-
# | 415 | No | Unsupported content type
|
|
488
|
-
# | 429 | No | Per-minute rate limit exceeded; honor Retry-After
|
|
489
|
-
# | 500 | No | Internal error
|
|
482
|
+
# | HTTP status | Billed? | Meaning |
|
|
483
|
+
# | ----------- | ----------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
|
484
|
+
# | 200 | Yes — 1 credit, or 2 credits with actions | Successful scrape, including a zero-length result when includeSelectors matched nothing |
|
|
485
|
+
# | 400 | No | Invalid input, skipped PDF, or the page could not be scraped. error_code WEBSITE_BLOCKED specifically means the site answered with an anti-bot challenge, CAPTCHA wall, or login shell instead of the page (even when the site returned HTTP 200) — retrying later or from another country sometimes succeeds |
|
|
486
|
+
# | 401 / 403 | No | Invalid/disabled key, insufficient permissions, or credits exhausted; inspect error_code |
|
|
487
|
+
# | 404 | No | Target page returned or fingerprinted as not found |
|
|
488
|
+
# | 408 | No | Request timed out |
|
|
489
|
+
# | 413 | No | Target content exceeds the maximum supported size (20 MB) |
|
|
490
|
+
# | 415 | No | Unsupported content type |
|
|
491
|
+
# | 429 | No | Per-minute rate limit exceeded; honor Retry-After |
|
|
492
|
+
# | 500 | No | Internal error |
|
|
490
493
|
#
|
|
491
|
-
# @overload web_scrape_md(url:, actions: nil, country: nil, exclude_selectors: nil, headers: nil, include_frames: nil, include_images: nil, include_links: nil, include_selectors: nil, max_age_ms: nil, pdf: nil, settle_animations: nil, shorten_base64_images: nil, tags: nil, timeout_ms: nil, use_main_content_only: nil, wait_for_ms: nil, zdr: nil, request_options: {})
|
|
494
|
+
# @overload web_scrape_md(url:, actions: nil, country: nil, exclude_selectors: nil, headers: nil, include_frames: nil, include_html: nil, include_images: nil, include_links: nil, include_selectors: nil, max_age_ms: nil, pdf: nil, settle_animations: nil, shorten_base64_images: nil, tags: nil, timeout_ms: nil, use_main_content_only: nil, wait_for_ms: nil, zdr: nil, request_options: {})
|
|
492
495
|
#
|
|
493
496
|
# @param url [String] Full URL to scrape into LLM usable Markdown (must include http:// or https:// pr
|
|
494
497
|
#
|
|
@@ -500,11 +503,13 @@ module ContextDev
|
|
|
500
503
|
#
|
|
501
504
|
# @param headers [Hash{Symbol=>String}] Optional outbound HTTP headers forwarded only to the target URL, sent as deep-ob
|
|
502
505
|
#
|
|
503
|
-
# @param include_frames [Boolean
|
|
506
|
+
# @param include_frames [Boolean] When true, the contents of iframes are rendered to Markdown.
|
|
507
|
+
#
|
|
508
|
+
# @param include_html [Boolean] When true, the response also includes an `html` field with the page HTML the Mar
|
|
504
509
|
#
|
|
505
|
-
# @param include_images [Boolean
|
|
510
|
+
# @param include_images [Boolean] Include image references in Markdown output
|
|
506
511
|
#
|
|
507
|
-
# @param include_links [Boolean
|
|
512
|
+
# @param include_links [Boolean] Preserve hyperlinks in Markdown output
|
|
508
513
|
#
|
|
509
514
|
# @param include_selectors [Array<String>, nil] CSS selectors. When provided, only matching HTML subtrees (and their descendants
|
|
510
515
|
#
|
|
@@ -512,15 +517,15 @@ module ContextDev
|
|
|
512
517
|
#
|
|
513
518
|
# @param pdf [ContextDev::Models::WebWebScrapeMdParams::Pdf] PDF parsing controls. Use start/end to limit text extraction and embedded-image
|
|
514
519
|
#
|
|
515
|
-
# @param settle_animations [Boolean
|
|
520
|
+
# @param settle_animations [Boolean] When true, waits briefly for CSS and transition animations to settle before conv
|
|
516
521
|
#
|
|
517
|
-
# @param shorten_base64_images [Boolean
|
|
522
|
+
# @param shorten_base64_images [Boolean] Shorten base64-encoded image data in the Markdown output
|
|
518
523
|
#
|
|
519
524
|
# @param tags [Array<String>] Optional comma-separated caller-defined tags for tracking this request. Tags are
|
|
520
525
|
#
|
|
521
526
|
# @param timeout_ms [Integer] Optional timeout in milliseconds for the request. If the request takes longer th
|
|
522
527
|
#
|
|
523
|
-
# @param use_main_content_only [Boolean
|
|
528
|
+
# @param use_main_content_only [Boolean] Extract only the main content of the page, excluding headers, footers, sidebars,
|
|
524
529
|
#
|
|
525
530
|
# @param wait_for_ms [Integer, nil] Optional browser wait time in milliseconds after initial page load before conver
|
|
526
531
|
#
|
|
@@ -540,6 +545,7 @@ module ContextDev
|
|
|
540
545
|
query: query.transform_keys(
|
|
541
546
|
exclude_selectors: "excludeSelectors",
|
|
542
547
|
include_frames: "includeFrames",
|
|
548
|
+
include_html: "includeHTML",
|
|
543
549
|
include_images: "includeImages",
|
|
544
550
|
include_links: "includeLinks",
|
|
545
551
|
include_selectors: "includeSelectors",
|
data/lib/context_dev/version.rb
CHANGED
data/lib/context_dev.rb
CHANGED
|
@@ -107,6 +107,8 @@ require_relative "context_dev/models/monitor_run_params"
|
|
|
107
107
|
require_relative "context_dev/models/monitor_run_response"
|
|
108
108
|
require_relative "context_dev/models/monitor_update_params"
|
|
109
109
|
require_relative "context_dev/models/monitor_update_response"
|
|
110
|
+
require_relative "context_dev/models/news_search_params"
|
|
111
|
+
require_relative "context_dev/models/news_search_response"
|
|
110
112
|
require_relative "context_dev/models/page_error_count"
|
|
111
113
|
require_relative "context_dev/models/parse_handle_params"
|
|
112
114
|
require_relative "context_dev/models/parse_handle_response"
|
|
@@ -143,6 +145,7 @@ require_relative "context_dev/resources/batch"
|
|
|
143
145
|
require_relative "context_dev/resources/brand"
|
|
144
146
|
require_relative "context_dev/resources/industry"
|
|
145
147
|
require_relative "context_dev/resources/monitors"
|
|
148
|
+
require_relative "context_dev/resources/news"
|
|
146
149
|
require_relative "context_dev/resources/parse"
|
|
147
150
|
require_relative "context_dev/resources/people"
|
|
148
151
|
require_relative "context_dev/resources/utility"
|
data/rbi/context_dev/client.rbi
CHANGED
|
@@ -45,6 +45,10 @@ module ContextDev
|
|
|
45
45
|
sig { returns(ContextDev::Resources::People) }
|
|
46
46
|
attr_reader :people
|
|
47
47
|
|
|
48
|
+
# Search live first-party RSS and free historical news data by company identity.
|
|
49
|
+
sig { returns(ContextDev::Resources::News) }
|
|
50
|
+
attr_reader :news
|
|
51
|
+
|
|
48
52
|
# @api private
|
|
49
53
|
sig { override.returns(T::Hash[String, String]) }
|
|
50
54
|
private def auth_headers
|
|
@@ -162,7 +162,8 @@ module ContextDev
|
|
|
162
162
|
sig { returns(String) }
|
|
163
163
|
attr_accessor :url
|
|
164
164
|
|
|
165
|
-
#
|
|
165
|
+
# Page HTML. Present on html batches, and on markdown batches submitted with
|
|
166
|
+
# `options.includeHTML`.
|
|
166
167
|
sig { returns(T.nilable(String)) }
|
|
167
168
|
attr_reader :html
|
|
168
169
|
|
|
@@ -223,7 +224,8 @@ module ContextDev
|
|
|
223
224
|
metadata:,
|
|
224
225
|
# URL as submitted, or as discovered by the crawl.
|
|
225
226
|
url:,
|
|
226
|
-
#
|
|
227
|
+
# Page HTML. Present on html batches, and on markdown batches submitted with
|
|
228
|
+
# `options.includeHTML`.
|
|
227
229
|
html: nil,
|
|
228
230
|
# Caller-supplied identifier echoed from submission.
|
|
229
231
|
item_id: nil,
|
|
@@ -351,6 +353,29 @@ module ContextDev
|
|
|
351
353
|
sig { params(favicon: String).void }
|
|
352
354
|
attr_writer :favicon
|
|
353
355
|
|
|
356
|
+
# Page headings (h1–h6) in document order, extracted from the unfiltered document.
|
|
357
|
+
# Capped at the first 500 headings. Omitted when the page has none.
|
|
358
|
+
sig do
|
|
359
|
+
returns(
|
|
360
|
+
T.nilable(
|
|
361
|
+
T::Array[
|
|
362
|
+
ContextDev::Models::BatchGetResultsResponse::Data::Ok::Metadata::Heading
|
|
363
|
+
]
|
|
364
|
+
)
|
|
365
|
+
)
|
|
366
|
+
end
|
|
367
|
+
attr_reader :headings
|
|
368
|
+
|
|
369
|
+
sig do
|
|
370
|
+
params(
|
|
371
|
+
headings:
|
|
372
|
+
T::Array[
|
|
373
|
+
ContextDev::Models::BatchGetResultsResponse::Data::Ok::Metadata::Heading::OrHash
|
|
374
|
+
]
|
|
375
|
+
).void
|
|
376
|
+
end
|
|
377
|
+
attr_writer :headings
|
|
378
|
+
|
|
354
379
|
# Primary resolved preview image from Open Graph, Twitter, or image metadata.
|
|
355
380
|
sig { returns(T.nilable(String)) }
|
|
356
381
|
attr_reader :image
|
|
@@ -480,6 +505,10 @@ module ContextDev
|
|
|
480
505
|
canonical_url: String,
|
|
481
506
|
description: String,
|
|
482
507
|
favicon: String,
|
|
508
|
+
headings:
|
|
509
|
+
T::Array[
|
|
510
|
+
ContextDev::Models::BatchGetResultsResponse::Data::Ok::Metadata::Heading::OrHash
|
|
511
|
+
],
|
|
483
512
|
image: String,
|
|
484
513
|
json_ld: T::Array[T::Hash[Symbol, T.anything]],
|
|
485
514
|
keywords: T::Array[String],
|
|
@@ -519,6 +548,9 @@ module ContextDev
|
|
|
519
548
|
description: nil,
|
|
520
549
|
# Resolved favicon URL, when present.
|
|
521
550
|
favicon: nil,
|
|
551
|
+
# Page headings (h1–h6) in document order, extracted from the unfiltered document.
|
|
552
|
+
# Capped at the first 500 headings. Omitted when the page has none.
|
|
553
|
+
headings: nil,
|
|
522
554
|
# Primary resolved preview image from Open Graph, Twitter, or image metadata.
|
|
523
555
|
image: nil,
|
|
524
556
|
# JSON-LD structured data blocks parsed from the page.
|
|
@@ -562,6 +594,10 @@ module ContextDev
|
|
|
562
594
|
canonical_url: String,
|
|
563
595
|
description: String,
|
|
564
596
|
favicon: String,
|
|
597
|
+
headings:
|
|
598
|
+
T::Array[
|
|
599
|
+
ContextDev::Models::BatchGetResultsResponse::Data::Ok::Metadata::Heading
|
|
600
|
+
],
|
|
565
601
|
image: String,
|
|
566
602
|
json_ld: T::Array[T::Hash[Symbol, T.anything]],
|
|
567
603
|
keywords: T::Array[String],
|
|
@@ -677,6 +713,39 @@ module ContextDev
|
|
|
677
713
|
end
|
|
678
714
|
end
|
|
679
715
|
|
|
716
|
+
class Heading < ContextDev::Internal::Type::BaseModel
|
|
717
|
+
OrHash =
|
|
718
|
+
T.type_alias do
|
|
719
|
+
T.any(
|
|
720
|
+
ContextDev::Models::BatchGetResultsResponse::Data::Ok::Metadata::Heading,
|
|
721
|
+
ContextDev::Internal::AnyHash
|
|
722
|
+
)
|
|
723
|
+
end
|
|
724
|
+
|
|
725
|
+
# Heading level, 1–6 (from h1–h6).
|
|
726
|
+
sig { returns(Integer) }
|
|
727
|
+
attr_accessor :level
|
|
728
|
+
|
|
729
|
+
# Heading text with whitespace collapsed, truncated to 1000 characters.
|
|
730
|
+
sig { returns(String) }
|
|
731
|
+
attr_accessor :text
|
|
732
|
+
|
|
733
|
+
sig do
|
|
734
|
+
params(level: Integer, text: String).returns(T.attached_class)
|
|
735
|
+
end
|
|
736
|
+
def self.new(
|
|
737
|
+
# Heading level, 1–6 (from h1–h6).
|
|
738
|
+
level:,
|
|
739
|
+
# Heading text with whitespace collapsed, truncated to 1000 characters.
|
|
740
|
+
text:
|
|
741
|
+
)
|
|
742
|
+
end
|
|
743
|
+
|
|
744
|
+
sig { override.returns({ level: Integer, text: String }) }
|
|
745
|
+
def to_hash
|
|
746
|
+
end
|
|
747
|
+
end
|
|
748
|
+
|
|
680
749
|
module OpenGraph
|
|
681
750
|
extend ContextDev::Internal::Type::Union
|
|
682
751
|
|