context.dev 2.9.0 → 2.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +13 -0
- data/README.md +1 -1
- data/lib/context_dev/client.rb +5 -0
- data/lib/context_dev/models/batch_get_results_response.rb +33 -3
- data/lib/context_dev/models/batch_submit_params.rb +44 -316
- data/lib/context_dev/models/brand_search_params.rb +41 -3
- data/lib/context_dev/models/news_search_params.rb +467 -0
- data/lib/context_dev/models/news_search_response.rb +238 -0
- data/lib/context_dev/models/parse_handle_params.rb +15 -144
- data/lib/context_dev/models/person_enrich_response.rb +61 -1
- data/lib/context_dev/models/utility_prefetch_params.rb +19 -16
- data/lib/context_dev/models/utility_prefetch_response.rb +4 -5
- data/lib/context_dev/models/web_screenshot_params.rb +3 -30
- data/lib/context_dev/models/web_web_crawl_md_response.rb +30 -1
- data/lib/context_dev/models/web_web_scrape_html_params.rb +15 -153
- data/lib/context_dev/models/web_web_scrape_html_response.rb +30 -1
- data/lib/context_dev/models/web_web_scrape_images_params.rb +12 -124
- data/lib/context_dev/models/web_web_scrape_md_params.rb +35 -238
- data/lib/context_dev/models/web_web_scrape_md_response.rb +41 -2
- data/lib/context_dev/models.rb +2 -0
- data/lib/context_dev/resources/brand.rb +10 -11
- data/lib/context_dev/resources/news.rb +51 -0
- data/lib/context_dev/resources/parse.rb +5 -5
- data/lib/context_dev/resources/utility.rb +7 -6
- data/lib/context_dev/resources/web.rb +26 -23
- data/lib/context_dev/version.rb +1 -1
- data/lib/context_dev.rb +3 -0
- data/rbi/context_dev/client.rbi +4 -0
- data/rbi/context_dev/models/batch_get_results_response.rbi +71 -2
- data/rbi/context_dev/models/batch_submit_params.rbi +58 -592
- data/rbi/context_dev/models/brand_search_params.rbi +71 -2
- data/rbi/context_dev/models/news_search_params.rbi +1294 -0
- data/rbi/context_dev/models/news_search_response.rbi +423 -0
- data/rbi/context_dev/models/parse_handle_params.rbi +20 -316
- data/rbi/context_dev/models/person_enrich_response.rbi +87 -0
- data/rbi/context_dev/models/utility_prefetch_params.rbi +24 -16
- data/rbi/context_dev/models/utility_prefetch_response.rbi +8 -6
- data/rbi/context_dev/models/web_screenshot_params.rbi +4 -71
- data/rbi/context_dev/models/web_web_crawl_md_response.rbi +67 -0
- data/rbi/context_dev/models/web_web_scrape_html_params.rbi +20 -356
- data/rbi/context_dev/models/web_web_scrape_html_response.rbi +65 -0
- data/rbi/context_dev/models/web_web_scrape_images_params.rbi +16 -287
- data/rbi/context_dev/models/web_web_scrape_md_params.rbi +47 -551
- data/rbi/context_dev/models/web_web_scrape_md_response.rbi +80 -0
- data/rbi/context_dev/models.rbi +2 -0
- data/rbi/context_dev/resources/brand.rbi +14 -9
- data/rbi/context_dev/resources/news.rbi +46 -0
- data/rbi/context_dev/resources/parse.rbi +5 -21
- data/rbi/context_dev/resources/utility.rbi +8 -6
- data/rbi/context_dev/resources/web.rbi +27 -66
- data/sig/context_dev/client.rbs +2 -0
- data/sig/context_dev/models/batch_get_results_response.rbs +21 -0
- data/sig/context_dev/models/batch_submit_params.rbs +54 -144
- data/sig/context_dev/models/brand_search_params.rbs +38 -1
- data/sig/context_dev/models/news_search_params.rbs +532 -0
- data/sig/context_dev/models/news_search_response.rbs +206 -0
- data/sig/context_dev/models/parse_handle_params.rbs +25 -90
- data/sig/context_dev/models/person_enrich_response.rbs +31 -0
- data/sig/context_dev/models/utility_prefetch_params.rbs +2 -1
- data/sig/context_dev/models/utility_prefetch_response.rbs +2 -1
- data/sig/context_dev/models/web_screenshot_params.rbs +5 -18
- data/sig/context_dev/models/web_web_crawl_md_response.rbs +21 -0
- data/sig/context_dev/models/web_web_scrape_html_params.rbs +24 -94
- data/sig/context_dev/models/web_web_scrape_html_response.rbs +21 -0
- data/sig/context_dev/models/web_web_scrape_images_params.rbs +20 -72
- data/sig/context_dev/models/web_web_scrape_md_params.rbs +46 -148
- data/sig/context_dev/models/web_web_scrape_md_response.rbs +28 -0
- data/sig/context_dev/models.rbs +2 -0
- data/sig/context_dev/resources/brand.rbs +3 -0
- data/sig/context_dev/resources/news.rbs +17 -0
- data/sig/context_dev/resources/parse.rbs +5 -5
- data/sig/context_dev/resources/web.rbs +12 -11
- metadata +11 -2
|
@@ -188,7 +188,7 @@ module ContextDev
|
|
|
188
188
|
#
|
|
189
189
|
# @param full_screenshot [Symbol, ContextDev::Models::WebScreenshotParams::FullScreenshot] Optional parameter to determine screenshot type. If 'true', takes a full page sc
|
|
190
190
|
#
|
|
191
|
-
# @param handle_cookie_popup [Boolean
|
|
191
|
+
# @param handle_cookie_popup [Boolean] Optional parameter to control cookie/consent popup handling. If 'true', we dismi
|
|
192
192
|
#
|
|
193
193
|
# @param max_age_ms [Integer, nil] Return a cached screenshot if a prior screenshot for the same parameters exists
|
|
194
194
|
#
|
|
@@ -359,7 +359,7 @@ module ContextDev
|
|
|
359
359
|
#
|
|
360
360
|
# @param headers [Hash{Symbol=>String}] Optional outbound HTTP headers forwarded only to the target URL, sent as deep-ob
|
|
361
361
|
#
|
|
362
|
-
# @param include_frames [Boolean
|
|
362
|
+
# @param include_frames [Boolean] When true, iframes are rendered inline into the returned HTML.
|
|
363
363
|
#
|
|
364
364
|
# @param include_selectors [Array<String>, nil] CSS selectors. When provided, only matching subtrees (and their descendants) are
|
|
365
365
|
#
|
|
@@ -367,13 +367,13 @@ module ContextDev
|
|
|
367
367
|
#
|
|
368
368
|
# @param pdf [ContextDev::Models::WebWebScrapeHTMLParams::Pdf] PDF parsing controls. Use start/end to limit text extraction and embedded-image
|
|
369
369
|
#
|
|
370
|
-
# @param settle_animations [Boolean
|
|
370
|
+
# @param settle_animations [Boolean] When true, waits briefly for CSS and transition animations to settle before extr
|
|
371
371
|
#
|
|
372
372
|
# @param tags [Array<String>] Optional comma-separated caller-defined tags for tracking this request. Tags are
|
|
373
373
|
#
|
|
374
374
|
# @param timeout_ms [Integer] Optional timeout in milliseconds for the request. If the request takes longer th
|
|
375
375
|
#
|
|
376
|
-
# @param use_main_content_only [Boolean
|
|
376
|
+
# @param use_main_content_only [Boolean] When true, return only the page's main content in the HTML response, excluding h
|
|
377
377
|
#
|
|
378
378
|
# @param wait_for_ms [Integer, nil] Optional browser wait time in milliseconds after initial page load. Min: 0. Max:
|
|
379
379
|
#
|
|
@@ -420,7 +420,7 @@ module ContextDev
|
|
|
420
420
|
#
|
|
421
421
|
# @param actions [Array<ContextDev::Models::WebWebScrapeImagesParams::Action::Wait, ContextDev::Models::WebWebScrapeImagesParams::Action::Perform>, nil] Optional browser actions executed in array order after the page loads and before
|
|
422
422
|
#
|
|
423
|
-
# @param dedupe [Boolean
|
|
423
|
+
# @param dedupe [Boolean] When true, visually duplicate images are removed: every image is loaded and perc
|
|
424
424
|
#
|
|
425
425
|
# @param enrichment [ContextDev::Models::WebWebScrapeImagesParams::Enrichment, nil] Optional per-image processing, sent as deep-object query params such as enrichme
|
|
426
426
|
#
|
|
@@ -476,19 +476,19 @@ module ContextDev
|
|
|
476
476
|
#
|
|
477
477
|
# ### Billing & errors
|
|
478
478
|
#
|
|
479
|
-
# | HTTP status | Billed? | Meaning
|
|
480
|
-
# | ----------- | ----------------------------------------- |
|
|
481
|
-
# | 200 | Yes — 1 credit, or 2 credits with actions | Successful scrape, including a zero-length result when includeSelectors matched nothing
|
|
482
|
-
# | 400 | No | Invalid input, skipped PDF, or the page could not be scraped
|
|
483
|
-
# | 401 / 403 | No | Invalid/disabled key, insufficient permissions, or credits exhausted; inspect error_code
|
|
484
|
-
# | 404 | No | Target page returned or fingerprinted as not found
|
|
485
|
-
# | 408 | No | Request timed out
|
|
486
|
-
# | 413 | No | Target content exceeds the maximum supported size (20 MB)
|
|
487
|
-
# | 415 | No | Unsupported content type
|
|
488
|
-
# | 429 | No | Per-minute rate limit exceeded; honor Retry-After
|
|
489
|
-
# | 500 | No | Internal error
|
|
479
|
+
# | HTTP status | Billed? | Meaning |
|
|
480
|
+
# | ----------- | ----------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
|
481
|
+
# | 200 | Yes — 1 credit, or 2 credits with actions | Successful scrape, including a zero-length result when includeSelectors matched nothing |
|
|
482
|
+
# | 400 | No | Invalid input, skipped PDF, or the page could not be scraped. error_code WEBSITE_BLOCKED specifically means the site answered with an anti-bot challenge, CAPTCHA wall, or login shell instead of the page (even when the site returned HTTP 200) — retrying later or from another country sometimes succeeds |
|
|
483
|
+
# | 401 / 403 | No | Invalid/disabled key, insufficient permissions, or credits exhausted; inspect error_code |
|
|
484
|
+
# | 404 | No | Target page returned or fingerprinted as not found |
|
|
485
|
+
# | 408 | No | Request timed out |
|
|
486
|
+
# | 413 | No | Target content exceeds the maximum supported size (20 MB) |
|
|
487
|
+
# | 415 | No | Unsupported content type |
|
|
488
|
+
# | 429 | No | Per-minute rate limit exceeded; honor Retry-After |
|
|
489
|
+
# | 500 | No | Internal error |
|
|
490
490
|
#
|
|
491
|
-
# @overload web_scrape_md(url:, actions: nil, country: nil, exclude_selectors: nil, headers: nil, include_frames: nil, include_images: nil, include_links: nil, include_selectors: nil, max_age_ms: nil, pdf: nil, settle_animations: nil, shorten_base64_images: nil, tags: nil, timeout_ms: nil, use_main_content_only: nil, wait_for_ms: nil, zdr: nil, request_options: {})
|
|
491
|
+
# @overload web_scrape_md(url:, actions: nil, country: nil, exclude_selectors: nil, headers: nil, include_frames: nil, include_html: nil, include_images: nil, include_links: nil, include_selectors: nil, max_age_ms: nil, pdf: nil, settle_animations: nil, shorten_base64_images: nil, tags: nil, timeout_ms: nil, use_main_content_only: nil, wait_for_ms: nil, zdr: nil, request_options: {})
|
|
492
492
|
#
|
|
493
493
|
# @param url [String] Full URL to scrape into LLM usable Markdown (must include http:// or https:// pr
|
|
494
494
|
#
|
|
@@ -500,11 +500,13 @@ module ContextDev
|
|
|
500
500
|
#
|
|
501
501
|
# @param headers [Hash{Symbol=>String}] Optional outbound HTTP headers forwarded only to the target URL, sent as deep-ob
|
|
502
502
|
#
|
|
503
|
-
# @param include_frames [Boolean
|
|
503
|
+
# @param include_frames [Boolean] When true, the contents of iframes are rendered to Markdown.
|
|
504
|
+
#
|
|
505
|
+
# @param include_html [Boolean] When true, the response also includes an `html` field with the page HTML the Mar
|
|
504
506
|
#
|
|
505
|
-
# @param include_images [Boolean
|
|
507
|
+
# @param include_images [Boolean] Include image references in Markdown output
|
|
506
508
|
#
|
|
507
|
-
# @param include_links [Boolean
|
|
509
|
+
# @param include_links [Boolean] Preserve hyperlinks in Markdown output
|
|
508
510
|
#
|
|
509
511
|
# @param include_selectors [Array<String>, nil] CSS selectors. When provided, only matching HTML subtrees (and their descendants
|
|
510
512
|
#
|
|
@@ -512,15 +514,15 @@ module ContextDev
|
|
|
512
514
|
#
|
|
513
515
|
# @param pdf [ContextDev::Models::WebWebScrapeMdParams::Pdf] PDF parsing controls. Use start/end to limit text extraction and embedded-image
|
|
514
516
|
#
|
|
515
|
-
# @param settle_animations [Boolean
|
|
517
|
+
# @param settle_animations [Boolean] When true, waits briefly for CSS and transition animations to settle before conv
|
|
516
518
|
#
|
|
517
|
-
# @param shorten_base64_images [Boolean
|
|
519
|
+
# @param shorten_base64_images [Boolean] Shorten base64-encoded image data in the Markdown output
|
|
518
520
|
#
|
|
519
521
|
# @param tags [Array<String>] Optional comma-separated caller-defined tags for tracking this request. Tags are
|
|
520
522
|
#
|
|
521
523
|
# @param timeout_ms [Integer] Optional timeout in milliseconds for the request. If the request takes longer th
|
|
522
524
|
#
|
|
523
|
-
# @param use_main_content_only [Boolean
|
|
525
|
+
# @param use_main_content_only [Boolean] Extract only the main content of the page, excluding headers, footers, sidebars,
|
|
524
526
|
#
|
|
525
527
|
# @param wait_for_ms [Integer, nil] Optional browser wait time in milliseconds after initial page load before conver
|
|
526
528
|
#
|
|
@@ -540,6 +542,7 @@ module ContextDev
|
|
|
540
542
|
query: query.transform_keys(
|
|
541
543
|
exclude_selectors: "excludeSelectors",
|
|
542
544
|
include_frames: "includeFrames",
|
|
545
|
+
include_html: "includeHTML",
|
|
543
546
|
include_images: "includeImages",
|
|
544
547
|
include_links: "includeLinks",
|
|
545
548
|
include_selectors: "includeSelectors",
|
data/lib/context_dev/version.rb
CHANGED
data/lib/context_dev.rb
CHANGED
|
@@ -107,6 +107,8 @@ require_relative "context_dev/models/monitor_run_params"
|
|
|
107
107
|
require_relative "context_dev/models/monitor_run_response"
|
|
108
108
|
require_relative "context_dev/models/monitor_update_params"
|
|
109
109
|
require_relative "context_dev/models/monitor_update_response"
|
|
110
|
+
require_relative "context_dev/models/news_search_params"
|
|
111
|
+
require_relative "context_dev/models/news_search_response"
|
|
110
112
|
require_relative "context_dev/models/page_error_count"
|
|
111
113
|
require_relative "context_dev/models/parse_handle_params"
|
|
112
114
|
require_relative "context_dev/models/parse_handle_response"
|
|
@@ -143,6 +145,7 @@ require_relative "context_dev/resources/batch"
|
|
|
143
145
|
require_relative "context_dev/resources/brand"
|
|
144
146
|
require_relative "context_dev/resources/industry"
|
|
145
147
|
require_relative "context_dev/resources/monitors"
|
|
148
|
+
require_relative "context_dev/resources/news"
|
|
146
149
|
require_relative "context_dev/resources/parse"
|
|
147
150
|
require_relative "context_dev/resources/people"
|
|
148
151
|
require_relative "context_dev/resources/utility"
|
data/rbi/context_dev/client.rbi
CHANGED
|
@@ -45,6 +45,10 @@ module ContextDev
|
|
|
45
45
|
sig { returns(ContextDev::Resources::People) }
|
|
46
46
|
attr_reader :people
|
|
47
47
|
|
|
48
|
+
# Search live first-party RSS and free historical news data by company identity.
|
|
49
|
+
sig { returns(ContextDev::Resources::News) }
|
|
50
|
+
attr_reader :news
|
|
51
|
+
|
|
48
52
|
# @api private
|
|
49
53
|
sig { override.returns(T::Hash[String, String]) }
|
|
50
54
|
private def auth_headers
|
|
@@ -162,7 +162,8 @@ module ContextDev
|
|
|
162
162
|
sig { returns(String) }
|
|
163
163
|
attr_accessor :url
|
|
164
164
|
|
|
165
|
-
#
|
|
165
|
+
# Page HTML. Present on html batches, and on markdown batches submitted with
|
|
166
|
+
# `options.includeHTML`.
|
|
166
167
|
sig { returns(T.nilable(String)) }
|
|
167
168
|
attr_reader :html
|
|
168
169
|
|
|
@@ -223,7 +224,8 @@ module ContextDev
|
|
|
223
224
|
metadata:,
|
|
224
225
|
# URL as submitted, or as discovered by the crawl.
|
|
225
226
|
url:,
|
|
226
|
-
#
|
|
227
|
+
# Page HTML. Present on html batches, and on markdown batches submitted with
|
|
228
|
+
# `options.includeHTML`.
|
|
227
229
|
html: nil,
|
|
228
230
|
# Caller-supplied identifier echoed from submission.
|
|
229
231
|
item_id: nil,
|
|
@@ -351,6 +353,29 @@ module ContextDev
|
|
|
351
353
|
sig { params(favicon: String).void }
|
|
352
354
|
attr_writer :favicon
|
|
353
355
|
|
|
356
|
+
# Page headings (h1–h6) in document order, extracted from the unfiltered document.
|
|
357
|
+
# Capped at the first 500 headings. Omitted when the page has none.
|
|
358
|
+
sig do
|
|
359
|
+
returns(
|
|
360
|
+
T.nilable(
|
|
361
|
+
T::Array[
|
|
362
|
+
ContextDev::Models::BatchGetResultsResponse::Data::Ok::Metadata::Heading
|
|
363
|
+
]
|
|
364
|
+
)
|
|
365
|
+
)
|
|
366
|
+
end
|
|
367
|
+
attr_reader :headings
|
|
368
|
+
|
|
369
|
+
sig do
|
|
370
|
+
params(
|
|
371
|
+
headings:
|
|
372
|
+
T::Array[
|
|
373
|
+
ContextDev::Models::BatchGetResultsResponse::Data::Ok::Metadata::Heading::OrHash
|
|
374
|
+
]
|
|
375
|
+
).void
|
|
376
|
+
end
|
|
377
|
+
attr_writer :headings
|
|
378
|
+
|
|
354
379
|
# Primary resolved preview image from Open Graph, Twitter, or image metadata.
|
|
355
380
|
sig { returns(T.nilable(String)) }
|
|
356
381
|
attr_reader :image
|
|
@@ -480,6 +505,10 @@ module ContextDev
|
|
|
480
505
|
canonical_url: String,
|
|
481
506
|
description: String,
|
|
482
507
|
favicon: String,
|
|
508
|
+
headings:
|
|
509
|
+
T::Array[
|
|
510
|
+
ContextDev::Models::BatchGetResultsResponse::Data::Ok::Metadata::Heading::OrHash
|
|
511
|
+
],
|
|
483
512
|
image: String,
|
|
484
513
|
json_ld: T::Array[T::Hash[Symbol, T.anything]],
|
|
485
514
|
keywords: T::Array[String],
|
|
@@ -519,6 +548,9 @@ module ContextDev
|
|
|
519
548
|
description: nil,
|
|
520
549
|
# Resolved favicon URL, when present.
|
|
521
550
|
favicon: nil,
|
|
551
|
+
# Page headings (h1–h6) in document order, extracted from the unfiltered document.
|
|
552
|
+
# Capped at the first 500 headings. Omitted when the page has none.
|
|
553
|
+
headings: nil,
|
|
522
554
|
# Primary resolved preview image from Open Graph, Twitter, or image metadata.
|
|
523
555
|
image: nil,
|
|
524
556
|
# JSON-LD structured data blocks parsed from the page.
|
|
@@ -562,6 +594,10 @@ module ContextDev
|
|
|
562
594
|
canonical_url: String,
|
|
563
595
|
description: String,
|
|
564
596
|
favicon: String,
|
|
597
|
+
headings:
|
|
598
|
+
T::Array[
|
|
599
|
+
ContextDev::Models::BatchGetResultsResponse::Data::Ok::Metadata::Heading
|
|
600
|
+
],
|
|
565
601
|
image: String,
|
|
566
602
|
json_ld: T::Array[T::Hash[Symbol, T.anything]],
|
|
567
603
|
keywords: T::Array[String],
|
|
@@ -677,6 +713,39 @@ module ContextDev
|
|
|
677
713
|
end
|
|
678
714
|
end
|
|
679
715
|
|
|
716
|
+
class Heading < ContextDev::Internal::Type::BaseModel
|
|
717
|
+
OrHash =
|
|
718
|
+
T.type_alias do
|
|
719
|
+
T.any(
|
|
720
|
+
ContextDev::Models::BatchGetResultsResponse::Data::Ok::Metadata::Heading,
|
|
721
|
+
ContextDev::Internal::AnyHash
|
|
722
|
+
)
|
|
723
|
+
end
|
|
724
|
+
|
|
725
|
+
# Heading level, 1–6 (from h1–h6).
|
|
726
|
+
sig { returns(Integer) }
|
|
727
|
+
attr_accessor :level
|
|
728
|
+
|
|
729
|
+
# Heading text with whitespace collapsed, truncated to 1000 characters.
|
|
730
|
+
sig { returns(String) }
|
|
731
|
+
attr_accessor :text
|
|
732
|
+
|
|
733
|
+
sig do
|
|
734
|
+
params(level: Integer, text: String).returns(T.attached_class)
|
|
735
|
+
end
|
|
736
|
+
def self.new(
|
|
737
|
+
# Heading level, 1–6 (from h1–h6).
|
|
738
|
+
level:,
|
|
739
|
+
# Heading text with whitespace collapsed, truncated to 1000 characters.
|
|
740
|
+
text:
|
|
741
|
+
)
|
|
742
|
+
end
|
|
743
|
+
|
|
744
|
+
sig { override.returns({ level: Integer, text: String }) }
|
|
745
|
+
def to_hash
|
|
746
|
+
end
|
|
747
|
+
end
|
|
748
|
+
|
|
680
749
|
module OpenGraph
|
|
681
750
|
extend ContextDev::Internal::Type::Union
|
|
682
751
|
|