context.dev 2.9.0 → 2.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +13 -0
- data/README.md +1 -1
- data/lib/context_dev/client.rb +5 -0
- data/lib/context_dev/models/batch_get_results_response.rb +33 -3
- data/lib/context_dev/models/batch_submit_params.rb +44 -316
- data/lib/context_dev/models/brand_search_params.rb +41 -3
- data/lib/context_dev/models/news_search_params.rb +467 -0
- data/lib/context_dev/models/news_search_response.rb +238 -0
- data/lib/context_dev/models/parse_handle_params.rb +15 -144
- data/lib/context_dev/models/person_enrich_response.rb +61 -1
- data/lib/context_dev/models/utility_prefetch_params.rb +19 -16
- data/lib/context_dev/models/utility_prefetch_response.rb +4 -5
- data/lib/context_dev/models/web_screenshot_params.rb +3 -30
- data/lib/context_dev/models/web_web_crawl_md_response.rb +30 -1
- data/lib/context_dev/models/web_web_scrape_html_params.rb +15 -153
- data/lib/context_dev/models/web_web_scrape_html_response.rb +30 -1
- data/lib/context_dev/models/web_web_scrape_images_params.rb +12 -124
- data/lib/context_dev/models/web_web_scrape_md_params.rb +35 -238
- data/lib/context_dev/models/web_web_scrape_md_response.rb +41 -2
- data/lib/context_dev/models.rb +2 -0
- data/lib/context_dev/resources/brand.rb +10 -11
- data/lib/context_dev/resources/news.rb +51 -0
- data/lib/context_dev/resources/parse.rb +5 -5
- data/lib/context_dev/resources/utility.rb +7 -6
- data/lib/context_dev/resources/web.rb +26 -23
- data/lib/context_dev/version.rb +1 -1
- data/lib/context_dev.rb +3 -0
- data/rbi/context_dev/client.rbi +4 -0
- data/rbi/context_dev/models/batch_get_results_response.rbi +71 -2
- data/rbi/context_dev/models/batch_submit_params.rbi +58 -592
- data/rbi/context_dev/models/brand_search_params.rbi +71 -2
- data/rbi/context_dev/models/news_search_params.rbi +1294 -0
- data/rbi/context_dev/models/news_search_response.rbi +423 -0
- data/rbi/context_dev/models/parse_handle_params.rbi +20 -316
- data/rbi/context_dev/models/person_enrich_response.rbi +87 -0
- data/rbi/context_dev/models/utility_prefetch_params.rbi +24 -16
- data/rbi/context_dev/models/utility_prefetch_response.rbi +8 -6
- data/rbi/context_dev/models/web_screenshot_params.rbi +4 -71
- data/rbi/context_dev/models/web_web_crawl_md_response.rbi +67 -0
- data/rbi/context_dev/models/web_web_scrape_html_params.rbi +20 -356
- data/rbi/context_dev/models/web_web_scrape_html_response.rbi +65 -0
- data/rbi/context_dev/models/web_web_scrape_images_params.rbi +16 -287
- data/rbi/context_dev/models/web_web_scrape_md_params.rbi +47 -551
- data/rbi/context_dev/models/web_web_scrape_md_response.rbi +80 -0
- data/rbi/context_dev/models.rbi +2 -0
- data/rbi/context_dev/resources/brand.rbi +14 -9
- data/rbi/context_dev/resources/news.rbi +46 -0
- data/rbi/context_dev/resources/parse.rbi +5 -21
- data/rbi/context_dev/resources/utility.rbi +8 -6
- data/rbi/context_dev/resources/web.rbi +27 -66
- data/sig/context_dev/client.rbs +2 -0
- data/sig/context_dev/models/batch_get_results_response.rbs +21 -0
- data/sig/context_dev/models/batch_submit_params.rbs +54 -144
- data/sig/context_dev/models/brand_search_params.rbs +38 -1
- data/sig/context_dev/models/news_search_params.rbs +532 -0
- data/sig/context_dev/models/news_search_response.rbs +206 -0
- data/sig/context_dev/models/parse_handle_params.rbs +25 -90
- data/sig/context_dev/models/person_enrich_response.rbs +31 -0
- data/sig/context_dev/models/utility_prefetch_params.rbs +2 -1
- data/sig/context_dev/models/utility_prefetch_response.rbs +2 -1
- data/sig/context_dev/models/web_screenshot_params.rbs +5 -18
- data/sig/context_dev/models/web_web_crawl_md_response.rbs +21 -0
- data/sig/context_dev/models/web_web_scrape_html_params.rbs +24 -94
- data/sig/context_dev/models/web_web_scrape_html_response.rbs +21 -0
- data/sig/context_dev/models/web_web_scrape_images_params.rbs +20 -72
- data/sig/context_dev/models/web_web_scrape_md_params.rbs +46 -148
- data/sig/context_dev/models/web_web_scrape_md_response.rbs +28 -0
- data/sig/context_dev/models.rbs +2 -0
- data/sig/context_dev/resources/brand.rbs +3 -0
- data/sig/context_dev/resources/news.rbs +17 -0
- data/sig/context_dev/resources/parse.rbs +5 -5
- data/sig/context_dev/resources/web.rbs +12 -11
- metadata +11 -2
|
@@ -50,20 +50,28 @@ module ContextDev
|
|
|
50
50
|
# @!attribute include_frames
|
|
51
51
|
# When true, the contents of iframes are rendered to Markdown.
|
|
52
52
|
#
|
|
53
|
-
# @return [Boolean,
|
|
54
|
-
optional :include_frames,
|
|
53
|
+
# @return [Boolean, nil]
|
|
54
|
+
optional :include_frames, ContextDev::Internal::Type::Boolean
|
|
55
|
+
|
|
56
|
+
# @!attribute include_html
|
|
57
|
+
# When true, the response also includes an `html` field with the page HTML the
|
|
58
|
+
# Markdown was converted from — the same body the Scrape HTML endpoint returns for
|
|
59
|
+
# the equivalent request.
|
|
60
|
+
#
|
|
61
|
+
# @return [Boolean, nil]
|
|
62
|
+
optional :include_html, ContextDev::Internal::Type::Boolean
|
|
55
63
|
|
|
56
64
|
# @!attribute include_images
|
|
57
65
|
# Include image references in Markdown output
|
|
58
66
|
#
|
|
59
|
-
# @return [Boolean,
|
|
60
|
-
optional :include_images,
|
|
67
|
+
# @return [Boolean, nil]
|
|
68
|
+
optional :include_images, ContextDev::Internal::Type::Boolean
|
|
61
69
|
|
|
62
70
|
# @!attribute include_links
|
|
63
71
|
# Preserve hyperlinks in Markdown output
|
|
64
72
|
#
|
|
65
|
-
# @return [Boolean,
|
|
66
|
-
optional :include_links,
|
|
73
|
+
# @return [Boolean, nil]
|
|
74
|
+
optional :include_links, ContextDev::Internal::Type::Boolean
|
|
67
75
|
|
|
68
76
|
# @!attribute include_selectors
|
|
69
77
|
# CSS selectors. When provided, only matching HTML subtrees (and their
|
|
@@ -93,14 +101,14 @@ module ContextDev
|
|
|
93
101
|
# converting to Markdown. Defaults to false. This adds a bit of latency in
|
|
94
102
|
# exchange for more stable output on animated pages.
|
|
95
103
|
#
|
|
96
|
-
# @return [Boolean,
|
|
97
|
-
optional :settle_animations,
|
|
104
|
+
# @return [Boolean, nil]
|
|
105
|
+
optional :settle_animations, ContextDev::Internal::Type::Boolean
|
|
98
106
|
|
|
99
107
|
# @!attribute shorten_base64_images
|
|
100
108
|
# Shorten base64-encoded image data in the Markdown output
|
|
101
109
|
#
|
|
102
|
-
# @return [Boolean,
|
|
103
|
-
optional :shorten_base64_images,
|
|
110
|
+
# @return [Boolean, nil]
|
|
111
|
+
optional :shorten_base64_images, ContextDev::Internal::Type::Boolean
|
|
104
112
|
|
|
105
113
|
# @!attribute tags
|
|
106
114
|
# Optional comma-separated caller-defined tags for tracking this request. Tags are
|
|
@@ -122,8 +130,8 @@ module ContextDev
|
|
|
122
130
|
# Extract only the main content of the page, excluding headers, footers, sidebars,
|
|
123
131
|
# and navigation
|
|
124
132
|
#
|
|
125
|
-
# @return [Boolean,
|
|
126
|
-
optional :use_main_content_only,
|
|
133
|
+
# @return [Boolean, nil]
|
|
134
|
+
optional :use_main_content_only, ContextDev::Internal::Type::Boolean
|
|
127
135
|
|
|
128
136
|
# @!attribute wait_for_ms
|
|
129
137
|
# Optional browser wait time in milliseconds after initial page load before
|
|
@@ -141,7 +149,7 @@ module ContextDev
|
|
|
141
149
|
# @return [Symbol, ContextDev::Models::WebWebScrapeMdParams::Zdr, nil]
|
|
142
150
|
optional :zdr, enum: -> { ContextDev::WebWebScrapeMdParams::Zdr }
|
|
143
151
|
|
|
144
|
-
# @!method initialize(url:, actions: nil, country: nil, exclude_selectors: nil, headers: nil, include_frames: nil, include_images: nil, include_links: nil, include_selectors: nil, max_age_ms: nil, pdf: nil, settle_animations: nil, shorten_base64_images: nil, tags: nil, timeout_ms: nil, use_main_content_only: nil, wait_for_ms: nil, zdr: nil, request_options: {})
|
|
152
|
+
# @!method initialize(url:, actions: nil, country: nil, exclude_selectors: nil, headers: nil, include_frames: nil, include_html: nil, include_images: nil, include_links: nil, include_selectors: nil, max_age_ms: nil, pdf: nil, settle_animations: nil, shorten_base64_images: nil, tags: nil, timeout_ms: nil, use_main_content_only: nil, wait_for_ms: nil, zdr: nil, request_options: {})
|
|
145
153
|
# Some parameter documentations has been truncated, see
|
|
146
154
|
# {ContextDev::Models::WebWebScrapeMdParams} for more details.
|
|
147
155
|
#
|
|
@@ -155,11 +163,13 @@ module ContextDev
|
|
|
155
163
|
#
|
|
156
164
|
# @param headers [Hash{Symbol=>String}] Optional outbound HTTP headers forwarded only to the target URL, sent as deep-ob
|
|
157
165
|
#
|
|
158
|
-
# @param include_frames [Boolean
|
|
166
|
+
# @param include_frames [Boolean] When true, the contents of iframes are rendered to Markdown.
|
|
159
167
|
#
|
|
160
|
-
# @param
|
|
168
|
+
# @param include_html [Boolean] When true, the response also includes an `html` field with the page HTML the Mar
|
|
161
169
|
#
|
|
162
|
-
# @param
|
|
170
|
+
# @param include_images [Boolean] Include image references in Markdown output
|
|
171
|
+
#
|
|
172
|
+
# @param include_links [Boolean] Preserve hyperlinks in Markdown output
|
|
163
173
|
#
|
|
164
174
|
# @param include_selectors [Array<String>, nil] CSS selectors. When provided, only matching HTML subtrees (and their descendants
|
|
165
175
|
#
|
|
@@ -167,15 +177,15 @@ module ContextDev
|
|
|
167
177
|
#
|
|
168
178
|
# @param pdf [ContextDev::Models::WebWebScrapeMdParams::Pdf] PDF parsing controls. Use start/end to limit text extraction and embedded-image
|
|
169
179
|
#
|
|
170
|
-
# @param settle_animations [Boolean
|
|
180
|
+
# @param settle_animations [Boolean] When true, waits briefly for CSS and transition animations to settle before conv
|
|
171
181
|
#
|
|
172
|
-
# @param shorten_base64_images [Boolean
|
|
182
|
+
# @param shorten_base64_images [Boolean] Shorten base64-encoded image data in the Markdown output
|
|
173
183
|
#
|
|
174
184
|
# @param tags [Array<String>] Optional comma-separated caller-defined tags for tracking this request. Tags are
|
|
175
185
|
#
|
|
176
186
|
# @param timeout_ms [Integer] Optional timeout in milliseconds for the request. If the request takes longer th
|
|
177
187
|
#
|
|
178
|
-
# @param use_main_content_only [Boolean
|
|
188
|
+
# @param use_main_content_only [Boolean] Extract only the main content of the page, excluding headers, footers, sidebars,
|
|
179
189
|
#
|
|
180
190
|
# @param wait_for_ms [Integer, nil] Optional browser wait time in milliseconds after initial page load before conver
|
|
181
191
|
#
|
|
@@ -450,81 +460,6 @@ module ContextDev
|
|
|
450
460
|
# @return [Array<Symbol>]
|
|
451
461
|
end
|
|
452
462
|
|
|
453
|
-
# When true, the contents of iframes are rendered to Markdown.
|
|
454
|
-
module IncludeFrames
|
|
455
|
-
extend ContextDev::Internal::Type::Union
|
|
456
|
-
|
|
457
|
-
variant ContextDev::Internal::Type::Boolean
|
|
458
|
-
|
|
459
|
-
variant const: -> { ContextDev::Models::WebWebScrapeMdParams::IncludeFrames::TRUE }
|
|
460
|
-
|
|
461
|
-
variant const: -> { ContextDev::Models::WebWebScrapeMdParams::IncludeFrames::FALSE }
|
|
462
|
-
|
|
463
|
-
# @!method self.variants
|
|
464
|
-
# @return [Array(Boolean, Symbol)]
|
|
465
|
-
|
|
466
|
-
define_sorbet_constant!(:Variants) do
|
|
467
|
-
T.type_alias { T.any(T::Boolean, ContextDev::WebWebScrapeMdParams::IncludeFrames::TaggedSymbol) }
|
|
468
|
-
end
|
|
469
|
-
|
|
470
|
-
# @!group
|
|
471
|
-
|
|
472
|
-
TRUE = :true
|
|
473
|
-
FALSE = :false
|
|
474
|
-
|
|
475
|
-
# @!endgroup
|
|
476
|
-
end
|
|
477
|
-
|
|
478
|
-
# Include image references in Markdown output
|
|
479
|
-
module IncludeImages
|
|
480
|
-
extend ContextDev::Internal::Type::Union
|
|
481
|
-
|
|
482
|
-
variant ContextDev::Internal::Type::Boolean
|
|
483
|
-
|
|
484
|
-
variant const: -> { ContextDev::Models::WebWebScrapeMdParams::IncludeImages::TRUE }
|
|
485
|
-
|
|
486
|
-
variant const: -> { ContextDev::Models::WebWebScrapeMdParams::IncludeImages::FALSE }
|
|
487
|
-
|
|
488
|
-
# @!method self.variants
|
|
489
|
-
# @return [Array(Boolean, Symbol)]
|
|
490
|
-
|
|
491
|
-
define_sorbet_constant!(:Variants) do
|
|
492
|
-
T.type_alias { T.any(T::Boolean, ContextDev::WebWebScrapeMdParams::IncludeImages::TaggedSymbol) }
|
|
493
|
-
end
|
|
494
|
-
|
|
495
|
-
# @!group
|
|
496
|
-
|
|
497
|
-
TRUE = :true
|
|
498
|
-
FALSE = :false
|
|
499
|
-
|
|
500
|
-
# @!endgroup
|
|
501
|
-
end
|
|
502
|
-
|
|
503
|
-
# Preserve hyperlinks in Markdown output
|
|
504
|
-
module IncludeLinks
|
|
505
|
-
extend ContextDev::Internal::Type::Union
|
|
506
|
-
|
|
507
|
-
variant ContextDev::Internal::Type::Boolean
|
|
508
|
-
|
|
509
|
-
variant const: -> { ContextDev::Models::WebWebScrapeMdParams::IncludeLinks::TRUE }
|
|
510
|
-
|
|
511
|
-
variant const: -> { ContextDev::Models::WebWebScrapeMdParams::IncludeLinks::FALSE }
|
|
512
|
-
|
|
513
|
-
# @!method self.variants
|
|
514
|
-
# @return [Array(Boolean, Symbol)]
|
|
515
|
-
|
|
516
|
-
define_sorbet_constant!(:Variants) do
|
|
517
|
-
T.type_alias { T.any(T::Boolean, ContextDev::WebWebScrapeMdParams::IncludeLinks::TaggedSymbol) }
|
|
518
|
-
end
|
|
519
|
-
|
|
520
|
-
# @!group
|
|
521
|
-
|
|
522
|
-
TRUE = :true
|
|
523
|
-
FALSE = :false
|
|
524
|
-
|
|
525
|
-
# @!endgroup
|
|
526
|
-
end
|
|
527
|
-
|
|
528
463
|
class Pdf < ContextDev::Internal::Type::BaseModel
|
|
529
464
|
# @!attribute end_
|
|
530
465
|
# Last 1-based PDF page to parse. When omitted, parsing ends at the last page.
|
|
@@ -539,17 +474,15 @@ module ContextDev
|
|
|
539
474
|
# text layer keep it. Billed at 1 credit per page OCR actually recovered, on top
|
|
540
475
|
# of the base request cost. When false, no OCR runs.
|
|
541
476
|
#
|
|
542
|
-
# @return [Boolean,
|
|
543
|
-
optional :ocr,
|
|
477
|
+
# @return [Boolean, nil]
|
|
478
|
+
optional :ocr, ContextDev::Internal::Type::Boolean
|
|
544
479
|
|
|
545
480
|
# @!attribute should_parse
|
|
546
481
|
# When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
|
|
547
482
|
# a 400 PDF_SKIPPED is returned.
|
|
548
483
|
#
|
|
549
|
-
# @return [Boolean,
|
|
550
|
-
optional :should_parse,
|
|
551
|
-
union: -> { ContextDev::WebWebScrapeMdParams::Pdf::ShouldParse },
|
|
552
|
-
api_name: :shouldParse
|
|
484
|
+
# @return [Boolean, nil]
|
|
485
|
+
optional :should_parse, ContextDev::Internal::Type::Boolean, api_name: :shouldParse
|
|
553
486
|
|
|
554
487
|
# @!attribute start
|
|
555
488
|
# First 1-based PDF page to parse. When omitted, parsing starts at the first page.
|
|
@@ -566,147 +499,11 @@ module ContextDev
|
|
|
566
499
|
#
|
|
567
500
|
# @param end_ [Integer] Last 1-based PDF page to parse. When omitted, parsing ends at the last page. Mus
|
|
568
501
|
#
|
|
569
|
-
# @param ocr [Boolean
|
|
502
|
+
# @param ocr [Boolean] When true, OCR the selected PDF pages that have no usable text layer (scans), re
|
|
570
503
|
#
|
|
571
|
-
# @param should_parse [Boolean
|
|
504
|
+
# @param should_parse [Boolean] When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
|
|
572
505
|
#
|
|
573
506
|
# @param start [Integer] First 1-based PDF page to parse. When omitted, parsing starts at the first page.
|
|
574
|
-
|
|
575
|
-
# When true, OCR the selected PDF pages that have no usable text layer (scans),
|
|
576
|
-
# replacing each recovered page's text with the OCR result while pages with a real
|
|
577
|
-
# text layer keep it. Billed at 1 credit per page OCR actually recovered, on top
|
|
578
|
-
# of the base request cost. When false, no OCR runs.
|
|
579
|
-
#
|
|
580
|
-
# @see ContextDev::Models::WebWebScrapeMdParams::Pdf#ocr
|
|
581
|
-
module Ocr
|
|
582
|
-
extend ContextDev::Internal::Type::Union
|
|
583
|
-
|
|
584
|
-
variant ContextDev::Internal::Type::Boolean
|
|
585
|
-
|
|
586
|
-
variant const: -> { ContextDev::Models::WebWebScrapeMdParams::Pdf::Ocr::TRUE }
|
|
587
|
-
|
|
588
|
-
variant const: -> { ContextDev::Models::WebWebScrapeMdParams::Pdf::Ocr::FALSE }
|
|
589
|
-
|
|
590
|
-
# @!method self.variants
|
|
591
|
-
# @return [Array(Boolean, Symbol)]
|
|
592
|
-
|
|
593
|
-
define_sorbet_constant!(:Variants) do
|
|
594
|
-
T.type_alias { T.any(T::Boolean, ContextDev::WebWebScrapeMdParams::Pdf::Ocr::TaggedSymbol) }
|
|
595
|
-
end
|
|
596
|
-
|
|
597
|
-
# @!group
|
|
598
|
-
|
|
599
|
-
TRUE = :true
|
|
600
|
-
FALSE = :false
|
|
601
|
-
|
|
602
|
-
# @!endgroup
|
|
603
|
-
end
|
|
604
|
-
|
|
605
|
-
# When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
|
|
606
|
-
# a 400 PDF_SKIPPED is returned.
|
|
607
|
-
#
|
|
608
|
-
# @see ContextDev::Models::WebWebScrapeMdParams::Pdf#should_parse
|
|
609
|
-
module ShouldParse
|
|
610
|
-
extend ContextDev::Internal::Type::Union
|
|
611
|
-
|
|
612
|
-
variant ContextDev::Internal::Type::Boolean
|
|
613
|
-
|
|
614
|
-
variant const: -> { ContextDev::Models::WebWebScrapeMdParams::Pdf::ShouldParse::TRUE }
|
|
615
|
-
|
|
616
|
-
variant const: -> { ContextDev::Models::WebWebScrapeMdParams::Pdf::ShouldParse::FALSE }
|
|
617
|
-
|
|
618
|
-
# @!method self.variants
|
|
619
|
-
# @return [Array(Boolean, Symbol)]
|
|
620
|
-
|
|
621
|
-
define_sorbet_constant!(:Variants) do
|
|
622
|
-
T.type_alias { T.any(T::Boolean, ContextDev::WebWebScrapeMdParams::Pdf::ShouldParse::TaggedSymbol) }
|
|
623
|
-
end
|
|
624
|
-
|
|
625
|
-
# @!group
|
|
626
|
-
|
|
627
|
-
TRUE = :true
|
|
628
|
-
FALSE = :false
|
|
629
|
-
|
|
630
|
-
# @!endgroup
|
|
631
|
-
end
|
|
632
|
-
end
|
|
633
|
-
|
|
634
|
-
# When true, waits briefly for CSS and transition animations to settle before
|
|
635
|
-
# converting to Markdown. Defaults to false. This adds a bit of latency in
|
|
636
|
-
# exchange for more stable output on animated pages.
|
|
637
|
-
module SettleAnimations
|
|
638
|
-
extend ContextDev::Internal::Type::Union
|
|
639
|
-
|
|
640
|
-
variant ContextDev::Internal::Type::Boolean
|
|
641
|
-
|
|
642
|
-
variant const: -> { ContextDev::Models::WebWebScrapeMdParams::SettleAnimations::TRUE }
|
|
643
|
-
|
|
644
|
-
variant const: -> { ContextDev::Models::WebWebScrapeMdParams::SettleAnimations::FALSE }
|
|
645
|
-
|
|
646
|
-
# @!method self.variants
|
|
647
|
-
# @return [Array(Boolean, Symbol)]
|
|
648
|
-
|
|
649
|
-
define_sorbet_constant!(:Variants) do
|
|
650
|
-
T.type_alias { T.any(T::Boolean, ContextDev::WebWebScrapeMdParams::SettleAnimations::TaggedSymbol) }
|
|
651
|
-
end
|
|
652
|
-
|
|
653
|
-
# @!group
|
|
654
|
-
|
|
655
|
-
TRUE = :true
|
|
656
|
-
FALSE = :false
|
|
657
|
-
|
|
658
|
-
# @!endgroup
|
|
659
|
-
end
|
|
660
|
-
|
|
661
|
-
# Shorten base64-encoded image data in the Markdown output
|
|
662
|
-
module ShortenBase64Images
|
|
663
|
-
extend ContextDev::Internal::Type::Union
|
|
664
|
-
|
|
665
|
-
variant ContextDev::Internal::Type::Boolean
|
|
666
|
-
|
|
667
|
-
variant const: -> { ContextDev::Models::WebWebScrapeMdParams::ShortenBase64Images::TRUE }
|
|
668
|
-
|
|
669
|
-
variant const: -> { ContextDev::Models::WebWebScrapeMdParams::ShortenBase64Images::FALSE }
|
|
670
|
-
|
|
671
|
-
# @!method self.variants
|
|
672
|
-
# @return [Array(Boolean, Symbol)]
|
|
673
|
-
|
|
674
|
-
define_sorbet_constant!(:Variants) do
|
|
675
|
-
T.type_alias { T.any(T::Boolean, ContextDev::WebWebScrapeMdParams::ShortenBase64Images::TaggedSymbol) }
|
|
676
|
-
end
|
|
677
|
-
|
|
678
|
-
# @!group
|
|
679
|
-
|
|
680
|
-
TRUE = :true
|
|
681
|
-
FALSE = :false
|
|
682
|
-
|
|
683
|
-
# @!endgroup
|
|
684
|
-
end
|
|
685
|
-
|
|
686
|
-
# Extract only the main content of the page, excluding headers, footers, sidebars,
|
|
687
|
-
# and navigation
|
|
688
|
-
module UseMainContentOnly
|
|
689
|
-
extend ContextDev::Internal::Type::Union
|
|
690
|
-
|
|
691
|
-
variant ContextDev::Internal::Type::Boolean
|
|
692
|
-
|
|
693
|
-
variant const: -> { ContextDev::Models::WebWebScrapeMdParams::UseMainContentOnly::TRUE }
|
|
694
|
-
|
|
695
|
-
variant const: -> { ContextDev::Models::WebWebScrapeMdParams::UseMainContentOnly::FALSE }
|
|
696
|
-
|
|
697
|
-
# @!method self.variants
|
|
698
|
-
# @return [Array(Boolean, Symbol)]
|
|
699
|
-
|
|
700
|
-
define_sorbet_constant!(:Variants) do
|
|
701
|
-
T.type_alias { T.any(T::Boolean, ContextDev::WebWebScrapeMdParams::UseMainContentOnly::TaggedSymbol) }
|
|
702
|
-
end
|
|
703
|
-
|
|
704
|
-
# @!group
|
|
705
|
-
|
|
706
|
-
TRUE = :true
|
|
707
|
-
FALSE = :false
|
|
708
|
-
|
|
709
|
-
# @!endgroup
|
|
710
507
|
end
|
|
711
508
|
|
|
712
509
|
# Set to enabled to bypass shared caches and omit request and response content
|
|
@@ -51,6 +51,14 @@ module ContextDev
|
|
|
51
51
|
# @return [Boolean, nil]
|
|
52
52
|
optional :actions_html_stale, ContextDev::Internal::Type::Boolean, api_name: :actionsHtmlStale
|
|
53
53
|
|
|
54
|
+
# @!attribute html
|
|
55
|
+
# Only present when includeHTML=true: the page HTML the Markdown was converted
|
|
56
|
+
# from — the same body the Scrape HTML endpoint returns for the equivalent
|
|
57
|
+
# request.
|
|
58
|
+
#
|
|
59
|
+
# @return [String, nil]
|
|
60
|
+
optional :html, String
|
|
61
|
+
|
|
54
62
|
# @!attribute key_metadata
|
|
55
63
|
# Metadata about the API key used for the request. Included in every response
|
|
56
64
|
# whenever a valid API key is provided, even when the response status is not 200.
|
|
@@ -58,7 +66,7 @@ module ContextDev
|
|
|
58
66
|
# @return [ContextDev::Models::WebWebScrapeMdResponse::KeyMetadata, nil]
|
|
59
67
|
optional :key_metadata, -> { ContextDev::Models::WebWebScrapeMdResponse::KeyMetadata }
|
|
60
68
|
|
|
61
|
-
# @!method initialize(content_length:, markdown:, metadata:, success:, url:, actions_applied: nil, actions_html_stale: nil, key_metadata: nil)
|
|
69
|
+
# @!method initialize(content_length:, markdown:, metadata:, success:, url:, actions_applied: nil, actions_html_stale: nil, html: nil, key_metadata: nil)
|
|
62
70
|
# Some parameter documentations has been truncated, see
|
|
63
71
|
# {ContextDev::Models::WebWebScrapeMdResponse} for more details.
|
|
64
72
|
#
|
|
@@ -76,6 +84,8 @@ module ContextDev
|
|
|
76
84
|
#
|
|
77
85
|
# @param actions_html_stale [Boolean] True when an action was applied but the returned content could not be refreshed
|
|
78
86
|
#
|
|
87
|
+
# @param html [String] Only present when includeHTML=true: the page HTML the Markdown was converted fro
|
|
88
|
+
#
|
|
79
89
|
# @param key_metadata [ContextDev::Models::WebWebScrapeMdResponse::KeyMetadata] Metadata about the API key used for the request. Included in every response when
|
|
80
90
|
|
|
81
91
|
# @see ContextDev::Models::WebWebScrapeMdResponse#metadata
|
|
@@ -132,6 +142,14 @@ module ContextDev
|
|
|
132
142
|
# @return [String, nil]
|
|
133
143
|
optional :favicon, String
|
|
134
144
|
|
|
145
|
+
# @!attribute headings
|
|
146
|
+
# Page headings (h1–h6) in document order, extracted from the unfiltered document.
|
|
147
|
+
# Capped at the first 500 headings. Omitted when the page has none.
|
|
148
|
+
#
|
|
149
|
+
# @return [Array<ContextDev::Models::WebWebScrapeMdResponse::Metadata::Heading>, nil]
|
|
150
|
+
optional :headings,
|
|
151
|
+
-> { ContextDev::Internal::Type::ArrayOf[ContextDev::Models::WebWebScrapeMdResponse::Metadata::Heading] }
|
|
152
|
+
|
|
135
153
|
# @!attribute image
|
|
136
154
|
# Primary resolved preview image from Open Graph, Twitter, or image metadata.
|
|
137
155
|
#
|
|
@@ -203,7 +221,7 @@ module ContextDev
|
|
|
203
221
|
optional :twitter,
|
|
204
222
|
-> { ContextDev::Internal::Type::HashOf[union: ContextDev::Models::WebWebScrapeMdResponse::Metadata::Twitter] }
|
|
205
223
|
|
|
206
|
-
# @!method initialize(final_url:, source_url:, additional_meta: nil, alternates: nil, author: nil, canonical_url: nil, description: nil, favicon: nil, image: nil, json_ld: nil, keywords: nil, language: nil, modified_time: nil, open_graph: nil, published_time: nil, robots: nil, site_name: nil, title: nil, twitter: nil)
|
|
224
|
+
# @!method initialize(final_url:, source_url:, additional_meta: nil, alternates: nil, author: nil, canonical_url: nil, description: nil, favicon: nil, headings: nil, image: nil, json_ld: nil, keywords: nil, language: nil, modified_time: nil, open_graph: nil, published_time: nil, robots: nil, site_name: nil, title: nil, twitter: nil)
|
|
207
225
|
# Some parameter documentations has been truncated, see
|
|
208
226
|
# {ContextDev::Models::WebWebScrapeMdResponse::Metadata} for more details.
|
|
209
227
|
#
|
|
@@ -225,6 +243,8 @@ module ContextDev
|
|
|
225
243
|
#
|
|
226
244
|
# @param favicon [String] Resolved favicon URL, when present.
|
|
227
245
|
#
|
|
246
|
+
# @param headings [Array<ContextDev::Models::WebWebScrapeMdResponse::Metadata::Heading>] Page headings (h1–h6) in document order, extracted from the unfiltered document.
|
|
247
|
+
#
|
|
228
248
|
# @param image [String] Primary resolved preview image from Open Graph, Twitter, or image metadata.
|
|
229
249
|
#
|
|
230
250
|
# @param json_ld [Array<Hash{Symbol=>Object}>] JSON-LD structured data blocks parsed from the page.
|
|
@@ -296,6 +316,25 @@ module ContextDev
|
|
|
296
316
|
# @param type [String] Alternate resource MIME type, when present.
|
|
297
317
|
end
|
|
298
318
|
|
|
319
|
+
class Heading < ContextDev::Internal::Type::BaseModel
|
|
320
|
+
# @!attribute level
|
|
321
|
+
# Heading level, 1–6 (from h1–h6).
|
|
322
|
+
#
|
|
323
|
+
# @return [Integer]
|
|
324
|
+
required :level, Integer
|
|
325
|
+
|
|
326
|
+
# @!attribute text
|
|
327
|
+
# Heading text with whitespace collapsed, truncated to 1000 characters.
|
|
328
|
+
#
|
|
329
|
+
# @return [String]
|
|
330
|
+
required :text, String
|
|
331
|
+
|
|
332
|
+
# @!method initialize(level:, text:)
|
|
333
|
+
# @param level [Integer] Heading level, 1–6 (from h1–h6).
|
|
334
|
+
#
|
|
335
|
+
# @param text [String] Heading text with whitespace collapsed, truncated to 1000 characters.
|
|
336
|
+
end
|
|
337
|
+
|
|
299
338
|
module OpenGraph
|
|
300
339
|
extend ContextDev::Internal::Type::Union
|
|
301
340
|
|
data/lib/context_dev/models.rb
CHANGED
|
@@ -97,6 +97,8 @@ module ContextDev
|
|
|
97
97
|
|
|
98
98
|
MonitorUpdateParams = ContextDev::Models::MonitorUpdateParams
|
|
99
99
|
|
|
100
|
+
NewsSearchParams = ContextDev::Models::NewsSearchParams
|
|
101
|
+
|
|
100
102
|
PageErrorCount = ContextDev::Models::PageErrorCount
|
|
101
103
|
|
|
102
104
|
ParseHandleParams = ContextDev::Models::ParseHandleParams
|
|
@@ -68,21 +68,20 @@ module ContextDev
|
|
|
68
68
|
# Some parameter documentations has been truncated, see
|
|
69
69
|
# {ContextDev::Models::BrandSearchParams} for more details.
|
|
70
70
|
#
|
|
71
|
-
# Search brands by name or domain
|
|
72
|
-
# (domain, name, logo). Name matches rank ahead of domain matches; within each
|
|
73
|
-
# group the most popular brands come first: by Tranco rank, then market cap for
|
|
74
|
-
# brands outside the Tranco list, with text relevance breaking ties. Matching is
|
|
75
|
-
# prefix-based with no typo tolerance, so it is suited to autocomplete. Only
|
|
76
|
-
# brands already in the Context.dev index are returned — use /brand/retrieve to
|
|
77
|
-
# fetch (and index) a specific domain. Free on Pro and Scale plans; costs 1 credit
|
|
78
|
-
# per request on the Free and Starter plans.
|
|
71
|
+
# Search indexed brands by name or domain
|
|
79
72
|
#
|
|
80
|
-
# @overload search(query:, tags: nil, request_options: {})
|
|
73
|
+
# @overload search(query:, autocomplete: nil, query_by: nil, tags: nil, typo_tolerance: nil, request_options: {})
|
|
81
74
|
#
|
|
82
|
-
# @param query [String] Search term, matched against
|
|
75
|
+
# @param query [String] Search term, matched against the fields selected by queryBy (e.g. 'nike', 'nike.
|
|
76
|
+
#
|
|
77
|
+
# @param autocomplete [Boolean] Whether the search term matches by prefix, so partial words match as they are ty
|
|
78
|
+
#
|
|
79
|
+
# @param query_by [Array<Symbol, ContextDev::Models::BrandSearchParams::QueryBy>] Fields to match the search term against, as a comma-separated list or repeated p
|
|
83
80
|
#
|
|
84
81
|
# @param tags [Array<String>] Optional comma-separated caller-defined tags for tracking this request. Tags are
|
|
85
82
|
#
|
|
83
|
+
# @param typo_tolerance [Integer] Maximum number of typos tolerated when matching, from 0 to 2. Defaults to 0 (no
|
|
84
|
+
#
|
|
86
85
|
# @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
|
|
87
86
|
#
|
|
88
87
|
# @return [ContextDev::Models::BrandSearchResponse]
|
|
@@ -94,7 +93,7 @@ module ContextDev
|
|
|
94
93
|
@client.request(
|
|
95
94
|
method: :get,
|
|
96
95
|
path: "brand/search",
|
|
97
|
-
query: query,
|
|
96
|
+
query: query.transform_keys(query_by: "queryBy", typo_tolerance: "typoTolerance"),
|
|
98
97
|
model: ContextDev::Models::BrandSearchResponse,
|
|
99
98
|
options: options
|
|
100
99
|
)
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module ContextDev
|
|
4
|
+
module Resources
|
|
5
|
+
# Search live first-party RSS and free historical news data by company identity.
|
|
6
|
+
class News
|
|
7
|
+
# Searches live and historical company news for one company, identified in
|
|
8
|
+
# searchBy by name, domain, ticker (optionally disambiguated by exchange), or
|
|
9
|
+
# ISIN. Results can be filtered by publisher domain, publisher country, article
|
|
10
|
+
# language, article type, and published-at date, and include stable story IDs,
|
|
11
|
+
# source metadata, verified entity relevance, and cursor pagination.
|
|
12
|
+
#
|
|
13
|
+
# @overload search(search_by:, cursor: nil, filter_by: nil, limit: nil, sort_by: nil, tags: nil, request_options: {})
|
|
14
|
+
#
|
|
15
|
+
# @param search_by [ContextDev::Models::NewsSearchParams::SearchBy] What to search for.
|
|
16
|
+
#
|
|
17
|
+
# @param cursor [String, nil] Opaque next_cursor from the previous response, or null for the first page.
|
|
18
|
+
#
|
|
19
|
+
# @param filter_by [ContextDev::Models::NewsSearchParams::FilterBy] Optional result filters.
|
|
20
|
+
#
|
|
21
|
+
# @param limit [Integer] Maximum results to return. Defaults to 10.
|
|
22
|
+
#
|
|
23
|
+
# @param sort_by [ContextDev::Models::NewsSearchParams::SortBy] Result ordering. Defaults to newest.
|
|
24
|
+
#
|
|
25
|
+
# @param tags [Array<String>] Optional tags for tracking usage. Up to 20 tags, each 1 to 50 characters.
|
|
26
|
+
#
|
|
27
|
+
# @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
|
|
28
|
+
#
|
|
29
|
+
# @return [ContextDev::Models::NewsSearchResponse]
|
|
30
|
+
#
|
|
31
|
+
# @see ContextDev::Models::NewsSearchParams
|
|
32
|
+
def search(params)
|
|
33
|
+
parsed, options = ContextDev::NewsSearchParams.dump_request(params)
|
|
34
|
+
@client.request(
|
|
35
|
+
method: :post,
|
|
36
|
+
path: "news/search",
|
|
37
|
+
body: parsed,
|
|
38
|
+
model: ContextDev::Models::NewsSearchResponse,
|
|
39
|
+
options: options
|
|
40
|
+
)
|
|
41
|
+
end
|
|
42
|
+
|
|
43
|
+
# @api private
|
|
44
|
+
#
|
|
45
|
+
# @param client [ContextDev::Client]
|
|
46
|
+
def initialize(client:)
|
|
47
|
+
@client = client
|
|
48
|
+
end
|
|
49
|
+
end
|
|
50
|
+
end
|
|
51
|
+
end
|
|
@@ -17,19 +17,19 @@ module ContextDev
|
|
|
17
17
|
#
|
|
18
18
|
# @param extension [Symbol, ContextDev::Models::ParseHandleParams::Extension] Query param: Optional file extension hint, such as pdf, docx, xlsx, pptx, html,
|
|
19
19
|
#
|
|
20
|
-
# @param include_images [Boolean
|
|
20
|
+
# @param include_images [Boolean] Query param: Include image references in Markdown output
|
|
21
21
|
#
|
|
22
|
-
# @param include_links [Boolean
|
|
22
|
+
# @param include_links [Boolean] Query param: Preserve hyperlinks in Markdown output
|
|
23
23
|
#
|
|
24
|
-
# @param ocr [Boolean
|
|
24
|
+
# @param ocr [Boolean] Query param: When true for PDF inputs, OCR the selected pages that have no usabl
|
|
25
25
|
#
|
|
26
26
|
# @param pdf [ContextDev::Models::ParseHandleParams::Pdf] Query param: PDF page-range options as a JSON object, e.g. {"start": 2, "end": 5
|
|
27
27
|
#
|
|
28
|
-
# @param shorten_base64_images [Boolean
|
|
28
|
+
# @param shorten_base64_images [Boolean] Query param: Shorten base64-encoded image data in the Markdown output
|
|
29
29
|
#
|
|
30
30
|
# @param tags [Array<String>] Query param: Optional comma-separated caller-defined tags for tracking this requ
|
|
31
31
|
#
|
|
32
|
-
# @param use_main_content_only [Boolean
|
|
32
|
+
# @param use_main_content_only [Boolean] Query param: Extract only the main content from HTML-like inputs
|
|
33
33
|
#
|
|
34
34
|
# @param zdr [Symbol, ContextDev::Models::ParseHandleParams::Zdr] Query param: Set to enabled to bypass shared caches and omit request and respons
|
|
35
35
|
#
|
|
@@ -6,16 +6,17 @@ module ContextDev
|
|
|
6
6
|
# Some parameter documentations has been truncated, see
|
|
7
7
|
# {ContextDev::Models::UtilityPrefetchParams} for more details.
|
|
8
8
|
#
|
|
9
|
-
# Signal that you may fetch
|
|
10
|
-
#
|
|
11
|
-
# one lookup key: a domain,
|
|
12
|
-
#
|
|
9
|
+
# Signal that you may fetch data soon to improve latency. The type field selects
|
|
10
|
+
# what to prefetch ('brand' queues a brand data fetch, 'styleguide' queues a
|
|
11
|
+
# styleguide extraction) and identifier carries exactly one lookup key: a domain,
|
|
12
|
+
# or an email whose domain is extracted and validated (free email providers and
|
|
13
|
+
# disposable email addresses are not allowed).
|
|
13
14
|
#
|
|
14
15
|
# @overload prefetch(identifier:, type:, tags: nil, timeout_ms: nil, request_options: {})
|
|
15
16
|
#
|
|
16
|
-
# @param identifier [ContextDev::Models::UtilityPrefetchParams::Identifier::UtilityPrefetchDomainIdentifier, ContextDev::Models::UtilityPrefetchParams::Identifier::UtilityPrefetchEmailIdentifier] Identifier of the
|
|
17
|
+
# @param identifier [ContextDev::Models::UtilityPrefetchParams::Identifier::UtilityPrefetchDomainIdentifier, ContextDev::Models::UtilityPrefetchParams::Identifier::UtilityPrefetchEmailIdentifier] Identifier of the target to prefetch. Provide exactly one of domain or email.
|
|
17
18
|
#
|
|
18
|
-
# @param type [Symbol, ContextDev::Models::UtilityPrefetchParams::Type] What to prefetch
|
|
19
|
+
# @param type [Symbol, ContextDev::Models::UtilityPrefetchParams::Type] What to prefetch: 'brand' warms the brand data cache, 'styleguide' warms the sty
|
|
19
20
|
#
|
|
20
21
|
# @param tags [Array<String>] Optional tags for tracking usage. Up to 20 tags, each 1 to 50 characters.
|
|
21
22
|
#
|