context.dev 2.9.0 → 2.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +23 -0
- data/README.md +1 -1
- data/lib/context_dev/client.rb +5 -0
- data/lib/context_dev/models/batch_get_results_response.rb +33 -3
- data/lib/context_dev/models/batch_submit_params.rb +44 -316
- data/lib/context_dev/models/brand_retrieve_response.rb +29 -1
- data/lib/context_dev/models/brand_retrieve_simplified_response.rb +30 -1
- data/lib/context_dev/models/brand_search_params.rb +41 -3
- data/lib/context_dev/models/news_search_params.rb +467 -0
- data/lib/context_dev/models/news_search_response.rb +284 -0
- data/lib/context_dev/models/parse_handle_params.rb +15 -144
- data/lib/context_dev/models/person_enrich_response.rb +61 -1
- data/lib/context_dev/models/utility_prefetch_params.rb +19 -16
- data/lib/context_dev/models/utility_prefetch_response.rb +4 -5
- data/lib/context_dev/models/web_screenshot_params.rb +16 -31
- data/lib/context_dev/models/web_web_crawl_md_response.rb +30 -1
- data/lib/context_dev/models/web_web_scrape_html_params.rb +15 -153
- data/lib/context_dev/models/web_web_scrape_html_response.rb +30 -1
- data/lib/context_dev/models/web_web_scrape_images_params.rb +12 -124
- data/lib/context_dev/models/web_web_scrape_md_params.rb +35 -238
- data/lib/context_dev/models/web_web_scrape_md_response.rb +41 -2
- data/lib/context_dev/models.rb +2 -0
- data/lib/context_dev/resources/brand.rb +10 -11
- data/lib/context_dev/resources/news.rb +51 -0
- data/lib/context_dev/resources/parse.rb +5 -5
- data/lib/context_dev/resources/utility.rb +7 -6
- data/lib/context_dev/resources/web.rb +30 -24
- data/lib/context_dev/version.rb +1 -1
- data/lib/context_dev.rb +3 -0
- data/rbi/context_dev/client.rbi +4 -0
- data/rbi/context_dev/models/batch_get_results_response.rbi +71 -2
- data/rbi/context_dev/models/batch_submit_params.rbi +58 -592
- data/rbi/context_dev/models/brand_retrieve_response.rbi +80 -3
- data/rbi/context_dev/models/brand_retrieve_simplified_response.rbi +80 -3
- data/rbi/context_dev/models/brand_search_params.rbi +71 -2
- data/rbi/context_dev/models/news_search_params.rbi +1294 -0
- data/rbi/context_dev/models/news_search_response.rbi +489 -0
- data/rbi/context_dev/models/parse_handle_params.rbi +20 -316
- data/rbi/context_dev/models/person_enrich_response.rbi +87 -0
- data/rbi/context_dev/models/utility_prefetch_params.rbi +24 -16
- data/rbi/context_dev/models/utility_prefetch_response.rbi +8 -6
- data/rbi/context_dev/models/web_screenshot_params.rbi +23 -71
- data/rbi/context_dev/models/web_web_crawl_md_response.rbi +67 -0
- data/rbi/context_dev/models/web_web_scrape_html_params.rbi +20 -356
- data/rbi/context_dev/models/web_web_scrape_html_response.rbi +65 -0
- data/rbi/context_dev/models/web_web_scrape_images_params.rbi +16 -287
- data/rbi/context_dev/models/web_web_scrape_md_params.rbi +47 -551
- data/rbi/context_dev/models/web_web_scrape_md_response.rbi +80 -0
- data/rbi/context_dev/models.rbi +2 -0
- data/rbi/context_dev/resources/brand.rbi +14 -9
- data/rbi/context_dev/resources/news.rbi +46 -0
- data/rbi/context_dev/resources/parse.rbi +5 -21
- data/rbi/context_dev/resources/utility.rbi +8 -6
- data/rbi/context_dev/resources/web.rbi +34 -66
- data/sig/context_dev/client.rbs +2 -0
- data/sig/context_dev/models/batch_get_results_response.rbs +21 -0
- data/sig/context_dev/models/batch_submit_params.rbs +54 -144
- data/sig/context_dev/models/brand_retrieve_response.rbs +33 -3
- data/sig/context_dev/models/brand_retrieve_simplified_response.rbs +33 -3
- data/sig/context_dev/models/brand_search_params.rbs +38 -1
- data/sig/context_dev/models/news_search_params.rbs +532 -0
- data/sig/context_dev/models/news_search_response.rbs +206 -0
- data/sig/context_dev/models/parse_handle_params.rbs +25 -90
- data/sig/context_dev/models/person_enrich_response.rbs +31 -0
- data/sig/context_dev/models/utility_prefetch_params.rbs +2 -1
- data/sig/context_dev/models/utility_prefetch_response.rbs +2 -1
- data/sig/context_dev/models/web_screenshot_params.rbs +12 -18
- data/sig/context_dev/models/web_web_crawl_md_response.rbs +21 -0
- data/sig/context_dev/models/web_web_scrape_html_params.rbs +24 -94
- data/sig/context_dev/models/web_web_scrape_html_response.rbs +21 -0
- data/sig/context_dev/models/web_web_scrape_images_params.rbs +20 -72
- data/sig/context_dev/models/web_web_scrape_md_params.rbs +46 -148
- data/sig/context_dev/models/web_web_scrape_md_response.rbs +28 -0
- data/sig/context_dev/models.rbs +2 -0
- data/sig/context_dev/resources/brand.rbs +3 -0
- data/sig/context_dev/resources/news.rbs +17 -0
- data/sig/context_dev/resources/parse.rbs +5 -5
- data/sig/context_dev/resources/web.rbs +13 -11
- metadata +11 -2
|
@@ -67,27 +67,10 @@ module ContextDev
|
|
|
67
67
|
attr_writer :headers
|
|
68
68
|
|
|
69
69
|
# When true, iframes are rendered inline into the returned HTML.
|
|
70
|
-
sig
|
|
71
|
-
returns(
|
|
72
|
-
T.nilable(
|
|
73
|
-
T.any(
|
|
74
|
-
T::Boolean,
|
|
75
|
-
ContextDev::WebWebScrapeHTMLParams::IncludeFrames::OrSymbol
|
|
76
|
-
)
|
|
77
|
-
)
|
|
78
|
-
)
|
|
79
|
-
end
|
|
70
|
+
sig { returns(T.nilable(T::Boolean)) }
|
|
80
71
|
attr_reader :include_frames
|
|
81
72
|
|
|
82
|
-
sig
|
|
83
|
-
params(
|
|
84
|
-
include_frames:
|
|
85
|
-
T.any(
|
|
86
|
-
T::Boolean,
|
|
87
|
-
ContextDev::WebWebScrapeHTMLParams::IncludeFrames::OrSymbol
|
|
88
|
-
)
|
|
89
|
-
).void
|
|
90
|
-
end
|
|
73
|
+
sig { params(include_frames: T::Boolean).void }
|
|
91
74
|
attr_writer :include_frames
|
|
92
75
|
|
|
93
76
|
# CSS selectors. When provided, only matching subtrees (and their descendants) are
|
|
@@ -113,27 +96,10 @@ module ContextDev
|
|
|
113
96
|
# When true, waits briefly for CSS and transition animations to settle before
|
|
114
97
|
# extracting HTML. Defaults to false. This adds a bit of latency in exchange for
|
|
115
98
|
# more stable output on animated pages.
|
|
116
|
-
sig
|
|
117
|
-
returns(
|
|
118
|
-
T.nilable(
|
|
119
|
-
T.any(
|
|
120
|
-
T::Boolean,
|
|
121
|
-
ContextDev::WebWebScrapeHTMLParams::SettleAnimations::OrSymbol
|
|
122
|
-
)
|
|
123
|
-
)
|
|
124
|
-
)
|
|
125
|
-
end
|
|
99
|
+
sig { returns(T.nilable(T::Boolean)) }
|
|
126
100
|
attr_reader :settle_animations
|
|
127
101
|
|
|
128
|
-
sig
|
|
129
|
-
params(
|
|
130
|
-
settle_animations:
|
|
131
|
-
T.any(
|
|
132
|
-
T::Boolean,
|
|
133
|
-
ContextDev::WebWebScrapeHTMLParams::SettleAnimations::OrSymbol
|
|
134
|
-
)
|
|
135
|
-
).void
|
|
136
|
-
end
|
|
102
|
+
sig { params(settle_animations: T::Boolean).void }
|
|
137
103
|
attr_writer :settle_animations
|
|
138
104
|
|
|
139
105
|
# Optional comma-separated caller-defined tags for tracking this request. Tags are
|
|
@@ -156,27 +122,10 @@ module ContextDev
|
|
|
156
122
|
|
|
157
123
|
# When true, return only the page's main content in the HTML response, excluding
|
|
158
124
|
# headers, footers, sidebars, and navigation when detectable.
|
|
159
|
-
sig
|
|
160
|
-
returns(
|
|
161
|
-
T.nilable(
|
|
162
|
-
T.any(
|
|
163
|
-
T::Boolean,
|
|
164
|
-
ContextDev::WebWebScrapeHTMLParams::UseMainContentOnly::OrSymbol
|
|
165
|
-
)
|
|
166
|
-
)
|
|
167
|
-
)
|
|
168
|
-
end
|
|
125
|
+
sig { returns(T.nilable(T::Boolean)) }
|
|
169
126
|
attr_reader :use_main_content_only
|
|
170
127
|
|
|
171
|
-
sig
|
|
172
|
-
params(
|
|
173
|
-
use_main_content_only:
|
|
174
|
-
T.any(
|
|
175
|
-
T::Boolean,
|
|
176
|
-
ContextDev::WebWebScrapeHTMLParams::UseMainContentOnly::OrSymbol
|
|
177
|
-
)
|
|
178
|
-
).void
|
|
179
|
-
end
|
|
128
|
+
sig { params(use_main_content_only: T::Boolean).void }
|
|
180
129
|
attr_writer :use_main_content_only
|
|
181
130
|
|
|
182
131
|
# Optional browser wait time in milliseconds after initial page load. Min: 0. Max:
|
|
@@ -213,26 +162,14 @@ module ContextDev
|
|
|
213
162
|
country: ContextDev::WebWebScrapeHTMLParams::Country::OrSymbol,
|
|
214
163
|
exclude_selectors: T.nilable(T::Array[String]),
|
|
215
164
|
headers: T::Hash[Symbol, String],
|
|
216
|
-
include_frames:
|
|
217
|
-
T.any(
|
|
218
|
-
T::Boolean,
|
|
219
|
-
ContextDev::WebWebScrapeHTMLParams::IncludeFrames::OrSymbol
|
|
220
|
-
),
|
|
165
|
+
include_frames: T::Boolean,
|
|
221
166
|
include_selectors: T.nilable(T::Array[String]),
|
|
222
167
|
max_age_ms: T.nilable(Integer),
|
|
223
168
|
pdf: ContextDev::WebWebScrapeHTMLParams::Pdf::OrHash,
|
|
224
|
-
settle_animations:
|
|
225
|
-
T.any(
|
|
226
|
-
T::Boolean,
|
|
227
|
-
ContextDev::WebWebScrapeHTMLParams::SettleAnimations::OrSymbol
|
|
228
|
-
),
|
|
169
|
+
settle_animations: T::Boolean,
|
|
229
170
|
tags: T::Array[String],
|
|
230
171
|
timeout_ms: Integer,
|
|
231
|
-
use_main_content_only:
|
|
232
|
-
T.any(
|
|
233
|
-
T::Boolean,
|
|
234
|
-
ContextDev::WebWebScrapeHTMLParams::UseMainContentOnly::OrSymbol
|
|
235
|
-
),
|
|
172
|
+
use_main_content_only: T::Boolean,
|
|
236
173
|
wait_for_ms: T.nilable(Integer),
|
|
237
174
|
zdr: ContextDev::WebWebScrapeHTMLParams::Zdr::OrSymbol,
|
|
238
175
|
request_options: ContextDev::RequestOptions::OrHash
|
|
@@ -312,26 +249,14 @@ module ContextDev
|
|
|
312
249
|
country: ContextDev::WebWebScrapeHTMLParams::Country::OrSymbol,
|
|
313
250
|
exclude_selectors: T.nilable(T::Array[String]),
|
|
314
251
|
headers: T::Hash[Symbol, String],
|
|
315
|
-
include_frames:
|
|
316
|
-
T.any(
|
|
317
|
-
T::Boolean,
|
|
318
|
-
ContextDev::WebWebScrapeHTMLParams::IncludeFrames::OrSymbol
|
|
319
|
-
),
|
|
252
|
+
include_frames: T::Boolean,
|
|
320
253
|
include_selectors: T.nilable(T::Array[String]),
|
|
321
254
|
max_age_ms: T.nilable(Integer),
|
|
322
255
|
pdf: ContextDev::WebWebScrapeHTMLParams::Pdf,
|
|
323
|
-
settle_animations:
|
|
324
|
-
T.any(
|
|
325
|
-
T::Boolean,
|
|
326
|
-
ContextDev::WebWebScrapeHTMLParams::SettleAnimations::OrSymbol
|
|
327
|
-
),
|
|
256
|
+
settle_animations: T::Boolean,
|
|
328
257
|
tags: T::Array[String],
|
|
329
258
|
timeout_ms: Integer,
|
|
330
|
-
use_main_content_only:
|
|
331
|
-
T.any(
|
|
332
|
-
T::Boolean,
|
|
333
|
-
ContextDev::WebWebScrapeHTMLParams::UseMainContentOnly::OrSymbol
|
|
334
|
-
),
|
|
259
|
+
use_main_content_only: T::Boolean,
|
|
335
260
|
wait_for_ms: T.nilable(Integer),
|
|
336
261
|
zdr: ContextDev::WebWebScrapeHTMLParams::Zdr::OrSymbol,
|
|
337
262
|
request_options: ContextDev::RequestOptions
|
|
@@ -844,46 +769,6 @@ module ContextDev
|
|
|
844
769
|
end
|
|
845
770
|
end
|
|
846
771
|
|
|
847
|
-
# When true, iframes are rendered inline into the returned HTML.
|
|
848
|
-
module IncludeFrames
|
|
849
|
-
extend ContextDev::Internal::Type::Union
|
|
850
|
-
|
|
851
|
-
Variants =
|
|
852
|
-
T.type_alias do
|
|
853
|
-
T.any(
|
|
854
|
-
T::Boolean,
|
|
855
|
-
ContextDev::WebWebScrapeHTMLParams::IncludeFrames::TaggedSymbol
|
|
856
|
-
)
|
|
857
|
-
end
|
|
858
|
-
|
|
859
|
-
sig do
|
|
860
|
-
override.returns(
|
|
861
|
-
T::Array[
|
|
862
|
-
ContextDev::WebWebScrapeHTMLParams::IncludeFrames::Variants
|
|
863
|
-
]
|
|
864
|
-
)
|
|
865
|
-
end
|
|
866
|
-
def self.variants
|
|
867
|
-
end
|
|
868
|
-
|
|
869
|
-
TaggedSymbol =
|
|
870
|
-
T.type_alias do
|
|
871
|
-
T.all(Symbol, ContextDev::WebWebScrapeHTMLParams::IncludeFrames)
|
|
872
|
-
end
|
|
873
|
-
OrSymbol = T.type_alias { T.any(Symbol, String) }
|
|
874
|
-
|
|
875
|
-
TRUE =
|
|
876
|
-
T.let(
|
|
877
|
-
:true,
|
|
878
|
-
ContextDev::WebWebScrapeHTMLParams::IncludeFrames::TaggedSymbol
|
|
879
|
-
)
|
|
880
|
-
FALSE =
|
|
881
|
-
T.let(
|
|
882
|
-
:false,
|
|
883
|
-
ContextDev::WebWebScrapeHTMLParams::IncludeFrames::TaggedSymbol
|
|
884
|
-
)
|
|
885
|
-
end
|
|
886
|
-
|
|
887
772
|
class Pdf < ContextDev::Internal::Type::BaseModel
|
|
888
773
|
OrHash =
|
|
889
774
|
T.type_alias do
|
|
@@ -905,52 +790,18 @@ module ContextDev
|
|
|
905
790
|
# replacing each recovered page's text with the OCR result while pages with a real
|
|
906
791
|
# text layer keep it. Billed at 1 credit per page OCR actually recovered, on top
|
|
907
792
|
# of the base request cost. When false, no OCR runs.
|
|
908
|
-
sig
|
|
909
|
-
returns(
|
|
910
|
-
T.nilable(
|
|
911
|
-
T.any(
|
|
912
|
-
T::Boolean,
|
|
913
|
-
ContextDev::WebWebScrapeHTMLParams::Pdf::Ocr::OrSymbol
|
|
914
|
-
)
|
|
915
|
-
)
|
|
916
|
-
)
|
|
917
|
-
end
|
|
793
|
+
sig { returns(T.nilable(T::Boolean)) }
|
|
918
794
|
attr_reader :ocr
|
|
919
795
|
|
|
920
|
-
sig
|
|
921
|
-
params(
|
|
922
|
-
ocr:
|
|
923
|
-
T.any(
|
|
924
|
-
T::Boolean,
|
|
925
|
-
ContextDev::WebWebScrapeHTMLParams::Pdf::Ocr::OrSymbol
|
|
926
|
-
)
|
|
927
|
-
).void
|
|
928
|
-
end
|
|
796
|
+
sig { params(ocr: T::Boolean).void }
|
|
929
797
|
attr_writer :ocr
|
|
930
798
|
|
|
931
799
|
# When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
|
|
932
800
|
# a 400 PDF_SKIPPED is returned.
|
|
933
|
-
sig
|
|
934
|
-
returns(
|
|
935
|
-
T.nilable(
|
|
936
|
-
T.any(
|
|
937
|
-
T::Boolean,
|
|
938
|
-
ContextDev::WebWebScrapeHTMLParams::Pdf::ShouldParse::OrSymbol
|
|
939
|
-
)
|
|
940
|
-
)
|
|
941
|
-
)
|
|
942
|
-
end
|
|
801
|
+
sig { returns(T.nilable(T::Boolean)) }
|
|
943
802
|
attr_reader :should_parse
|
|
944
803
|
|
|
945
|
-
sig
|
|
946
|
-
params(
|
|
947
|
-
should_parse:
|
|
948
|
-
T.any(
|
|
949
|
-
T::Boolean,
|
|
950
|
-
ContextDev::WebWebScrapeHTMLParams::Pdf::ShouldParse::OrSymbol
|
|
951
|
-
)
|
|
952
|
-
).void
|
|
953
|
-
end
|
|
804
|
+
sig { params(should_parse: T::Boolean).void }
|
|
954
805
|
attr_writer :should_parse
|
|
955
806
|
|
|
956
807
|
# First 1-based PDF page to parse. When omitted, parsing starts at the first page.
|
|
@@ -965,16 +816,8 @@ module ContextDev
|
|
|
965
816
|
sig do
|
|
966
817
|
params(
|
|
967
818
|
end_: Integer,
|
|
968
|
-
ocr:
|
|
969
|
-
|
|
970
|
-
T::Boolean,
|
|
971
|
-
ContextDev::WebWebScrapeHTMLParams::Pdf::Ocr::OrSymbol
|
|
972
|
-
),
|
|
973
|
-
should_parse:
|
|
974
|
-
T.any(
|
|
975
|
-
T::Boolean,
|
|
976
|
-
ContextDev::WebWebScrapeHTMLParams::Pdf::ShouldParse::OrSymbol
|
|
977
|
-
),
|
|
819
|
+
ocr: T::Boolean,
|
|
820
|
+
should_parse: T::Boolean,
|
|
978
821
|
start: Integer
|
|
979
822
|
).returns(T.attached_class)
|
|
980
823
|
end
|
|
@@ -999,193 +842,14 @@ module ContextDev
|
|
|
999
842
|
override.returns(
|
|
1000
843
|
{
|
|
1001
844
|
end_: Integer,
|
|
1002
|
-
ocr:
|
|
1003
|
-
|
|
1004
|
-
T::Boolean,
|
|
1005
|
-
ContextDev::WebWebScrapeHTMLParams::Pdf::Ocr::OrSymbol
|
|
1006
|
-
),
|
|
1007
|
-
should_parse:
|
|
1008
|
-
T.any(
|
|
1009
|
-
T::Boolean,
|
|
1010
|
-
ContextDev::WebWebScrapeHTMLParams::Pdf::ShouldParse::OrSymbol
|
|
1011
|
-
),
|
|
845
|
+
ocr: T::Boolean,
|
|
846
|
+
should_parse: T::Boolean,
|
|
1012
847
|
start: Integer
|
|
1013
848
|
}
|
|
1014
849
|
)
|
|
1015
850
|
end
|
|
1016
851
|
def to_hash
|
|
1017
852
|
end
|
|
1018
|
-
|
|
1019
|
-
# When true, OCR the selected PDF pages that have no usable text layer (scans),
|
|
1020
|
-
# replacing each recovered page's text with the OCR result while pages with a real
|
|
1021
|
-
# text layer keep it. Billed at 1 credit per page OCR actually recovered, on top
|
|
1022
|
-
# of the base request cost. When false, no OCR runs.
|
|
1023
|
-
module Ocr
|
|
1024
|
-
extend ContextDev::Internal::Type::Union
|
|
1025
|
-
|
|
1026
|
-
Variants =
|
|
1027
|
-
T.type_alias do
|
|
1028
|
-
T.any(
|
|
1029
|
-
T::Boolean,
|
|
1030
|
-
ContextDev::WebWebScrapeHTMLParams::Pdf::Ocr::TaggedSymbol
|
|
1031
|
-
)
|
|
1032
|
-
end
|
|
1033
|
-
|
|
1034
|
-
sig do
|
|
1035
|
-
override.returns(
|
|
1036
|
-
T::Array[ContextDev::WebWebScrapeHTMLParams::Pdf::Ocr::Variants]
|
|
1037
|
-
)
|
|
1038
|
-
end
|
|
1039
|
-
def self.variants
|
|
1040
|
-
end
|
|
1041
|
-
|
|
1042
|
-
TaggedSymbol =
|
|
1043
|
-
T.type_alias do
|
|
1044
|
-
T.all(Symbol, ContextDev::WebWebScrapeHTMLParams::Pdf::Ocr)
|
|
1045
|
-
end
|
|
1046
|
-
OrSymbol = T.type_alias { T.any(Symbol, String) }
|
|
1047
|
-
|
|
1048
|
-
TRUE =
|
|
1049
|
-
T.let(
|
|
1050
|
-
:true,
|
|
1051
|
-
ContextDev::WebWebScrapeHTMLParams::Pdf::Ocr::TaggedSymbol
|
|
1052
|
-
)
|
|
1053
|
-
FALSE =
|
|
1054
|
-
T.let(
|
|
1055
|
-
:false,
|
|
1056
|
-
ContextDev::WebWebScrapeHTMLParams::Pdf::Ocr::TaggedSymbol
|
|
1057
|
-
)
|
|
1058
|
-
end
|
|
1059
|
-
|
|
1060
|
-
# When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
|
|
1061
|
-
# a 400 PDF_SKIPPED is returned.
|
|
1062
|
-
module ShouldParse
|
|
1063
|
-
extend ContextDev::Internal::Type::Union
|
|
1064
|
-
|
|
1065
|
-
Variants =
|
|
1066
|
-
T.type_alias do
|
|
1067
|
-
T.any(
|
|
1068
|
-
T::Boolean,
|
|
1069
|
-
ContextDev::WebWebScrapeHTMLParams::Pdf::ShouldParse::TaggedSymbol
|
|
1070
|
-
)
|
|
1071
|
-
end
|
|
1072
|
-
|
|
1073
|
-
sig do
|
|
1074
|
-
override.returns(
|
|
1075
|
-
T::Array[
|
|
1076
|
-
ContextDev::WebWebScrapeHTMLParams::Pdf::ShouldParse::Variants
|
|
1077
|
-
]
|
|
1078
|
-
)
|
|
1079
|
-
end
|
|
1080
|
-
def self.variants
|
|
1081
|
-
end
|
|
1082
|
-
|
|
1083
|
-
TaggedSymbol =
|
|
1084
|
-
T.type_alias do
|
|
1085
|
-
T.all(
|
|
1086
|
-
Symbol,
|
|
1087
|
-
ContextDev::WebWebScrapeHTMLParams::Pdf::ShouldParse
|
|
1088
|
-
)
|
|
1089
|
-
end
|
|
1090
|
-
OrSymbol = T.type_alias { T.any(Symbol, String) }
|
|
1091
|
-
|
|
1092
|
-
TRUE =
|
|
1093
|
-
T.let(
|
|
1094
|
-
:true,
|
|
1095
|
-
ContextDev::WebWebScrapeHTMLParams::Pdf::ShouldParse::TaggedSymbol
|
|
1096
|
-
)
|
|
1097
|
-
FALSE =
|
|
1098
|
-
T.let(
|
|
1099
|
-
:false,
|
|
1100
|
-
ContextDev::WebWebScrapeHTMLParams::Pdf::ShouldParse::TaggedSymbol
|
|
1101
|
-
)
|
|
1102
|
-
end
|
|
1103
|
-
end
|
|
1104
|
-
|
|
1105
|
-
# When true, waits briefly for CSS and transition animations to settle before
|
|
1106
|
-
# extracting HTML. Defaults to false. This adds a bit of latency in exchange for
|
|
1107
|
-
# more stable output on animated pages.
|
|
1108
|
-
module SettleAnimations
|
|
1109
|
-
extend ContextDev::Internal::Type::Union
|
|
1110
|
-
|
|
1111
|
-
Variants =
|
|
1112
|
-
T.type_alias do
|
|
1113
|
-
T.any(
|
|
1114
|
-
T::Boolean,
|
|
1115
|
-
ContextDev::WebWebScrapeHTMLParams::SettleAnimations::TaggedSymbol
|
|
1116
|
-
)
|
|
1117
|
-
end
|
|
1118
|
-
|
|
1119
|
-
sig do
|
|
1120
|
-
override.returns(
|
|
1121
|
-
T::Array[
|
|
1122
|
-
ContextDev::WebWebScrapeHTMLParams::SettleAnimations::Variants
|
|
1123
|
-
]
|
|
1124
|
-
)
|
|
1125
|
-
end
|
|
1126
|
-
def self.variants
|
|
1127
|
-
end
|
|
1128
|
-
|
|
1129
|
-
TaggedSymbol =
|
|
1130
|
-
T.type_alias do
|
|
1131
|
-
T.all(Symbol, ContextDev::WebWebScrapeHTMLParams::SettleAnimations)
|
|
1132
|
-
end
|
|
1133
|
-
OrSymbol = T.type_alias { T.any(Symbol, String) }
|
|
1134
|
-
|
|
1135
|
-
TRUE =
|
|
1136
|
-
T.let(
|
|
1137
|
-
:true,
|
|
1138
|
-
ContextDev::WebWebScrapeHTMLParams::SettleAnimations::TaggedSymbol
|
|
1139
|
-
)
|
|
1140
|
-
FALSE =
|
|
1141
|
-
T.let(
|
|
1142
|
-
:false,
|
|
1143
|
-
ContextDev::WebWebScrapeHTMLParams::SettleAnimations::TaggedSymbol
|
|
1144
|
-
)
|
|
1145
|
-
end
|
|
1146
|
-
|
|
1147
|
-
# When true, return only the page's main content in the HTML response, excluding
|
|
1148
|
-
# headers, footers, sidebars, and navigation when detectable.
|
|
1149
|
-
module UseMainContentOnly
|
|
1150
|
-
extend ContextDev::Internal::Type::Union
|
|
1151
|
-
|
|
1152
|
-
Variants =
|
|
1153
|
-
T.type_alias do
|
|
1154
|
-
T.any(
|
|
1155
|
-
T::Boolean,
|
|
1156
|
-
ContextDev::WebWebScrapeHTMLParams::UseMainContentOnly::TaggedSymbol
|
|
1157
|
-
)
|
|
1158
|
-
end
|
|
1159
|
-
|
|
1160
|
-
sig do
|
|
1161
|
-
override.returns(
|
|
1162
|
-
T::Array[
|
|
1163
|
-
ContextDev::WebWebScrapeHTMLParams::UseMainContentOnly::Variants
|
|
1164
|
-
]
|
|
1165
|
-
)
|
|
1166
|
-
end
|
|
1167
|
-
def self.variants
|
|
1168
|
-
end
|
|
1169
|
-
|
|
1170
|
-
TaggedSymbol =
|
|
1171
|
-
T.type_alias do
|
|
1172
|
-
T.all(
|
|
1173
|
-
Symbol,
|
|
1174
|
-
ContextDev::WebWebScrapeHTMLParams::UseMainContentOnly
|
|
1175
|
-
)
|
|
1176
|
-
end
|
|
1177
|
-
OrSymbol = T.type_alias { T.any(Symbol, String) }
|
|
1178
|
-
|
|
1179
|
-
TRUE =
|
|
1180
|
-
T.let(
|
|
1181
|
-
:true,
|
|
1182
|
-
ContextDev::WebWebScrapeHTMLParams::UseMainContentOnly::TaggedSymbol
|
|
1183
|
-
)
|
|
1184
|
-
FALSE =
|
|
1185
|
-
T.let(
|
|
1186
|
-
:false,
|
|
1187
|
-
ContextDev::WebWebScrapeHTMLParams::UseMainContentOnly::TaggedSymbol
|
|
1188
|
-
)
|
|
1189
853
|
end
|
|
1190
854
|
|
|
1191
855
|
# Set to enabled to bypass shared caches and omit request and response content
|
|
@@ -259,6 +259,29 @@ module ContextDev
|
|
|
259
259
|
sig { params(favicon: String).void }
|
|
260
260
|
attr_writer :favicon
|
|
261
261
|
|
|
262
|
+
# Page headings (h1–h6) in document order, extracted from the unfiltered document.
|
|
263
|
+
# Capped at the first 500 headings. Omitted when the page has none.
|
|
264
|
+
sig do
|
|
265
|
+
returns(
|
|
266
|
+
T.nilable(
|
|
267
|
+
T::Array[
|
|
268
|
+
ContextDev::Models::WebWebScrapeHTMLResponse::Metadata::Heading
|
|
269
|
+
]
|
|
270
|
+
)
|
|
271
|
+
)
|
|
272
|
+
end
|
|
273
|
+
attr_reader :headings
|
|
274
|
+
|
|
275
|
+
sig do
|
|
276
|
+
params(
|
|
277
|
+
headings:
|
|
278
|
+
T::Array[
|
|
279
|
+
ContextDev::Models::WebWebScrapeHTMLResponse::Metadata::Heading::OrHash
|
|
280
|
+
]
|
|
281
|
+
).void
|
|
282
|
+
end
|
|
283
|
+
attr_writer :headings
|
|
284
|
+
|
|
262
285
|
# Primary resolved preview image from Open Graph, Twitter, or image metadata.
|
|
263
286
|
sig { returns(T.nilable(String)) }
|
|
264
287
|
attr_reader :image
|
|
@@ -388,6 +411,10 @@ module ContextDev
|
|
|
388
411
|
canonical_url: String,
|
|
389
412
|
description: String,
|
|
390
413
|
favicon: String,
|
|
414
|
+
headings:
|
|
415
|
+
T::Array[
|
|
416
|
+
ContextDev::Models::WebWebScrapeHTMLResponse::Metadata::Heading::OrHash
|
|
417
|
+
],
|
|
391
418
|
image: String,
|
|
392
419
|
json_ld: T::Array[T::Hash[Symbol, T.anything]],
|
|
393
420
|
keywords: T::Array[String],
|
|
@@ -427,6 +454,9 @@ module ContextDev
|
|
|
427
454
|
description: nil,
|
|
428
455
|
# Resolved favicon URL, when present.
|
|
429
456
|
favicon: nil,
|
|
457
|
+
# Page headings (h1–h6) in document order, extracted from the unfiltered document.
|
|
458
|
+
# Capped at the first 500 headings. Omitted when the page has none.
|
|
459
|
+
headings: nil,
|
|
430
460
|
# Primary resolved preview image from Open Graph, Twitter, or image metadata.
|
|
431
461
|
image: nil,
|
|
432
462
|
# JSON-LD structured data blocks parsed from the page.
|
|
@@ -470,6 +500,10 @@ module ContextDev
|
|
|
470
500
|
canonical_url: String,
|
|
471
501
|
description: String,
|
|
472
502
|
favicon: String,
|
|
503
|
+
headings:
|
|
504
|
+
T::Array[
|
|
505
|
+
ContextDev::Models::WebWebScrapeHTMLResponse::Metadata::Heading
|
|
506
|
+
],
|
|
473
507
|
image: String,
|
|
474
508
|
json_ld: T::Array[T::Hash[Symbol, T.anything]],
|
|
475
509
|
keywords: T::Array[String],
|
|
@@ -580,6 +614,37 @@ module ContextDev
|
|
|
580
614
|
end
|
|
581
615
|
end
|
|
582
616
|
|
|
617
|
+
class Heading < ContextDev::Internal::Type::BaseModel
|
|
618
|
+
OrHash =
|
|
619
|
+
T.type_alias do
|
|
620
|
+
T.any(
|
|
621
|
+
ContextDev::Models::WebWebScrapeHTMLResponse::Metadata::Heading,
|
|
622
|
+
ContextDev::Internal::AnyHash
|
|
623
|
+
)
|
|
624
|
+
end
|
|
625
|
+
|
|
626
|
+
# Heading level, 1–6 (from h1–h6).
|
|
627
|
+
sig { returns(Integer) }
|
|
628
|
+
attr_accessor :level
|
|
629
|
+
|
|
630
|
+
# Heading text with whitespace collapsed, truncated to 1000 characters.
|
|
631
|
+
sig { returns(String) }
|
|
632
|
+
attr_accessor :text
|
|
633
|
+
|
|
634
|
+
sig { params(level: Integer, text: String).returns(T.attached_class) }
|
|
635
|
+
def self.new(
|
|
636
|
+
# Heading level, 1–6 (from h1–h6).
|
|
637
|
+
level:,
|
|
638
|
+
# Heading text with whitespace collapsed, truncated to 1000 characters.
|
|
639
|
+
text:
|
|
640
|
+
)
|
|
641
|
+
end
|
|
642
|
+
|
|
643
|
+
sig { override.returns({ level: Integer, text: String }) }
|
|
644
|
+
def to_hash
|
|
645
|
+
end
|
|
646
|
+
end
|
|
647
|
+
|
|
583
648
|
module OpenGraph
|
|
584
649
|
extend ContextDev::Internal::Type::Union
|
|
585
650
|
|