context.dev 2.9.0 → 2.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +23 -0
- data/README.md +1 -1
- data/lib/context_dev/client.rb +5 -0
- data/lib/context_dev/models/batch_get_results_response.rb +33 -3
- data/lib/context_dev/models/batch_submit_params.rb +44 -316
- data/lib/context_dev/models/brand_retrieve_response.rb +29 -1
- data/lib/context_dev/models/brand_retrieve_simplified_response.rb +30 -1
- data/lib/context_dev/models/brand_search_params.rb +41 -3
- data/lib/context_dev/models/news_search_params.rb +467 -0
- data/lib/context_dev/models/news_search_response.rb +284 -0
- data/lib/context_dev/models/parse_handle_params.rb +15 -144
- data/lib/context_dev/models/person_enrich_response.rb +61 -1
- data/lib/context_dev/models/utility_prefetch_params.rb +19 -16
- data/lib/context_dev/models/utility_prefetch_response.rb +4 -5
- data/lib/context_dev/models/web_screenshot_params.rb +16 -31
- data/lib/context_dev/models/web_web_crawl_md_response.rb +30 -1
- data/lib/context_dev/models/web_web_scrape_html_params.rb +15 -153
- data/lib/context_dev/models/web_web_scrape_html_response.rb +30 -1
- data/lib/context_dev/models/web_web_scrape_images_params.rb +12 -124
- data/lib/context_dev/models/web_web_scrape_md_params.rb +35 -238
- data/lib/context_dev/models/web_web_scrape_md_response.rb +41 -2
- data/lib/context_dev/models.rb +2 -0
- data/lib/context_dev/resources/brand.rb +10 -11
- data/lib/context_dev/resources/news.rb +51 -0
- data/lib/context_dev/resources/parse.rb +5 -5
- data/lib/context_dev/resources/utility.rb +7 -6
- data/lib/context_dev/resources/web.rb +30 -24
- data/lib/context_dev/version.rb +1 -1
- data/lib/context_dev.rb +3 -0
- data/rbi/context_dev/client.rbi +4 -0
- data/rbi/context_dev/models/batch_get_results_response.rbi +71 -2
- data/rbi/context_dev/models/batch_submit_params.rbi +58 -592
- data/rbi/context_dev/models/brand_retrieve_response.rbi +80 -3
- data/rbi/context_dev/models/brand_retrieve_simplified_response.rbi +80 -3
- data/rbi/context_dev/models/brand_search_params.rbi +71 -2
- data/rbi/context_dev/models/news_search_params.rbi +1294 -0
- data/rbi/context_dev/models/news_search_response.rbi +489 -0
- data/rbi/context_dev/models/parse_handle_params.rbi +20 -316
- data/rbi/context_dev/models/person_enrich_response.rbi +87 -0
- data/rbi/context_dev/models/utility_prefetch_params.rbi +24 -16
- data/rbi/context_dev/models/utility_prefetch_response.rbi +8 -6
- data/rbi/context_dev/models/web_screenshot_params.rbi +23 -71
- data/rbi/context_dev/models/web_web_crawl_md_response.rbi +67 -0
- data/rbi/context_dev/models/web_web_scrape_html_params.rbi +20 -356
- data/rbi/context_dev/models/web_web_scrape_html_response.rbi +65 -0
- data/rbi/context_dev/models/web_web_scrape_images_params.rbi +16 -287
- data/rbi/context_dev/models/web_web_scrape_md_params.rbi +47 -551
- data/rbi/context_dev/models/web_web_scrape_md_response.rbi +80 -0
- data/rbi/context_dev/models.rbi +2 -0
- data/rbi/context_dev/resources/brand.rbi +14 -9
- data/rbi/context_dev/resources/news.rbi +46 -0
- data/rbi/context_dev/resources/parse.rbi +5 -21
- data/rbi/context_dev/resources/utility.rbi +8 -6
- data/rbi/context_dev/resources/web.rbi +34 -66
- data/sig/context_dev/client.rbs +2 -0
- data/sig/context_dev/models/batch_get_results_response.rbs +21 -0
- data/sig/context_dev/models/batch_submit_params.rbs +54 -144
- data/sig/context_dev/models/brand_retrieve_response.rbs +33 -3
- data/sig/context_dev/models/brand_retrieve_simplified_response.rbs +33 -3
- data/sig/context_dev/models/brand_search_params.rbs +38 -1
- data/sig/context_dev/models/news_search_params.rbs +532 -0
- data/sig/context_dev/models/news_search_response.rbs +206 -0
- data/sig/context_dev/models/parse_handle_params.rbs +25 -90
- data/sig/context_dev/models/person_enrich_response.rbs +31 -0
- data/sig/context_dev/models/utility_prefetch_params.rbs +2 -1
- data/sig/context_dev/models/utility_prefetch_response.rbs +2 -1
- data/sig/context_dev/models/web_screenshot_params.rbs +12 -18
- data/sig/context_dev/models/web_web_crawl_md_response.rbs +21 -0
- data/sig/context_dev/models/web_web_scrape_html_params.rbs +24 -94
- data/sig/context_dev/models/web_web_scrape_html_response.rbs +21 -0
- data/sig/context_dev/models/web_web_scrape_images_params.rbs +20 -72
- data/sig/context_dev/models/web_web_scrape_md_params.rbs +46 -148
- data/sig/context_dev/models/web_web_scrape_md_response.rbs +28 -0
- data/sig/context_dev/models.rbs +2 -0
- data/sig/context_dev/resources/brand.rbs +3 -0
- data/sig/context_dev/resources/news.rbs +17 -0
- data/sig/context_dev/resources/parse.rbs +5 -5
- data/sig/context_dev/resources/web.rbs +13 -11
- metadata +11 -2
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
SHA256:
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
3
|
+
metadata.gz: 366622603821e809f6f9eedf42ab688325b500d9dcae8e9cea5ff059edf08d53
|
|
4
|
+
data.tar.gz: deb733e2b4bb0a4def45544a438a143f799bb74e1cd51412fd4881d7f4cd83fc
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: fc24b4ae331e3ac3097c98f525783b79062620de5c4bd761569358ffe8a33b092da8a102baf2e777e0912d56fa981d11aa066daae83024e50d8684c5d07225d5
|
|
7
|
+
data.tar.gz: 2ccea91ca85c5a826930ceaae09cd8eb04cfef0dc8cdac14c7b616007fa0a1fff6d0079993d192781afdcdd23f8558030faf5b287a2952fe5438b611f65ae476
|
data/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,28 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## 2.11.0 (2026-08-18)
|
|
4
|
+
|
|
5
|
+
Full Changelog: [v2.10.0...v2.11.0](https://github.com/context-dot-dev/context-ruby-sdk/compare/v2.10.0...v2.11.0)
|
|
6
|
+
|
|
7
|
+
### Features
|
|
8
|
+
|
|
9
|
+
* **api:** api update ([9bff49f](https://github.com/context-dot-dev/context-ruby-sdk/commit/9bff49f9a966fd40691f2cd7a41ba96ec7679243))
|
|
10
|
+
* **api:** api update ([5b19e0c](https://github.com/context-dot-dev/context-ruby-sdk/commit/5b19e0ce1f7ad615a92e28440bec6ab3703ffe02))
|
|
11
|
+
* **api:** api update ([f4ccf22](https://github.com/context-dot-dev/context-ruby-sdk/commit/f4ccf220f130672950d3231a6b9952c166168ef9))
|
|
12
|
+
|
|
13
|
+
## 2.10.0 (2026-08-17)
|
|
14
|
+
|
|
15
|
+
Full Changelog: [v2.9.0...v2.10.0](https://github.com/context-dot-dev/context-ruby-sdk/compare/v2.9.0...v2.10.0)
|
|
16
|
+
|
|
17
|
+
### Features
|
|
18
|
+
|
|
19
|
+
* **api:** api update ([6f90ebf](https://github.com/context-dot-dev/context-ruby-sdk/commit/6f90ebff6164a8daf2f6c4bf8d1492ade1ad2779))
|
|
20
|
+
* **api:** api update ([95cbdde](https://github.com/context-dot-dev/context-ruby-sdk/commit/95cbddeb3a3351a03c394709dfeaf3fb5c98be92))
|
|
21
|
+
* **api:** api update ([97520df](https://github.com/context-dot-dev/context-ruby-sdk/commit/97520df52a798122460e2f219d7b582d2f6830da))
|
|
22
|
+
* **api:** api update ([2b73772](https://github.com/context-dot-dev/context-ruby-sdk/commit/2b737722e962cf7a5aba20a2b596d379f7f2bce0))
|
|
23
|
+
* **api:** api update ([630b415](https://github.com/context-dot-dev/context-ruby-sdk/commit/630b4156f12004d2da154e26003271cd143ccbb8))
|
|
24
|
+
* **api:** manual updates ([54ad450](https://github.com/context-dot-dev/context-ruby-sdk/commit/54ad450f892542378cec0ce5430d1436e1437a4d))
|
|
25
|
+
|
|
3
26
|
## 2.9.0 (2026-08-07)
|
|
4
27
|
|
|
5
28
|
Full Changelog: [v2.8.0...v2.9.0](https://github.com/context-dot-dev/context-ruby-sdk/compare/v2.8.0...v2.9.0)
|
data/README.md
CHANGED
data/lib/context_dev/client.rb
CHANGED
|
@@ -50,6 +50,10 @@ module ContextDev
|
|
|
50
50
|
# @return [ContextDev::Resources::People]
|
|
51
51
|
attr_reader :people
|
|
52
52
|
|
|
53
|
+
# Search live first-party RSS and free historical news data by company identity.
|
|
54
|
+
# @return [ContextDev::Resources::News]
|
|
55
|
+
attr_reader :news
|
|
56
|
+
|
|
53
57
|
# @api private
|
|
54
58
|
#
|
|
55
59
|
# @return [Hash{String=>String}]
|
|
@@ -120,6 +124,7 @@ module ContextDev
|
|
|
120
124
|
@monitors = ContextDev::Resources::Monitors.new(client: self)
|
|
121
125
|
@batch = ContextDev::Resources::Batch.new(client: self)
|
|
122
126
|
@people = ContextDev::Resources::People.new(client: self)
|
|
127
|
+
@news = ContextDev::Resources::News.new(client: self)
|
|
123
128
|
end
|
|
124
129
|
end
|
|
125
130
|
end
|
|
@@ -86,7 +86,8 @@ module ContextDev
|
|
|
86
86
|
required :url, String
|
|
87
87
|
|
|
88
88
|
# @!attribute html
|
|
89
|
-
#
|
|
89
|
+
# Page HTML. Present on html batches, and on markdown batches submitted with
|
|
90
|
+
# `options.includeHTML`.
|
|
90
91
|
#
|
|
91
92
|
# @return [String, nil]
|
|
92
93
|
optional :html, String
|
|
@@ -130,7 +131,7 @@ module ContextDev
|
|
|
130
131
|
#
|
|
131
132
|
# @param url [String] URL as submitted, or as discovered by the crawl.
|
|
132
133
|
#
|
|
133
|
-
# @param html [String]
|
|
134
|
+
# @param html [String] Page HTML. Present on html batches, and on markdown batches submitted with `opti
|
|
134
135
|
#
|
|
135
136
|
# @param item_id [String] Caller-supplied identifier echoed from submission.
|
|
136
137
|
#
|
|
@@ -196,6 +197,14 @@ module ContextDev
|
|
|
196
197
|
# @return [String, nil]
|
|
197
198
|
optional :favicon, String
|
|
198
199
|
|
|
200
|
+
# @!attribute headings
|
|
201
|
+
# Page headings (h1–h6) in document order, extracted from the unfiltered document.
|
|
202
|
+
# Capped at the first 500 headings. Omitted when the page has none.
|
|
203
|
+
#
|
|
204
|
+
# @return [Array<ContextDev::Models::BatchGetResultsResponse::Data::Ok::Metadata::Heading>, nil]
|
|
205
|
+
optional :headings,
|
|
206
|
+
-> { ContextDev::Internal::Type::ArrayOf[ContextDev::Models::BatchGetResultsResponse::Data::Ok::Metadata::Heading] }
|
|
207
|
+
|
|
199
208
|
# @!attribute image
|
|
200
209
|
# Primary resolved preview image from Open Graph, Twitter, or image metadata.
|
|
201
210
|
#
|
|
@@ -267,7 +276,7 @@ module ContextDev
|
|
|
267
276
|
optional :twitter,
|
|
268
277
|
-> { ContextDev::Internal::Type::HashOf[union: ContextDev::Models::BatchGetResultsResponse::Data::Ok::Metadata::Twitter] }
|
|
269
278
|
|
|
270
|
-
# @!method initialize(final_url:, source_url:, additional_meta: nil, alternates: nil, author: nil, canonical_url: nil, description: nil, favicon: nil, image: nil, json_ld: nil, keywords: nil, language: nil, modified_time: nil, open_graph: nil, published_time: nil, robots: nil, site_name: nil, title: nil, twitter: nil)
|
|
279
|
+
# @!method initialize(final_url:, source_url:, additional_meta: nil, alternates: nil, author: nil, canonical_url: nil, description: nil, favicon: nil, headings: nil, image: nil, json_ld: nil, keywords: nil, language: nil, modified_time: nil, open_graph: nil, published_time: nil, robots: nil, site_name: nil, title: nil, twitter: nil)
|
|
271
280
|
# Some parameter documentations has been truncated, see
|
|
272
281
|
# {ContextDev::Models::BatchGetResultsResponse::Data::Ok::Metadata} for more
|
|
273
282
|
# details.
|
|
@@ -290,6 +299,8 @@ module ContextDev
|
|
|
290
299
|
#
|
|
291
300
|
# @param favicon [String] Resolved favicon URL, when present.
|
|
292
301
|
#
|
|
302
|
+
# @param headings [Array<ContextDev::Models::BatchGetResultsResponse::Data::Ok::Metadata::Heading>] Page headings (h1–h6) in document order, extracted from the unfiltered document.
|
|
303
|
+
#
|
|
293
304
|
# @param image [String] Primary resolved preview image from Open Graph, Twitter, or image metadata.
|
|
294
305
|
#
|
|
295
306
|
# @param json_ld [Array<Hash{Symbol=>Object}>] JSON-LD structured data blocks parsed from the page.
|
|
@@ -361,6 +372,25 @@ module ContextDev
|
|
|
361
372
|
# @param type [String] Alternate resource MIME type, when present.
|
|
362
373
|
end
|
|
363
374
|
|
|
375
|
+
class Heading < ContextDev::Internal::Type::BaseModel
|
|
376
|
+
# @!attribute level
|
|
377
|
+
# Heading level, 1–6 (from h1–h6).
|
|
378
|
+
#
|
|
379
|
+
# @return [Integer]
|
|
380
|
+
required :level, Integer
|
|
381
|
+
|
|
382
|
+
# @!attribute text
|
|
383
|
+
# Heading text with whitespace collapsed, truncated to 1000 characters.
|
|
384
|
+
#
|
|
385
|
+
# @return [String]
|
|
386
|
+
required :text, String
|
|
387
|
+
|
|
388
|
+
# @!method initialize(level:, text:)
|
|
389
|
+
# @param level [Integer] Heading level, 1–6 (from h1–h6).
|
|
390
|
+
#
|
|
391
|
+
# @param text [String] Heading text with whitespace collapsed, truncated to 1000 characters.
|
|
392
|
+
end
|
|
393
|
+
|
|
364
394
|
module OpenGraph
|
|
365
395
|
extend ContextDev::Internal::Type::Union
|
|
366
396
|
|
|
@@ -175,6 +175,13 @@ module ContextDev
|
|
|
175
175
|
api_name: :excludeSelectors,
|
|
176
176
|
nil?: true
|
|
177
177
|
|
|
178
|
+
# @!attribute include_html
|
|
179
|
+
# Also include each page's HTML in its result record, as an `html` field alongside
|
|
180
|
+
# the Markdown.
|
|
181
|
+
#
|
|
182
|
+
# @return [Boolean, nil]
|
|
183
|
+
optional :include_html, ContextDev::Internal::Type::Boolean, api_name: :includeHTML
|
|
184
|
+
|
|
178
185
|
# @!attribute include_images
|
|
179
186
|
# Include image references in the Markdown.
|
|
180
187
|
#
|
|
@@ -241,7 +248,7 @@ module ContextDev
|
|
|
241
248
|
# @return [Integer, nil]
|
|
242
249
|
optional :wait_for_ms, Integer, api_name: :waitForMs
|
|
243
250
|
|
|
244
|
-
# @!method initialize(country: nil, exclude_selectors: nil, include_images: nil, include_links: nil, include_selectors: nil, max_age_ms: nil, pdf: nil, settle_animations: nil, shorten_base64_images: nil, use_main_content_only: nil, wait_for_ms: nil)
|
|
251
|
+
# @!method initialize(country: nil, exclude_selectors: nil, include_html: nil, include_images: nil, include_links: nil, include_selectors: nil, max_age_ms: nil, pdf: nil, settle_animations: nil, shorten_base64_images: nil, use_main_content_only: nil, wait_for_ms: nil)
|
|
245
252
|
# Some parameter documentations has been truncated, see
|
|
246
253
|
# {ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::Options}
|
|
247
254
|
# for more details.
|
|
@@ -252,6 +259,8 @@ module ContextDev
|
|
|
252
259
|
#
|
|
253
260
|
# @param exclude_selectors [Array<String>, nil] Remove elements matching these CSS selectors. Applied after `includeSelectors`,
|
|
254
261
|
#
|
|
262
|
+
# @param include_html [Boolean] Also include each page's HTML in its result record, as an `html` field alongside
|
|
263
|
+
#
|
|
255
264
|
# @param include_images [Boolean] Include image references in the Markdown.
|
|
256
265
|
#
|
|
257
266
|
# @param include_links [Boolean] Include links in the Markdown.
|
|
@@ -501,20 +510,15 @@ module ContextDev
|
|
|
501
510
|
# text layer keep it. Billed at 1 credit per page OCR actually recovered, on top
|
|
502
511
|
# of the base request cost. When false, no OCR runs.
|
|
503
512
|
#
|
|
504
|
-
# @return [Boolean,
|
|
505
|
-
optional :ocr,
|
|
506
|
-
union: -> { ContextDev::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf::Ocr }
|
|
513
|
+
# @return [Boolean, nil]
|
|
514
|
+
optional :ocr, ContextDev::Internal::Type::Boolean
|
|
507
515
|
|
|
508
516
|
# @!attribute should_parse
|
|
509
517
|
# When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
|
|
510
518
|
# a 400 PDF_SKIPPED is returned.
|
|
511
519
|
#
|
|
512
|
-
# @return [Boolean,
|
|
513
|
-
optional :should_parse,
|
|
514
|
-
union: -> {
|
|
515
|
-
ContextDev::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf::ShouldParse
|
|
516
|
-
},
|
|
517
|
-
api_name: :shouldParse
|
|
520
|
+
# @return [Boolean, nil]
|
|
521
|
+
optional :should_parse, ContextDev::Internal::Type::Boolean, api_name: :shouldParse
|
|
518
522
|
|
|
519
523
|
# @!attribute start
|
|
520
524
|
# First 1-based PDF page to parse. When omitted, parsing starts at the first page.
|
|
@@ -532,79 +536,11 @@ module ContextDev
|
|
|
532
536
|
#
|
|
533
537
|
# @param end_ [Integer] Last 1-based PDF page to parse. When omitted, parsing ends at the last page. Mus
|
|
534
538
|
#
|
|
535
|
-
# @param ocr [Boolean
|
|
539
|
+
# @param ocr [Boolean] When true, OCR the selected PDF pages that have no usable text layer (scans), re
|
|
536
540
|
#
|
|
537
|
-
# @param should_parse [Boolean
|
|
541
|
+
# @param should_parse [Boolean] When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
|
|
538
542
|
#
|
|
539
543
|
# @param start [Integer] First 1-based PDF page to parse. When omitted, parsing starts at the first page.
|
|
540
|
-
|
|
541
|
-
# When true, OCR the selected PDF pages that have no usable text layer (scans),
|
|
542
|
-
# replacing each recovered page's text with the OCR result while pages with a real
|
|
543
|
-
# text layer keep it. Billed at 1 credit per page OCR actually recovered, on top
|
|
544
|
-
# of the base request cost. When false, no OCR runs.
|
|
545
|
-
#
|
|
546
|
-
# @see ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf#ocr
|
|
547
|
-
module Ocr
|
|
548
|
-
extend ContextDev::Internal::Type::Union
|
|
549
|
-
|
|
550
|
-
variant ContextDev::Internal::Type::Boolean
|
|
551
|
-
|
|
552
|
-
variant const: -> { ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf::Ocr::TRUE }
|
|
553
|
-
|
|
554
|
-
variant const: -> { ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf::Ocr::FALSE }
|
|
555
|
-
|
|
556
|
-
# @!method self.variants
|
|
557
|
-
# @return [Array(Boolean, Symbol)]
|
|
558
|
-
|
|
559
|
-
define_sorbet_constant!(:Variants) do
|
|
560
|
-
T.type_alias do
|
|
561
|
-
T.any(
|
|
562
|
-
T::Boolean,
|
|
563
|
-
ContextDev::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf::Ocr::TaggedSymbol
|
|
564
|
-
)
|
|
565
|
-
end
|
|
566
|
-
end
|
|
567
|
-
|
|
568
|
-
# @!group
|
|
569
|
-
|
|
570
|
-
TRUE = :true
|
|
571
|
-
FALSE = :false
|
|
572
|
-
|
|
573
|
-
# @!endgroup
|
|
574
|
-
end
|
|
575
|
-
|
|
576
|
-
# When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
|
|
577
|
-
# a 400 PDF_SKIPPED is returned.
|
|
578
|
-
#
|
|
579
|
-
# @see ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf#should_parse
|
|
580
|
-
module ShouldParse
|
|
581
|
-
extend ContextDev::Internal::Type::Union
|
|
582
|
-
|
|
583
|
-
variant ContextDev::Internal::Type::Boolean
|
|
584
|
-
|
|
585
|
-
variant const: -> { ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf::ShouldParse::TRUE }
|
|
586
|
-
|
|
587
|
-
variant const: -> { ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf::ShouldParse::FALSE }
|
|
588
|
-
|
|
589
|
-
# @!method self.variants
|
|
590
|
-
# @return [Array(Boolean, Symbol)]
|
|
591
|
-
|
|
592
|
-
define_sorbet_constant!(:Variants) do
|
|
593
|
-
T.type_alias do
|
|
594
|
-
T.any(
|
|
595
|
-
T::Boolean,
|
|
596
|
-
ContextDev::BatchSubmitParams::Input::Scrape::Data::Markdown::Options::Pdf::ShouldParse::TaggedSymbol
|
|
597
|
-
)
|
|
598
|
-
end
|
|
599
|
-
end
|
|
600
|
-
|
|
601
|
-
# @!group
|
|
602
|
-
|
|
603
|
-
TRUE = :true
|
|
604
|
-
FALSE = :false
|
|
605
|
-
|
|
606
|
-
# @!endgroup
|
|
607
|
-
end
|
|
608
544
|
end
|
|
609
545
|
end
|
|
610
546
|
end
|
|
@@ -991,19 +927,15 @@ module ContextDev
|
|
|
991
927
|
# text layer keep it. Billed at 1 credit per page OCR actually recovered, on top
|
|
992
928
|
# of the base request cost. When false, no OCR runs.
|
|
993
929
|
#
|
|
994
|
-
# @return [Boolean,
|
|
995
|
-
optional :ocr,
|
|
930
|
+
# @return [Boolean, nil]
|
|
931
|
+
optional :ocr, ContextDev::Internal::Type::Boolean
|
|
996
932
|
|
|
997
933
|
# @!attribute should_parse
|
|
998
934
|
# When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
|
|
999
935
|
# a 400 PDF_SKIPPED is returned.
|
|
1000
936
|
#
|
|
1001
|
-
# @return [Boolean,
|
|
1002
|
-
optional :should_parse,
|
|
1003
|
-
union: -> {
|
|
1004
|
-
ContextDev::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf::ShouldParse
|
|
1005
|
-
},
|
|
1006
|
-
api_name: :shouldParse
|
|
937
|
+
# @return [Boolean, nil]
|
|
938
|
+
optional :should_parse, ContextDev::Internal::Type::Boolean, api_name: :shouldParse
|
|
1007
939
|
|
|
1008
940
|
# @!attribute start
|
|
1009
941
|
# First 1-based PDF page to parse. When omitted, parsing starts at the first page.
|
|
@@ -1021,79 +953,11 @@ module ContextDev
|
|
|
1021
953
|
#
|
|
1022
954
|
# @param end_ [Integer] Last 1-based PDF page to parse. When omitted, parsing ends at the last page. Mus
|
|
1023
955
|
#
|
|
1024
|
-
# @param ocr [Boolean
|
|
956
|
+
# @param ocr [Boolean] When true, OCR the selected PDF pages that have no usable text layer (scans), re
|
|
1025
957
|
#
|
|
1026
|
-
# @param should_parse [Boolean
|
|
958
|
+
# @param should_parse [Boolean] When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
|
|
1027
959
|
#
|
|
1028
960
|
# @param start [Integer] First 1-based PDF page to parse. When omitted, parsing starts at the first page.
|
|
1029
|
-
|
|
1030
|
-
# When true, OCR the selected PDF pages that have no usable text layer (scans),
|
|
1031
|
-
# replacing each recovered page's text with the OCR result while pages with a real
|
|
1032
|
-
# text layer keep it. Billed at 1 credit per page OCR actually recovered, on top
|
|
1033
|
-
# of the base request cost. When false, no OCR runs.
|
|
1034
|
-
#
|
|
1035
|
-
# @see ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf#ocr
|
|
1036
|
-
module Ocr
|
|
1037
|
-
extend ContextDev::Internal::Type::Union
|
|
1038
|
-
|
|
1039
|
-
variant ContextDev::Internal::Type::Boolean
|
|
1040
|
-
|
|
1041
|
-
variant const: -> { ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf::Ocr::TRUE }
|
|
1042
|
-
|
|
1043
|
-
variant const: -> { ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf::Ocr::FALSE }
|
|
1044
|
-
|
|
1045
|
-
# @!method self.variants
|
|
1046
|
-
# @return [Array(Boolean, Symbol)]
|
|
1047
|
-
|
|
1048
|
-
define_sorbet_constant!(:Variants) do
|
|
1049
|
-
T.type_alias do
|
|
1050
|
-
T.any(
|
|
1051
|
-
T::Boolean,
|
|
1052
|
-
ContextDev::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf::Ocr::TaggedSymbol
|
|
1053
|
-
)
|
|
1054
|
-
end
|
|
1055
|
-
end
|
|
1056
|
-
|
|
1057
|
-
# @!group
|
|
1058
|
-
|
|
1059
|
-
TRUE = :true
|
|
1060
|
-
FALSE = :false
|
|
1061
|
-
|
|
1062
|
-
# @!endgroup
|
|
1063
|
-
end
|
|
1064
|
-
|
|
1065
|
-
# When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
|
|
1066
|
-
# a 400 PDF_SKIPPED is returned.
|
|
1067
|
-
#
|
|
1068
|
-
# @see ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf#should_parse
|
|
1069
|
-
module ShouldParse
|
|
1070
|
-
extend ContextDev::Internal::Type::Union
|
|
1071
|
-
|
|
1072
|
-
variant ContextDev::Internal::Type::Boolean
|
|
1073
|
-
|
|
1074
|
-
variant const: -> { ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf::ShouldParse::TRUE }
|
|
1075
|
-
|
|
1076
|
-
variant const: -> { ContextDev::Models::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf::ShouldParse::FALSE }
|
|
1077
|
-
|
|
1078
|
-
# @!method self.variants
|
|
1079
|
-
# @return [Array(Boolean, Symbol)]
|
|
1080
|
-
|
|
1081
|
-
define_sorbet_constant!(:Variants) do
|
|
1082
|
-
T.type_alias do
|
|
1083
|
-
T.any(
|
|
1084
|
-
T::Boolean,
|
|
1085
|
-
ContextDev::BatchSubmitParams::Input::Scrape::Data::HTML::Options::Pdf::ShouldParse::TaggedSymbol
|
|
1086
|
-
)
|
|
1087
|
-
end
|
|
1088
|
-
end
|
|
1089
|
-
|
|
1090
|
-
# @!group
|
|
1091
|
-
|
|
1092
|
-
TRUE = :true
|
|
1093
|
-
FALSE = :false
|
|
1094
|
-
|
|
1095
|
-
# @!endgroup
|
|
1096
|
-
end
|
|
1097
961
|
end
|
|
1098
962
|
end
|
|
1099
963
|
end
|
|
@@ -1333,6 +1197,13 @@ module ContextDev
|
|
|
1333
1197
|
api_name: :excludeSelectors,
|
|
1334
1198
|
nil?: true
|
|
1335
1199
|
|
|
1200
|
+
# @!attribute include_html
|
|
1201
|
+
# Also include each page's HTML in its result record, as an `html` field alongside
|
|
1202
|
+
# the Markdown.
|
|
1203
|
+
#
|
|
1204
|
+
# @return [Boolean, nil]
|
|
1205
|
+
optional :include_html, ContextDev::Internal::Type::Boolean, api_name: :includeHTML
|
|
1206
|
+
|
|
1336
1207
|
# @!attribute include_images
|
|
1337
1208
|
# Include image references in the Markdown.
|
|
1338
1209
|
#
|
|
@@ -1399,7 +1270,7 @@ module ContextDev
|
|
|
1399
1270
|
# @return [Integer, nil]
|
|
1400
1271
|
optional :wait_for_ms, Integer, api_name: :waitForMs
|
|
1401
1272
|
|
|
1402
|
-
# @!method initialize(country: nil, exclude_selectors: nil, include_images: nil, include_links: nil, include_selectors: nil, max_age_ms: nil, pdf: nil, settle_animations: nil, shorten_base64_images: nil, use_main_content_only: nil, wait_for_ms: nil)
|
|
1273
|
+
# @!method initialize(country: nil, exclude_selectors: nil, include_html: nil, include_images: nil, include_links: nil, include_selectors: nil, max_age_ms: nil, pdf: nil, settle_animations: nil, shorten_base64_images: nil, use_main_content_only: nil, wait_for_ms: nil)
|
|
1403
1274
|
# Some parameter documentations has been truncated, see
|
|
1404
1275
|
# {ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Options}
|
|
1405
1276
|
# for more details.
|
|
@@ -1410,6 +1281,8 @@ module ContextDev
|
|
|
1410
1281
|
#
|
|
1411
1282
|
# @param exclude_selectors [Array<String>, nil] Remove elements matching these CSS selectors. Applied after `includeSelectors`,
|
|
1412
1283
|
#
|
|
1284
|
+
# @param include_html [Boolean] Also include each page's HTML in its result record, as an `html` field alongside
|
|
1285
|
+
#
|
|
1413
1286
|
# @param include_images [Boolean] Include image references in the Markdown.
|
|
1414
1287
|
#
|
|
1415
1288
|
# @param include_links [Boolean] Include links in the Markdown.
|
|
@@ -1659,20 +1532,15 @@ module ContextDev
|
|
|
1659
1532
|
# text layer keep it. Billed at 1 credit per page OCR actually recovered, on top
|
|
1660
1533
|
# of the base request cost. When false, no OCR runs.
|
|
1661
1534
|
#
|
|
1662
|
-
# @return [Boolean,
|
|
1663
|
-
optional :ocr,
|
|
1664
|
-
union: -> { ContextDev::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf::Ocr }
|
|
1535
|
+
# @return [Boolean, nil]
|
|
1536
|
+
optional :ocr, ContextDev::Internal::Type::Boolean
|
|
1665
1537
|
|
|
1666
1538
|
# @!attribute should_parse
|
|
1667
1539
|
# When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
|
|
1668
1540
|
# a 400 PDF_SKIPPED is returned.
|
|
1669
1541
|
#
|
|
1670
|
-
# @return [Boolean,
|
|
1671
|
-
optional :should_parse,
|
|
1672
|
-
union: -> {
|
|
1673
|
-
ContextDev::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf::ShouldParse
|
|
1674
|
-
},
|
|
1675
|
-
api_name: :shouldParse
|
|
1542
|
+
# @return [Boolean, nil]
|
|
1543
|
+
optional :should_parse, ContextDev::Internal::Type::Boolean, api_name: :shouldParse
|
|
1676
1544
|
|
|
1677
1545
|
# @!attribute start
|
|
1678
1546
|
# First 1-based PDF page to parse. When omitted, parsing starts at the first page.
|
|
@@ -1690,79 +1558,11 @@ module ContextDev
|
|
|
1690
1558
|
#
|
|
1691
1559
|
# @param end_ [Integer] Last 1-based PDF page to parse. When omitted, parsing ends at the last page. Mus
|
|
1692
1560
|
#
|
|
1693
|
-
# @param ocr [Boolean
|
|
1561
|
+
# @param ocr [Boolean] When true, OCR the selected PDF pages that have no usable text layer (scans), re
|
|
1694
1562
|
#
|
|
1695
|
-
# @param should_parse [Boolean
|
|
1563
|
+
# @param should_parse [Boolean] When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
|
|
1696
1564
|
#
|
|
1697
1565
|
# @param start [Integer] First 1-based PDF page to parse. When omitted, parsing starts at the first page.
|
|
1698
|
-
|
|
1699
|
-
# When true, OCR the selected PDF pages that have no usable text layer (scans),
|
|
1700
|
-
# replacing each recovered page's text with the OCR result while pages with a real
|
|
1701
|
-
# text layer keep it. Billed at 1 credit per page OCR actually recovered, on top
|
|
1702
|
-
# of the base request cost. When false, no OCR runs.
|
|
1703
|
-
#
|
|
1704
|
-
# @see ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf#ocr
|
|
1705
|
-
module Ocr
|
|
1706
|
-
extend ContextDev::Internal::Type::Union
|
|
1707
|
-
|
|
1708
|
-
variant ContextDev::Internal::Type::Boolean
|
|
1709
|
-
|
|
1710
|
-
variant const: -> { ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf::Ocr::TRUE }
|
|
1711
|
-
|
|
1712
|
-
variant const: -> { ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf::Ocr::FALSE }
|
|
1713
|
-
|
|
1714
|
-
# @!method self.variants
|
|
1715
|
-
# @return [Array(Boolean, Symbol)]
|
|
1716
|
-
|
|
1717
|
-
define_sorbet_constant!(:Variants) do
|
|
1718
|
-
T.type_alias do
|
|
1719
|
-
T.any(
|
|
1720
|
-
T::Boolean,
|
|
1721
|
-
ContextDev::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf::Ocr::TaggedSymbol
|
|
1722
|
-
)
|
|
1723
|
-
end
|
|
1724
|
-
end
|
|
1725
|
-
|
|
1726
|
-
# @!group
|
|
1727
|
-
|
|
1728
|
-
TRUE = :true
|
|
1729
|
-
FALSE = :false
|
|
1730
|
-
|
|
1731
|
-
# @!endgroup
|
|
1732
|
-
end
|
|
1733
|
-
|
|
1734
|
-
# When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
|
|
1735
|
-
# a 400 PDF_SKIPPED is returned.
|
|
1736
|
-
#
|
|
1737
|
-
# @see ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf#should_parse
|
|
1738
|
-
module ShouldParse
|
|
1739
|
-
extend ContextDev::Internal::Type::Union
|
|
1740
|
-
|
|
1741
|
-
variant ContextDev::Internal::Type::Boolean
|
|
1742
|
-
|
|
1743
|
-
variant const: -> { ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf::ShouldParse::TRUE }
|
|
1744
|
-
|
|
1745
|
-
variant const: -> { ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf::ShouldParse::FALSE }
|
|
1746
|
-
|
|
1747
|
-
# @!method self.variants
|
|
1748
|
-
# @return [Array(Boolean, Symbol)]
|
|
1749
|
-
|
|
1750
|
-
define_sorbet_constant!(:Variants) do
|
|
1751
|
-
T.type_alias do
|
|
1752
|
-
T.any(
|
|
1753
|
-
T::Boolean,
|
|
1754
|
-
ContextDev::BatchSubmitParams::Input::Crawl::Data::Markdown::Options::Pdf::ShouldParse::TaggedSymbol
|
|
1755
|
-
)
|
|
1756
|
-
end
|
|
1757
|
-
end
|
|
1758
|
-
|
|
1759
|
-
# @!group
|
|
1760
|
-
|
|
1761
|
-
TRUE = :true
|
|
1762
|
-
FALSE = :false
|
|
1763
|
-
|
|
1764
|
-
# @!endgroup
|
|
1765
|
-
end
|
|
1766
1566
|
end
|
|
1767
1567
|
end
|
|
1768
1568
|
end
|
|
@@ -2262,19 +2062,15 @@ module ContextDev
|
|
|
2262
2062
|
# text layer keep it. Billed at 1 credit per page OCR actually recovered, on top
|
|
2263
2063
|
# of the base request cost. When false, no OCR runs.
|
|
2264
2064
|
#
|
|
2265
|
-
# @return [Boolean,
|
|
2266
|
-
optional :ocr,
|
|
2065
|
+
# @return [Boolean, nil]
|
|
2066
|
+
optional :ocr, ContextDev::Internal::Type::Boolean
|
|
2267
2067
|
|
|
2268
2068
|
# @!attribute should_parse
|
|
2269
2069
|
# When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
|
|
2270
2070
|
# a 400 PDF_SKIPPED is returned.
|
|
2271
2071
|
#
|
|
2272
|
-
# @return [Boolean,
|
|
2273
|
-
optional :should_parse,
|
|
2274
|
-
union: -> {
|
|
2275
|
-
ContextDev::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf::ShouldParse
|
|
2276
|
-
},
|
|
2277
|
-
api_name: :shouldParse
|
|
2072
|
+
# @return [Boolean, nil]
|
|
2073
|
+
optional :should_parse, ContextDev::Internal::Type::Boolean, api_name: :shouldParse
|
|
2278
2074
|
|
|
2279
2075
|
# @!attribute start
|
|
2280
2076
|
# First 1-based PDF page to parse. When omitted, parsing starts at the first page.
|
|
@@ -2292,79 +2088,11 @@ module ContextDev
|
|
|
2292
2088
|
#
|
|
2293
2089
|
# @param end_ [Integer] Last 1-based PDF page to parse. When omitted, parsing ends at the last page. Mus
|
|
2294
2090
|
#
|
|
2295
|
-
# @param ocr [Boolean
|
|
2091
|
+
# @param ocr [Boolean] When true, OCR the selected PDF pages that have no usable text layer (scans), re
|
|
2296
2092
|
#
|
|
2297
|
-
# @param should_parse [Boolean
|
|
2093
|
+
# @param should_parse [Boolean] When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
|
|
2298
2094
|
#
|
|
2299
2095
|
# @param start [Integer] First 1-based PDF page to parse. When omitted, parsing starts at the first page.
|
|
2300
|
-
|
|
2301
|
-
# When true, OCR the selected PDF pages that have no usable text layer (scans),
|
|
2302
|
-
# replacing each recovered page's text with the OCR result while pages with a real
|
|
2303
|
-
# text layer keep it. Billed at 1 credit per page OCR actually recovered, on top
|
|
2304
|
-
# of the base request cost. When false, no OCR runs.
|
|
2305
|
-
#
|
|
2306
|
-
# @see ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf#ocr
|
|
2307
|
-
module Ocr
|
|
2308
|
-
extend ContextDev::Internal::Type::Union
|
|
2309
|
-
|
|
2310
|
-
variant ContextDev::Internal::Type::Boolean
|
|
2311
|
-
|
|
2312
|
-
variant const: -> { ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf::Ocr::TRUE }
|
|
2313
|
-
|
|
2314
|
-
variant const: -> { ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf::Ocr::FALSE }
|
|
2315
|
-
|
|
2316
|
-
# @!method self.variants
|
|
2317
|
-
# @return [Array(Boolean, Symbol)]
|
|
2318
|
-
|
|
2319
|
-
define_sorbet_constant!(:Variants) do
|
|
2320
|
-
T.type_alias do
|
|
2321
|
-
T.any(
|
|
2322
|
-
T::Boolean,
|
|
2323
|
-
ContextDev::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf::Ocr::TaggedSymbol
|
|
2324
|
-
)
|
|
2325
|
-
end
|
|
2326
|
-
end
|
|
2327
|
-
|
|
2328
|
-
# @!group
|
|
2329
|
-
|
|
2330
|
-
TRUE = :true
|
|
2331
|
-
FALSE = :false
|
|
2332
|
-
|
|
2333
|
-
# @!endgroup
|
|
2334
|
-
end
|
|
2335
|
-
|
|
2336
|
-
# When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
|
|
2337
|
-
# a 400 PDF_SKIPPED is returned.
|
|
2338
|
-
#
|
|
2339
|
-
# @see ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf#should_parse
|
|
2340
|
-
module ShouldParse
|
|
2341
|
-
extend ContextDev::Internal::Type::Union
|
|
2342
|
-
|
|
2343
|
-
variant ContextDev::Internal::Type::Boolean
|
|
2344
|
-
|
|
2345
|
-
variant const: -> { ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf::ShouldParse::TRUE }
|
|
2346
|
-
|
|
2347
|
-
variant const: -> { ContextDev::Models::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf::ShouldParse::FALSE }
|
|
2348
|
-
|
|
2349
|
-
# @!method self.variants
|
|
2350
|
-
# @return [Array(Boolean, Symbol)]
|
|
2351
|
-
|
|
2352
|
-
define_sorbet_constant!(:Variants) do
|
|
2353
|
-
T.type_alias do
|
|
2354
|
-
T.any(
|
|
2355
|
-
T::Boolean,
|
|
2356
|
-
ContextDev::BatchSubmitParams::Input::Crawl::Data::HTML::Options::Pdf::ShouldParse::TaggedSymbol
|
|
2357
|
-
)
|
|
2358
|
-
end
|
|
2359
|
-
end
|
|
2360
|
-
|
|
2361
|
-
# @!group
|
|
2362
|
-
|
|
2363
|
-
TRUE = :true
|
|
2364
|
-
FALSE = :false
|
|
2365
|
-
|
|
2366
|
-
# @!endgroup
|
|
2367
|
-
end
|
|
2368
2096
|
end
|
|
2369
2097
|
end
|
|
2370
2098
|
end
|