context.dev 1.24.0 → 1.26.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +18 -0
- data/README.md +1 -1
- data/lib/context_dev/internal/type/base_model.rb +5 -5
- data/lib/context_dev/models/web_extract_params.rb +148 -0
- data/lib/context_dev/models/web_extract_response.rb +83 -0
- data/lib/context_dev/models.rb +2 -0
- data/lib/context_dev/resources/web.rb +46 -0
- data/lib/context_dev/version.rb +1 -1
- data/lib/context_dev.rb +2 -0
- data/rbi/context_dev/models/web_extract_params.rbi +232 -0
- data/rbi/context_dev/models/web_extract_response.rbi +134 -0
- data/rbi/context_dev/models.rbi +2 -0
- data/rbi/context_dev/resources/web.rbi +57 -0
- data/sig/context_dev/models/web_extract_params.rbs +120 -0
- data/sig/context_dev/models/web_extract_response.rbs +77 -0
- data/sig/context_dev/models.rbs +2 -0
- data/sig/context_dev/resources/web.rbs +15 -0
- metadata +8 -2
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
SHA256:
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
3
|
+
metadata.gz: bb1eb49bd2c33309910f98a48b129a07a32707a85e5a832277226ee555d22780
|
|
4
|
+
data.tar.gz: 4c7410af11f9f6f69145a70f37e1c613fe9a90d6b2ea99acc061c10ae9119716
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: 13e5b25d98899a354e90fdf559ea5b3ed8d1140a543f538a7909691a92ecb49917e87398f2da8b3759aad2751487c02e4b89e4738f15febe9bc10995825c4660
|
|
7
|
+
data.tar.gz: b82dee66dd6c01d5a4edb347097a978708714164cbe4460b7da394f0440b297afcc602bf52cc3cf597e3fe8638559b953bafe0288512e883f1d3b2957fc62cea
|
data/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,23 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## 1.26.0 (2026-06-01)
|
|
4
|
+
|
|
5
|
+
Full Changelog: [v1.25.0...v1.26.0](https://github.com/context-dot-dev/context-ruby-sdk/compare/v1.25.0...v1.26.0)
|
|
6
|
+
|
|
7
|
+
### Features
|
|
8
|
+
|
|
9
|
+
* **api:** api update ([d442168](https://github.com/context-dot-dev/context-ruby-sdk/commit/d442168b67accce12463e36f03630f155c1c7be3))
|
|
10
|
+
* **api:** api update ([f3a97c1](https://github.com/context-dot-dev/context-ruby-sdk/commit/f3a97c19ef01be5e6cf1ce7ecfa26c972edfdf33))
|
|
11
|
+
* **api:** api update ([8d8abf7](https://github.com/context-dot-dev/context-ruby-sdk/commit/8d8abf7c1600bce17ea26aa245f41b2e3199f722))
|
|
12
|
+
|
|
13
|
+
## 1.25.0 (2026-05-31)
|
|
14
|
+
|
|
15
|
+
Full Changelog: [v1.24.0...v1.25.0](https://github.com/context-dot-dev/context-ruby-sdk/compare/v1.24.0...v1.25.0)
|
|
16
|
+
|
|
17
|
+
### Features
|
|
18
|
+
|
|
19
|
+
* **api:** manual updates ([688e5f5](https://github.com/context-dot-dev/context-ruby-sdk/commit/688e5f56137aeac7c5219383f241a459173f961f))
|
|
20
|
+
|
|
3
21
|
## 1.24.0 (2026-05-30)
|
|
4
22
|
|
|
5
23
|
Full Changelog: [v1.23.0...v1.24.0](https://github.com/context-dot-dev/context-ruby-sdk/compare/v1.23.0...v1.24.0)
|
data/README.md
CHANGED
|
@@ -438,11 +438,11 @@ module ContextDev
|
|
|
438
438
|
# @return [Hash{Symbol=>Object}]
|
|
439
439
|
#
|
|
440
440
|
# @example
|
|
441
|
-
# # `
|
|
442
|
-
#
|
|
443
|
-
#
|
|
444
|
-
#
|
|
445
|
-
#
|
|
441
|
+
# # `web_extract_response` is a `ContextDev::Models::WebExtractResponse`
|
|
442
|
+
# web_extract_response => {
|
|
443
|
+
# data: data,
|
|
444
|
+
# metadata: metadata,
|
|
445
|
+
# status: status
|
|
446
446
|
# }
|
|
447
447
|
def deconstruct_keys(keys)
|
|
448
448
|
(keys || self.class.known_fields.keys)
|
|
@@ -0,0 +1,148 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module ContextDev
|
|
4
|
+
module Models
|
|
5
|
+
# @see ContextDev::Resources::Web#extract
|
|
6
|
+
class WebExtractParams < ContextDev::Internal::Type::BaseModel
|
|
7
|
+
extend ContextDev::Internal::Type::RequestParameters::Converter
|
|
8
|
+
include ContextDev::Internal::Type::RequestParameters
|
|
9
|
+
|
|
10
|
+
# @!attribute schema
|
|
11
|
+
# JSON Schema for the returned data object. TypeScript Zod users can pass a JSON
|
|
12
|
+
# Schema generated from a Zod object; Python users can pass the equivalent JSON
|
|
13
|
+
# Schema object.
|
|
14
|
+
#
|
|
15
|
+
# @return [Hash{Symbol=>Object}]
|
|
16
|
+
required :schema, ContextDev::Internal::Type::HashOf[ContextDev::Internal::Type::Unknown]
|
|
17
|
+
|
|
18
|
+
# @!attribute url
|
|
19
|
+
# The starting website URL to crawl and extract from. Must include http:// or
|
|
20
|
+
# https://.
|
|
21
|
+
#
|
|
22
|
+
# @return [String]
|
|
23
|
+
required :url, String
|
|
24
|
+
|
|
25
|
+
# @!attribute fact_check
|
|
26
|
+
# When true, every returned value must be grounded in facts stated on the page;
|
|
27
|
+
# fields that cannot be supported by the page are returned as null/empty. When
|
|
28
|
+
# false (default), the model may make reasonable inferences and derivations from
|
|
29
|
+
# the page content (e.g. ideal customer, competitor analysis, recommendations)
|
|
30
|
+
# while keeping verifiable specifics (names, quotes, URLs, dates, metrics)
|
|
31
|
+
# faithful to the source.
|
|
32
|
+
#
|
|
33
|
+
# @return [Boolean, nil]
|
|
34
|
+
optional :fact_check, ContextDev::Internal::Type::Boolean, api_name: :factCheck
|
|
35
|
+
|
|
36
|
+
# @!attribute follow_subdomains
|
|
37
|
+
# When true, follow links on subdomains of the starting URL's domain.
|
|
38
|
+
#
|
|
39
|
+
# @return [Boolean, nil]
|
|
40
|
+
optional :follow_subdomains, ContextDev::Internal::Type::Boolean, api_name: :followSubdomains
|
|
41
|
+
|
|
42
|
+
# @!attribute include_frames
|
|
43
|
+
# When true, iframe contents are included in Markdown before extraction.
|
|
44
|
+
#
|
|
45
|
+
# @return [Boolean, nil]
|
|
46
|
+
optional :include_frames, ContextDev::Internal::Type::Boolean, api_name: :includeFrames
|
|
47
|
+
|
|
48
|
+
# @!attribute instructions
|
|
49
|
+
# Optional extraction guidance, such as which facts to prioritize or how to
|
|
50
|
+
# interpret fields in the schema.
|
|
51
|
+
#
|
|
52
|
+
# @return [String, nil]
|
|
53
|
+
optional :instructions, String
|
|
54
|
+
|
|
55
|
+
# @!attribute max_age_ms
|
|
56
|
+
# Return cached scrape results if a prior scrape for the same parameters is
|
|
57
|
+
# younger than this many milliseconds. Defaults to 7 days (604800000 ms).
|
|
58
|
+
#
|
|
59
|
+
# @return [Integer, nil]
|
|
60
|
+
optional :max_age_ms, Integer, api_name: :maxAgeMs
|
|
61
|
+
|
|
62
|
+
# @!attribute pdf
|
|
63
|
+
#
|
|
64
|
+
# @return [ContextDev::Models::WebExtractParams::Pdf, nil]
|
|
65
|
+
optional :pdf, -> { ContextDev::WebExtractParams::Pdf }
|
|
66
|
+
|
|
67
|
+
# @!attribute stop_after_ms
|
|
68
|
+
# Soft time budget for the crawl in milliseconds.
|
|
69
|
+
#
|
|
70
|
+
# @return [Integer, nil]
|
|
71
|
+
optional :stop_after_ms, Integer, api_name: :stopAfterMs
|
|
72
|
+
|
|
73
|
+
# @!attribute timeout_ms
|
|
74
|
+
# Optional timeout in milliseconds for the request. If the request takes longer
|
|
75
|
+
# than this value, it will be aborted with a 408 status code. Maximum allowed
|
|
76
|
+
# value is 300000ms (5 minutes).
|
|
77
|
+
#
|
|
78
|
+
# @return [Integer, nil]
|
|
79
|
+
optional :timeout_ms, Integer, api_name: :timeoutMS
|
|
80
|
+
|
|
81
|
+
# @!attribute wait_for_ms
|
|
82
|
+
# Optional browser wait time in milliseconds after initial page load for each
|
|
83
|
+
# crawled page.
|
|
84
|
+
#
|
|
85
|
+
# @return [Integer, nil]
|
|
86
|
+
optional :wait_for_ms, Integer, api_name: :waitForMs
|
|
87
|
+
|
|
88
|
+
# @!method initialize(schema:, url:, fact_check: nil, follow_subdomains: nil, include_frames: nil, instructions: nil, max_age_ms: nil, pdf: nil, stop_after_ms: nil, timeout_ms: nil, wait_for_ms: nil, request_options: {})
|
|
89
|
+
# Some parameter documentations has been truncated, see
|
|
90
|
+
# {ContextDev::Models::WebExtractParams} for more details.
|
|
91
|
+
#
|
|
92
|
+
# @param schema [Hash{Symbol=>Object}] JSON Schema for the returned data object. TypeScript Zod users can pass a JSON S
|
|
93
|
+
#
|
|
94
|
+
# @param url [String] The starting website URL to crawl and extract from. Must include http:// or http
|
|
95
|
+
#
|
|
96
|
+
# @param fact_check [Boolean] When true, every returned value must be grounded in facts stated on the page; fi
|
|
97
|
+
#
|
|
98
|
+
# @param follow_subdomains [Boolean] When true, follow links on subdomains of the starting URL's domain.
|
|
99
|
+
#
|
|
100
|
+
# @param include_frames [Boolean] When true, iframe contents are included in Markdown before extraction.
|
|
101
|
+
#
|
|
102
|
+
# @param instructions [String] Optional extraction guidance, such as which facts to prioritize or how to interp
|
|
103
|
+
#
|
|
104
|
+
# @param max_age_ms [Integer] Return cached scrape results if a prior scrape for the same parameters is younge
|
|
105
|
+
#
|
|
106
|
+
# @param pdf [ContextDev::Models::WebExtractParams::Pdf]
|
|
107
|
+
#
|
|
108
|
+
# @param stop_after_ms [Integer] Soft time budget for the crawl in milliseconds.
|
|
109
|
+
#
|
|
110
|
+
# @param timeout_ms [Integer] Optional timeout in milliseconds for the request. If the request takes longer th
|
|
111
|
+
#
|
|
112
|
+
# @param wait_for_ms [Integer] Optional browser wait time in milliseconds after initial page load for each craw
|
|
113
|
+
#
|
|
114
|
+
# @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}]
|
|
115
|
+
|
|
116
|
+
class Pdf < ContextDev::Internal::Type::BaseModel
|
|
117
|
+
# @!attribute end_
|
|
118
|
+
# Last 1-based PDF page to parse. Must be greater than or equal to start when both
|
|
119
|
+
# are provided.
|
|
120
|
+
#
|
|
121
|
+
# @return [Integer, nil]
|
|
122
|
+
optional :end_, Integer, api_name: :end
|
|
123
|
+
|
|
124
|
+
# @!attribute should_parse
|
|
125
|
+
# When true, PDF pages are fetched and parsed. When false, PDF pages are skipped.
|
|
126
|
+
#
|
|
127
|
+
# @return [Boolean, nil]
|
|
128
|
+
optional :should_parse, ContextDev::Internal::Type::Boolean, api_name: :shouldParse
|
|
129
|
+
|
|
130
|
+
# @!attribute start
|
|
131
|
+
# First 1-based PDF page to parse.
|
|
132
|
+
#
|
|
133
|
+
# @return [Integer, nil]
|
|
134
|
+
optional :start, Integer
|
|
135
|
+
|
|
136
|
+
# @!method initialize(end_: nil, should_parse: nil, start: nil)
|
|
137
|
+
# Some parameter documentations has been truncated, see
|
|
138
|
+
# {ContextDev::Models::WebExtractParams::Pdf} for more details.
|
|
139
|
+
#
|
|
140
|
+
# @param end_ [Integer] Last 1-based PDF page to parse. Must be greater than or equal to start when both
|
|
141
|
+
#
|
|
142
|
+
# @param should_parse [Boolean] When true, PDF pages are fetched and parsed. When false, PDF pages are skipped.
|
|
143
|
+
#
|
|
144
|
+
# @param start [Integer] First 1-based PDF page to parse.
|
|
145
|
+
end
|
|
146
|
+
end
|
|
147
|
+
end
|
|
148
|
+
end
|
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module ContextDev
|
|
4
|
+
module Models
|
|
5
|
+
# @see ContextDev::Resources::Web#extract
|
|
6
|
+
class WebExtractResponse < ContextDev::Internal::Type::BaseModel
|
|
7
|
+
# @!attribute data
|
|
8
|
+
# Extracted data matching the request schema
|
|
9
|
+
#
|
|
10
|
+
# @return [Hash{Symbol=>Object}]
|
|
11
|
+
required :data, ContextDev::Internal::Type::HashOf[ContextDev::Internal::Type::Unknown]
|
|
12
|
+
|
|
13
|
+
# @!attribute metadata
|
|
14
|
+
#
|
|
15
|
+
# @return [ContextDev::Models::WebExtractResponse::Metadata]
|
|
16
|
+
required :metadata, -> { ContextDev::Models::WebExtractResponse::Metadata }
|
|
17
|
+
|
|
18
|
+
# @!attribute status
|
|
19
|
+
# Status of the response, e.g., 'ok'
|
|
20
|
+
#
|
|
21
|
+
# @return [String]
|
|
22
|
+
required :status, String
|
|
23
|
+
|
|
24
|
+
# @!attribute url
|
|
25
|
+
# The starting URL that was analyzed
|
|
26
|
+
#
|
|
27
|
+
# @return [String]
|
|
28
|
+
required :url, String
|
|
29
|
+
|
|
30
|
+
# @!attribute urls_analyzed
|
|
31
|
+
# List of URLs whose Markdown was used for extraction
|
|
32
|
+
#
|
|
33
|
+
# @return [Array<String>]
|
|
34
|
+
required :urls_analyzed, ContextDev::Internal::Type::ArrayOf[String]
|
|
35
|
+
|
|
36
|
+
# @!method initialize(data:, metadata:, status:, url:, urls_analyzed:)
|
|
37
|
+
# @param data [Hash{Symbol=>Object}] Extracted data matching the request schema
|
|
38
|
+
#
|
|
39
|
+
# @param metadata [ContextDev::Models::WebExtractResponse::Metadata]
|
|
40
|
+
#
|
|
41
|
+
# @param status [String] Status of the response, e.g., 'ok'
|
|
42
|
+
#
|
|
43
|
+
# @param url [String] The starting URL that was analyzed
|
|
44
|
+
#
|
|
45
|
+
# @param urls_analyzed [Array<String>] List of URLs whose Markdown was used for extraction
|
|
46
|
+
|
|
47
|
+
# @see ContextDev::Models::WebExtractResponse#metadata
|
|
48
|
+
class Metadata < ContextDev::Internal::Type::BaseModel
|
|
49
|
+
# @!attribute max_crawl_depth
|
|
50
|
+
#
|
|
51
|
+
# @return [Integer]
|
|
52
|
+
required :max_crawl_depth, Integer, api_name: :maxCrawlDepth
|
|
53
|
+
|
|
54
|
+
# @!attribute num_failed
|
|
55
|
+
#
|
|
56
|
+
# @return [Integer]
|
|
57
|
+
required :num_failed, Integer, api_name: :numFailed
|
|
58
|
+
|
|
59
|
+
# @!attribute num_skipped
|
|
60
|
+
#
|
|
61
|
+
# @return [Integer]
|
|
62
|
+
required :num_skipped, Integer, api_name: :numSkipped
|
|
63
|
+
|
|
64
|
+
# @!attribute num_succeeded
|
|
65
|
+
#
|
|
66
|
+
# @return [Integer]
|
|
67
|
+
required :num_succeeded, Integer, api_name: :numSucceeded
|
|
68
|
+
|
|
69
|
+
# @!attribute num_urls
|
|
70
|
+
#
|
|
71
|
+
# @return [Integer]
|
|
72
|
+
required :num_urls, Integer, api_name: :numUrls
|
|
73
|
+
|
|
74
|
+
# @!method initialize(max_crawl_depth:, num_failed:, num_skipped:, num_succeeded:, num_urls:)
|
|
75
|
+
# @param max_crawl_depth [Integer]
|
|
76
|
+
# @param num_failed [Integer]
|
|
77
|
+
# @param num_skipped [Integer]
|
|
78
|
+
# @param num_succeeded [Integer]
|
|
79
|
+
# @param num_urls [Integer]
|
|
80
|
+
end
|
|
81
|
+
end
|
|
82
|
+
end
|
|
83
|
+
end
|
data/lib/context_dev/models.rb
CHANGED
|
@@ -69,6 +69,8 @@ module ContextDev
|
|
|
69
69
|
|
|
70
70
|
WebExtractFontsParams = ContextDev::Models::WebExtractFontsParams
|
|
71
71
|
|
|
72
|
+
WebExtractParams = ContextDev::Models::WebExtractParams
|
|
73
|
+
|
|
72
74
|
WebExtractStyleguideParams = ContextDev::Models::WebExtractStyleguideParams
|
|
73
75
|
|
|
74
76
|
WebScreenshotParams = ContextDev::Models::WebScreenshotParams
|
|
@@ -3,6 +3,52 @@
|
|
|
3
3
|
module ContextDev
|
|
4
4
|
module Resources
|
|
5
5
|
class Web
|
|
6
|
+
# Some parameter documentations has been truncated, see
|
|
7
|
+
# {ContextDev::Models::WebExtractParams} for more details.
|
|
8
|
+
#
|
|
9
|
+
# Crawl a website, use the provided JSON Schema and instructions to prioritize
|
|
10
|
+
# relevant internal links, and extract structured data from the selected pages.
|
|
11
|
+
#
|
|
12
|
+
# @overload extract(schema:, url:, fact_check: nil, follow_subdomains: nil, include_frames: nil, instructions: nil, max_age_ms: nil, pdf: nil, stop_after_ms: nil, timeout_ms: nil, wait_for_ms: nil, request_options: {})
|
|
13
|
+
#
|
|
14
|
+
# @param schema [Hash{Symbol=>Object}] JSON Schema for the returned data object. TypeScript Zod users can pass a JSON S
|
|
15
|
+
#
|
|
16
|
+
# @param url [String] The starting website URL to crawl and extract from. Must include http:// or http
|
|
17
|
+
#
|
|
18
|
+
# @param fact_check [Boolean] When true, every returned value must be grounded in facts stated on the page; fi
|
|
19
|
+
#
|
|
20
|
+
# @param follow_subdomains [Boolean] When true, follow links on subdomains of the starting URL's domain.
|
|
21
|
+
#
|
|
22
|
+
# @param include_frames [Boolean] When true, iframe contents are included in Markdown before extraction.
|
|
23
|
+
#
|
|
24
|
+
# @param instructions [String] Optional extraction guidance, such as which facts to prioritize or how to interp
|
|
25
|
+
#
|
|
26
|
+
# @param max_age_ms [Integer] Return cached scrape results if a prior scrape for the same parameters is younge
|
|
27
|
+
#
|
|
28
|
+
# @param pdf [ContextDev::Models::WebExtractParams::Pdf]
|
|
29
|
+
#
|
|
30
|
+
# @param stop_after_ms [Integer] Soft time budget for the crawl in milliseconds.
|
|
31
|
+
#
|
|
32
|
+
# @param timeout_ms [Integer] Optional timeout in milliseconds for the request. If the request takes longer th
|
|
33
|
+
#
|
|
34
|
+
# @param wait_for_ms [Integer] Optional browser wait time in milliseconds after initial page load for each craw
|
|
35
|
+
#
|
|
36
|
+
# @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
|
|
37
|
+
#
|
|
38
|
+
# @return [ContextDev::Models::WebExtractResponse]
|
|
39
|
+
#
|
|
40
|
+
# @see ContextDev::Models::WebExtractParams
|
|
41
|
+
def extract(params)
|
|
42
|
+
parsed, options = ContextDev::WebExtractParams.dump_request(params)
|
|
43
|
+
@client.request(
|
|
44
|
+
method: :post,
|
|
45
|
+
path: "web/extract",
|
|
46
|
+
body: parsed,
|
|
47
|
+
model: ContextDev::Models::WebExtractResponse,
|
|
48
|
+
options: options
|
|
49
|
+
)
|
|
50
|
+
end
|
|
51
|
+
|
|
6
52
|
# Some parameter documentations has been truncated, see
|
|
7
53
|
# {ContextDev::Models::WebExtractFontsParams} for more details.
|
|
8
54
|
#
|
data/lib/context_dev/version.rb
CHANGED
data/lib/context_dev.rb
CHANGED
|
@@ -82,6 +82,8 @@ require_relative "context_dev/models/utility_prefetch_params"
|
|
|
82
82
|
require_relative "context_dev/models/utility_prefetch_response"
|
|
83
83
|
require_relative "context_dev/models/web_extract_fonts_params"
|
|
84
84
|
require_relative "context_dev/models/web_extract_fonts_response"
|
|
85
|
+
require_relative "context_dev/models/web_extract_params"
|
|
86
|
+
require_relative "context_dev/models/web_extract_response"
|
|
85
87
|
require_relative "context_dev/models/web_extract_styleguide_params"
|
|
86
88
|
require_relative "context_dev/models/web_extract_styleguide_response"
|
|
87
89
|
require_relative "context_dev/models/web_screenshot_params"
|
|
@@ -0,0 +1,232 @@
|
|
|
1
|
+
# typed: strong
|
|
2
|
+
|
|
3
|
+
module ContextDev
|
|
4
|
+
module Models
|
|
5
|
+
class WebExtractParams < ContextDev::Internal::Type::BaseModel
|
|
6
|
+
extend ContextDev::Internal::Type::RequestParameters::Converter
|
|
7
|
+
include ContextDev::Internal::Type::RequestParameters
|
|
8
|
+
|
|
9
|
+
OrHash =
|
|
10
|
+
T.type_alias do
|
|
11
|
+
T.any(ContextDev::WebExtractParams, ContextDev::Internal::AnyHash)
|
|
12
|
+
end
|
|
13
|
+
|
|
14
|
+
# JSON Schema for the returned data object. TypeScript Zod users can pass a JSON
|
|
15
|
+
# Schema generated from a Zod object; Python users can pass the equivalent JSON
|
|
16
|
+
# Schema object.
|
|
17
|
+
sig { returns(T::Hash[Symbol, T.anything]) }
|
|
18
|
+
attr_accessor :schema
|
|
19
|
+
|
|
20
|
+
# The starting website URL to crawl and extract from. Must include http:// or
|
|
21
|
+
# https://.
|
|
22
|
+
sig { returns(String) }
|
|
23
|
+
attr_accessor :url
|
|
24
|
+
|
|
25
|
+
# When true, every returned value must be grounded in facts stated on the page;
|
|
26
|
+
# fields that cannot be supported by the page are returned as null/empty. When
|
|
27
|
+
# false (default), the model may make reasonable inferences and derivations from
|
|
28
|
+
# the page content (e.g. ideal customer, competitor analysis, recommendations)
|
|
29
|
+
# while keeping verifiable specifics (names, quotes, URLs, dates, metrics)
|
|
30
|
+
# faithful to the source.
|
|
31
|
+
sig { returns(T.nilable(T::Boolean)) }
|
|
32
|
+
attr_reader :fact_check
|
|
33
|
+
|
|
34
|
+
sig { params(fact_check: T::Boolean).void }
|
|
35
|
+
attr_writer :fact_check
|
|
36
|
+
|
|
37
|
+
# When true, follow links on subdomains of the starting URL's domain.
|
|
38
|
+
sig { returns(T.nilable(T::Boolean)) }
|
|
39
|
+
attr_reader :follow_subdomains
|
|
40
|
+
|
|
41
|
+
sig { params(follow_subdomains: T::Boolean).void }
|
|
42
|
+
attr_writer :follow_subdomains
|
|
43
|
+
|
|
44
|
+
# When true, iframe contents are included in Markdown before extraction.
|
|
45
|
+
sig { returns(T.nilable(T::Boolean)) }
|
|
46
|
+
attr_reader :include_frames
|
|
47
|
+
|
|
48
|
+
sig { params(include_frames: T::Boolean).void }
|
|
49
|
+
attr_writer :include_frames
|
|
50
|
+
|
|
51
|
+
# Optional extraction guidance, such as which facts to prioritize or how to
|
|
52
|
+
# interpret fields in the schema.
|
|
53
|
+
sig { returns(T.nilable(String)) }
|
|
54
|
+
attr_reader :instructions
|
|
55
|
+
|
|
56
|
+
sig { params(instructions: String).void }
|
|
57
|
+
attr_writer :instructions
|
|
58
|
+
|
|
59
|
+
# Return cached scrape results if a prior scrape for the same parameters is
|
|
60
|
+
# younger than this many milliseconds. Defaults to 7 days (604800000 ms).
|
|
61
|
+
sig { returns(T.nilable(Integer)) }
|
|
62
|
+
attr_reader :max_age_ms
|
|
63
|
+
|
|
64
|
+
sig { params(max_age_ms: Integer).void }
|
|
65
|
+
attr_writer :max_age_ms
|
|
66
|
+
|
|
67
|
+
sig { returns(T.nilable(ContextDev::WebExtractParams::Pdf)) }
|
|
68
|
+
attr_reader :pdf
|
|
69
|
+
|
|
70
|
+
sig { params(pdf: ContextDev::WebExtractParams::Pdf::OrHash).void }
|
|
71
|
+
attr_writer :pdf
|
|
72
|
+
|
|
73
|
+
# Soft time budget for the crawl in milliseconds.
|
|
74
|
+
sig { returns(T.nilable(Integer)) }
|
|
75
|
+
attr_reader :stop_after_ms
|
|
76
|
+
|
|
77
|
+
sig { params(stop_after_ms: Integer).void }
|
|
78
|
+
attr_writer :stop_after_ms
|
|
79
|
+
|
|
80
|
+
# Optional timeout in milliseconds for the request. If the request takes longer
|
|
81
|
+
# than this value, it will be aborted with a 408 status code. Maximum allowed
|
|
82
|
+
# value is 300000ms (5 minutes).
|
|
83
|
+
sig { returns(T.nilable(Integer)) }
|
|
84
|
+
attr_reader :timeout_ms
|
|
85
|
+
|
|
86
|
+
sig { params(timeout_ms: Integer).void }
|
|
87
|
+
attr_writer :timeout_ms
|
|
88
|
+
|
|
89
|
+
# Optional browser wait time in milliseconds after initial page load for each
|
|
90
|
+
# crawled page.
|
|
91
|
+
sig { returns(T.nilable(Integer)) }
|
|
92
|
+
attr_reader :wait_for_ms
|
|
93
|
+
|
|
94
|
+
sig { params(wait_for_ms: Integer).void }
|
|
95
|
+
attr_writer :wait_for_ms
|
|
96
|
+
|
|
97
|
+
sig do
|
|
98
|
+
params(
|
|
99
|
+
schema: T::Hash[Symbol, T.anything],
|
|
100
|
+
url: String,
|
|
101
|
+
fact_check: T::Boolean,
|
|
102
|
+
follow_subdomains: T::Boolean,
|
|
103
|
+
include_frames: T::Boolean,
|
|
104
|
+
instructions: String,
|
|
105
|
+
max_age_ms: Integer,
|
|
106
|
+
pdf: ContextDev::WebExtractParams::Pdf::OrHash,
|
|
107
|
+
stop_after_ms: Integer,
|
|
108
|
+
timeout_ms: Integer,
|
|
109
|
+
wait_for_ms: Integer,
|
|
110
|
+
request_options: ContextDev::RequestOptions::OrHash
|
|
111
|
+
).returns(T.attached_class)
|
|
112
|
+
end
|
|
113
|
+
def self.new(
|
|
114
|
+
# JSON Schema for the returned data object. TypeScript Zod users can pass a JSON
|
|
115
|
+
# Schema generated from a Zod object; Python users can pass the equivalent JSON
|
|
116
|
+
# Schema object.
|
|
117
|
+
schema:,
|
|
118
|
+
# The starting website URL to crawl and extract from. Must include http:// or
|
|
119
|
+
# https://.
|
|
120
|
+
url:,
|
|
121
|
+
# When true, every returned value must be grounded in facts stated on the page;
|
|
122
|
+
# fields that cannot be supported by the page are returned as null/empty. When
|
|
123
|
+
# false (default), the model may make reasonable inferences and derivations from
|
|
124
|
+
# the page content (e.g. ideal customer, competitor analysis, recommendations)
|
|
125
|
+
# while keeping verifiable specifics (names, quotes, URLs, dates, metrics)
|
|
126
|
+
# faithful to the source.
|
|
127
|
+
fact_check: nil,
|
|
128
|
+
# When true, follow links on subdomains of the starting URL's domain.
|
|
129
|
+
follow_subdomains: nil,
|
|
130
|
+
# When true, iframe contents are included in Markdown before extraction.
|
|
131
|
+
include_frames: nil,
|
|
132
|
+
# Optional extraction guidance, such as which facts to prioritize or how to
|
|
133
|
+
# interpret fields in the schema.
|
|
134
|
+
instructions: nil,
|
|
135
|
+
# Return cached scrape results if a prior scrape for the same parameters is
|
|
136
|
+
# younger than this many milliseconds. Defaults to 7 days (604800000 ms).
|
|
137
|
+
max_age_ms: nil,
|
|
138
|
+
pdf: nil,
|
|
139
|
+
# Soft time budget for the crawl in milliseconds.
|
|
140
|
+
stop_after_ms: nil,
|
|
141
|
+
# Optional timeout in milliseconds for the request. If the request takes longer
|
|
142
|
+
# than this value, it will be aborted with a 408 status code. Maximum allowed
|
|
143
|
+
# value is 300000ms (5 minutes).
|
|
144
|
+
timeout_ms: nil,
|
|
145
|
+
# Optional browser wait time in milliseconds after initial page load for each
|
|
146
|
+
# crawled page.
|
|
147
|
+
wait_for_ms: nil,
|
|
148
|
+
request_options: {}
|
|
149
|
+
)
|
|
150
|
+
end
|
|
151
|
+
|
|
152
|
+
sig do
|
|
153
|
+
override.returns(
|
|
154
|
+
{
|
|
155
|
+
schema: T::Hash[Symbol, T.anything],
|
|
156
|
+
url: String,
|
|
157
|
+
fact_check: T::Boolean,
|
|
158
|
+
follow_subdomains: T::Boolean,
|
|
159
|
+
include_frames: T::Boolean,
|
|
160
|
+
instructions: String,
|
|
161
|
+
max_age_ms: Integer,
|
|
162
|
+
pdf: ContextDev::WebExtractParams::Pdf,
|
|
163
|
+
stop_after_ms: Integer,
|
|
164
|
+
timeout_ms: Integer,
|
|
165
|
+
wait_for_ms: Integer,
|
|
166
|
+
request_options: ContextDev::RequestOptions
|
|
167
|
+
}
|
|
168
|
+
)
|
|
169
|
+
end
|
|
170
|
+
def to_hash
|
|
171
|
+
end
|
|
172
|
+
|
|
173
|
+
class Pdf < ContextDev::Internal::Type::BaseModel
|
|
174
|
+
OrHash =
|
|
175
|
+
T.type_alias do
|
|
176
|
+
T.any(
|
|
177
|
+
ContextDev::WebExtractParams::Pdf,
|
|
178
|
+
ContextDev::Internal::AnyHash
|
|
179
|
+
)
|
|
180
|
+
end
|
|
181
|
+
|
|
182
|
+
# Last 1-based PDF page to parse. Must be greater than or equal to start when both
|
|
183
|
+
# are provided.
|
|
184
|
+
sig { returns(T.nilable(Integer)) }
|
|
185
|
+
attr_reader :end_
|
|
186
|
+
|
|
187
|
+
sig { params(end_: Integer).void }
|
|
188
|
+
attr_writer :end_
|
|
189
|
+
|
|
190
|
+
# When true, PDF pages are fetched and parsed. When false, PDF pages are skipped.
|
|
191
|
+
sig { returns(T.nilable(T::Boolean)) }
|
|
192
|
+
attr_reader :should_parse
|
|
193
|
+
|
|
194
|
+
sig { params(should_parse: T::Boolean).void }
|
|
195
|
+
attr_writer :should_parse
|
|
196
|
+
|
|
197
|
+
# First 1-based PDF page to parse.
|
|
198
|
+
sig { returns(T.nilable(Integer)) }
|
|
199
|
+
attr_reader :start
|
|
200
|
+
|
|
201
|
+
sig { params(start: Integer).void }
|
|
202
|
+
attr_writer :start
|
|
203
|
+
|
|
204
|
+
sig do
|
|
205
|
+
params(
|
|
206
|
+
end_: Integer,
|
|
207
|
+
should_parse: T::Boolean,
|
|
208
|
+
start: Integer
|
|
209
|
+
).returns(T.attached_class)
|
|
210
|
+
end
|
|
211
|
+
def self.new(
|
|
212
|
+
# Last 1-based PDF page to parse. Must be greater than or equal to start when both
|
|
213
|
+
# are provided.
|
|
214
|
+
end_: nil,
|
|
215
|
+
# When true, PDF pages are fetched and parsed. When false, PDF pages are skipped.
|
|
216
|
+
should_parse: nil,
|
|
217
|
+
# First 1-based PDF page to parse.
|
|
218
|
+
start: nil
|
|
219
|
+
)
|
|
220
|
+
end
|
|
221
|
+
|
|
222
|
+
sig do
|
|
223
|
+
override.returns(
|
|
224
|
+
{ end_: Integer, should_parse: T::Boolean, start: Integer }
|
|
225
|
+
)
|
|
226
|
+
end
|
|
227
|
+
def to_hash
|
|
228
|
+
end
|
|
229
|
+
end
|
|
230
|
+
end
|
|
231
|
+
end
|
|
232
|
+
end
|
|
@@ -0,0 +1,134 @@
|
|
|
1
|
+
# typed: strong
|
|
2
|
+
|
|
3
|
+
module ContextDev
|
|
4
|
+
module Models
|
|
5
|
+
class WebExtractResponse < ContextDev::Internal::Type::BaseModel
|
|
6
|
+
OrHash =
|
|
7
|
+
T.type_alias do
|
|
8
|
+
T.any(
|
|
9
|
+
ContextDev::Models::WebExtractResponse,
|
|
10
|
+
ContextDev::Internal::AnyHash
|
|
11
|
+
)
|
|
12
|
+
end
|
|
13
|
+
|
|
14
|
+
# Extracted data matching the request schema
|
|
15
|
+
sig { returns(T::Hash[Symbol, T.anything]) }
|
|
16
|
+
attr_accessor :data
|
|
17
|
+
|
|
18
|
+
sig { returns(ContextDev::Models::WebExtractResponse::Metadata) }
|
|
19
|
+
attr_reader :metadata
|
|
20
|
+
|
|
21
|
+
sig do
|
|
22
|
+
params(
|
|
23
|
+
metadata: ContextDev::Models::WebExtractResponse::Metadata::OrHash
|
|
24
|
+
).void
|
|
25
|
+
end
|
|
26
|
+
attr_writer :metadata
|
|
27
|
+
|
|
28
|
+
# Status of the response, e.g., 'ok'
|
|
29
|
+
sig { returns(String) }
|
|
30
|
+
attr_accessor :status
|
|
31
|
+
|
|
32
|
+
# The starting URL that was analyzed
|
|
33
|
+
sig { returns(String) }
|
|
34
|
+
attr_accessor :url
|
|
35
|
+
|
|
36
|
+
# List of URLs whose Markdown was used for extraction
|
|
37
|
+
sig { returns(T::Array[String]) }
|
|
38
|
+
attr_accessor :urls_analyzed
|
|
39
|
+
|
|
40
|
+
sig do
|
|
41
|
+
params(
|
|
42
|
+
data: T::Hash[Symbol, T.anything],
|
|
43
|
+
metadata: ContextDev::Models::WebExtractResponse::Metadata::OrHash,
|
|
44
|
+
status: String,
|
|
45
|
+
url: String,
|
|
46
|
+
urls_analyzed: T::Array[String]
|
|
47
|
+
).returns(T.attached_class)
|
|
48
|
+
end
|
|
49
|
+
def self.new(
|
|
50
|
+
# Extracted data matching the request schema
|
|
51
|
+
data:,
|
|
52
|
+
metadata:,
|
|
53
|
+
# Status of the response, e.g., 'ok'
|
|
54
|
+
status:,
|
|
55
|
+
# The starting URL that was analyzed
|
|
56
|
+
url:,
|
|
57
|
+
# List of URLs whose Markdown was used for extraction
|
|
58
|
+
urls_analyzed:
|
|
59
|
+
)
|
|
60
|
+
end
|
|
61
|
+
|
|
62
|
+
sig do
|
|
63
|
+
override.returns(
|
|
64
|
+
{
|
|
65
|
+
data: T::Hash[Symbol, T.anything],
|
|
66
|
+
metadata: ContextDev::Models::WebExtractResponse::Metadata,
|
|
67
|
+
status: String,
|
|
68
|
+
url: String,
|
|
69
|
+
urls_analyzed: T::Array[String]
|
|
70
|
+
}
|
|
71
|
+
)
|
|
72
|
+
end
|
|
73
|
+
def to_hash
|
|
74
|
+
end
|
|
75
|
+
|
|
76
|
+
class Metadata < ContextDev::Internal::Type::BaseModel
|
|
77
|
+
OrHash =
|
|
78
|
+
T.type_alias do
|
|
79
|
+
T.any(
|
|
80
|
+
ContextDev::Models::WebExtractResponse::Metadata,
|
|
81
|
+
ContextDev::Internal::AnyHash
|
|
82
|
+
)
|
|
83
|
+
end
|
|
84
|
+
|
|
85
|
+
sig { returns(Integer) }
|
|
86
|
+
attr_accessor :max_crawl_depth
|
|
87
|
+
|
|
88
|
+
sig { returns(Integer) }
|
|
89
|
+
attr_accessor :num_failed
|
|
90
|
+
|
|
91
|
+
sig { returns(Integer) }
|
|
92
|
+
attr_accessor :num_skipped
|
|
93
|
+
|
|
94
|
+
sig { returns(Integer) }
|
|
95
|
+
attr_accessor :num_succeeded
|
|
96
|
+
|
|
97
|
+
sig { returns(Integer) }
|
|
98
|
+
attr_accessor :num_urls
|
|
99
|
+
|
|
100
|
+
sig do
|
|
101
|
+
params(
|
|
102
|
+
max_crawl_depth: Integer,
|
|
103
|
+
num_failed: Integer,
|
|
104
|
+
num_skipped: Integer,
|
|
105
|
+
num_succeeded: Integer,
|
|
106
|
+
num_urls: Integer
|
|
107
|
+
).returns(T.attached_class)
|
|
108
|
+
end
|
|
109
|
+
def self.new(
|
|
110
|
+
max_crawl_depth:,
|
|
111
|
+
num_failed:,
|
|
112
|
+
num_skipped:,
|
|
113
|
+
num_succeeded:,
|
|
114
|
+
num_urls:
|
|
115
|
+
)
|
|
116
|
+
end
|
|
117
|
+
|
|
118
|
+
sig do
|
|
119
|
+
override.returns(
|
|
120
|
+
{
|
|
121
|
+
max_crawl_depth: Integer,
|
|
122
|
+
num_failed: Integer,
|
|
123
|
+
num_skipped: Integer,
|
|
124
|
+
num_succeeded: Integer,
|
|
125
|
+
num_urls: Integer
|
|
126
|
+
}
|
|
127
|
+
)
|
|
128
|
+
end
|
|
129
|
+
def to_hash
|
|
130
|
+
end
|
|
131
|
+
end
|
|
132
|
+
end
|
|
133
|
+
end
|
|
134
|
+
end
|
data/rbi/context_dev/models.rbi
CHANGED
|
@@ -34,6 +34,8 @@ module ContextDev
|
|
|
34
34
|
|
|
35
35
|
WebExtractFontsParams = ContextDev::Models::WebExtractFontsParams
|
|
36
36
|
|
|
37
|
+
WebExtractParams = ContextDev::Models::WebExtractParams
|
|
38
|
+
|
|
37
39
|
WebExtractStyleguideParams = ContextDev::Models::WebExtractStyleguideParams
|
|
38
40
|
|
|
39
41
|
WebScreenshotParams = ContextDev::Models::WebScreenshotParams
|
|
@@ -3,6 +3,63 @@
|
|
|
3
3
|
module ContextDev
|
|
4
4
|
module Resources
|
|
5
5
|
class Web
|
|
6
|
+
# Crawl a website, use the provided JSON Schema and instructions to prioritize
|
|
7
|
+
# relevant internal links, and extract structured data from the selected pages.
|
|
8
|
+
sig do
|
|
9
|
+
params(
|
|
10
|
+
schema: T::Hash[Symbol, T.anything],
|
|
11
|
+
url: String,
|
|
12
|
+
fact_check: T::Boolean,
|
|
13
|
+
follow_subdomains: T::Boolean,
|
|
14
|
+
include_frames: T::Boolean,
|
|
15
|
+
instructions: String,
|
|
16
|
+
max_age_ms: Integer,
|
|
17
|
+
pdf: ContextDev::WebExtractParams::Pdf::OrHash,
|
|
18
|
+
stop_after_ms: Integer,
|
|
19
|
+
timeout_ms: Integer,
|
|
20
|
+
wait_for_ms: Integer,
|
|
21
|
+
request_options: ContextDev::RequestOptions::OrHash
|
|
22
|
+
).returns(ContextDev::Models::WebExtractResponse)
|
|
23
|
+
end
|
|
24
|
+
def extract(
|
|
25
|
+
# JSON Schema for the returned data object. TypeScript Zod users can pass a JSON
|
|
26
|
+
# Schema generated from a Zod object; Python users can pass the equivalent JSON
|
|
27
|
+
# Schema object.
|
|
28
|
+
schema:,
|
|
29
|
+
# The starting website URL to crawl and extract from. Must include http:// or
|
|
30
|
+
# https://.
|
|
31
|
+
url:,
|
|
32
|
+
# When true, every returned value must be grounded in facts stated on the page;
|
|
33
|
+
# fields that cannot be supported by the page are returned as null/empty. When
|
|
34
|
+
# false (default), the model may make reasonable inferences and derivations from
|
|
35
|
+
# the page content (e.g. ideal customer, competitor analysis, recommendations)
|
|
36
|
+
# while keeping verifiable specifics (names, quotes, URLs, dates, metrics)
|
|
37
|
+
# faithful to the source.
|
|
38
|
+
fact_check: nil,
|
|
39
|
+
# When true, follow links on subdomains of the starting URL's domain.
|
|
40
|
+
follow_subdomains: nil,
|
|
41
|
+
# When true, iframe contents are included in Markdown before extraction.
|
|
42
|
+
include_frames: nil,
|
|
43
|
+
# Optional extraction guidance, such as which facts to prioritize or how to
|
|
44
|
+
# interpret fields in the schema.
|
|
45
|
+
instructions: nil,
|
|
46
|
+
# Return cached scrape results if a prior scrape for the same parameters is
|
|
47
|
+
# younger than this many milliseconds. Defaults to 7 days (604800000 ms).
|
|
48
|
+
max_age_ms: nil,
|
|
49
|
+
pdf: nil,
|
|
50
|
+
# Soft time budget for the crawl in milliseconds.
|
|
51
|
+
stop_after_ms: nil,
|
|
52
|
+
# Optional timeout in milliseconds for the request. If the request takes longer
|
|
53
|
+
# than this value, it will be aborted with a 408 status code. Maximum allowed
|
|
54
|
+
# value is 300000ms (5 minutes).
|
|
55
|
+
timeout_ms: nil,
|
|
56
|
+
# Optional browser wait time in milliseconds after initial page load for each
|
|
57
|
+
# crawled page.
|
|
58
|
+
wait_for_ms: nil,
|
|
59
|
+
request_options: {}
|
|
60
|
+
)
|
|
61
|
+
end
|
|
62
|
+
|
|
6
63
|
# Scrape font information from a website including font families, usage
|
|
7
64
|
# statistics, fallbacks, and element/word counts.
|
|
8
65
|
sig do
|
|
@@ -0,0 +1,120 @@
|
|
|
1
|
+
module ContextDev
|
|
2
|
+
module Models
|
|
3
|
+
type web_extract_params =
|
|
4
|
+
{
|
|
5
|
+
schema: ::Hash[Symbol, top],
|
|
6
|
+
url: String,
|
|
7
|
+
fact_check: bool,
|
|
8
|
+
follow_subdomains: bool,
|
|
9
|
+
include_frames: bool,
|
|
10
|
+
instructions: String,
|
|
11
|
+
max_age_ms: Integer,
|
|
12
|
+
pdf: ContextDev::WebExtractParams::Pdf,
|
|
13
|
+
stop_after_ms: Integer,
|
|
14
|
+
timeout_ms: Integer,
|
|
15
|
+
wait_for_ms: Integer
|
|
16
|
+
}
|
|
17
|
+
& ContextDev::Internal::Type::request_parameters
|
|
18
|
+
|
|
19
|
+
class WebExtractParams < ContextDev::Internal::Type::BaseModel
|
|
20
|
+
extend ContextDev::Internal::Type::RequestParameters::Converter
|
|
21
|
+
include ContextDev::Internal::Type::RequestParameters
|
|
22
|
+
|
|
23
|
+
attr_accessor schema: ::Hash[Symbol, top]
|
|
24
|
+
|
|
25
|
+
attr_accessor url: String
|
|
26
|
+
|
|
27
|
+
attr_reader fact_check: bool?
|
|
28
|
+
|
|
29
|
+
def fact_check=: (bool) -> bool
|
|
30
|
+
|
|
31
|
+
attr_reader follow_subdomains: bool?
|
|
32
|
+
|
|
33
|
+
def follow_subdomains=: (bool) -> bool
|
|
34
|
+
|
|
35
|
+
attr_reader include_frames: bool?
|
|
36
|
+
|
|
37
|
+
def include_frames=: (bool) -> bool
|
|
38
|
+
|
|
39
|
+
attr_reader instructions: String?
|
|
40
|
+
|
|
41
|
+
def instructions=: (String) -> String
|
|
42
|
+
|
|
43
|
+
attr_reader max_age_ms: Integer?
|
|
44
|
+
|
|
45
|
+
def max_age_ms=: (Integer) -> Integer
|
|
46
|
+
|
|
47
|
+
attr_reader pdf: ContextDev::WebExtractParams::Pdf?
|
|
48
|
+
|
|
49
|
+
def pdf=: (
|
|
50
|
+
ContextDev::WebExtractParams::Pdf
|
|
51
|
+
) -> ContextDev::WebExtractParams::Pdf
|
|
52
|
+
|
|
53
|
+
attr_reader stop_after_ms: Integer?
|
|
54
|
+
|
|
55
|
+
def stop_after_ms=: (Integer) -> Integer
|
|
56
|
+
|
|
57
|
+
attr_reader timeout_ms: Integer?
|
|
58
|
+
|
|
59
|
+
def timeout_ms=: (Integer) -> Integer
|
|
60
|
+
|
|
61
|
+
attr_reader wait_for_ms: Integer?
|
|
62
|
+
|
|
63
|
+
def wait_for_ms=: (Integer) -> Integer
|
|
64
|
+
|
|
65
|
+
def initialize: (
|
|
66
|
+
schema: ::Hash[Symbol, top],
|
|
67
|
+
url: String,
|
|
68
|
+
?fact_check: bool,
|
|
69
|
+
?follow_subdomains: bool,
|
|
70
|
+
?include_frames: bool,
|
|
71
|
+
?instructions: String,
|
|
72
|
+
?max_age_ms: Integer,
|
|
73
|
+
?pdf: ContextDev::WebExtractParams::Pdf,
|
|
74
|
+
?stop_after_ms: Integer,
|
|
75
|
+
?timeout_ms: Integer,
|
|
76
|
+
?wait_for_ms: Integer,
|
|
77
|
+
?request_options: ContextDev::request_opts
|
|
78
|
+
) -> void
|
|
79
|
+
|
|
80
|
+
def to_hash: -> {
|
|
81
|
+
schema: ::Hash[Symbol, top],
|
|
82
|
+
url: String,
|
|
83
|
+
fact_check: bool,
|
|
84
|
+
follow_subdomains: bool,
|
|
85
|
+
include_frames: bool,
|
|
86
|
+
instructions: String,
|
|
87
|
+
max_age_ms: Integer,
|
|
88
|
+
pdf: ContextDev::WebExtractParams::Pdf,
|
|
89
|
+
stop_after_ms: Integer,
|
|
90
|
+
timeout_ms: Integer,
|
|
91
|
+
wait_for_ms: Integer,
|
|
92
|
+
request_options: ContextDev::RequestOptions
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
type pdf = { end_: Integer, should_parse: bool, start: Integer }
|
|
96
|
+
|
|
97
|
+
class Pdf < ContextDev::Internal::Type::BaseModel
|
|
98
|
+
attr_reader end_: Integer?
|
|
99
|
+
|
|
100
|
+
def end_=: (Integer) -> Integer
|
|
101
|
+
|
|
102
|
+
attr_reader should_parse: bool?
|
|
103
|
+
|
|
104
|
+
def should_parse=: (bool) -> bool
|
|
105
|
+
|
|
106
|
+
attr_reader start: Integer?
|
|
107
|
+
|
|
108
|
+
def start=: (Integer) -> Integer
|
|
109
|
+
|
|
110
|
+
def initialize: (
|
|
111
|
+
?end_: Integer,
|
|
112
|
+
?should_parse: bool,
|
|
113
|
+
?start: Integer
|
|
114
|
+
) -> void
|
|
115
|
+
|
|
116
|
+
def to_hash: -> { end_: Integer, should_parse: bool, start: Integer }
|
|
117
|
+
end
|
|
118
|
+
end
|
|
119
|
+
end
|
|
120
|
+
end
|
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
module ContextDev
|
|
2
|
+
module Models
|
|
3
|
+
type web_extract_response =
|
|
4
|
+
{
|
|
5
|
+
data: ::Hash[Symbol, top],
|
|
6
|
+
metadata: ContextDev::Models::WebExtractResponse::Metadata,
|
|
7
|
+
status: String,
|
|
8
|
+
url: String,
|
|
9
|
+
urls_analyzed: ::Array[String]
|
|
10
|
+
}
|
|
11
|
+
|
|
12
|
+
class WebExtractResponse < ContextDev::Internal::Type::BaseModel
|
|
13
|
+
attr_accessor data: ::Hash[Symbol, top]
|
|
14
|
+
|
|
15
|
+
attr_accessor metadata: ContextDev::Models::WebExtractResponse::Metadata
|
|
16
|
+
|
|
17
|
+
attr_accessor status: String
|
|
18
|
+
|
|
19
|
+
attr_accessor url: String
|
|
20
|
+
|
|
21
|
+
attr_accessor urls_analyzed: ::Array[String]
|
|
22
|
+
|
|
23
|
+
def initialize: (
|
|
24
|
+
data: ::Hash[Symbol, top],
|
|
25
|
+
metadata: ContextDev::Models::WebExtractResponse::Metadata,
|
|
26
|
+
status: String,
|
|
27
|
+
url: String,
|
|
28
|
+
urls_analyzed: ::Array[String]
|
|
29
|
+
) -> void
|
|
30
|
+
|
|
31
|
+
def to_hash: -> {
|
|
32
|
+
data: ::Hash[Symbol, top],
|
|
33
|
+
metadata: ContextDev::Models::WebExtractResponse::Metadata,
|
|
34
|
+
status: String,
|
|
35
|
+
url: String,
|
|
36
|
+
urls_analyzed: ::Array[String]
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
type metadata =
|
|
40
|
+
{
|
|
41
|
+
max_crawl_depth: Integer,
|
|
42
|
+
num_failed: Integer,
|
|
43
|
+
num_skipped: Integer,
|
|
44
|
+
num_succeeded: Integer,
|
|
45
|
+
num_urls: Integer
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
class Metadata < ContextDev::Internal::Type::BaseModel
|
|
49
|
+
attr_accessor max_crawl_depth: Integer
|
|
50
|
+
|
|
51
|
+
attr_accessor num_failed: Integer
|
|
52
|
+
|
|
53
|
+
attr_accessor num_skipped: Integer
|
|
54
|
+
|
|
55
|
+
attr_accessor num_succeeded: Integer
|
|
56
|
+
|
|
57
|
+
attr_accessor num_urls: Integer
|
|
58
|
+
|
|
59
|
+
def initialize: (
|
|
60
|
+
max_crawl_depth: Integer,
|
|
61
|
+
num_failed: Integer,
|
|
62
|
+
num_skipped: Integer,
|
|
63
|
+
num_succeeded: Integer,
|
|
64
|
+
num_urls: Integer
|
|
65
|
+
) -> void
|
|
66
|
+
|
|
67
|
+
def to_hash: -> {
|
|
68
|
+
max_crawl_depth: Integer,
|
|
69
|
+
num_failed: Integer,
|
|
70
|
+
num_skipped: Integer,
|
|
71
|
+
num_succeeded: Integer,
|
|
72
|
+
num_urls: Integer
|
|
73
|
+
}
|
|
74
|
+
end
|
|
75
|
+
end
|
|
76
|
+
end
|
|
77
|
+
end
|
data/sig/context_dev/models.rbs
CHANGED
|
@@ -29,6 +29,8 @@ module ContextDev
|
|
|
29
29
|
|
|
30
30
|
class WebExtractFontsParams = ContextDev::Models::WebExtractFontsParams
|
|
31
31
|
|
|
32
|
+
class WebExtractParams = ContextDev::Models::WebExtractParams
|
|
33
|
+
|
|
32
34
|
class WebExtractStyleguideParams = ContextDev::Models::WebExtractStyleguideParams
|
|
33
35
|
|
|
34
36
|
class WebScreenshotParams = ContextDev::Models::WebScreenshotParams
|
|
@@ -1,6 +1,21 @@
|
|
|
1
1
|
module ContextDev
|
|
2
2
|
module Resources
|
|
3
3
|
class Web
|
|
4
|
+
def extract: (
|
|
5
|
+
schema: ::Hash[Symbol, top],
|
|
6
|
+
url: String,
|
|
7
|
+
?fact_check: bool,
|
|
8
|
+
?follow_subdomains: bool,
|
|
9
|
+
?include_frames: bool,
|
|
10
|
+
?instructions: String,
|
|
11
|
+
?max_age_ms: Integer,
|
|
12
|
+
?pdf: ContextDev::WebExtractParams::Pdf,
|
|
13
|
+
?stop_after_ms: Integer,
|
|
14
|
+
?timeout_ms: Integer,
|
|
15
|
+
?wait_for_ms: Integer,
|
|
16
|
+
?request_options: ContextDev::request_opts
|
|
17
|
+
) -> ContextDev::Models::WebExtractResponse
|
|
18
|
+
|
|
4
19
|
def extract_fonts: (
|
|
5
20
|
?direct_url: String,
|
|
6
21
|
?domain: String,
|
metadata
CHANGED
|
@@ -1,14 +1,14 @@
|
|
|
1
1
|
--- !ruby/object:Gem::Specification
|
|
2
2
|
name: context.dev
|
|
3
3
|
version: !ruby/object:Gem::Version
|
|
4
|
-
version: 1.
|
|
4
|
+
version: 1.26.0
|
|
5
5
|
platform: ruby
|
|
6
6
|
authors:
|
|
7
7
|
- Context Dev
|
|
8
8
|
autorequire:
|
|
9
9
|
bindir: bin
|
|
10
10
|
cert_chain: []
|
|
11
|
-
date: 2026-
|
|
11
|
+
date: 2026-06-01 00:00:00.000000000 Z
|
|
12
12
|
dependencies:
|
|
13
13
|
- !ruby/object:Gem::Dependency
|
|
14
14
|
name: cgi
|
|
@@ -99,6 +99,8 @@ files:
|
|
|
99
99
|
- lib/context_dev/models/utility_prefetch_response.rb
|
|
100
100
|
- lib/context_dev/models/web_extract_fonts_params.rb
|
|
101
101
|
- lib/context_dev/models/web_extract_fonts_response.rb
|
|
102
|
+
- lib/context_dev/models/web_extract_params.rb
|
|
103
|
+
- lib/context_dev/models/web_extract_response.rb
|
|
102
104
|
- lib/context_dev/models/web_extract_styleguide_params.rb
|
|
103
105
|
- lib/context_dev/models/web_extract_styleguide_response.rb
|
|
104
106
|
- lib/context_dev/models/web_screenshot_params.rb
|
|
@@ -172,6 +174,8 @@ files:
|
|
|
172
174
|
- rbi/context_dev/models/utility_prefetch_response.rbi
|
|
173
175
|
- rbi/context_dev/models/web_extract_fonts_params.rbi
|
|
174
176
|
- rbi/context_dev/models/web_extract_fonts_response.rbi
|
|
177
|
+
- rbi/context_dev/models/web_extract_params.rbi
|
|
178
|
+
- rbi/context_dev/models/web_extract_response.rbi
|
|
175
179
|
- rbi/context_dev/models/web_extract_styleguide_params.rbi
|
|
176
180
|
- rbi/context_dev/models/web_extract_styleguide_response.rbi
|
|
177
181
|
- rbi/context_dev/models/web_screenshot_params.rbi
|
|
@@ -244,6 +248,8 @@ files:
|
|
|
244
248
|
- sig/context_dev/models/utility_prefetch_response.rbs
|
|
245
249
|
- sig/context_dev/models/web_extract_fonts_params.rbs
|
|
246
250
|
- sig/context_dev/models/web_extract_fonts_response.rbs
|
|
251
|
+
- sig/context_dev/models/web_extract_params.rbs
|
|
252
|
+
- sig/context_dev/models/web_extract_response.rbs
|
|
247
253
|
- sig/context_dev/models/web_extract_styleguide_params.rbs
|
|
248
254
|
- sig/context_dev/models/web_extract_styleguide_response.rbs
|
|
249
255
|
- sig/context_dev/models/web_screenshot_params.rbs
|