context.dev 1.24.0 → 1.26.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
checksums.yaml CHANGED
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  SHA256:
3
- metadata.gz: 18c7d842ed7b7ad15534738c63bcd5c85e4423fce2f7d91f5a66f08d148705a0
4
- data.tar.gz: '067911c3ee2689b6aaff65fd02c2ff6a3ddc6bd7ab8557c436465da33a66c6b7'
3
+ metadata.gz: bb1eb49bd2c33309910f98a48b129a07a32707a85e5a832277226ee555d22780
4
+ data.tar.gz: 4c7410af11f9f6f69145a70f37e1c613fe9a90d6b2ea99acc061c10ae9119716
5
5
  SHA512:
6
- metadata.gz: 63387a950cc87f056744ac02ec23a7b1566d6111189fcee2a78297ad21e3a0e6510c965ae81258a3a427651ec5290d029a0ccab716d464cb8779a63329d7a43d
7
- data.tar.gz: 19785a765c206092c2857e6aca4c645c13e491296c23cf5b4236a97898cf8182ad7f37db1d2e95f2b5e1010c94b32fa7861959e11546f89252cfa15bf03efb23
6
+ metadata.gz: 13e5b25d98899a354e90fdf559ea5b3ed8d1140a543f538a7909691a92ecb49917e87398f2da8b3759aad2751487c02e4b89e4738f15febe9bc10995825c4660
7
+ data.tar.gz: b82dee66dd6c01d5a4edb347097a978708714164cbe4460b7da394f0440b297afcc602bf52cc3cf597e3fe8638559b953bafe0288512e883f1d3b2957fc62cea
data/CHANGELOG.md CHANGED
@@ -1,5 +1,23 @@
1
1
  # Changelog
2
2
 
3
+ ## 1.26.0 (2026-06-01)
4
+
5
+ Full Changelog: [v1.25.0...v1.26.0](https://github.com/context-dot-dev/context-ruby-sdk/compare/v1.25.0...v1.26.0)
6
+
7
+ ### Features
8
+
9
+ * **api:** api update ([d442168](https://github.com/context-dot-dev/context-ruby-sdk/commit/d442168b67accce12463e36f03630f155c1c7be3))
10
+ * **api:** api update ([f3a97c1](https://github.com/context-dot-dev/context-ruby-sdk/commit/f3a97c19ef01be5e6cf1ce7ecfa26c972edfdf33))
11
+ * **api:** api update ([8d8abf7](https://github.com/context-dot-dev/context-ruby-sdk/commit/8d8abf7c1600bce17ea26aa245f41b2e3199f722))
12
+
13
+ ## 1.25.0 (2026-05-31)
14
+
15
+ Full Changelog: [v1.24.0...v1.25.0](https://github.com/context-dot-dev/context-ruby-sdk/compare/v1.24.0...v1.25.0)
16
+
17
+ ### Features
18
+
19
+ * **api:** manual updates ([688e5f5](https://github.com/context-dot-dev/context-ruby-sdk/commit/688e5f56137aeac7c5219383f241a459173f961f))
20
+
3
21
  ## 1.24.0 (2026-05-30)
4
22
 
5
23
  Full Changelog: [v1.23.0...v1.24.0](https://github.com/context-dot-dev/context-ruby-sdk/compare/v1.23.0...v1.24.0)
data/README.md CHANGED
@@ -26,7 +26,7 @@ To use this gem, install via Bundler by adding the following to your application
26
26
  <!-- x-release-please-start-version -->
27
27
 
28
28
  ```ruby
29
- gem "context.dev", "~> 1.24.0"
29
+ gem "context.dev", "~> 1.26.0"
30
30
  ```
31
31
 
32
32
  <!-- x-release-please-end -->
@@ -438,11 +438,11 @@ module ContextDev
438
438
  # @return [Hash{Symbol=>Object}]
439
439
  #
440
440
  # @example
441
- # # `web_extract_fonts_response` is a `ContextDev::Models::WebExtractFontsResponse`
442
- # web_extract_fonts_response => {
443
- # code: code,
444
- # domain: domain,
445
- # fonts: fonts
441
+ # # `web_extract_response` is a `ContextDev::Models::WebExtractResponse`
442
+ # web_extract_response => {
443
+ # data: data,
444
+ # metadata: metadata,
445
+ # status: status
446
446
  # }
447
447
  def deconstruct_keys(keys)
448
448
  (keys || self.class.known_fields.keys)
@@ -0,0 +1,148 @@
1
+ # frozen_string_literal: true
2
+
3
+ module ContextDev
4
+ module Models
5
+ # @see ContextDev::Resources::Web#extract
6
+ class WebExtractParams < ContextDev::Internal::Type::BaseModel
7
+ extend ContextDev::Internal::Type::RequestParameters::Converter
8
+ include ContextDev::Internal::Type::RequestParameters
9
+
10
+ # @!attribute schema
11
+ # JSON Schema for the returned data object. TypeScript Zod users can pass a JSON
12
+ # Schema generated from a Zod object; Python users can pass the equivalent JSON
13
+ # Schema object.
14
+ #
15
+ # @return [Hash{Symbol=>Object}]
16
+ required :schema, ContextDev::Internal::Type::HashOf[ContextDev::Internal::Type::Unknown]
17
+
18
+ # @!attribute url
19
+ # The starting website URL to crawl and extract from. Must include http:// or
20
+ # https://.
21
+ #
22
+ # @return [String]
23
+ required :url, String
24
+
25
+ # @!attribute fact_check
26
+ # When true, every returned value must be grounded in facts stated on the page;
27
+ # fields that cannot be supported by the page are returned as null/empty. When
28
+ # false (default), the model may make reasonable inferences and derivations from
29
+ # the page content (e.g. ideal customer, competitor analysis, recommendations)
30
+ # while keeping verifiable specifics (names, quotes, URLs, dates, metrics)
31
+ # faithful to the source.
32
+ #
33
+ # @return [Boolean, nil]
34
+ optional :fact_check, ContextDev::Internal::Type::Boolean, api_name: :factCheck
35
+
36
+ # @!attribute follow_subdomains
37
+ # When true, follow links on subdomains of the starting URL's domain.
38
+ #
39
+ # @return [Boolean, nil]
40
+ optional :follow_subdomains, ContextDev::Internal::Type::Boolean, api_name: :followSubdomains
41
+
42
+ # @!attribute include_frames
43
+ # When true, iframe contents are included in Markdown before extraction.
44
+ #
45
+ # @return [Boolean, nil]
46
+ optional :include_frames, ContextDev::Internal::Type::Boolean, api_name: :includeFrames
47
+
48
+ # @!attribute instructions
49
+ # Optional extraction guidance, such as which facts to prioritize or how to
50
+ # interpret fields in the schema.
51
+ #
52
+ # @return [String, nil]
53
+ optional :instructions, String
54
+
55
+ # @!attribute max_age_ms
56
+ # Return cached scrape results if a prior scrape for the same parameters is
57
+ # younger than this many milliseconds. Defaults to 7 days (604800000 ms).
58
+ #
59
+ # @return [Integer, nil]
60
+ optional :max_age_ms, Integer, api_name: :maxAgeMs
61
+
62
+ # @!attribute pdf
63
+ #
64
+ # @return [ContextDev::Models::WebExtractParams::Pdf, nil]
65
+ optional :pdf, -> { ContextDev::WebExtractParams::Pdf }
66
+
67
+ # @!attribute stop_after_ms
68
+ # Soft time budget for the crawl in milliseconds.
69
+ #
70
+ # @return [Integer, nil]
71
+ optional :stop_after_ms, Integer, api_name: :stopAfterMs
72
+
73
+ # @!attribute timeout_ms
74
+ # Optional timeout in milliseconds for the request. If the request takes longer
75
+ # than this value, it will be aborted with a 408 status code. Maximum allowed
76
+ # value is 300000ms (5 minutes).
77
+ #
78
+ # @return [Integer, nil]
79
+ optional :timeout_ms, Integer, api_name: :timeoutMS
80
+
81
+ # @!attribute wait_for_ms
82
+ # Optional browser wait time in milliseconds after initial page load for each
83
+ # crawled page.
84
+ #
85
+ # @return [Integer, nil]
86
+ optional :wait_for_ms, Integer, api_name: :waitForMs
87
+
88
+ # @!method initialize(schema:, url:, fact_check: nil, follow_subdomains: nil, include_frames: nil, instructions: nil, max_age_ms: nil, pdf: nil, stop_after_ms: nil, timeout_ms: nil, wait_for_ms: nil, request_options: {})
89
+ # Some parameter documentations has been truncated, see
90
+ # {ContextDev::Models::WebExtractParams} for more details.
91
+ #
92
+ # @param schema [Hash{Symbol=>Object}] JSON Schema for the returned data object. TypeScript Zod users can pass a JSON S
93
+ #
94
+ # @param url [String] The starting website URL to crawl and extract from. Must include http:// or http
95
+ #
96
+ # @param fact_check [Boolean] When true, every returned value must be grounded in facts stated on the page; fi
97
+ #
98
+ # @param follow_subdomains [Boolean] When true, follow links on subdomains of the starting URL's domain.
99
+ #
100
+ # @param include_frames [Boolean] When true, iframe contents are included in Markdown before extraction.
101
+ #
102
+ # @param instructions [String] Optional extraction guidance, such as which facts to prioritize or how to interp
103
+ #
104
+ # @param max_age_ms [Integer] Return cached scrape results if a prior scrape for the same parameters is younge
105
+ #
106
+ # @param pdf [ContextDev::Models::WebExtractParams::Pdf]
107
+ #
108
+ # @param stop_after_ms [Integer] Soft time budget for the crawl in milliseconds.
109
+ #
110
+ # @param timeout_ms [Integer] Optional timeout in milliseconds for the request. If the request takes longer th
111
+ #
112
+ # @param wait_for_ms [Integer] Optional browser wait time in milliseconds after initial page load for each craw
113
+ #
114
+ # @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}]
115
+
116
+ class Pdf < ContextDev::Internal::Type::BaseModel
117
+ # @!attribute end_
118
+ # Last 1-based PDF page to parse. Must be greater than or equal to start when both
119
+ # are provided.
120
+ #
121
+ # @return [Integer, nil]
122
+ optional :end_, Integer, api_name: :end
123
+
124
+ # @!attribute should_parse
125
+ # When true, PDF pages are fetched and parsed. When false, PDF pages are skipped.
126
+ #
127
+ # @return [Boolean, nil]
128
+ optional :should_parse, ContextDev::Internal::Type::Boolean, api_name: :shouldParse
129
+
130
+ # @!attribute start
131
+ # First 1-based PDF page to parse.
132
+ #
133
+ # @return [Integer, nil]
134
+ optional :start, Integer
135
+
136
+ # @!method initialize(end_: nil, should_parse: nil, start: nil)
137
+ # Some parameter documentations has been truncated, see
138
+ # {ContextDev::Models::WebExtractParams::Pdf} for more details.
139
+ #
140
+ # @param end_ [Integer] Last 1-based PDF page to parse. Must be greater than or equal to start when both
141
+ #
142
+ # @param should_parse [Boolean] When true, PDF pages are fetched and parsed. When false, PDF pages are skipped.
143
+ #
144
+ # @param start [Integer] First 1-based PDF page to parse.
145
+ end
146
+ end
147
+ end
148
+ end
@@ -0,0 +1,83 @@
1
+ # frozen_string_literal: true
2
+
3
+ module ContextDev
4
+ module Models
5
+ # @see ContextDev::Resources::Web#extract
6
+ class WebExtractResponse < ContextDev::Internal::Type::BaseModel
7
+ # @!attribute data
8
+ # Extracted data matching the request schema
9
+ #
10
+ # @return [Hash{Symbol=>Object}]
11
+ required :data, ContextDev::Internal::Type::HashOf[ContextDev::Internal::Type::Unknown]
12
+
13
+ # @!attribute metadata
14
+ #
15
+ # @return [ContextDev::Models::WebExtractResponse::Metadata]
16
+ required :metadata, -> { ContextDev::Models::WebExtractResponse::Metadata }
17
+
18
+ # @!attribute status
19
+ # Status of the response, e.g., 'ok'
20
+ #
21
+ # @return [String]
22
+ required :status, String
23
+
24
+ # @!attribute url
25
+ # The starting URL that was analyzed
26
+ #
27
+ # @return [String]
28
+ required :url, String
29
+
30
+ # @!attribute urls_analyzed
31
+ # List of URLs whose Markdown was used for extraction
32
+ #
33
+ # @return [Array<String>]
34
+ required :urls_analyzed, ContextDev::Internal::Type::ArrayOf[String]
35
+
36
+ # @!method initialize(data:, metadata:, status:, url:, urls_analyzed:)
37
+ # @param data [Hash{Symbol=>Object}] Extracted data matching the request schema
38
+ #
39
+ # @param metadata [ContextDev::Models::WebExtractResponse::Metadata]
40
+ #
41
+ # @param status [String] Status of the response, e.g., 'ok'
42
+ #
43
+ # @param url [String] The starting URL that was analyzed
44
+ #
45
+ # @param urls_analyzed [Array<String>] List of URLs whose Markdown was used for extraction
46
+
47
+ # @see ContextDev::Models::WebExtractResponse#metadata
48
+ class Metadata < ContextDev::Internal::Type::BaseModel
49
+ # @!attribute max_crawl_depth
50
+ #
51
+ # @return [Integer]
52
+ required :max_crawl_depth, Integer, api_name: :maxCrawlDepth
53
+
54
+ # @!attribute num_failed
55
+ #
56
+ # @return [Integer]
57
+ required :num_failed, Integer, api_name: :numFailed
58
+
59
+ # @!attribute num_skipped
60
+ #
61
+ # @return [Integer]
62
+ required :num_skipped, Integer, api_name: :numSkipped
63
+
64
+ # @!attribute num_succeeded
65
+ #
66
+ # @return [Integer]
67
+ required :num_succeeded, Integer, api_name: :numSucceeded
68
+
69
+ # @!attribute num_urls
70
+ #
71
+ # @return [Integer]
72
+ required :num_urls, Integer, api_name: :numUrls
73
+
74
+ # @!method initialize(max_crawl_depth:, num_failed:, num_skipped:, num_succeeded:, num_urls:)
75
+ # @param max_crawl_depth [Integer]
76
+ # @param num_failed [Integer]
77
+ # @param num_skipped [Integer]
78
+ # @param num_succeeded [Integer]
79
+ # @param num_urls [Integer]
80
+ end
81
+ end
82
+ end
83
+ end
@@ -69,6 +69,8 @@ module ContextDev
69
69
 
70
70
  WebExtractFontsParams = ContextDev::Models::WebExtractFontsParams
71
71
 
72
+ WebExtractParams = ContextDev::Models::WebExtractParams
73
+
72
74
  WebExtractStyleguideParams = ContextDev::Models::WebExtractStyleguideParams
73
75
 
74
76
  WebScreenshotParams = ContextDev::Models::WebScreenshotParams
@@ -3,6 +3,52 @@
3
3
  module ContextDev
4
4
  module Resources
5
5
  class Web
6
+ # Some parameter documentations has been truncated, see
7
+ # {ContextDev::Models::WebExtractParams} for more details.
8
+ #
9
+ # Crawl a website, use the provided JSON Schema and instructions to prioritize
10
+ # relevant internal links, and extract structured data from the selected pages.
11
+ #
12
+ # @overload extract(schema:, url:, fact_check: nil, follow_subdomains: nil, include_frames: nil, instructions: nil, max_age_ms: nil, pdf: nil, stop_after_ms: nil, timeout_ms: nil, wait_for_ms: nil, request_options: {})
13
+ #
14
+ # @param schema [Hash{Symbol=>Object}] JSON Schema for the returned data object. TypeScript Zod users can pass a JSON S
15
+ #
16
+ # @param url [String] The starting website URL to crawl and extract from. Must include http:// or http
17
+ #
18
+ # @param fact_check [Boolean] When true, every returned value must be grounded in facts stated on the page; fi
19
+ #
20
+ # @param follow_subdomains [Boolean] When true, follow links on subdomains of the starting URL's domain.
21
+ #
22
+ # @param include_frames [Boolean] When true, iframe contents are included in Markdown before extraction.
23
+ #
24
+ # @param instructions [String] Optional extraction guidance, such as which facts to prioritize or how to interp
25
+ #
26
+ # @param max_age_ms [Integer] Return cached scrape results if a prior scrape for the same parameters is younge
27
+ #
28
+ # @param pdf [ContextDev::Models::WebExtractParams::Pdf]
29
+ #
30
+ # @param stop_after_ms [Integer] Soft time budget for the crawl in milliseconds.
31
+ #
32
+ # @param timeout_ms [Integer] Optional timeout in milliseconds for the request. If the request takes longer th
33
+ #
34
+ # @param wait_for_ms [Integer] Optional browser wait time in milliseconds after initial page load for each craw
35
+ #
36
+ # @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
37
+ #
38
+ # @return [ContextDev::Models::WebExtractResponse]
39
+ #
40
+ # @see ContextDev::Models::WebExtractParams
41
+ def extract(params)
42
+ parsed, options = ContextDev::WebExtractParams.dump_request(params)
43
+ @client.request(
44
+ method: :post,
45
+ path: "web/extract",
46
+ body: parsed,
47
+ model: ContextDev::Models::WebExtractResponse,
48
+ options: options
49
+ )
50
+ end
51
+
6
52
  # Some parameter documentations has been truncated, see
7
53
  # {ContextDev::Models::WebExtractFontsParams} for more details.
8
54
  #
@@ -1,5 +1,5 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  module ContextDev
4
- VERSION = "1.24.0"
4
+ VERSION = "1.26.0"
5
5
  end
data/lib/context_dev.rb CHANGED
@@ -82,6 +82,8 @@ require_relative "context_dev/models/utility_prefetch_params"
82
82
  require_relative "context_dev/models/utility_prefetch_response"
83
83
  require_relative "context_dev/models/web_extract_fonts_params"
84
84
  require_relative "context_dev/models/web_extract_fonts_response"
85
+ require_relative "context_dev/models/web_extract_params"
86
+ require_relative "context_dev/models/web_extract_response"
85
87
  require_relative "context_dev/models/web_extract_styleguide_params"
86
88
  require_relative "context_dev/models/web_extract_styleguide_response"
87
89
  require_relative "context_dev/models/web_screenshot_params"
@@ -0,0 +1,232 @@
1
+ # typed: strong
2
+
3
+ module ContextDev
4
+ module Models
5
+ class WebExtractParams < ContextDev::Internal::Type::BaseModel
6
+ extend ContextDev::Internal::Type::RequestParameters::Converter
7
+ include ContextDev::Internal::Type::RequestParameters
8
+
9
+ OrHash =
10
+ T.type_alias do
11
+ T.any(ContextDev::WebExtractParams, ContextDev::Internal::AnyHash)
12
+ end
13
+
14
+ # JSON Schema for the returned data object. TypeScript Zod users can pass a JSON
15
+ # Schema generated from a Zod object; Python users can pass the equivalent JSON
16
+ # Schema object.
17
+ sig { returns(T::Hash[Symbol, T.anything]) }
18
+ attr_accessor :schema
19
+
20
+ # The starting website URL to crawl and extract from. Must include http:// or
21
+ # https://.
22
+ sig { returns(String) }
23
+ attr_accessor :url
24
+
25
+ # When true, every returned value must be grounded in facts stated on the page;
26
+ # fields that cannot be supported by the page are returned as null/empty. When
27
+ # false (default), the model may make reasonable inferences and derivations from
28
+ # the page content (e.g. ideal customer, competitor analysis, recommendations)
29
+ # while keeping verifiable specifics (names, quotes, URLs, dates, metrics)
30
+ # faithful to the source.
31
+ sig { returns(T.nilable(T::Boolean)) }
32
+ attr_reader :fact_check
33
+
34
+ sig { params(fact_check: T::Boolean).void }
35
+ attr_writer :fact_check
36
+
37
+ # When true, follow links on subdomains of the starting URL's domain.
38
+ sig { returns(T.nilable(T::Boolean)) }
39
+ attr_reader :follow_subdomains
40
+
41
+ sig { params(follow_subdomains: T::Boolean).void }
42
+ attr_writer :follow_subdomains
43
+
44
+ # When true, iframe contents are included in Markdown before extraction.
45
+ sig { returns(T.nilable(T::Boolean)) }
46
+ attr_reader :include_frames
47
+
48
+ sig { params(include_frames: T::Boolean).void }
49
+ attr_writer :include_frames
50
+
51
+ # Optional extraction guidance, such as which facts to prioritize or how to
52
+ # interpret fields in the schema.
53
+ sig { returns(T.nilable(String)) }
54
+ attr_reader :instructions
55
+
56
+ sig { params(instructions: String).void }
57
+ attr_writer :instructions
58
+
59
+ # Return cached scrape results if a prior scrape for the same parameters is
60
+ # younger than this many milliseconds. Defaults to 7 days (604800000 ms).
61
+ sig { returns(T.nilable(Integer)) }
62
+ attr_reader :max_age_ms
63
+
64
+ sig { params(max_age_ms: Integer).void }
65
+ attr_writer :max_age_ms
66
+
67
+ sig { returns(T.nilable(ContextDev::WebExtractParams::Pdf)) }
68
+ attr_reader :pdf
69
+
70
+ sig { params(pdf: ContextDev::WebExtractParams::Pdf::OrHash).void }
71
+ attr_writer :pdf
72
+
73
+ # Soft time budget for the crawl in milliseconds.
74
+ sig { returns(T.nilable(Integer)) }
75
+ attr_reader :stop_after_ms
76
+
77
+ sig { params(stop_after_ms: Integer).void }
78
+ attr_writer :stop_after_ms
79
+
80
+ # Optional timeout in milliseconds for the request. If the request takes longer
81
+ # than this value, it will be aborted with a 408 status code. Maximum allowed
82
+ # value is 300000ms (5 minutes).
83
+ sig { returns(T.nilable(Integer)) }
84
+ attr_reader :timeout_ms
85
+
86
+ sig { params(timeout_ms: Integer).void }
87
+ attr_writer :timeout_ms
88
+
89
+ # Optional browser wait time in milliseconds after initial page load for each
90
+ # crawled page.
91
+ sig { returns(T.nilable(Integer)) }
92
+ attr_reader :wait_for_ms
93
+
94
+ sig { params(wait_for_ms: Integer).void }
95
+ attr_writer :wait_for_ms
96
+
97
+ sig do
98
+ params(
99
+ schema: T::Hash[Symbol, T.anything],
100
+ url: String,
101
+ fact_check: T::Boolean,
102
+ follow_subdomains: T::Boolean,
103
+ include_frames: T::Boolean,
104
+ instructions: String,
105
+ max_age_ms: Integer,
106
+ pdf: ContextDev::WebExtractParams::Pdf::OrHash,
107
+ stop_after_ms: Integer,
108
+ timeout_ms: Integer,
109
+ wait_for_ms: Integer,
110
+ request_options: ContextDev::RequestOptions::OrHash
111
+ ).returns(T.attached_class)
112
+ end
113
+ def self.new(
114
+ # JSON Schema for the returned data object. TypeScript Zod users can pass a JSON
115
+ # Schema generated from a Zod object; Python users can pass the equivalent JSON
116
+ # Schema object.
117
+ schema:,
118
+ # The starting website URL to crawl and extract from. Must include http:// or
119
+ # https://.
120
+ url:,
121
+ # When true, every returned value must be grounded in facts stated on the page;
122
+ # fields that cannot be supported by the page are returned as null/empty. When
123
+ # false (default), the model may make reasonable inferences and derivations from
124
+ # the page content (e.g. ideal customer, competitor analysis, recommendations)
125
+ # while keeping verifiable specifics (names, quotes, URLs, dates, metrics)
126
+ # faithful to the source.
127
+ fact_check: nil,
128
+ # When true, follow links on subdomains of the starting URL's domain.
129
+ follow_subdomains: nil,
130
+ # When true, iframe contents are included in Markdown before extraction.
131
+ include_frames: nil,
132
+ # Optional extraction guidance, such as which facts to prioritize or how to
133
+ # interpret fields in the schema.
134
+ instructions: nil,
135
+ # Return cached scrape results if a prior scrape for the same parameters is
136
+ # younger than this many milliseconds. Defaults to 7 days (604800000 ms).
137
+ max_age_ms: nil,
138
+ pdf: nil,
139
+ # Soft time budget for the crawl in milliseconds.
140
+ stop_after_ms: nil,
141
+ # Optional timeout in milliseconds for the request. If the request takes longer
142
+ # than this value, it will be aborted with a 408 status code. Maximum allowed
143
+ # value is 300000ms (5 minutes).
144
+ timeout_ms: nil,
145
+ # Optional browser wait time in milliseconds after initial page load for each
146
+ # crawled page.
147
+ wait_for_ms: nil,
148
+ request_options: {}
149
+ )
150
+ end
151
+
152
+ sig do
153
+ override.returns(
154
+ {
155
+ schema: T::Hash[Symbol, T.anything],
156
+ url: String,
157
+ fact_check: T::Boolean,
158
+ follow_subdomains: T::Boolean,
159
+ include_frames: T::Boolean,
160
+ instructions: String,
161
+ max_age_ms: Integer,
162
+ pdf: ContextDev::WebExtractParams::Pdf,
163
+ stop_after_ms: Integer,
164
+ timeout_ms: Integer,
165
+ wait_for_ms: Integer,
166
+ request_options: ContextDev::RequestOptions
167
+ }
168
+ )
169
+ end
170
+ def to_hash
171
+ end
172
+
173
+ class Pdf < ContextDev::Internal::Type::BaseModel
174
+ OrHash =
175
+ T.type_alias do
176
+ T.any(
177
+ ContextDev::WebExtractParams::Pdf,
178
+ ContextDev::Internal::AnyHash
179
+ )
180
+ end
181
+
182
+ # Last 1-based PDF page to parse. Must be greater than or equal to start when both
183
+ # are provided.
184
+ sig { returns(T.nilable(Integer)) }
185
+ attr_reader :end_
186
+
187
+ sig { params(end_: Integer).void }
188
+ attr_writer :end_
189
+
190
+ # When true, PDF pages are fetched and parsed. When false, PDF pages are skipped.
191
+ sig { returns(T.nilable(T::Boolean)) }
192
+ attr_reader :should_parse
193
+
194
+ sig { params(should_parse: T::Boolean).void }
195
+ attr_writer :should_parse
196
+
197
+ # First 1-based PDF page to parse.
198
+ sig { returns(T.nilable(Integer)) }
199
+ attr_reader :start
200
+
201
+ sig { params(start: Integer).void }
202
+ attr_writer :start
203
+
204
+ sig do
205
+ params(
206
+ end_: Integer,
207
+ should_parse: T::Boolean,
208
+ start: Integer
209
+ ).returns(T.attached_class)
210
+ end
211
+ def self.new(
212
+ # Last 1-based PDF page to parse. Must be greater than or equal to start when both
213
+ # are provided.
214
+ end_: nil,
215
+ # When true, PDF pages are fetched and parsed. When false, PDF pages are skipped.
216
+ should_parse: nil,
217
+ # First 1-based PDF page to parse.
218
+ start: nil
219
+ )
220
+ end
221
+
222
+ sig do
223
+ override.returns(
224
+ { end_: Integer, should_parse: T::Boolean, start: Integer }
225
+ )
226
+ end
227
+ def to_hash
228
+ end
229
+ end
230
+ end
231
+ end
232
+ end
@@ -0,0 +1,134 @@
1
+ # typed: strong
2
+
3
+ module ContextDev
4
+ module Models
5
+ class WebExtractResponse < ContextDev::Internal::Type::BaseModel
6
+ OrHash =
7
+ T.type_alias do
8
+ T.any(
9
+ ContextDev::Models::WebExtractResponse,
10
+ ContextDev::Internal::AnyHash
11
+ )
12
+ end
13
+
14
+ # Extracted data matching the request schema
15
+ sig { returns(T::Hash[Symbol, T.anything]) }
16
+ attr_accessor :data
17
+
18
+ sig { returns(ContextDev::Models::WebExtractResponse::Metadata) }
19
+ attr_reader :metadata
20
+
21
+ sig do
22
+ params(
23
+ metadata: ContextDev::Models::WebExtractResponse::Metadata::OrHash
24
+ ).void
25
+ end
26
+ attr_writer :metadata
27
+
28
+ # Status of the response, e.g., 'ok'
29
+ sig { returns(String) }
30
+ attr_accessor :status
31
+
32
+ # The starting URL that was analyzed
33
+ sig { returns(String) }
34
+ attr_accessor :url
35
+
36
+ # List of URLs whose Markdown was used for extraction
37
+ sig { returns(T::Array[String]) }
38
+ attr_accessor :urls_analyzed
39
+
40
+ sig do
41
+ params(
42
+ data: T::Hash[Symbol, T.anything],
43
+ metadata: ContextDev::Models::WebExtractResponse::Metadata::OrHash,
44
+ status: String,
45
+ url: String,
46
+ urls_analyzed: T::Array[String]
47
+ ).returns(T.attached_class)
48
+ end
49
+ def self.new(
50
+ # Extracted data matching the request schema
51
+ data:,
52
+ metadata:,
53
+ # Status of the response, e.g., 'ok'
54
+ status:,
55
+ # The starting URL that was analyzed
56
+ url:,
57
+ # List of URLs whose Markdown was used for extraction
58
+ urls_analyzed:
59
+ )
60
+ end
61
+
62
+ sig do
63
+ override.returns(
64
+ {
65
+ data: T::Hash[Symbol, T.anything],
66
+ metadata: ContextDev::Models::WebExtractResponse::Metadata,
67
+ status: String,
68
+ url: String,
69
+ urls_analyzed: T::Array[String]
70
+ }
71
+ )
72
+ end
73
+ def to_hash
74
+ end
75
+
76
+ class Metadata < ContextDev::Internal::Type::BaseModel
77
+ OrHash =
78
+ T.type_alias do
79
+ T.any(
80
+ ContextDev::Models::WebExtractResponse::Metadata,
81
+ ContextDev::Internal::AnyHash
82
+ )
83
+ end
84
+
85
+ sig { returns(Integer) }
86
+ attr_accessor :max_crawl_depth
87
+
88
+ sig { returns(Integer) }
89
+ attr_accessor :num_failed
90
+
91
+ sig { returns(Integer) }
92
+ attr_accessor :num_skipped
93
+
94
+ sig { returns(Integer) }
95
+ attr_accessor :num_succeeded
96
+
97
+ sig { returns(Integer) }
98
+ attr_accessor :num_urls
99
+
100
+ sig do
101
+ params(
102
+ max_crawl_depth: Integer,
103
+ num_failed: Integer,
104
+ num_skipped: Integer,
105
+ num_succeeded: Integer,
106
+ num_urls: Integer
107
+ ).returns(T.attached_class)
108
+ end
109
+ def self.new(
110
+ max_crawl_depth:,
111
+ num_failed:,
112
+ num_skipped:,
113
+ num_succeeded:,
114
+ num_urls:
115
+ )
116
+ end
117
+
118
+ sig do
119
+ override.returns(
120
+ {
121
+ max_crawl_depth: Integer,
122
+ num_failed: Integer,
123
+ num_skipped: Integer,
124
+ num_succeeded: Integer,
125
+ num_urls: Integer
126
+ }
127
+ )
128
+ end
129
+ def to_hash
130
+ end
131
+ end
132
+ end
133
+ end
134
+ end
@@ -34,6 +34,8 @@ module ContextDev
34
34
 
35
35
  WebExtractFontsParams = ContextDev::Models::WebExtractFontsParams
36
36
 
37
+ WebExtractParams = ContextDev::Models::WebExtractParams
38
+
37
39
  WebExtractStyleguideParams = ContextDev::Models::WebExtractStyleguideParams
38
40
 
39
41
  WebScreenshotParams = ContextDev::Models::WebScreenshotParams
@@ -3,6 +3,63 @@
3
3
  module ContextDev
4
4
  module Resources
5
5
  class Web
6
+ # Crawl a website, use the provided JSON Schema and instructions to prioritize
7
+ # relevant internal links, and extract structured data from the selected pages.
8
+ sig do
9
+ params(
10
+ schema: T::Hash[Symbol, T.anything],
11
+ url: String,
12
+ fact_check: T::Boolean,
13
+ follow_subdomains: T::Boolean,
14
+ include_frames: T::Boolean,
15
+ instructions: String,
16
+ max_age_ms: Integer,
17
+ pdf: ContextDev::WebExtractParams::Pdf::OrHash,
18
+ stop_after_ms: Integer,
19
+ timeout_ms: Integer,
20
+ wait_for_ms: Integer,
21
+ request_options: ContextDev::RequestOptions::OrHash
22
+ ).returns(ContextDev::Models::WebExtractResponse)
23
+ end
24
+ def extract(
25
+ # JSON Schema for the returned data object. TypeScript Zod users can pass a JSON
26
+ # Schema generated from a Zod object; Python users can pass the equivalent JSON
27
+ # Schema object.
28
+ schema:,
29
+ # The starting website URL to crawl and extract from. Must include http:// or
30
+ # https://.
31
+ url:,
32
+ # When true, every returned value must be grounded in facts stated on the page;
33
+ # fields that cannot be supported by the page are returned as null/empty. When
34
+ # false (default), the model may make reasonable inferences and derivations from
35
+ # the page content (e.g. ideal customer, competitor analysis, recommendations)
36
+ # while keeping verifiable specifics (names, quotes, URLs, dates, metrics)
37
+ # faithful to the source.
38
+ fact_check: nil,
39
+ # When true, follow links on subdomains of the starting URL's domain.
40
+ follow_subdomains: nil,
41
+ # When true, iframe contents are included in Markdown before extraction.
42
+ include_frames: nil,
43
+ # Optional extraction guidance, such as which facts to prioritize or how to
44
+ # interpret fields in the schema.
45
+ instructions: nil,
46
+ # Return cached scrape results if a prior scrape for the same parameters is
47
+ # younger than this many milliseconds. Defaults to 7 days (604800000 ms).
48
+ max_age_ms: nil,
49
+ pdf: nil,
50
+ # Soft time budget for the crawl in milliseconds.
51
+ stop_after_ms: nil,
52
+ # Optional timeout in milliseconds for the request. If the request takes longer
53
+ # than this value, it will be aborted with a 408 status code. Maximum allowed
54
+ # value is 300000ms (5 minutes).
55
+ timeout_ms: nil,
56
+ # Optional browser wait time in milliseconds after initial page load for each
57
+ # crawled page.
58
+ wait_for_ms: nil,
59
+ request_options: {}
60
+ )
61
+ end
62
+
6
63
  # Scrape font information from a website including font families, usage
7
64
  # statistics, fallbacks, and element/word counts.
8
65
  sig do
@@ -0,0 +1,120 @@
1
+ module ContextDev
2
+ module Models
3
+ type web_extract_params =
4
+ {
5
+ schema: ::Hash[Symbol, top],
6
+ url: String,
7
+ fact_check: bool,
8
+ follow_subdomains: bool,
9
+ include_frames: bool,
10
+ instructions: String,
11
+ max_age_ms: Integer,
12
+ pdf: ContextDev::WebExtractParams::Pdf,
13
+ stop_after_ms: Integer,
14
+ timeout_ms: Integer,
15
+ wait_for_ms: Integer
16
+ }
17
+ & ContextDev::Internal::Type::request_parameters
18
+
19
+ class WebExtractParams < ContextDev::Internal::Type::BaseModel
20
+ extend ContextDev::Internal::Type::RequestParameters::Converter
21
+ include ContextDev::Internal::Type::RequestParameters
22
+
23
+ attr_accessor schema: ::Hash[Symbol, top]
24
+
25
+ attr_accessor url: String
26
+
27
+ attr_reader fact_check: bool?
28
+
29
+ def fact_check=: (bool) -> bool
30
+
31
+ attr_reader follow_subdomains: bool?
32
+
33
+ def follow_subdomains=: (bool) -> bool
34
+
35
+ attr_reader include_frames: bool?
36
+
37
+ def include_frames=: (bool) -> bool
38
+
39
+ attr_reader instructions: String?
40
+
41
+ def instructions=: (String) -> String
42
+
43
+ attr_reader max_age_ms: Integer?
44
+
45
+ def max_age_ms=: (Integer) -> Integer
46
+
47
+ attr_reader pdf: ContextDev::WebExtractParams::Pdf?
48
+
49
+ def pdf=: (
50
+ ContextDev::WebExtractParams::Pdf
51
+ ) -> ContextDev::WebExtractParams::Pdf
52
+
53
+ attr_reader stop_after_ms: Integer?
54
+
55
+ def stop_after_ms=: (Integer) -> Integer
56
+
57
+ attr_reader timeout_ms: Integer?
58
+
59
+ def timeout_ms=: (Integer) -> Integer
60
+
61
+ attr_reader wait_for_ms: Integer?
62
+
63
+ def wait_for_ms=: (Integer) -> Integer
64
+
65
+ def initialize: (
66
+ schema: ::Hash[Symbol, top],
67
+ url: String,
68
+ ?fact_check: bool,
69
+ ?follow_subdomains: bool,
70
+ ?include_frames: bool,
71
+ ?instructions: String,
72
+ ?max_age_ms: Integer,
73
+ ?pdf: ContextDev::WebExtractParams::Pdf,
74
+ ?stop_after_ms: Integer,
75
+ ?timeout_ms: Integer,
76
+ ?wait_for_ms: Integer,
77
+ ?request_options: ContextDev::request_opts
78
+ ) -> void
79
+
80
+ def to_hash: -> {
81
+ schema: ::Hash[Symbol, top],
82
+ url: String,
83
+ fact_check: bool,
84
+ follow_subdomains: bool,
85
+ include_frames: bool,
86
+ instructions: String,
87
+ max_age_ms: Integer,
88
+ pdf: ContextDev::WebExtractParams::Pdf,
89
+ stop_after_ms: Integer,
90
+ timeout_ms: Integer,
91
+ wait_for_ms: Integer,
92
+ request_options: ContextDev::RequestOptions
93
+ }
94
+
95
+ type pdf = { end_: Integer, should_parse: bool, start: Integer }
96
+
97
+ class Pdf < ContextDev::Internal::Type::BaseModel
98
+ attr_reader end_: Integer?
99
+
100
+ def end_=: (Integer) -> Integer
101
+
102
+ attr_reader should_parse: bool?
103
+
104
+ def should_parse=: (bool) -> bool
105
+
106
+ attr_reader start: Integer?
107
+
108
+ def start=: (Integer) -> Integer
109
+
110
+ def initialize: (
111
+ ?end_: Integer,
112
+ ?should_parse: bool,
113
+ ?start: Integer
114
+ ) -> void
115
+
116
+ def to_hash: -> { end_: Integer, should_parse: bool, start: Integer }
117
+ end
118
+ end
119
+ end
120
+ end
@@ -0,0 +1,77 @@
1
+ module ContextDev
2
+ module Models
3
+ type web_extract_response =
4
+ {
5
+ data: ::Hash[Symbol, top],
6
+ metadata: ContextDev::Models::WebExtractResponse::Metadata,
7
+ status: String,
8
+ url: String,
9
+ urls_analyzed: ::Array[String]
10
+ }
11
+
12
+ class WebExtractResponse < ContextDev::Internal::Type::BaseModel
13
+ attr_accessor data: ::Hash[Symbol, top]
14
+
15
+ attr_accessor metadata: ContextDev::Models::WebExtractResponse::Metadata
16
+
17
+ attr_accessor status: String
18
+
19
+ attr_accessor url: String
20
+
21
+ attr_accessor urls_analyzed: ::Array[String]
22
+
23
+ def initialize: (
24
+ data: ::Hash[Symbol, top],
25
+ metadata: ContextDev::Models::WebExtractResponse::Metadata,
26
+ status: String,
27
+ url: String,
28
+ urls_analyzed: ::Array[String]
29
+ ) -> void
30
+
31
+ def to_hash: -> {
32
+ data: ::Hash[Symbol, top],
33
+ metadata: ContextDev::Models::WebExtractResponse::Metadata,
34
+ status: String,
35
+ url: String,
36
+ urls_analyzed: ::Array[String]
37
+ }
38
+
39
+ type metadata =
40
+ {
41
+ max_crawl_depth: Integer,
42
+ num_failed: Integer,
43
+ num_skipped: Integer,
44
+ num_succeeded: Integer,
45
+ num_urls: Integer
46
+ }
47
+
48
+ class Metadata < ContextDev::Internal::Type::BaseModel
49
+ attr_accessor max_crawl_depth: Integer
50
+
51
+ attr_accessor num_failed: Integer
52
+
53
+ attr_accessor num_skipped: Integer
54
+
55
+ attr_accessor num_succeeded: Integer
56
+
57
+ attr_accessor num_urls: Integer
58
+
59
+ def initialize: (
60
+ max_crawl_depth: Integer,
61
+ num_failed: Integer,
62
+ num_skipped: Integer,
63
+ num_succeeded: Integer,
64
+ num_urls: Integer
65
+ ) -> void
66
+
67
+ def to_hash: -> {
68
+ max_crawl_depth: Integer,
69
+ num_failed: Integer,
70
+ num_skipped: Integer,
71
+ num_succeeded: Integer,
72
+ num_urls: Integer
73
+ }
74
+ end
75
+ end
76
+ end
77
+ end
@@ -29,6 +29,8 @@ module ContextDev
29
29
 
30
30
  class WebExtractFontsParams = ContextDev::Models::WebExtractFontsParams
31
31
 
32
+ class WebExtractParams = ContextDev::Models::WebExtractParams
33
+
32
34
  class WebExtractStyleguideParams = ContextDev::Models::WebExtractStyleguideParams
33
35
 
34
36
  class WebScreenshotParams = ContextDev::Models::WebScreenshotParams
@@ -1,6 +1,21 @@
1
1
  module ContextDev
2
2
  module Resources
3
3
  class Web
4
+ def extract: (
5
+ schema: ::Hash[Symbol, top],
6
+ url: String,
7
+ ?fact_check: bool,
8
+ ?follow_subdomains: bool,
9
+ ?include_frames: bool,
10
+ ?instructions: String,
11
+ ?max_age_ms: Integer,
12
+ ?pdf: ContextDev::WebExtractParams::Pdf,
13
+ ?stop_after_ms: Integer,
14
+ ?timeout_ms: Integer,
15
+ ?wait_for_ms: Integer,
16
+ ?request_options: ContextDev::request_opts
17
+ ) -> ContextDev::Models::WebExtractResponse
18
+
4
19
  def extract_fonts: (
5
20
  ?direct_url: String,
6
21
  ?domain: String,
metadata CHANGED
@@ -1,14 +1,14 @@
1
1
  --- !ruby/object:Gem::Specification
2
2
  name: context.dev
3
3
  version: !ruby/object:Gem::Version
4
- version: 1.24.0
4
+ version: 1.26.0
5
5
  platform: ruby
6
6
  authors:
7
7
  - Context Dev
8
8
  autorequire:
9
9
  bindir: bin
10
10
  cert_chain: []
11
- date: 2026-05-30 00:00:00.000000000 Z
11
+ date: 2026-06-01 00:00:00.000000000 Z
12
12
  dependencies:
13
13
  - !ruby/object:Gem::Dependency
14
14
  name: cgi
@@ -99,6 +99,8 @@ files:
99
99
  - lib/context_dev/models/utility_prefetch_response.rb
100
100
  - lib/context_dev/models/web_extract_fonts_params.rb
101
101
  - lib/context_dev/models/web_extract_fonts_response.rb
102
+ - lib/context_dev/models/web_extract_params.rb
103
+ - lib/context_dev/models/web_extract_response.rb
102
104
  - lib/context_dev/models/web_extract_styleguide_params.rb
103
105
  - lib/context_dev/models/web_extract_styleguide_response.rb
104
106
  - lib/context_dev/models/web_screenshot_params.rb
@@ -172,6 +174,8 @@ files:
172
174
  - rbi/context_dev/models/utility_prefetch_response.rbi
173
175
  - rbi/context_dev/models/web_extract_fonts_params.rbi
174
176
  - rbi/context_dev/models/web_extract_fonts_response.rbi
177
+ - rbi/context_dev/models/web_extract_params.rbi
178
+ - rbi/context_dev/models/web_extract_response.rbi
175
179
  - rbi/context_dev/models/web_extract_styleguide_params.rbi
176
180
  - rbi/context_dev/models/web_extract_styleguide_response.rbi
177
181
  - rbi/context_dev/models/web_screenshot_params.rbi
@@ -244,6 +248,8 @@ files:
244
248
  - sig/context_dev/models/utility_prefetch_response.rbs
245
249
  - sig/context_dev/models/web_extract_fonts_params.rbs
246
250
  - sig/context_dev/models/web_extract_fonts_response.rbs
251
+ - sig/context_dev/models/web_extract_params.rbs
252
+ - sig/context_dev/models/web_extract_response.rbs
247
253
  - sig/context_dev/models/web_extract_styleguide_params.rbs
248
254
  - sig/context_dev/models/web_extract_styleguide_response.rbs
249
255
  - sig/context_dev/models/web_screenshot_params.rbs