context.dev 2.2.0 → 2.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +22 -0
- data/README.md +9 -9
- data/lib/context_dev/client.rb +7 -2
- data/lib/context_dev/internal/type/base_model.rb +5 -5
- data/lib/context_dev/models/monitor_create_params.rb +28 -3
- data/lib/context_dev/models/monitor_create_response.rb +96 -4
- data/lib/context_dev/models/monitor_list_account_runs_response.rb +19 -98
- data/lib/context_dev/models/monitor_list_response.rb +100 -4
- data/lib/context_dev/models/monitor_list_runs_response.rb +19 -96
- data/lib/context_dev/models/monitor_retrieve_change_response.rb +10 -10
- data/lib/context_dev/models/monitor_retrieve_response.rb +97 -4
- data/lib/context_dev/models/monitor_update_params.rb +28 -3
- data/lib/context_dev/models/monitor_update_response.rb +96 -4
- data/lib/context_dev/models/parse_handle_params.rb +194 -0
- data/lib/context_dev/models/parse_handle_response.rb +132 -0
- data/lib/context_dev/models/web_web_crawl_md_params.rb +16 -6
- data/lib/context_dev/models/web_web_scrape_html_params.rb +16 -6
- data/lib/context_dev/models/web_web_scrape_md_params.rb +16 -6
- data/lib/context_dev/models/web_web_scrape_md_response.rb +11 -1
- data/lib/context_dev/models/webhook_delivery.rb +109 -0
- data/lib/context_dev/models.rb +4 -0
- data/lib/context_dev/resources/monitors.rb +3 -2
- data/lib/context_dev/resources/parse.rb +59 -0
- data/lib/context_dev/resources/web.rb +19 -4
- data/lib/context_dev/version.rb +1 -1
- data/lib/context_dev.rb +4 -0
- data/rbi/context_dev/client.rbi +6 -2
- data/rbi/context_dev/models/monitor_create_params.rbi +85 -5
- data/rbi/context_dev/models/monitor_create_response.rbi +234 -6
- data/rbi/context_dev/models/monitor_list_account_runs_response.rbi +27 -192
- data/rbi/context_dev/models/monitor_list_response.rbi +235 -5
- data/rbi/context_dev/models/monitor_list_runs_response.rbi +27 -192
- data/rbi/context_dev/models/monitor_retrieve_change_response.rbi +12 -15
- data/rbi/context_dev/models/monitor_retrieve_response.rbi +234 -6
- data/rbi/context_dev/models/monitor_update_params.rbi +85 -5
- data/rbi/context_dev/models/monitor_update_response.rbi +234 -6
- data/rbi/context_dev/models/parse_handle_params.rbi +339 -0
- data/rbi/context_dev/models/parse_handle_response.rbi +377 -0
- data/rbi/context_dev/models/web_web_crawl_md_params.rbi +26 -7
- data/rbi/context_dev/models/web_web_scrape_html_params.rbi +26 -7
- data/rbi/context_dev/models/web_web_scrape_md_params.rbi +26 -7
- data/rbi/context_dev/models/web_web_scrape_md_response.rbi +12 -0
- data/rbi/context_dev/models/webhook_delivery.rbi +172 -0
- data/rbi/context_dev/models.rbi +4 -0
- data/rbi/context_dev/resources/monitors.rbi +3 -2
- data/rbi/context_dev/resources/parse.rbi +58 -0
- data/rbi/context_dev/resources/web.rbi +22 -7
- data/sig/context_dev/client.rbs +2 -0
- data/sig/context_dev/models/monitor_create_params.rbs +32 -3
- data/sig/context_dev/models/monitor_create_response.rbs +85 -6
- data/sig/context_dev/models/monitor_list_account_runs_response.rbs +15 -68
- data/sig/context_dev/models/monitor_list_response.rbs +85 -6
- data/sig/context_dev/models/monitor_list_runs_response.rbs +15 -68
- data/sig/context_dev/models/monitor_retrieve_change_response.rbs +8 -10
- data/sig/context_dev/models/monitor_retrieve_response.rbs +85 -6
- data/sig/context_dev/models/monitor_update_params.rbs +32 -3
- data/sig/context_dev/models/monitor_update_response.rbs +85 -6
- data/sig/context_dev/models/parse_handle_params.rbs +244 -0
- data/sig/context_dev/models/parse_handle_response.rbs +159 -0
- data/sig/context_dev/models/web_web_crawl_md_params.rbs +13 -2
- data/sig/context_dev/models/web_web_scrape_html_params.rbs +13 -2
- data/sig/context_dev/models/web_web_scrape_md_params.rbs +13 -2
- data/sig/context_dev/models/web_web_scrape_md_response.rbs +5 -0
- data/sig/context_dev/models/webhook_delivery.rbs +81 -0
- data/sig/context_dev/models.rbs +4 -0
- data/sig/context_dev/resources/parse.rbs +19 -0
- metadata +14 -2
|
@@ -0,0 +1,194 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module ContextDev
|
|
4
|
+
module Models
|
|
5
|
+
# @see ContextDev::Resources::Parse#handle
|
|
6
|
+
class ParseHandleParams < ContextDev::Internal::Type::BaseModel
|
|
7
|
+
extend ContextDev::Internal::Type::RequestParameters::Converter
|
|
8
|
+
include ContextDev::Internal::Type::RequestParameters
|
|
9
|
+
|
|
10
|
+
# @!attribute body
|
|
11
|
+
#
|
|
12
|
+
# @return [Pathname, StringIO, IO, String, ContextDev::FilePart]
|
|
13
|
+
required :body, ContextDev::Internal::Type::FileInput
|
|
14
|
+
|
|
15
|
+
# @!attribute extension
|
|
16
|
+
# Optional file extension hint. Case-insensitive; a leading dot is accepted (e.g.
|
|
17
|
+
# ".pdf").
|
|
18
|
+
#
|
|
19
|
+
# @return [Symbol, ContextDev::Models::ParseHandleParams::Extension, nil]
|
|
20
|
+
optional :extension, enum: -> { ContextDev::ParseHandleParams::Extension }
|
|
21
|
+
|
|
22
|
+
# @!attribute include_images
|
|
23
|
+
# Include image references in Markdown output
|
|
24
|
+
#
|
|
25
|
+
# @return [Boolean, nil]
|
|
26
|
+
optional :include_images, ContextDev::Internal::Type::Boolean
|
|
27
|
+
|
|
28
|
+
# @!attribute include_links
|
|
29
|
+
# Preserve hyperlinks in Markdown output
|
|
30
|
+
#
|
|
31
|
+
# @return [Boolean, nil]
|
|
32
|
+
optional :include_links, ContextDev::Internal::Type::Boolean
|
|
33
|
+
|
|
34
|
+
# @!attribute ocr
|
|
35
|
+
# Gates all OCR. When true, PDFs get embedded-image OCR (recognized text inserted
|
|
36
|
+
# at each image's position in page reading order, preserving the text layer;
|
|
37
|
+
# pdf.start/pdf.end limit the page range), scanned PDFs with no text layer get
|
|
38
|
+
# full-document OCR, and raster images get their visible text transcribed. When
|
|
39
|
+
# false, no OCR runs: scanned PDFs may yield no content and images return only
|
|
40
|
+
# format/dimension metadata. Calls where OCR actually runs cost 5 credits instead
|
|
41
|
+
# of 1.
|
|
42
|
+
#
|
|
43
|
+
# @return [Boolean, nil]
|
|
44
|
+
optional :ocr, ContextDev::Internal::Type::Boolean
|
|
45
|
+
|
|
46
|
+
# @!attribute pdf
|
|
47
|
+
# PDF page-range controls. Use start/end to limit parsing (and OCR when ocr=true)
|
|
48
|
+
# to an inclusive 1-based page range.
|
|
49
|
+
#
|
|
50
|
+
# @return [ContextDev::Models::ParseHandleParams::Pdf, nil]
|
|
51
|
+
optional :pdf, -> { ContextDev::ParseHandleParams::Pdf }
|
|
52
|
+
|
|
53
|
+
# @!attribute shorten_base64_images
|
|
54
|
+
# Shorten base64-encoded image data in the Markdown output
|
|
55
|
+
#
|
|
56
|
+
# @return [Boolean, nil]
|
|
57
|
+
optional :shorten_base64_images, ContextDev::Internal::Type::Boolean
|
|
58
|
+
|
|
59
|
+
# @!attribute use_main_content_only
|
|
60
|
+
# Extract only the main content from HTML-like inputs
|
|
61
|
+
#
|
|
62
|
+
# @return [Boolean, nil]
|
|
63
|
+
optional :use_main_content_only, ContextDev::Internal::Type::Boolean
|
|
64
|
+
|
|
65
|
+
# @!method initialize(body:, extension: nil, include_images: nil, include_links: nil, ocr: nil, pdf: nil, shorten_base64_images: nil, use_main_content_only: nil, request_options: {})
|
|
66
|
+
# Some parameter documentations has been truncated, see
|
|
67
|
+
# {ContextDev::Models::ParseHandleParams} for more details.
|
|
68
|
+
#
|
|
69
|
+
# @param body [Pathname, StringIO, IO, String, ContextDev::FilePart]
|
|
70
|
+
#
|
|
71
|
+
# @param extension [Symbol, ContextDev::Models::ParseHandleParams::Extension] Optional file extension hint. Case-insensitive; a leading dot is accepted (e.g.
|
|
72
|
+
#
|
|
73
|
+
# @param include_images [Boolean] Include image references in Markdown output
|
|
74
|
+
#
|
|
75
|
+
# @param include_links [Boolean] Preserve hyperlinks in Markdown output
|
|
76
|
+
#
|
|
77
|
+
# @param ocr [Boolean] Gates all OCR. When true, PDFs get embedded-image OCR (recognized text inserted
|
|
78
|
+
#
|
|
79
|
+
# @param pdf [ContextDev::Models::ParseHandleParams::Pdf] PDF page-range controls. Use start/end to limit parsing (and OCR when ocr=true)
|
|
80
|
+
#
|
|
81
|
+
# @param shorten_base64_images [Boolean] Shorten base64-encoded image data in the Markdown output
|
|
82
|
+
#
|
|
83
|
+
# @param use_main_content_only [Boolean] Extract only the main content from HTML-like inputs
|
|
84
|
+
#
|
|
85
|
+
# @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}]
|
|
86
|
+
|
|
87
|
+
# Optional file extension hint. Case-insensitive; a leading dot is accepted (e.g.
|
|
88
|
+
# ".pdf").
|
|
89
|
+
module Extension
|
|
90
|
+
extend ContextDev::Internal::Type::Enum
|
|
91
|
+
|
|
92
|
+
TXT = :txt
|
|
93
|
+
TEXT = :text
|
|
94
|
+
MD = :md
|
|
95
|
+
MARKDOWN = :markdown
|
|
96
|
+
HTML = :html
|
|
97
|
+
HTM = :htm
|
|
98
|
+
XHTML = :xhtml
|
|
99
|
+
XML = :xml
|
|
100
|
+
RSS = :rss
|
|
101
|
+
ATOM = :atom
|
|
102
|
+
CSV = :csv
|
|
103
|
+
TSV = :tsv
|
|
104
|
+
YAML = :yaml
|
|
105
|
+
YML = :yml
|
|
106
|
+
PY = :py
|
|
107
|
+
JAVA = :java
|
|
108
|
+
JS = :js
|
|
109
|
+
JSX = :jsx
|
|
110
|
+
MJS = :mjs
|
|
111
|
+
CJS = :cjs
|
|
112
|
+
JSON = :json
|
|
113
|
+
JSONL = :jsonl
|
|
114
|
+
NDJSON = :ndjson
|
|
115
|
+
PHP = :php
|
|
116
|
+
SH = :sh
|
|
117
|
+
BASH = :bash
|
|
118
|
+
ZSH = :zsh
|
|
119
|
+
FISH = :fish
|
|
120
|
+
RB = :rb
|
|
121
|
+
TS = :ts
|
|
122
|
+
TSX = :tsx
|
|
123
|
+
RTF = :rtf
|
|
124
|
+
SRT = :srt
|
|
125
|
+
CSS = :css
|
|
126
|
+
SCSS = :scss
|
|
127
|
+
LESS = :less
|
|
128
|
+
STYL = :styl
|
|
129
|
+
SASS = :sass
|
|
130
|
+
SVG = :svg
|
|
131
|
+
PDF = :pdf
|
|
132
|
+
DOCX = :docx
|
|
133
|
+
DOC = :doc
|
|
134
|
+
XLSX = :xlsx
|
|
135
|
+
XLSM = :xlsm
|
|
136
|
+
XLSB = :xlsb
|
|
137
|
+
XLTX = :xltx
|
|
138
|
+
XLTM = :xltm
|
|
139
|
+
XLS = :xls
|
|
140
|
+
PPTX = :pptx
|
|
141
|
+
PPTM = :pptm
|
|
142
|
+
PPSX = :ppsx
|
|
143
|
+
PPSM = :ppsm
|
|
144
|
+
POTX = :potx
|
|
145
|
+
POTM = :potm
|
|
146
|
+
PPT = :ppt
|
|
147
|
+
PPS = :pps
|
|
148
|
+
POT = :pot
|
|
149
|
+
JPG = :jpg
|
|
150
|
+
JPEG = :jpeg
|
|
151
|
+
JPE = :jpe
|
|
152
|
+
PNG = :png
|
|
153
|
+
GIF = :gif
|
|
154
|
+
BMP = :bmp
|
|
155
|
+
TIFF = :tiff
|
|
156
|
+
TIF = :tif
|
|
157
|
+
WEBP = :webp
|
|
158
|
+
PPM = :ppm
|
|
159
|
+
PBM = :pbm
|
|
160
|
+
PGM = :pgm
|
|
161
|
+
PNM = :pnm
|
|
162
|
+
|
|
163
|
+
# @!method self.values
|
|
164
|
+
# @return [Array<Symbol>]
|
|
165
|
+
end
|
|
166
|
+
|
|
167
|
+
class Pdf < ContextDev::Internal::Type::BaseModel
|
|
168
|
+
# @!attribute end_
|
|
169
|
+
# Last 1-based PDF page to parse. When omitted, parsing ends at the last page.
|
|
170
|
+
# Must be greater than or equal to start when both are provided.
|
|
171
|
+
#
|
|
172
|
+
# @return [Integer, nil]
|
|
173
|
+
optional :end_, Integer, api_name: :end
|
|
174
|
+
|
|
175
|
+
# @!attribute start
|
|
176
|
+
# First 1-based PDF page to parse. When omitted, parsing starts at the first page.
|
|
177
|
+
#
|
|
178
|
+
# @return [Integer, nil]
|
|
179
|
+
optional :start, Integer
|
|
180
|
+
|
|
181
|
+
# @!method initialize(end_: nil, start: nil)
|
|
182
|
+
# Some parameter documentations has been truncated, see
|
|
183
|
+
# {ContextDev::Models::ParseHandleParams::Pdf} for more details.
|
|
184
|
+
#
|
|
185
|
+
# PDF page-range controls. Use start/end to limit parsing (and OCR when ocr=true)
|
|
186
|
+
# to an inclusive 1-based page range.
|
|
187
|
+
#
|
|
188
|
+
# @param end_ [Integer] Last 1-based PDF page to parse. When omitted, parsing ends at the last page. Mus
|
|
189
|
+
#
|
|
190
|
+
# @param start [Integer] First 1-based PDF page to parse. When omitted, parsing starts at the first page.
|
|
191
|
+
end
|
|
192
|
+
end
|
|
193
|
+
end
|
|
194
|
+
end
|
|
@@ -0,0 +1,132 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module ContextDev
|
|
4
|
+
module Models
|
|
5
|
+
# @see ContextDev::Resources::Parse#handle
|
|
6
|
+
class ParseHandleResponse < ContextDev::Internal::Type::BaseModel
|
|
7
|
+
# @!attribute markdown
|
|
8
|
+
# Input bytes converted to GitHub Flavored Markdown
|
|
9
|
+
#
|
|
10
|
+
# @return [String]
|
|
11
|
+
required :markdown, String
|
|
12
|
+
|
|
13
|
+
# @!attribute success
|
|
14
|
+
# Indicates success
|
|
15
|
+
#
|
|
16
|
+
# @return [Boolean, ContextDev::Models::ParseHandleResponse::Success]
|
|
17
|
+
required :success, enum: -> { ContextDev::Models::ParseHandleResponse::Success }
|
|
18
|
+
|
|
19
|
+
# @!attribute type
|
|
20
|
+
# Detected content type used for parsing
|
|
21
|
+
#
|
|
22
|
+
# @return [Symbol, ContextDev::Models::ParseHandleResponse::Type]
|
|
23
|
+
required :type, enum: -> { ContextDev::Models::ParseHandleResponse::Type }
|
|
24
|
+
|
|
25
|
+
# @!attribute key_metadata
|
|
26
|
+
# Metadata about the API key used for the request. Included in every response
|
|
27
|
+
# whenever a valid API key is provided, even when the response status is not 200.
|
|
28
|
+
#
|
|
29
|
+
# @return [ContextDev::Models::ParseHandleResponse::KeyMetadata, nil]
|
|
30
|
+
optional :key_metadata, -> { ContextDev::Models::ParseHandleResponse::KeyMetadata }
|
|
31
|
+
|
|
32
|
+
# @!method initialize(markdown:, success:, type:, key_metadata: nil)
|
|
33
|
+
# Some parameter documentations has been truncated, see
|
|
34
|
+
# {ContextDev::Models::ParseHandleResponse} for more details.
|
|
35
|
+
#
|
|
36
|
+
# @param markdown [String] Input bytes converted to GitHub Flavored Markdown
|
|
37
|
+
#
|
|
38
|
+
# @param success [Boolean, ContextDev::Models::ParseHandleResponse::Success] Indicates success
|
|
39
|
+
#
|
|
40
|
+
# @param type [Symbol, ContextDev::Models::ParseHandleResponse::Type] Detected content type used for parsing
|
|
41
|
+
#
|
|
42
|
+
# @param key_metadata [ContextDev::Models::ParseHandleResponse::KeyMetadata] Metadata about the API key used for the request. Included in every response when
|
|
43
|
+
|
|
44
|
+
# Indicates success
|
|
45
|
+
#
|
|
46
|
+
# @see ContextDev::Models::ParseHandleResponse#success
|
|
47
|
+
module Success
|
|
48
|
+
extend ContextDev::Internal::Type::Enum
|
|
49
|
+
|
|
50
|
+
TRUE = true
|
|
51
|
+
|
|
52
|
+
# @!method self.values
|
|
53
|
+
# @return [Array<Boolean>]
|
|
54
|
+
end
|
|
55
|
+
|
|
56
|
+
# Detected content type used for parsing
|
|
57
|
+
#
|
|
58
|
+
# @see ContextDev::Models::ParseHandleResponse#type
|
|
59
|
+
module Type
|
|
60
|
+
extend ContextDev::Internal::Type::Enum
|
|
61
|
+
|
|
62
|
+
HTML = :html
|
|
63
|
+
XML = :xml
|
|
64
|
+
JSON = :json
|
|
65
|
+
JSONL = :jsonl
|
|
66
|
+
TEXT = :text
|
|
67
|
+
CSV = :csv
|
|
68
|
+
TSV = :tsv
|
|
69
|
+
MARKDOWN = :markdown
|
|
70
|
+
YAML = :yaml
|
|
71
|
+
PYTHON = :python
|
|
72
|
+
JAVA = :java
|
|
73
|
+
JAVASCRIPT = :javascript
|
|
74
|
+
PHP = :php
|
|
75
|
+
SHELL = :shell
|
|
76
|
+
RUBY = :ruby
|
|
77
|
+
TYPESCRIPT = :typescript
|
|
78
|
+
RTF = :rtf
|
|
79
|
+
SRT = :srt
|
|
80
|
+
CSS = :css
|
|
81
|
+
SCSS = :scss
|
|
82
|
+
LESS = :less
|
|
83
|
+
STYLUS = :stylus
|
|
84
|
+
SASS = :sass
|
|
85
|
+
SVG = :svg
|
|
86
|
+
PDF = :pdf
|
|
87
|
+
DOCX = :docx
|
|
88
|
+
DOC = :doc
|
|
89
|
+
XLSX = :xlsx
|
|
90
|
+
XLS = :xls
|
|
91
|
+
PPTX = :pptx
|
|
92
|
+
PPT = :ppt
|
|
93
|
+
JPG = :jpg
|
|
94
|
+
PNG = :png
|
|
95
|
+
GIF = :gif
|
|
96
|
+
BMP = :bmp
|
|
97
|
+
TIFF = :tiff
|
|
98
|
+
WEBP = :webp
|
|
99
|
+
PPM = :ppm
|
|
100
|
+
PBM = :pbm
|
|
101
|
+
PGM = :pgm
|
|
102
|
+
PNM = :pnm
|
|
103
|
+
|
|
104
|
+
# @!method self.values
|
|
105
|
+
# @return [Array<Symbol>]
|
|
106
|
+
end
|
|
107
|
+
|
|
108
|
+
# @see ContextDev::Models::ParseHandleResponse#key_metadata
|
|
109
|
+
class KeyMetadata < ContextDev::Internal::Type::BaseModel
|
|
110
|
+
# @!attribute credits_consumed
|
|
111
|
+
# The number of credits consumed by this request.
|
|
112
|
+
#
|
|
113
|
+
# @return [Integer]
|
|
114
|
+
required :credits_consumed, Integer
|
|
115
|
+
|
|
116
|
+
# @!attribute credits_remaining
|
|
117
|
+
# The number of credits remaining for your organization after this request.
|
|
118
|
+
#
|
|
119
|
+
# @return [Integer]
|
|
120
|
+
required :credits_remaining, Integer
|
|
121
|
+
|
|
122
|
+
# @!method initialize(credits_consumed:, credits_remaining:)
|
|
123
|
+
# Metadata about the API key used for the request. Included in every response
|
|
124
|
+
# whenever a valid API key is provided, even when the response status is not 200.
|
|
125
|
+
#
|
|
126
|
+
# @param credits_consumed [Integer] The number of credits consumed by this request.
|
|
127
|
+
#
|
|
128
|
+
# @param credits_remaining [Integer] The number of credits remaining for your organization after this request.
|
|
129
|
+
end
|
|
130
|
+
end
|
|
131
|
+
end
|
|
132
|
+
end
|
|
@@ -86,8 +86,8 @@ module ContextDev
|
|
|
86
86
|
optional :max_pages, Integer, api_name: :maxPages
|
|
87
87
|
|
|
88
88
|
# @!attribute pdf
|
|
89
|
-
# PDF parsing controls. Use start/end to limit text extraction and
|
|
90
|
-
# inclusive 1-based page range.
|
|
89
|
+
# PDF parsing controls. Use start/end to limit text extraction and embedded-image
|
|
90
|
+
# detection/OCR to an inclusive 1-based page range.
|
|
91
91
|
#
|
|
92
92
|
# @return [ContextDev::Models::WebWebCrawlMdParams::Pdf, nil]
|
|
93
93
|
optional :pdf, -> { ContextDev::WebWebCrawlMdParams::Pdf }
|
|
@@ -169,7 +169,7 @@ module ContextDev
|
|
|
169
169
|
#
|
|
170
170
|
# @param max_pages [Integer] Maximum number of pages to crawl. Hard cap: 500.
|
|
171
171
|
#
|
|
172
|
-
# @param pdf [ContextDev::Models::WebWebCrawlMdParams::Pdf] PDF parsing controls. Use start/end to limit text extraction and
|
|
172
|
+
# @param pdf [ContextDev::Models::WebWebCrawlMdParams::Pdf] PDF parsing controls. Use start/end to limit text extraction and embedded-image
|
|
173
173
|
#
|
|
174
174
|
# @param settle_animations [Boolean] When true, waits briefly for CSS and transition animations to settle before extr
|
|
175
175
|
#
|
|
@@ -410,6 +410,14 @@ module ContextDev
|
|
|
410
410
|
# @return [Integer, nil]
|
|
411
411
|
optional :end_, Integer, api_name: :end
|
|
412
412
|
|
|
413
|
+
# @!attribute ocr
|
|
414
|
+
# When true, detect and OCR images embedded in the selected PDF pages, inserting
|
|
415
|
+
# recognized text at each image's position in page reading order while preserving
|
|
416
|
+
# the PDF text layer. This is separate from automatic scanned-PDF OCR fallback.
|
|
417
|
+
#
|
|
418
|
+
# @return [Boolean, nil]
|
|
419
|
+
optional :ocr, ContextDev::Internal::Type::Boolean
|
|
420
|
+
|
|
413
421
|
# @!attribute should_parse
|
|
414
422
|
# When true, PDF pages are fetched and parsed. When false, PDF pages are skipped
|
|
415
423
|
# entirely (not included in results and not counted as failures).
|
|
@@ -423,15 +431,17 @@ module ContextDev
|
|
|
423
431
|
# @return [Integer, nil]
|
|
424
432
|
optional :start, Integer
|
|
425
433
|
|
|
426
|
-
# @!method initialize(end_: nil, should_parse: nil, start: nil)
|
|
434
|
+
# @!method initialize(end_: nil, ocr: nil, should_parse: nil, start: nil)
|
|
427
435
|
# Some parameter documentations has been truncated, see
|
|
428
436
|
# {ContextDev::Models::WebWebCrawlMdParams::Pdf} for more details.
|
|
429
437
|
#
|
|
430
|
-
# PDF parsing controls. Use start/end to limit text extraction and
|
|
431
|
-
# inclusive 1-based page range.
|
|
438
|
+
# PDF parsing controls. Use start/end to limit text extraction and embedded-image
|
|
439
|
+
# detection/OCR to an inclusive 1-based page range.
|
|
432
440
|
#
|
|
433
441
|
# @param end_ [Integer] Last 1-based PDF page to parse. When omitted, parsing ends at the last page. Mus
|
|
434
442
|
#
|
|
443
|
+
# @param ocr [Boolean] When true, detect and OCR images embedded in the selected PDF pages, inserting r
|
|
444
|
+
#
|
|
435
445
|
# @param should_parse [Boolean] When true, PDF pages are fetched and parsed. When false, PDF pages are skipped e
|
|
436
446
|
#
|
|
437
447
|
# @param start [Integer] First 1-based PDF page to parse. When omitted, parsing starts at the first page.
|
|
@@ -59,8 +59,8 @@ module ContextDev
|
|
|
59
59
|
optional :max_age_ms, Integer
|
|
60
60
|
|
|
61
61
|
# @!attribute pdf
|
|
62
|
-
# PDF parsing controls. Use start/end to limit text extraction and
|
|
63
|
-
# inclusive 1-based page range.
|
|
62
|
+
# PDF parsing controls. Use start/end to limit text extraction and embedded-image
|
|
63
|
+
# detection/OCR to an inclusive 1-based page range.
|
|
64
64
|
#
|
|
65
65
|
# @return [ContextDev::Models::WebWebScrapeHTMLParams::Pdf, nil]
|
|
66
66
|
optional :pdf, -> { ContextDev::WebWebScrapeHTMLParams::Pdf }
|
|
@@ -113,7 +113,7 @@ module ContextDev
|
|
|
113
113
|
#
|
|
114
114
|
# @param max_age_ms [Integer] Return a cached result if a prior scrape for the same parameters exists and is y
|
|
115
115
|
#
|
|
116
|
-
# @param pdf [ContextDev::Models::WebWebScrapeHTMLParams::Pdf] PDF parsing controls. Use start/end to limit text extraction and
|
|
116
|
+
# @param pdf [ContextDev::Models::WebWebScrapeHTMLParams::Pdf] PDF parsing controls. Use start/end to limit text extraction and embedded-image
|
|
117
117
|
#
|
|
118
118
|
# @param settle_animations [Boolean] When true, waits briefly for CSS and transition animations to settle before extr
|
|
119
119
|
#
|
|
@@ -347,6 +347,14 @@ module ContextDev
|
|
|
347
347
|
# @return [Integer, nil]
|
|
348
348
|
optional :end_, Integer, api_name: :end
|
|
349
349
|
|
|
350
|
+
# @!attribute ocr
|
|
351
|
+
# When true, detect and OCR images embedded in the selected PDF pages, inserting
|
|
352
|
+
# recognized text at each image's position in page reading order while preserving
|
|
353
|
+
# the PDF text layer. This is separate from automatic scanned-PDF OCR fallback.
|
|
354
|
+
#
|
|
355
|
+
# @return [Boolean, nil]
|
|
356
|
+
optional :ocr, ContextDev::Internal::Type::Boolean
|
|
357
|
+
|
|
350
358
|
# @!attribute should_parse
|
|
351
359
|
# When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
|
|
352
360
|
# a 400 WEBSITE_ACCESS_ERROR is returned.
|
|
@@ -360,15 +368,17 @@ module ContextDev
|
|
|
360
368
|
# @return [Integer, nil]
|
|
361
369
|
optional :start, Integer
|
|
362
370
|
|
|
363
|
-
# @!method initialize(end_: nil, should_parse: nil, start: nil)
|
|
371
|
+
# @!method initialize(end_: nil, ocr: nil, should_parse: nil, start: nil)
|
|
364
372
|
# Some parameter documentations has been truncated, see
|
|
365
373
|
# {ContextDev::Models::WebWebScrapeHTMLParams::Pdf} for more details.
|
|
366
374
|
#
|
|
367
|
-
# PDF parsing controls. Use start/end to limit text extraction and
|
|
368
|
-
# inclusive 1-based page range.
|
|
375
|
+
# PDF parsing controls. Use start/end to limit text extraction and embedded-image
|
|
376
|
+
# detection/OCR to an inclusive 1-based page range.
|
|
369
377
|
#
|
|
370
378
|
# @param end_ [Integer] Last 1-based PDF page to parse. When omitted, parsing ends at the last page. Mus
|
|
371
379
|
#
|
|
380
|
+
# @param ocr [Boolean] When true, detect and OCR images embedded in the selected PDF pages, inserting r
|
|
381
|
+
#
|
|
372
382
|
# @param should_parse [Boolean] When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
|
|
373
383
|
#
|
|
374
384
|
# @param start [Integer] First 1-based PDF page to parse. When omitted, parsing starts at the first page.
|
|
@@ -72,8 +72,8 @@ module ContextDev
|
|
|
72
72
|
optional :max_age_ms, Integer
|
|
73
73
|
|
|
74
74
|
# @!attribute pdf
|
|
75
|
-
# PDF parsing controls. Use start/end to limit text extraction and
|
|
76
|
-
# inclusive 1-based page range.
|
|
75
|
+
# PDF parsing controls. Use start/end to limit text extraction and embedded-image
|
|
76
|
+
# detection/OCR to an inclusive 1-based page range.
|
|
77
77
|
#
|
|
78
78
|
# @return [ContextDev::Models::WebWebScrapeMdParams::Pdf, nil]
|
|
79
79
|
optional :pdf, -> { ContextDev::WebWebScrapeMdParams::Pdf }
|
|
@@ -136,7 +136,7 @@ module ContextDev
|
|
|
136
136
|
#
|
|
137
137
|
# @param max_age_ms [Integer] Return a cached result if a prior scrape for the same parameters exists and is y
|
|
138
138
|
#
|
|
139
|
-
# @param pdf [ContextDev::Models::WebWebScrapeMdParams::Pdf] PDF parsing controls. Use start/end to limit text extraction and
|
|
139
|
+
# @param pdf [ContextDev::Models::WebWebScrapeMdParams::Pdf] PDF parsing controls. Use start/end to limit text extraction and embedded-image
|
|
140
140
|
#
|
|
141
141
|
# @param settle_animations [Boolean] When true, waits briefly for CSS and transition animations to settle before conv
|
|
142
142
|
#
|
|
@@ -372,6 +372,14 @@ module ContextDev
|
|
|
372
372
|
# @return [Integer, nil]
|
|
373
373
|
optional :end_, Integer, api_name: :end
|
|
374
374
|
|
|
375
|
+
# @!attribute ocr
|
|
376
|
+
# When true, detect and OCR images embedded in the selected PDF pages, inserting
|
|
377
|
+
# recognized text at each image's position in page reading order while preserving
|
|
378
|
+
# the PDF text layer. This is separate from automatic scanned-PDF OCR fallback.
|
|
379
|
+
#
|
|
380
|
+
# @return [Boolean, nil]
|
|
381
|
+
optional :ocr, ContextDev::Internal::Type::Boolean
|
|
382
|
+
|
|
375
383
|
# @!attribute should_parse
|
|
376
384
|
# When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
|
|
377
385
|
# a 400 WEBSITE_ACCESS_ERROR is returned.
|
|
@@ -385,15 +393,17 @@ module ContextDev
|
|
|
385
393
|
# @return [Integer, nil]
|
|
386
394
|
optional :start, Integer
|
|
387
395
|
|
|
388
|
-
# @!method initialize(end_: nil, should_parse: nil, start: nil)
|
|
396
|
+
# @!method initialize(end_: nil, ocr: nil, should_parse: nil, start: nil)
|
|
389
397
|
# Some parameter documentations has been truncated, see
|
|
390
398
|
# {ContextDev::Models::WebWebScrapeMdParams::Pdf} for more details.
|
|
391
399
|
#
|
|
392
|
-
# PDF parsing controls. Use start/end to limit text extraction and
|
|
393
|
-
# inclusive 1-based page range.
|
|
400
|
+
# PDF parsing controls. Use start/end to limit text extraction and embedded-image
|
|
401
|
+
# detection/OCR to an inclusive 1-based page range.
|
|
394
402
|
#
|
|
395
403
|
# @param end_ [Integer] Last 1-based PDF page to parse. When omitted, parsing ends at the last page. Mus
|
|
396
404
|
#
|
|
405
|
+
# @param ocr [Boolean] When true, detect and OCR images embedded in the selected PDF pages, inserting r
|
|
406
|
+
#
|
|
397
407
|
# @param should_parse [Boolean] When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
|
|
398
408
|
#
|
|
399
409
|
# @param start [Integer] First 1-based PDF page to parse. When omitted, parsing starts at the first page.
|
|
@@ -4,6 +4,14 @@ module ContextDev
|
|
|
4
4
|
module Models
|
|
5
5
|
# @see ContextDev::Resources::Web#web_scrape_md
|
|
6
6
|
class WebWebScrapeMdResponse < ContextDev::Internal::Type::BaseModel
|
|
7
|
+
# @!attribute content_length
|
|
8
|
+
# UTF-8 byte length of the returned Markdown. Use 0 to identify an empty result
|
|
9
|
+
# and compare small values against your workload's minimum useful-content
|
|
10
|
+
# threshold.
|
|
11
|
+
#
|
|
12
|
+
# @return [Integer]
|
|
13
|
+
required :content_length, Integer, api_name: :contentLength
|
|
14
|
+
|
|
7
15
|
# @!attribute markdown
|
|
8
16
|
# Page content converted to GitHub Flavored Markdown
|
|
9
17
|
#
|
|
@@ -35,10 +43,12 @@ module ContextDev
|
|
|
35
43
|
# @return [ContextDev::Models::WebWebScrapeMdResponse::KeyMetadata, nil]
|
|
36
44
|
optional :key_metadata, -> { ContextDev::Models::WebWebScrapeMdResponse::KeyMetadata }
|
|
37
45
|
|
|
38
|
-
# @!method initialize(markdown:, metadata:, success:, url:, key_metadata: nil)
|
|
46
|
+
# @!method initialize(content_length:, markdown:, metadata:, success:, url:, key_metadata: nil)
|
|
39
47
|
# Some parameter documentations has been truncated, see
|
|
40
48
|
# {ContextDev::Models::WebWebScrapeMdResponse} for more details.
|
|
41
49
|
#
|
|
50
|
+
# @param content_length [Integer] UTF-8 byte length of the returned Markdown. Use 0 to identify an empty result an
|
|
51
|
+
#
|
|
42
52
|
# @param markdown [String] Page content converted to GitHub Flavored Markdown
|
|
43
53
|
#
|
|
44
54
|
# @param metadata [ContextDev::Models::WebWebScrapeMdResponse::Metadata] Metadata extracted from the scraped page HTML.
|
|
@@ -0,0 +1,109 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module ContextDev
|
|
4
|
+
module Models
|
|
5
|
+
class WebhookDelivery < ContextDev::Internal::Type::BaseModel
|
|
6
|
+
# @!attribute attempted_at
|
|
7
|
+
#
|
|
8
|
+
# @return [Time]
|
|
9
|
+
required :attempted_at, Time
|
|
10
|
+
|
|
11
|
+
# @!attribute error
|
|
12
|
+
#
|
|
13
|
+
# @return [ContextDev::Models::WebhookDelivery::Error, nil]
|
|
14
|
+
required :error, -> { ContextDev::WebhookDelivery::Error }, nil?: true
|
|
15
|
+
|
|
16
|
+
# @!attribute event
|
|
17
|
+
# The event this delivery carried. Deliveries recorded before event selection
|
|
18
|
+
# existed report change.detected.
|
|
19
|
+
#
|
|
20
|
+
# @return [Symbol, ContextDev::Models::WebhookDelivery::Event]
|
|
21
|
+
required :event, enum: -> { ContextDev::WebhookDelivery::Event }
|
|
22
|
+
|
|
23
|
+
# @!attribute event_id
|
|
24
|
+
# Identifier sent in the X-Context-Id header.
|
|
25
|
+
#
|
|
26
|
+
# @return [String]
|
|
27
|
+
required :event_id, String
|
|
28
|
+
|
|
29
|
+
# @!attribute http_status
|
|
30
|
+
# The endpoint's final HTTP response status, or null when no response was
|
|
31
|
+
# received.
|
|
32
|
+
#
|
|
33
|
+
# @return [Integer, nil]
|
|
34
|
+
required :http_status, Integer, nil?: true
|
|
35
|
+
|
|
36
|
+
# @!attribute status
|
|
37
|
+
# Delivery outcome. delivered means any 2xx response; rejected means a non-2xx
|
|
38
|
+
# response; failed means no HTTP response was received; skipped_unsafe_url means
|
|
39
|
+
# the URL failed the public-endpoint safety check.
|
|
40
|
+
#
|
|
41
|
+
# @return [Symbol, ContextDev::Models::WebhookDelivery::Status]
|
|
42
|
+
required :status, enum: -> { ContextDev::WebhookDelivery::Status }
|
|
43
|
+
|
|
44
|
+
# @!method initialize(attempted_at:, error:, event:, event_id:, http_status:, status:)
|
|
45
|
+
# Some parameter documentations has been truncated, see
|
|
46
|
+
# {ContextDev::Models::WebhookDelivery} for more details.
|
|
47
|
+
#
|
|
48
|
+
# @param attempted_at [Time]
|
|
49
|
+
#
|
|
50
|
+
# @param error [ContextDev::Models::WebhookDelivery::Error, nil]
|
|
51
|
+
#
|
|
52
|
+
# @param event [Symbol, ContextDev::Models::WebhookDelivery::Event] The event this delivery carried. Deliveries recorded before event selection exis
|
|
53
|
+
#
|
|
54
|
+
# @param event_id [String] Identifier sent in the X-Context-Id header.
|
|
55
|
+
#
|
|
56
|
+
# @param http_status [Integer, nil] The endpoint's final HTTP response status, or null when no response was received
|
|
57
|
+
#
|
|
58
|
+
# @param status [Symbol, ContextDev::Models::WebhookDelivery::Status] Delivery outcome. delivered means any 2xx response; rejected means a non-2xx res
|
|
59
|
+
|
|
60
|
+
# @see ContextDev::Models::WebhookDelivery#error
|
|
61
|
+
class Error < ContextDev::Internal::Type::BaseModel
|
|
62
|
+
# @!attribute code
|
|
63
|
+
#
|
|
64
|
+
# @return [String]
|
|
65
|
+
required :code, String
|
|
66
|
+
|
|
67
|
+
# @!attribute message
|
|
68
|
+
#
|
|
69
|
+
# @return [String]
|
|
70
|
+
required :message, String
|
|
71
|
+
|
|
72
|
+
# @!method initialize(code:, message:)
|
|
73
|
+
# @param code [String]
|
|
74
|
+
# @param message [String]
|
|
75
|
+
end
|
|
76
|
+
|
|
77
|
+
# The event this delivery carried. Deliveries recorded before event selection
|
|
78
|
+
# existed report change.detected.
|
|
79
|
+
#
|
|
80
|
+
# @see ContextDev::Models::WebhookDelivery#event
|
|
81
|
+
module Event
|
|
82
|
+
extend ContextDev::Internal::Type::Enum
|
|
83
|
+
|
|
84
|
+
CHANGE_DETECTED = :"change.detected"
|
|
85
|
+
RUN_COMPLETED = :"run.completed"
|
|
86
|
+
|
|
87
|
+
# @!method self.values
|
|
88
|
+
# @return [Array<Symbol>]
|
|
89
|
+
end
|
|
90
|
+
|
|
91
|
+
# Delivery outcome. delivered means any 2xx response; rejected means a non-2xx
|
|
92
|
+
# response; failed means no HTTP response was received; skipped_unsafe_url means
|
|
93
|
+
# the URL failed the public-endpoint safety check.
|
|
94
|
+
#
|
|
95
|
+
# @see ContextDev::Models::WebhookDelivery#status
|
|
96
|
+
module Status
|
|
97
|
+
extend ContextDev::Internal::Type::Enum
|
|
98
|
+
|
|
99
|
+
DELIVERED = :delivered
|
|
100
|
+
REJECTED = :rejected
|
|
101
|
+
FAILED = :failed
|
|
102
|
+
SKIPPED_UNSAFE_URL = :skipped_unsafe_url
|
|
103
|
+
|
|
104
|
+
# @!method self.values
|
|
105
|
+
# @return [Array<Symbol>]
|
|
106
|
+
end
|
|
107
|
+
end
|
|
108
|
+
end
|
|
109
|
+
end
|
data/lib/context_dev/models.rb
CHANGED
|
@@ -73,6 +73,8 @@ module ContextDev
|
|
|
73
73
|
|
|
74
74
|
MonitorUpdateParams = ContextDev::Models::MonitorUpdateParams
|
|
75
75
|
|
|
76
|
+
ParseHandleParams = ContextDev::Models::ParseHandleParams
|
|
77
|
+
|
|
76
78
|
UtilityPrefetchParams = ContextDev::Models::UtilityPrefetchParams
|
|
77
79
|
|
|
78
80
|
WebExtractCompetitorsParams = ContextDev::Models::WebExtractCompetitorsParams
|
|
@@ -83,6 +85,8 @@ module ContextDev
|
|
|
83
85
|
|
|
84
86
|
WebExtractStyleguideParams = ContextDev::Models::WebExtractStyleguideParams
|
|
85
87
|
|
|
88
|
+
WebhookDelivery = ContextDev::Models::WebhookDelivery
|
|
89
|
+
|
|
86
90
|
WebScreenshotParams = ContextDev::Models::WebScreenshotParams
|
|
87
91
|
|
|
88
92
|
WebSearchParams = ContextDev::Models::WebSearchParams
|
|
@@ -3,8 +3,9 @@
|
|
|
3
3
|
module ContextDev
|
|
4
4
|
module Resources
|
|
5
5
|
# Monitor pages, sitemaps, and extracted website data for exact or semantic
|
|
6
|
-
# changes.
|
|
7
|
-
# MonitorsChangeDetectedWebhookPayload
|
|
6
|
+
# changes. Webhook payloads are documented by the
|
|
7
|
+
# MonitorsChangeDetectedWebhookPayload and MonitorsRunCompletedWebhookPayload
|
|
8
|
+
# schemas.
|
|
8
9
|
class Monitors
|
|
9
10
|
# Some parameter documentations has been truncated, see
|
|
10
11
|
# {ContextDev::Models::MonitorCreateParams} for more details.
|