context.dev 2.3.0 → 2.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
checksums.yaml CHANGED
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  SHA256:
3
- metadata.gz: adde8335edf19dff9d76bc4d66ae8a7a6593a7fe76b77400f1dc7f1d0306b545
4
- data.tar.gz: 06d1ea73b8472749a835d88c6bad7c433976e6f06b5e7e873b42a9309b25df02
3
+ metadata.gz: 6e76a9e12d2d02a5a2fabf5f621579b215d278c973e9b88eb9c032ea9c5fee8f
4
+ data.tar.gz: 5bd0dda46235f5c51359bb91ad1bc53df70f13cb2f49120b6b1b4c5a23958389
5
5
  SHA512:
6
- metadata.gz: d827c5bb02c124f85f5c36d0eee5870d6612e2973b550272074b3c0a6be77f5254ec42383ce8273feb02eb7ecedf0391a6678919004dc392548d4ee7f99a195b
7
- data.tar.gz: cc18411a99e00820eab476005adaacf0ac158cb8a0d0c4e88e9e566be24c8decb1e91e7610864a9568145b28c8a33e97396fa8d7e4f01bfbed76dccd4afccb99
6
+ metadata.gz: c2c06bbf5e65076e9867a7beb132b243ecea991290c8c2a8492a0de26936225c1fa7346f0486f9889aa41d1f3b96f5e544d69269d964523169611b2e8256e849
7
+ data.tar.gz: 4a4e5d37a6a84ce2854b1dba0eb854312964a144a914db69f418f796b29aeed0e54102ce9b0c3bd03877f26bec65ed5b4979e10675fa547707509d4c1d609334
data/CHANGELOG.md CHANGED
@@ -1,5 +1,13 @@
1
1
  # Changelog
2
2
 
3
+ ## 2.4.0 (2026-07-12)
4
+
5
+ Full Changelog: [v2.3.0...v2.4.0](https://github.com/context-dot-dev/context-ruby-sdk/compare/v2.3.0...v2.4.0)
6
+
7
+ ### Features
8
+
9
+ * **api:** api update ([426deb1](https://github.com/context-dot-dev/context-ruby-sdk/commit/426deb18249b214c3ccaa44c1e654e75ef666dd2))
10
+
3
11
  ## 2.3.0 (2026-07-12)
4
12
 
5
13
  Full Changelog: [v2.2.0...v2.3.0](https://github.com/context-dot-dev/context-ruby-sdk/compare/v2.2.0...v2.3.0)
data/README.md CHANGED
@@ -26,7 +26,7 @@ To use this gem, install via Bundler by adding the following to your application
26
26
  <!-- x-release-please-start-version -->
27
27
 
28
28
  ```ruby
29
- gem "context.dev", "~> 2.3.0"
29
+ gem "context.dev", "~> 2.4.0"
30
30
  ```
31
31
 
32
32
  <!-- x-release-please-end -->
@@ -213,25 +213,25 @@ context_dev.brand.retrieve(**params)
213
213
  Since this library does not depend on `sorbet-runtime`, it cannot provide [`T::Enum`](https://sorbet.org/docs/tenum) instances. Instead, we provide "tagged symbols" instead, which is always a primitive at runtime:
214
214
 
215
215
  ```ruby
216
- # :light
217
- puts(ContextDev::WebExtractStyleguideParams::ColorScheme::LIGHT)
216
+ # :txt
217
+ puts(ContextDev::ParseHandleParams::Extension::TXT)
218
218
 
219
- # Revealed type: `T.all(ContextDev::WebExtractStyleguideParams::ColorScheme, Symbol)`
220
- T.reveal_type(ContextDev::WebExtractStyleguideParams::ColorScheme::LIGHT)
219
+ # Revealed type: `T.all(ContextDev::ParseHandleParams::Extension, Symbol)`
220
+ T.reveal_type(ContextDev::ParseHandleParams::Extension::TXT)
221
221
  ```
222
222
 
223
223
  Enum parameters have a "relaxed" type, so you can either pass in enum constants or their literal value:
224
224
 
225
225
  ```ruby
226
226
  # Using the enum constants preserves the tagged type information:
227
- context_dev.web.extract_styleguide(
228
- color_scheme: ContextDev::WebExtractStyleguideParams::ColorScheme::LIGHT,
227
+ context_dev.parse.handle(
228
+ extension: ContextDev::ParseHandleParams::Extension::TXT,
229
229
  # …
230
230
  )
231
231
 
232
232
  # Literal values are also permissible:
233
- context_dev.web.extract_styleguide(
234
- color_scheme: :light,
233
+ context_dev.parse.handle(
234
+ extension: :txt,
235
235
  # …
236
236
  )
237
237
  ```
@@ -12,25 +12,12 @@ module ContextDev
12
12
  # @return [Pathname, StringIO, IO, String, ContextDev::FilePart]
13
13
  required :body, ContextDev::Internal::Type::FileInput
14
14
 
15
- # @!attribute base_url
16
- # Optional HTTP(S) source document URL used to resolve relative links and image
17
- # references. Relative references remain relative when omitted.
18
- #
19
- # @return [String, nil]
20
- optional :base_url, String
21
-
22
15
  # @!attribute extension
23
- # Optional file extension hint, such as pdf, docx, xlsx, pptx, html, json, csv,
24
- # md, py, rtf, jpg, png, or txt.
25
- #
26
- # @return [String, nil]
27
- optional :extension, String
28
-
29
- # @!attribute filename
30
- # Optional filename hint used to infer the extension when extension is omitted.
16
+ # Optional file extension hint. Case-insensitive; a leading dot is accepted (e.g.
17
+ # ".pdf").
31
18
  #
32
- # @return [String, nil]
33
- optional :filename, String
19
+ # @return [Symbol, ContextDev::Models::ParseHandleParams::Extension, nil]
20
+ optional :extension, enum: -> { ContextDev::ParseHandleParams::Extension }
34
21
 
35
22
  # @!attribute include_images
36
23
  # Include image references in Markdown output
@@ -45,26 +32,23 @@ module ContextDev
45
32
  optional :include_links, ContextDev::Internal::Type::Boolean
46
33
 
47
34
  # @!attribute ocr
48
- # When true for PDF inputs, detect and OCR images embedded in the selected pages,
49
- # inserting recognized text at each image's position in page reading order while
50
- # preserving the PDF text layer. pdfStart/pdfEnd limit the inclusive page range.
51
- # This is separate from automatic scanned-PDF OCR fallback.
35
+ # Gates all OCR. When true, PDFs get embedded-image OCR (recognized text inserted
36
+ # at each image's position in page reading order, preserving the text layer;
37
+ # pdf.start/pdf.end limit the page range), scanned PDFs with no text layer get
38
+ # full-document OCR, and raster images get their visible text transcribed. When
39
+ # false, no OCR runs: scanned PDFs may yield no content and images return only
40
+ # format/dimension metadata. Calls where OCR actually runs cost 5 credits instead
41
+ # of 1.
52
42
  #
53
43
  # @return [Boolean, nil]
54
44
  optional :ocr, ContextDev::Internal::Type::Boolean
55
45
 
56
- # @!attribute pdf_end
57
- # Last 1-based PDF page to parse. When omitted, parsing ends at the last page.
58
- # Must be greater than or equal to pdfStart when both are provided.
46
+ # @!attribute pdf
47
+ # PDF page-range controls. Use start/end to limit parsing (and OCR when ocr=true)
48
+ # to an inclusive 1-based page range.
59
49
  #
60
- # @return [Integer, nil]
61
- optional :pdf_end, Integer
62
-
63
- # @!attribute pdf_start
64
- # First 1-based PDF page to parse. When omitted, parsing starts at the first page.
65
- #
66
- # @return [Integer, nil]
67
- optional :pdf_start, Integer
50
+ # @return [ContextDev::Models::ParseHandleParams::Pdf, nil]
51
+ optional :pdf, -> { ContextDev::ParseHandleParams::Pdf }
68
52
 
69
53
  # @!attribute shorten_base64_images
70
54
  # Shorten base64-encoded image data in the Markdown output
@@ -78,33 +62,133 @@ module ContextDev
78
62
  # @return [Boolean, nil]
79
63
  optional :use_main_content_only, ContextDev::Internal::Type::Boolean
80
64
 
81
- # @!method initialize(body:, base_url: nil, extension: nil, filename: nil, include_images: nil, include_links: nil, ocr: nil, pdf_end: nil, pdf_start: nil, shorten_base64_images: nil, use_main_content_only: nil, request_options: {})
65
+ # @!method initialize(body:, extension: nil, include_images: nil, include_links: nil, ocr: nil, pdf: nil, shorten_base64_images: nil, use_main_content_only: nil, request_options: {})
82
66
  # Some parameter documentations has been truncated, see
83
67
  # {ContextDev::Models::ParseHandleParams} for more details.
84
68
  #
85
69
  # @param body [Pathname, StringIO, IO, String, ContextDev::FilePart]
86
70
  #
87
- # @param base_url [String] Optional HTTP(S) source document URL used to resolve relative links and image re
88
- #
89
- # @param extension [String] Optional file extension hint, such as pdf, docx, xlsx, pptx, html, json, csv, md
90
- #
91
- # @param filename [String] Optional filename hint used to infer the extension when extension is omitted.
71
+ # @param extension [Symbol, ContextDev::Models::ParseHandleParams::Extension] Optional file extension hint. Case-insensitive; a leading dot is accepted (e.g.
92
72
  #
93
73
  # @param include_images [Boolean] Include image references in Markdown output
94
74
  #
95
75
  # @param include_links [Boolean] Preserve hyperlinks in Markdown output
96
76
  #
97
- # @param ocr [Boolean] When true for PDF inputs, detect and OCR images embedded in the selected pages,
77
+ # @param ocr [Boolean] Gates all OCR. When true, PDFs get embedded-image OCR (recognized text inserted
98
78
  #
99
- # @param pdf_end [Integer] Last 1-based PDF page to parse. When omitted, parsing ends at the last page. Mus
100
- #
101
- # @param pdf_start [Integer] First 1-based PDF page to parse. When omitted, parsing starts at the first page.
79
+ # @param pdf [ContextDev::Models::ParseHandleParams::Pdf] PDF page-range controls. Use start/end to limit parsing (and OCR when ocr=true)
102
80
  #
103
81
  # @param shorten_base64_images [Boolean] Shorten base64-encoded image data in the Markdown output
104
82
  #
105
83
  # @param use_main_content_only [Boolean] Extract only the main content from HTML-like inputs
106
84
  #
107
85
  # @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}]
86
+
87
+ # Optional file extension hint. Case-insensitive; a leading dot is accepted (e.g.
88
+ # ".pdf").
89
+ module Extension
90
+ extend ContextDev::Internal::Type::Enum
91
+
92
+ TXT = :txt
93
+ TEXT = :text
94
+ MD = :md
95
+ MARKDOWN = :markdown
96
+ HTML = :html
97
+ HTM = :htm
98
+ XHTML = :xhtml
99
+ XML = :xml
100
+ RSS = :rss
101
+ ATOM = :atom
102
+ CSV = :csv
103
+ TSV = :tsv
104
+ YAML = :yaml
105
+ YML = :yml
106
+ PY = :py
107
+ JAVA = :java
108
+ JS = :js
109
+ JSX = :jsx
110
+ MJS = :mjs
111
+ CJS = :cjs
112
+ JSON = :json
113
+ JSONL = :jsonl
114
+ NDJSON = :ndjson
115
+ PHP = :php
116
+ SH = :sh
117
+ BASH = :bash
118
+ ZSH = :zsh
119
+ FISH = :fish
120
+ RB = :rb
121
+ TS = :ts
122
+ TSX = :tsx
123
+ RTF = :rtf
124
+ SRT = :srt
125
+ CSS = :css
126
+ SCSS = :scss
127
+ LESS = :less
128
+ STYL = :styl
129
+ SASS = :sass
130
+ SVG = :svg
131
+ PDF = :pdf
132
+ DOCX = :docx
133
+ DOC = :doc
134
+ XLSX = :xlsx
135
+ XLSM = :xlsm
136
+ XLSB = :xlsb
137
+ XLTX = :xltx
138
+ XLTM = :xltm
139
+ XLS = :xls
140
+ PPTX = :pptx
141
+ PPTM = :pptm
142
+ PPSX = :ppsx
143
+ PPSM = :ppsm
144
+ POTX = :potx
145
+ POTM = :potm
146
+ PPT = :ppt
147
+ PPS = :pps
148
+ POT = :pot
149
+ JPG = :jpg
150
+ JPEG = :jpeg
151
+ JPE = :jpe
152
+ PNG = :png
153
+ GIF = :gif
154
+ BMP = :bmp
155
+ TIFF = :tiff
156
+ TIF = :tif
157
+ WEBP = :webp
158
+ PPM = :ppm
159
+ PBM = :pbm
160
+ PGM = :pgm
161
+ PNM = :pnm
162
+
163
+ # @!method self.values
164
+ # @return [Array<Symbol>]
165
+ end
166
+
167
+ class Pdf < ContextDev::Internal::Type::BaseModel
168
+ # @!attribute end_
169
+ # Last 1-based PDF page to parse. When omitted, parsing ends at the last page.
170
+ # Must be greater than or equal to start when both are provided.
171
+ #
172
+ # @return [Integer, nil]
173
+ optional :end_, Integer, api_name: :end
174
+
175
+ # @!attribute start
176
+ # First 1-based PDF page to parse. When omitted, parsing starts at the first page.
177
+ #
178
+ # @return [Integer, nil]
179
+ optional :start, Integer
180
+
181
+ # @!method initialize(end_: nil, start: nil)
182
+ # Some parameter documentations has been truncated, see
183
+ # {ContextDev::Models::ParseHandleParams::Pdf} for more details.
184
+ #
185
+ # PDF page-range controls. Use start/end to limit parsing (and OCR when ocr=true)
186
+ # to an inclusive 1-based page range.
187
+ #
188
+ # @param end_ [Integer] Last 1-based PDF page to parse. When omitted, parsing ends at the last page. Mus
189
+ #
190
+ # @param start [Integer] First 1-based PDF page to parse. When omitted, parsing starts at the first page.
191
+ end
108
192
  end
109
193
  end
110
194
  end
@@ -7,27 +7,23 @@ module ContextDev
7
7
  # {ContextDev::Models::ParseHandleParams} for more details.
8
8
  #
9
9
  # Converts raw text, source code, web/data, PDF, Microsoft Office, and image bytes
10
- # into LLM-usable Markdown.
10
+ # into LLM-usable Markdown. The base request costs 1 credit. When OCR runs
11
+ # (requires ocr=true), the entire call costs 5 credits; ocr=true requests where no
12
+ # OCR ends up running still cost 1 credit.
11
13
  #
12
- # @overload handle(body:, base_url: nil, extension: nil, filename: nil, include_images: nil, include_links: nil, ocr: nil, pdf_end: nil, pdf_start: nil, shorten_base64_images: nil, use_main_content_only: nil, request_options: {})
14
+ # @overload handle(body:, extension: nil, include_images: nil, include_links: nil, ocr: nil, pdf: nil, shorten_base64_images: nil, use_main_content_only: nil, request_options: {})
13
15
  #
14
16
  # @param body [Pathname, StringIO, IO, String, ContextDev::FilePart] Body param
15
17
  #
16
- # @param base_url [String] Query param: Optional HTTP(S) source document URL used to resolve relative links
17
- #
18
- # @param extension [String] Query param: Optional file extension hint, such as pdf, docx, xlsx, pptx, html,
19
- #
20
- # @param filename [String] Query param: Optional filename hint used to infer the extension when extension i
18
+ # @param extension [Symbol, ContextDev::Models::ParseHandleParams::Extension] Query param: Optional file extension hint. Case-insensitive; a leading dot is ac
21
19
  #
22
20
  # @param include_images [Boolean] Query param: Include image references in Markdown output
23
21
  #
24
22
  # @param include_links [Boolean] Query param: Preserve hyperlinks in Markdown output
25
23
  #
26
- # @param ocr [Boolean] Query param: When true for PDF inputs, detect and OCR images embedded in the sel
27
- #
28
- # @param pdf_end [Integer] Query param: Last 1-based PDF page to parse. When omitted, parsing ends at the l
24
+ # @param ocr [Boolean] Query param: Gates all OCR. When true, PDFs get embedded-image OCR (recognized t
29
25
  #
30
- # @param pdf_start [Integer] Query param: First 1-based PDF page to parse. When omitted, parsing starts at th
26
+ # @param pdf [ContextDev::Models::ParseHandleParams::Pdf] Query param: PDF page-range controls. Use start/end to limit parsing (and OCR wh
31
27
  #
32
28
  # @param shorten_base64_images [Boolean] Query param: Shorten base64-encoded image data in the Markdown output
33
29
  #
@@ -1,5 +1,5 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  module ContextDev
4
- VERSION = "2.3.0"
4
+ VERSION = "2.4.0"
5
5
  end
@@ -14,29 +14,20 @@ module ContextDev
14
14
  sig { returns(ContextDev::Internal::FileInput) }
15
15
  attr_accessor :body
16
16
 
17
- # Optional HTTP(S) source document URL used to resolve relative links and image
18
- # references. Relative references remain relative when omitted.
19
- sig { returns(T.nilable(String)) }
20
- attr_reader :base_url
21
-
22
- sig { params(base_url: String).void }
23
- attr_writer :base_url
24
-
25
- # Optional file extension hint, such as pdf, docx, xlsx, pptx, html, json, csv,
26
- # md, py, rtf, jpg, png, or txt.
27
- sig { returns(T.nilable(String)) }
17
+ # Optional file extension hint. Case-insensitive; a leading dot is accepted (e.g.
18
+ # ".pdf").
19
+ sig do
20
+ returns(T.nilable(ContextDev::ParseHandleParams::Extension::OrSymbol))
21
+ end
28
22
  attr_reader :extension
29
23
 
30
- sig { params(extension: String).void }
24
+ sig do
25
+ params(
26
+ extension: ContextDev::ParseHandleParams::Extension::OrSymbol
27
+ ).void
28
+ end
31
29
  attr_writer :extension
32
30
 
33
- # Optional filename hint used to infer the extension when extension is omitted.
34
- sig { returns(T.nilable(String)) }
35
- attr_reader :filename
36
-
37
- sig { params(filename: String).void }
38
- attr_writer :filename
39
-
40
31
  # Include image references in Markdown output
41
32
  sig { returns(T.nilable(T::Boolean)) }
42
33
  attr_reader :include_images
@@ -51,30 +42,26 @@ module ContextDev
51
42
  sig { params(include_links: T::Boolean).void }
52
43
  attr_writer :include_links
53
44
 
54
- # When true for PDF inputs, detect and OCR images embedded in the selected pages,
55
- # inserting recognized text at each image's position in page reading order while
56
- # preserving the PDF text layer. pdfStart/pdfEnd limit the inclusive page range.
57
- # This is separate from automatic scanned-PDF OCR fallback.
45
+ # Gates all OCR. When true, PDFs get embedded-image OCR (recognized text inserted
46
+ # at each image's position in page reading order, preserving the text layer;
47
+ # pdf.start/pdf.end limit the page range), scanned PDFs with no text layer get
48
+ # full-document OCR, and raster images get their visible text transcribed. When
49
+ # false, no OCR runs: scanned PDFs may yield no content and images return only
50
+ # format/dimension metadata. Calls where OCR actually runs cost 5 credits instead
51
+ # of 1.
58
52
  sig { returns(T.nilable(T::Boolean)) }
59
53
  attr_reader :ocr
60
54
 
61
55
  sig { params(ocr: T::Boolean).void }
62
56
  attr_writer :ocr
63
57
 
64
- # Last 1-based PDF page to parse. When omitted, parsing ends at the last page.
65
- # Must be greater than or equal to pdfStart when both are provided.
66
- sig { returns(T.nilable(Integer)) }
67
- attr_reader :pdf_end
58
+ # PDF page-range controls. Use start/end to limit parsing (and OCR when ocr=true)
59
+ # to an inclusive 1-based page range.
60
+ sig { returns(T.nilable(ContextDev::ParseHandleParams::Pdf)) }
61
+ attr_reader :pdf
68
62
 
69
- sig { params(pdf_end: Integer).void }
70
- attr_writer :pdf_end
71
-
72
- # First 1-based PDF page to parse. When omitted, parsing starts at the first page.
73
- sig { returns(T.nilable(Integer)) }
74
- attr_reader :pdf_start
75
-
76
- sig { params(pdf_start: Integer).void }
77
- attr_writer :pdf_start
63
+ sig { params(pdf: ContextDev::ParseHandleParams::Pdf::OrHash).void }
64
+ attr_writer :pdf
78
65
 
79
66
  # Shorten base64-encoded image data in the Markdown output
80
67
  sig { returns(T.nilable(T::Boolean)) }
@@ -93,14 +80,11 @@ module ContextDev
93
80
  sig do
94
81
  params(
95
82
  body: ContextDev::Internal::FileInput,
96
- base_url: String,
97
- extension: String,
98
- filename: String,
83
+ extension: ContextDev::ParseHandleParams::Extension::OrSymbol,
99
84
  include_images: T::Boolean,
100
85
  include_links: T::Boolean,
101
86
  ocr: T::Boolean,
102
- pdf_end: Integer,
103
- pdf_start: Integer,
87
+ pdf: ContextDev::ParseHandleParams::Pdf::OrHash,
104
88
  shorten_base64_images: T::Boolean,
105
89
  use_main_content_only: T::Boolean,
106
90
  request_options: ContextDev::RequestOptions::OrHash
@@ -108,28 +92,24 @@ module ContextDev
108
92
  end
109
93
  def self.new(
110
94
  body:,
111
- # Optional HTTP(S) source document URL used to resolve relative links and image
112
- # references. Relative references remain relative when omitted.
113
- base_url: nil,
114
- # Optional file extension hint, such as pdf, docx, xlsx, pptx, html, json, csv,
115
- # md, py, rtf, jpg, png, or txt.
95
+ # Optional file extension hint. Case-insensitive; a leading dot is accepted (e.g.
96
+ # ".pdf").
116
97
  extension: nil,
117
- # Optional filename hint used to infer the extension when extension is omitted.
118
- filename: nil,
119
98
  # Include image references in Markdown output
120
99
  include_images: nil,
121
100
  # Preserve hyperlinks in Markdown output
122
101
  include_links: nil,
123
- # When true for PDF inputs, detect and OCR images embedded in the selected pages,
124
- # inserting recognized text at each image's position in page reading order while
125
- # preserving the PDF text layer. pdfStart/pdfEnd limit the inclusive page range.
126
- # This is separate from automatic scanned-PDF OCR fallback.
102
+ # Gates all OCR. When true, PDFs get embedded-image OCR (recognized text inserted
103
+ # at each image's position in page reading order, preserving the text layer;
104
+ # pdf.start/pdf.end limit the page range), scanned PDFs with no text layer get
105
+ # full-document OCR, and raster images get their visible text transcribed. When
106
+ # false, no OCR runs: scanned PDFs may yield no content and images return only
107
+ # format/dimension metadata. Calls where OCR actually runs cost 5 credits instead
108
+ # of 1.
127
109
  ocr: nil,
128
- # Last 1-based PDF page to parse. When omitted, parsing ends at the last page.
129
- # Must be greater than or equal to pdfStart when both are provided.
130
- pdf_end: nil,
131
- # First 1-based PDF page to parse. When omitted, parsing starts at the first page.
132
- pdf_start: nil,
110
+ # PDF page-range controls. Use start/end to limit parsing (and OCR when ocr=true)
111
+ # to an inclusive 1-based page range.
112
+ pdf: nil,
133
113
  # Shorten base64-encoded image data in the Markdown output
134
114
  shorten_base64_images: nil,
135
115
  # Extract only the main content from HTML-like inputs
@@ -142,14 +122,11 @@ module ContextDev
142
122
  override.returns(
143
123
  {
144
124
  body: ContextDev::Internal::FileInput,
145
- base_url: String,
146
- extension: String,
147
- filename: String,
125
+ extension: ContextDev::ParseHandleParams::Extension::OrSymbol,
148
126
  include_images: T::Boolean,
149
127
  include_links: T::Boolean,
150
128
  ocr: T::Boolean,
151
- pdf_end: Integer,
152
- pdf_start: Integer,
129
+ pdf: ContextDev::ParseHandleParams::Pdf,
153
130
  shorten_base64_images: T::Boolean,
154
131
  use_main_content_only: T::Boolean,
155
132
  request_options: ContextDev::RequestOptions
@@ -158,6 +135,205 @@ module ContextDev
158
135
  end
159
136
  def to_hash
160
137
  end
138
+
139
+ # Optional file extension hint. Case-insensitive; a leading dot is accepted (e.g.
140
+ # ".pdf").
141
+ module Extension
142
+ extend ContextDev::Internal::Type::Enum
143
+
144
+ TaggedSymbol =
145
+ T.type_alias do
146
+ T.all(Symbol, ContextDev::ParseHandleParams::Extension)
147
+ end
148
+ OrSymbol = T.type_alias { T.any(Symbol, String) }
149
+
150
+ TXT =
151
+ T.let(:txt, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
152
+ TEXT =
153
+ T.let(:text, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
154
+ MD = T.let(:md, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
155
+ MARKDOWN =
156
+ T.let(
157
+ :markdown,
158
+ ContextDev::ParseHandleParams::Extension::TaggedSymbol
159
+ )
160
+ HTML =
161
+ T.let(:html, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
162
+ HTM =
163
+ T.let(:htm, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
164
+ XHTML =
165
+ T.let(:xhtml, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
166
+ XML =
167
+ T.let(:xml, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
168
+ RSS =
169
+ T.let(:rss, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
170
+ ATOM =
171
+ T.let(:atom, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
172
+ CSV =
173
+ T.let(:csv, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
174
+ TSV =
175
+ T.let(:tsv, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
176
+ YAML =
177
+ T.let(:yaml, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
178
+ YML =
179
+ T.let(:yml, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
180
+ PY = T.let(:py, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
181
+ JAVA =
182
+ T.let(:java, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
183
+ JS = T.let(:js, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
184
+ JSX =
185
+ T.let(:jsx, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
186
+ MJS =
187
+ T.let(:mjs, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
188
+ CJS =
189
+ T.let(:cjs, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
190
+ JSON =
191
+ T.let(:json, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
192
+ JSONL =
193
+ T.let(:jsonl, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
194
+ NDJSON =
195
+ T.let(:ndjson, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
196
+ PHP =
197
+ T.let(:php, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
198
+ SH = T.let(:sh, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
199
+ BASH =
200
+ T.let(:bash, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
201
+ ZSH =
202
+ T.let(:zsh, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
203
+ FISH =
204
+ T.let(:fish, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
205
+ RB = T.let(:rb, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
206
+ TS = T.let(:ts, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
207
+ TSX =
208
+ T.let(:tsx, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
209
+ RTF =
210
+ T.let(:rtf, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
211
+ SRT =
212
+ T.let(:srt, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
213
+ CSS =
214
+ T.let(:css, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
215
+ SCSS =
216
+ T.let(:scss, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
217
+ LESS =
218
+ T.let(:less, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
219
+ STYL =
220
+ T.let(:styl, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
221
+ SASS =
222
+ T.let(:sass, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
223
+ SVG =
224
+ T.let(:svg, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
225
+ PDF =
226
+ T.let(:pdf, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
227
+ DOCX =
228
+ T.let(:docx, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
229
+ DOC =
230
+ T.let(:doc, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
231
+ XLSX =
232
+ T.let(:xlsx, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
233
+ XLSM =
234
+ T.let(:xlsm, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
235
+ XLSB =
236
+ T.let(:xlsb, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
237
+ XLTX =
238
+ T.let(:xltx, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
239
+ XLTM =
240
+ T.let(:xltm, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
241
+ XLS =
242
+ T.let(:xls, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
243
+ PPTX =
244
+ T.let(:pptx, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
245
+ PPTM =
246
+ T.let(:pptm, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
247
+ PPSX =
248
+ T.let(:ppsx, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
249
+ PPSM =
250
+ T.let(:ppsm, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
251
+ POTX =
252
+ T.let(:potx, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
253
+ POTM =
254
+ T.let(:potm, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
255
+ PPT =
256
+ T.let(:ppt, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
257
+ PPS =
258
+ T.let(:pps, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
259
+ POT =
260
+ T.let(:pot, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
261
+ JPG =
262
+ T.let(:jpg, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
263
+ JPEG =
264
+ T.let(:jpeg, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
265
+ JPE =
266
+ T.let(:jpe, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
267
+ PNG =
268
+ T.let(:png, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
269
+ GIF =
270
+ T.let(:gif, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
271
+ BMP =
272
+ T.let(:bmp, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
273
+ TIFF =
274
+ T.let(:tiff, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
275
+ TIF =
276
+ T.let(:tif, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
277
+ WEBP =
278
+ T.let(:webp, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
279
+ PPM =
280
+ T.let(:ppm, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
281
+ PBM =
282
+ T.let(:pbm, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
283
+ PGM =
284
+ T.let(:pgm, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
285
+ PNM =
286
+ T.let(:pnm, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
287
+
288
+ sig do
289
+ override.returns(
290
+ T::Array[ContextDev::ParseHandleParams::Extension::TaggedSymbol]
291
+ )
292
+ end
293
+ def self.values
294
+ end
295
+ end
296
+
297
+ class Pdf < ContextDev::Internal::Type::BaseModel
298
+ OrHash =
299
+ T.type_alias do
300
+ T.any(
301
+ ContextDev::ParseHandleParams::Pdf,
302
+ ContextDev::Internal::AnyHash
303
+ )
304
+ end
305
+
306
+ # Last 1-based PDF page to parse. When omitted, parsing ends at the last page.
307
+ # Must be greater than or equal to start when both are provided.
308
+ sig { returns(T.nilable(Integer)) }
309
+ attr_reader :end_
310
+
311
+ sig { params(end_: Integer).void }
312
+ attr_writer :end_
313
+
314
+ # First 1-based PDF page to parse. When omitted, parsing starts at the first page.
315
+ sig { returns(T.nilable(Integer)) }
316
+ attr_reader :start
317
+
318
+ sig { params(start: Integer).void }
319
+ attr_writer :start
320
+
321
+ # PDF page-range controls. Use start/end to limit parsing (and OCR when ocr=true)
322
+ # to an inclusive 1-based page range.
323
+ sig { params(end_: Integer, start: Integer).returns(T.attached_class) }
324
+ def self.new(
325
+ # Last 1-based PDF page to parse. When omitted, parsing ends at the last page.
326
+ # Must be greater than or equal to start when both are provided.
327
+ end_: nil,
328
+ # First 1-based PDF page to parse. When omitted, parsing starts at the first page.
329
+ start: nil
330
+ )
331
+ end
332
+
333
+ sig { override.returns({ end_: Integer, start: Integer }) }
334
+ def to_hash
335
+ end
336
+ end
161
337
  end
162
338
  end
163
339
  end
@@ -4,18 +4,17 @@ module ContextDev
4
4
  module Resources
5
5
  class Parse
6
6
  # Converts raw text, source code, web/data, PDF, Microsoft Office, and image bytes
7
- # into LLM-usable Markdown.
7
+ # into LLM-usable Markdown. The base request costs 1 credit. When OCR runs
8
+ # (requires ocr=true), the entire call costs 5 credits; ocr=true requests where no
9
+ # OCR ends up running still cost 1 credit.
8
10
  sig do
9
11
  params(
10
12
  body: ContextDev::Internal::FileInput,
11
- base_url: String,
12
- extension: String,
13
- filename: String,
13
+ extension: ContextDev::ParseHandleParams::Extension::OrSymbol,
14
14
  include_images: T::Boolean,
15
15
  include_links: T::Boolean,
16
16
  ocr: T::Boolean,
17
- pdf_end: Integer,
18
- pdf_start: Integer,
17
+ pdf: ContextDev::ParseHandleParams::Pdf::OrHash,
19
18
  shorten_base64_images: T::Boolean,
20
19
  use_main_content_only: T::Boolean,
21
20
  request_options: ContextDev::RequestOptions::OrHash
@@ -24,30 +23,24 @@ module ContextDev
24
23
  def handle(
25
24
  # Body param
26
25
  body:,
27
- # Query param: Optional HTTP(S) source document URL used to resolve relative links
28
- # and image references. Relative references remain relative when omitted.
29
- base_url: nil,
30
- # Query param: Optional file extension hint, such as pdf, docx, xlsx, pptx, html,
31
- # json, csv, md, py, rtf, jpg, png, or txt.
26
+ # Query param: Optional file extension hint. Case-insensitive; a leading dot is
27
+ # accepted (e.g. ".pdf").
32
28
  extension: nil,
33
- # Query param: Optional filename hint used to infer the extension when extension
34
- # is omitted.
35
- filename: nil,
36
29
  # Query param: Include image references in Markdown output
37
30
  include_images: nil,
38
31
  # Query param: Preserve hyperlinks in Markdown output
39
32
  include_links: nil,
40
- # Query param: When true for PDF inputs, detect and OCR images embedded in the
41
- # selected pages, inserting recognized text at each image's position in page
42
- # reading order while preserving the PDF text layer. pdfStart/pdfEnd limit the
43
- # inclusive page range. This is separate from automatic scanned-PDF OCR fallback.
33
+ # Query param: Gates all OCR. When true, PDFs get embedded-image OCR (recognized
34
+ # text inserted at each image's position in page reading order, preserving the
35
+ # text layer; pdf.start/pdf.end limit the page range), scanned PDFs with no text
36
+ # layer get full-document OCR, and raster images get their visible text
37
+ # transcribed. When false, no OCR runs: scanned PDFs may yield no content and
38
+ # images return only format/dimension metadata. Calls where OCR actually runs cost
39
+ # 5 credits instead of 1.
44
40
  ocr: nil,
45
- # Query param: Last 1-based PDF page to parse. When omitted, parsing ends at the
46
- # last page. Must be greater than or equal to pdfStart when both are provided.
47
- pdf_end: nil,
48
- # Query param: First 1-based PDF page to parse. When omitted, parsing starts at
49
- # the first page.
50
- pdf_start: nil,
41
+ # Query param: PDF page-range controls. Use start/end to limit parsing (and OCR
42
+ # when ocr=true) to an inclusive 1-based page range.
43
+ pdf: nil,
51
44
  # Query param: Shorten base64-encoded image data in the Markdown output
52
45
  shorten_base64_images: nil,
53
46
  # Query param: Extract only the main content from HTML-like inputs
@@ -3,14 +3,11 @@ module ContextDev
3
3
  type parse_handle_params =
4
4
  {
5
5
  body: ContextDev::Internal::file_input,
6
- base_url: String,
7
- extension: String,
8
- filename: String,
6
+ extension: ContextDev::Models::ParseHandleParams::extension,
9
7
  include_images: bool,
10
8
  include_links: bool,
11
9
  ocr: bool,
12
- pdf_end: Integer,
13
- pdf_start: Integer,
10
+ pdf: ContextDev::ParseHandleParams::Pdf,
14
11
  :shorten_base64_images => bool,
15
12
  use_main_content_only: bool
16
13
  }
@@ -22,17 +19,11 @@ module ContextDev
22
19
 
23
20
  attr_accessor body: ContextDev::Internal::file_input
24
21
 
25
- attr_reader base_url: String?
22
+ attr_reader extension: ContextDev::Models::ParseHandleParams::extension?
26
23
 
27
- def base_url=: (String) -> String
28
-
29
- attr_reader extension: String?
30
-
31
- def extension=: (String) -> String
32
-
33
- attr_reader filename: String?
34
-
35
- def filename=: (String) -> String
24
+ def extension=: (
25
+ ContextDev::Models::ParseHandleParams::extension
26
+ ) -> ContextDev::Models::ParseHandleParams::extension
36
27
 
37
28
  attr_reader include_images: bool?
38
29
 
@@ -46,13 +37,11 @@ module ContextDev
46
37
 
47
38
  def ocr=: (bool) -> bool
48
39
 
49
- attr_reader pdf_end: Integer?
40
+ attr_reader pdf: ContextDev::ParseHandleParams::Pdf?
50
41
 
51
- def pdf_end=: (Integer) -> Integer
52
-
53
- attr_reader pdf_start: Integer?
54
-
55
- def pdf_start=: (Integer) -> Integer
42
+ def pdf=: (
43
+ ContextDev::ParseHandleParams::Pdf
44
+ ) -> ContextDev::ParseHandleParams::Pdf
56
45
 
57
46
  attr_reader shorten_base64_images: bool?
58
47
 
@@ -64,14 +53,11 @@ module ContextDev
64
53
 
65
54
  def initialize: (
66
55
  body: ContextDev::Internal::file_input,
67
- ?base_url: String,
68
- ?extension: String,
69
- ?filename: String,
56
+ ?extension: ContextDev::Models::ParseHandleParams::extension,
70
57
  ?include_images: bool,
71
58
  ?include_links: bool,
72
59
  ?ocr: bool,
73
- ?pdf_end: Integer,
74
- ?pdf_start: Integer,
60
+ ?pdf: ContextDev::ParseHandleParams::Pdf,
75
61
  ?shorten_base64_images: bool,
76
62
  ?use_main_content_only: bool,
77
63
  ?request_options: ContextDev::request_opts
@@ -79,18 +65,180 @@ module ContextDev
79
65
 
80
66
  def to_hash: -> {
81
67
  body: ContextDev::Internal::file_input,
82
- base_url: String,
83
- extension: String,
84
- filename: String,
68
+ extension: ContextDev::Models::ParseHandleParams::extension,
85
69
  include_images: bool,
86
70
  include_links: bool,
87
71
  ocr: bool,
88
- pdf_end: Integer,
89
- pdf_start: Integer,
72
+ pdf: ContextDev::ParseHandleParams::Pdf,
90
73
  :shorten_base64_images => bool,
91
74
  use_main_content_only: bool,
92
75
  request_options: ContextDev::RequestOptions
93
76
  }
77
+
78
+ type extension =
79
+ :txt
80
+ | :text
81
+ | :md
82
+ | :markdown
83
+ | :html
84
+ | :htm
85
+ | :xhtml
86
+ | :xml
87
+ | :rss
88
+ | :atom
89
+ | :csv
90
+ | :tsv
91
+ | :yaml
92
+ | :yml
93
+ | :py
94
+ | :java
95
+ | :js
96
+ | :jsx
97
+ | :mjs
98
+ | :cjs
99
+ | :json
100
+ | :jsonl
101
+ | :ndjson
102
+ | :php
103
+ | :sh
104
+ | :bash
105
+ | :zsh
106
+ | :fish
107
+ | :rb
108
+ | :ts
109
+ | :tsx
110
+ | :rtf
111
+ | :srt
112
+ | :css
113
+ | :scss
114
+ | :less
115
+ | :styl
116
+ | :sass
117
+ | :svg
118
+ | :pdf
119
+ | :docx
120
+ | :doc
121
+ | :xlsx
122
+ | :xlsm
123
+ | :xlsb
124
+ | :xltx
125
+ | :xltm
126
+ | :xls
127
+ | :pptx
128
+ | :pptm
129
+ | :ppsx
130
+ | :ppsm
131
+ | :potx
132
+ | :potm
133
+ | :ppt
134
+ | :pps
135
+ | :pot
136
+ | :jpg
137
+ | :jpeg
138
+ | :jpe
139
+ | :png
140
+ | :gif
141
+ | :bmp
142
+ | :tiff
143
+ | :tif
144
+ | :webp
145
+ | :ppm
146
+ | :pbm
147
+ | :pgm
148
+ | :pnm
149
+
150
+ module Extension
151
+ extend ContextDev::Internal::Type::Enum
152
+
153
+ TXT: :txt
154
+ TEXT: :text
155
+ MD: :md
156
+ MARKDOWN: :markdown
157
+ HTML: :html
158
+ HTM: :htm
159
+ XHTML: :xhtml
160
+ XML: :xml
161
+ RSS: :rss
162
+ ATOM: :atom
163
+ CSV: :csv
164
+ TSV: :tsv
165
+ YAML: :yaml
166
+ YML: :yml
167
+ PY: :py
168
+ JAVA: :java
169
+ JS: :js
170
+ JSX: :jsx
171
+ MJS: :mjs
172
+ CJS: :cjs
173
+ JSON: :json
174
+ JSONL: :jsonl
175
+ NDJSON: :ndjson
176
+ PHP: :php
177
+ SH: :sh
178
+ BASH: :bash
179
+ ZSH: :zsh
180
+ FISH: :fish
181
+ RB: :rb
182
+ TS: :ts
183
+ TSX: :tsx
184
+ RTF: :rtf
185
+ SRT: :srt
186
+ CSS: :css
187
+ SCSS: :scss
188
+ LESS: :less
189
+ STYL: :styl
190
+ SASS: :sass
191
+ SVG: :svg
192
+ PDF: :pdf
193
+ DOCX: :docx
194
+ DOC: :doc
195
+ XLSX: :xlsx
196
+ XLSM: :xlsm
197
+ XLSB: :xlsb
198
+ XLTX: :xltx
199
+ XLTM: :xltm
200
+ XLS: :xls
201
+ PPTX: :pptx
202
+ PPTM: :pptm
203
+ PPSX: :ppsx
204
+ PPSM: :ppsm
205
+ POTX: :potx
206
+ POTM: :potm
207
+ PPT: :ppt
208
+ PPS: :pps
209
+ POT: :pot
210
+ JPG: :jpg
211
+ JPEG: :jpeg
212
+ JPE: :jpe
213
+ PNG: :png
214
+ GIF: :gif
215
+ BMP: :bmp
216
+ TIFF: :tiff
217
+ TIF: :tif
218
+ WEBP: :webp
219
+ PPM: :ppm
220
+ PBM: :pbm
221
+ PGM: :pgm
222
+ PNM: :pnm
223
+
224
+ def self?.values: -> ::Array[ContextDev::Models::ParseHandleParams::extension]
225
+ end
226
+
227
+ type pdf = { end_: Integer, start: Integer }
228
+
229
+ class Pdf < ContextDev::Internal::Type::BaseModel
230
+ attr_reader end_: Integer?
231
+
232
+ def end_=: (Integer) -> Integer
233
+
234
+ attr_reader start: Integer?
235
+
236
+ def start=: (Integer) -> Integer
237
+
238
+ def initialize: (?end_: Integer, ?start: Integer) -> void
239
+
240
+ def to_hash: -> { end_: Integer, start: Integer }
241
+ end
94
242
  end
95
243
  end
96
244
  end
@@ -3,14 +3,11 @@ module ContextDev
3
3
  class Parse
4
4
  def handle: (
5
5
  body: ContextDev::Internal::file_input,
6
- ?base_url: String,
7
- ?extension: String,
8
- ?filename: String,
6
+ ?extension: ContextDev::Models::ParseHandleParams::extension,
9
7
  ?include_images: bool,
10
8
  ?include_links: bool,
11
9
  ?ocr: bool,
12
- ?pdf_end: Integer,
13
- ?pdf_start: Integer,
10
+ ?pdf: ContextDev::ParseHandleParams::Pdf,
14
11
  ?shorten_base64_images: bool,
15
12
  ?use_main_content_only: bool,
16
13
  ?request_options: ContextDev::request_opts
metadata CHANGED
@@ -1,7 +1,7 @@
1
1
  --- !ruby/object:Gem::Specification
2
2
  name: context.dev
3
3
  version: !ruby/object:Gem::Version
4
- version: 2.3.0
4
+ version: 2.4.0
5
5
  platform: ruby
6
6
  authors:
7
7
  - Context Dev