context.dev 2.3.0 → 2.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +8 -0
- data/README.md +9 -9
- data/lib/context_dev/models/parse_handle_params.rb +126 -42
- data/lib/context_dev/resources/parse.rb +7 -11
- data/lib/context_dev/version.rb +1 -1
- data/rbi/context_dev/models/parse_handle_params.rbi +238 -62
- data/rbi/context_dev/resources/parse.rbi +17 -24
- data/sig/context_dev/models/parse_handle_params.rbs +179 -31
- data/sig/context_dev/resources/parse.rbs +2 -5
- metadata +1 -1
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
SHA256:
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
3
|
+
metadata.gz: 6e76a9e12d2d02a5a2fabf5f621579b215d278c973e9b88eb9c032ea9c5fee8f
|
|
4
|
+
data.tar.gz: 5bd0dda46235f5c51359bb91ad1bc53df70f13cb2f49120b6b1b4c5a23958389
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: c2c06bbf5e65076e9867a7beb132b243ecea991290c8c2a8492a0de26936225c1fa7346f0486f9889aa41d1f3b96f5e544d69269d964523169611b2e8256e849
|
|
7
|
+
data.tar.gz: 4a4e5d37a6a84ce2854b1dba0eb854312964a144a914db69f418f796b29aeed0e54102ce9b0c3bd03877f26bec65ed5b4979e10675fa547707509d4c1d609334
|
data/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,13 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## 2.4.0 (2026-07-12)
|
|
4
|
+
|
|
5
|
+
Full Changelog: [v2.3.0...v2.4.0](https://github.com/context-dot-dev/context-ruby-sdk/compare/v2.3.0...v2.4.0)
|
|
6
|
+
|
|
7
|
+
### Features
|
|
8
|
+
|
|
9
|
+
* **api:** api update ([426deb1](https://github.com/context-dot-dev/context-ruby-sdk/commit/426deb18249b214c3ccaa44c1e654e75ef666dd2))
|
|
10
|
+
|
|
3
11
|
## 2.3.0 (2026-07-12)
|
|
4
12
|
|
|
5
13
|
Full Changelog: [v2.2.0...v2.3.0](https://github.com/context-dot-dev/context-ruby-sdk/compare/v2.2.0...v2.3.0)
|
data/README.md
CHANGED
|
@@ -26,7 +26,7 @@ To use this gem, install via Bundler by adding the following to your application
|
|
|
26
26
|
<!-- x-release-please-start-version -->
|
|
27
27
|
|
|
28
28
|
```ruby
|
|
29
|
-
gem "context.dev", "~> 2.
|
|
29
|
+
gem "context.dev", "~> 2.4.0"
|
|
30
30
|
```
|
|
31
31
|
|
|
32
32
|
<!-- x-release-please-end -->
|
|
@@ -213,25 +213,25 @@ context_dev.brand.retrieve(**params)
|
|
|
213
213
|
Since this library does not depend on `sorbet-runtime`, it cannot provide [`T::Enum`](https://sorbet.org/docs/tenum) instances. Instead, we provide "tagged symbols" instead, which is always a primitive at runtime:
|
|
214
214
|
|
|
215
215
|
```ruby
|
|
216
|
-
# :
|
|
217
|
-
puts(ContextDev::
|
|
216
|
+
# :txt
|
|
217
|
+
puts(ContextDev::ParseHandleParams::Extension::TXT)
|
|
218
218
|
|
|
219
|
-
# Revealed type: `T.all(ContextDev::
|
|
220
|
-
T.reveal_type(ContextDev::
|
|
219
|
+
# Revealed type: `T.all(ContextDev::ParseHandleParams::Extension, Symbol)`
|
|
220
|
+
T.reveal_type(ContextDev::ParseHandleParams::Extension::TXT)
|
|
221
221
|
```
|
|
222
222
|
|
|
223
223
|
Enum parameters have a "relaxed" type, so you can either pass in enum constants or their literal value:
|
|
224
224
|
|
|
225
225
|
```ruby
|
|
226
226
|
# Using the enum constants preserves the tagged type information:
|
|
227
|
-
context_dev.
|
|
228
|
-
|
|
227
|
+
context_dev.parse.handle(
|
|
228
|
+
extension: ContextDev::ParseHandleParams::Extension::TXT,
|
|
229
229
|
# …
|
|
230
230
|
)
|
|
231
231
|
|
|
232
232
|
# Literal values are also permissible:
|
|
233
|
-
context_dev.
|
|
234
|
-
|
|
233
|
+
context_dev.parse.handle(
|
|
234
|
+
extension: :txt,
|
|
235
235
|
# …
|
|
236
236
|
)
|
|
237
237
|
```
|
|
@@ -12,25 +12,12 @@ module ContextDev
|
|
|
12
12
|
# @return [Pathname, StringIO, IO, String, ContextDev::FilePart]
|
|
13
13
|
required :body, ContextDev::Internal::Type::FileInput
|
|
14
14
|
|
|
15
|
-
# @!attribute base_url
|
|
16
|
-
# Optional HTTP(S) source document URL used to resolve relative links and image
|
|
17
|
-
# references. Relative references remain relative when omitted.
|
|
18
|
-
#
|
|
19
|
-
# @return [String, nil]
|
|
20
|
-
optional :base_url, String
|
|
21
|
-
|
|
22
15
|
# @!attribute extension
|
|
23
|
-
# Optional file extension hint
|
|
24
|
-
#
|
|
25
|
-
#
|
|
26
|
-
# @return [String, nil]
|
|
27
|
-
optional :extension, String
|
|
28
|
-
|
|
29
|
-
# @!attribute filename
|
|
30
|
-
# Optional filename hint used to infer the extension when extension is omitted.
|
|
16
|
+
# Optional file extension hint. Case-insensitive; a leading dot is accepted (e.g.
|
|
17
|
+
# ".pdf").
|
|
31
18
|
#
|
|
32
|
-
# @return [
|
|
33
|
-
optional :
|
|
19
|
+
# @return [Symbol, ContextDev::Models::ParseHandleParams::Extension, nil]
|
|
20
|
+
optional :extension, enum: -> { ContextDev::ParseHandleParams::Extension }
|
|
34
21
|
|
|
35
22
|
# @!attribute include_images
|
|
36
23
|
# Include image references in Markdown output
|
|
@@ -45,26 +32,23 @@ module ContextDev
|
|
|
45
32
|
optional :include_links, ContextDev::Internal::Type::Boolean
|
|
46
33
|
|
|
47
34
|
# @!attribute ocr
|
|
48
|
-
#
|
|
49
|
-
#
|
|
50
|
-
#
|
|
51
|
-
#
|
|
35
|
+
# Gates all OCR. When true, PDFs get embedded-image OCR (recognized text inserted
|
|
36
|
+
# at each image's position in page reading order, preserving the text layer;
|
|
37
|
+
# pdf.start/pdf.end limit the page range), scanned PDFs with no text layer get
|
|
38
|
+
# full-document OCR, and raster images get their visible text transcribed. When
|
|
39
|
+
# false, no OCR runs: scanned PDFs may yield no content and images return only
|
|
40
|
+
# format/dimension metadata. Calls where OCR actually runs cost 5 credits instead
|
|
41
|
+
# of 1.
|
|
52
42
|
#
|
|
53
43
|
# @return [Boolean, nil]
|
|
54
44
|
optional :ocr, ContextDev::Internal::Type::Boolean
|
|
55
45
|
|
|
56
|
-
# @!attribute
|
|
57
|
-
#
|
|
58
|
-
#
|
|
46
|
+
# @!attribute pdf
|
|
47
|
+
# PDF page-range controls. Use start/end to limit parsing (and OCR when ocr=true)
|
|
48
|
+
# to an inclusive 1-based page range.
|
|
59
49
|
#
|
|
60
|
-
# @return [
|
|
61
|
-
optional :
|
|
62
|
-
|
|
63
|
-
# @!attribute pdf_start
|
|
64
|
-
# First 1-based PDF page to parse. When omitted, parsing starts at the first page.
|
|
65
|
-
#
|
|
66
|
-
# @return [Integer, nil]
|
|
67
|
-
optional :pdf_start, Integer
|
|
50
|
+
# @return [ContextDev::Models::ParseHandleParams::Pdf, nil]
|
|
51
|
+
optional :pdf, -> { ContextDev::ParseHandleParams::Pdf }
|
|
68
52
|
|
|
69
53
|
# @!attribute shorten_base64_images
|
|
70
54
|
# Shorten base64-encoded image data in the Markdown output
|
|
@@ -78,33 +62,133 @@ module ContextDev
|
|
|
78
62
|
# @return [Boolean, nil]
|
|
79
63
|
optional :use_main_content_only, ContextDev::Internal::Type::Boolean
|
|
80
64
|
|
|
81
|
-
# @!method initialize(body:,
|
|
65
|
+
# @!method initialize(body:, extension: nil, include_images: nil, include_links: nil, ocr: nil, pdf: nil, shorten_base64_images: nil, use_main_content_only: nil, request_options: {})
|
|
82
66
|
# Some parameter documentations has been truncated, see
|
|
83
67
|
# {ContextDev::Models::ParseHandleParams} for more details.
|
|
84
68
|
#
|
|
85
69
|
# @param body [Pathname, StringIO, IO, String, ContextDev::FilePart]
|
|
86
70
|
#
|
|
87
|
-
# @param
|
|
88
|
-
#
|
|
89
|
-
# @param extension [String] Optional file extension hint, such as pdf, docx, xlsx, pptx, html, json, csv, md
|
|
90
|
-
#
|
|
91
|
-
# @param filename [String] Optional filename hint used to infer the extension when extension is omitted.
|
|
71
|
+
# @param extension [Symbol, ContextDev::Models::ParseHandleParams::Extension] Optional file extension hint. Case-insensitive; a leading dot is accepted (e.g.
|
|
92
72
|
#
|
|
93
73
|
# @param include_images [Boolean] Include image references in Markdown output
|
|
94
74
|
#
|
|
95
75
|
# @param include_links [Boolean] Preserve hyperlinks in Markdown output
|
|
96
76
|
#
|
|
97
|
-
# @param ocr [Boolean]
|
|
77
|
+
# @param ocr [Boolean] Gates all OCR. When true, PDFs get embedded-image OCR (recognized text inserted
|
|
98
78
|
#
|
|
99
|
-
# @param
|
|
100
|
-
#
|
|
101
|
-
# @param pdf_start [Integer] First 1-based PDF page to parse. When omitted, parsing starts at the first page.
|
|
79
|
+
# @param pdf [ContextDev::Models::ParseHandleParams::Pdf] PDF page-range controls. Use start/end to limit parsing (and OCR when ocr=true)
|
|
102
80
|
#
|
|
103
81
|
# @param shorten_base64_images [Boolean] Shorten base64-encoded image data in the Markdown output
|
|
104
82
|
#
|
|
105
83
|
# @param use_main_content_only [Boolean] Extract only the main content from HTML-like inputs
|
|
106
84
|
#
|
|
107
85
|
# @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}]
|
|
86
|
+
|
|
87
|
+
# Optional file extension hint. Case-insensitive; a leading dot is accepted (e.g.
|
|
88
|
+
# ".pdf").
|
|
89
|
+
module Extension
|
|
90
|
+
extend ContextDev::Internal::Type::Enum
|
|
91
|
+
|
|
92
|
+
TXT = :txt
|
|
93
|
+
TEXT = :text
|
|
94
|
+
MD = :md
|
|
95
|
+
MARKDOWN = :markdown
|
|
96
|
+
HTML = :html
|
|
97
|
+
HTM = :htm
|
|
98
|
+
XHTML = :xhtml
|
|
99
|
+
XML = :xml
|
|
100
|
+
RSS = :rss
|
|
101
|
+
ATOM = :atom
|
|
102
|
+
CSV = :csv
|
|
103
|
+
TSV = :tsv
|
|
104
|
+
YAML = :yaml
|
|
105
|
+
YML = :yml
|
|
106
|
+
PY = :py
|
|
107
|
+
JAVA = :java
|
|
108
|
+
JS = :js
|
|
109
|
+
JSX = :jsx
|
|
110
|
+
MJS = :mjs
|
|
111
|
+
CJS = :cjs
|
|
112
|
+
JSON = :json
|
|
113
|
+
JSONL = :jsonl
|
|
114
|
+
NDJSON = :ndjson
|
|
115
|
+
PHP = :php
|
|
116
|
+
SH = :sh
|
|
117
|
+
BASH = :bash
|
|
118
|
+
ZSH = :zsh
|
|
119
|
+
FISH = :fish
|
|
120
|
+
RB = :rb
|
|
121
|
+
TS = :ts
|
|
122
|
+
TSX = :tsx
|
|
123
|
+
RTF = :rtf
|
|
124
|
+
SRT = :srt
|
|
125
|
+
CSS = :css
|
|
126
|
+
SCSS = :scss
|
|
127
|
+
LESS = :less
|
|
128
|
+
STYL = :styl
|
|
129
|
+
SASS = :sass
|
|
130
|
+
SVG = :svg
|
|
131
|
+
PDF = :pdf
|
|
132
|
+
DOCX = :docx
|
|
133
|
+
DOC = :doc
|
|
134
|
+
XLSX = :xlsx
|
|
135
|
+
XLSM = :xlsm
|
|
136
|
+
XLSB = :xlsb
|
|
137
|
+
XLTX = :xltx
|
|
138
|
+
XLTM = :xltm
|
|
139
|
+
XLS = :xls
|
|
140
|
+
PPTX = :pptx
|
|
141
|
+
PPTM = :pptm
|
|
142
|
+
PPSX = :ppsx
|
|
143
|
+
PPSM = :ppsm
|
|
144
|
+
POTX = :potx
|
|
145
|
+
POTM = :potm
|
|
146
|
+
PPT = :ppt
|
|
147
|
+
PPS = :pps
|
|
148
|
+
POT = :pot
|
|
149
|
+
JPG = :jpg
|
|
150
|
+
JPEG = :jpeg
|
|
151
|
+
JPE = :jpe
|
|
152
|
+
PNG = :png
|
|
153
|
+
GIF = :gif
|
|
154
|
+
BMP = :bmp
|
|
155
|
+
TIFF = :tiff
|
|
156
|
+
TIF = :tif
|
|
157
|
+
WEBP = :webp
|
|
158
|
+
PPM = :ppm
|
|
159
|
+
PBM = :pbm
|
|
160
|
+
PGM = :pgm
|
|
161
|
+
PNM = :pnm
|
|
162
|
+
|
|
163
|
+
# @!method self.values
|
|
164
|
+
# @return [Array<Symbol>]
|
|
165
|
+
end
|
|
166
|
+
|
|
167
|
+
class Pdf < ContextDev::Internal::Type::BaseModel
|
|
168
|
+
# @!attribute end_
|
|
169
|
+
# Last 1-based PDF page to parse. When omitted, parsing ends at the last page.
|
|
170
|
+
# Must be greater than or equal to start when both are provided.
|
|
171
|
+
#
|
|
172
|
+
# @return [Integer, nil]
|
|
173
|
+
optional :end_, Integer, api_name: :end
|
|
174
|
+
|
|
175
|
+
# @!attribute start
|
|
176
|
+
# First 1-based PDF page to parse. When omitted, parsing starts at the first page.
|
|
177
|
+
#
|
|
178
|
+
# @return [Integer, nil]
|
|
179
|
+
optional :start, Integer
|
|
180
|
+
|
|
181
|
+
# @!method initialize(end_: nil, start: nil)
|
|
182
|
+
# Some parameter documentations has been truncated, see
|
|
183
|
+
# {ContextDev::Models::ParseHandleParams::Pdf} for more details.
|
|
184
|
+
#
|
|
185
|
+
# PDF page-range controls. Use start/end to limit parsing (and OCR when ocr=true)
|
|
186
|
+
# to an inclusive 1-based page range.
|
|
187
|
+
#
|
|
188
|
+
# @param end_ [Integer] Last 1-based PDF page to parse. When omitted, parsing ends at the last page. Mus
|
|
189
|
+
#
|
|
190
|
+
# @param start [Integer] First 1-based PDF page to parse. When omitted, parsing starts at the first page.
|
|
191
|
+
end
|
|
108
192
|
end
|
|
109
193
|
end
|
|
110
194
|
end
|
|
@@ -7,27 +7,23 @@ module ContextDev
|
|
|
7
7
|
# {ContextDev::Models::ParseHandleParams} for more details.
|
|
8
8
|
#
|
|
9
9
|
# Converts raw text, source code, web/data, PDF, Microsoft Office, and image bytes
|
|
10
|
-
# into LLM-usable Markdown.
|
|
10
|
+
# into LLM-usable Markdown. The base request costs 1 credit. When OCR runs
|
|
11
|
+
# (requires ocr=true), the entire call costs 5 credits; ocr=true requests where no
|
|
12
|
+
# OCR ends up running still cost 1 credit.
|
|
11
13
|
#
|
|
12
|
-
# @overload handle(body:,
|
|
14
|
+
# @overload handle(body:, extension: nil, include_images: nil, include_links: nil, ocr: nil, pdf: nil, shorten_base64_images: nil, use_main_content_only: nil, request_options: {})
|
|
13
15
|
#
|
|
14
16
|
# @param body [Pathname, StringIO, IO, String, ContextDev::FilePart] Body param
|
|
15
17
|
#
|
|
16
|
-
# @param
|
|
17
|
-
#
|
|
18
|
-
# @param extension [String] Query param: Optional file extension hint, such as pdf, docx, xlsx, pptx, html,
|
|
19
|
-
#
|
|
20
|
-
# @param filename [String] Query param: Optional filename hint used to infer the extension when extension i
|
|
18
|
+
# @param extension [Symbol, ContextDev::Models::ParseHandleParams::Extension] Query param: Optional file extension hint. Case-insensitive; a leading dot is ac
|
|
21
19
|
#
|
|
22
20
|
# @param include_images [Boolean] Query param: Include image references in Markdown output
|
|
23
21
|
#
|
|
24
22
|
# @param include_links [Boolean] Query param: Preserve hyperlinks in Markdown output
|
|
25
23
|
#
|
|
26
|
-
# @param ocr [Boolean] Query param:
|
|
27
|
-
#
|
|
28
|
-
# @param pdf_end [Integer] Query param: Last 1-based PDF page to parse. When omitted, parsing ends at the l
|
|
24
|
+
# @param ocr [Boolean] Query param: Gates all OCR. When true, PDFs get embedded-image OCR (recognized t
|
|
29
25
|
#
|
|
30
|
-
# @param
|
|
26
|
+
# @param pdf [ContextDev::Models::ParseHandleParams::Pdf] Query param: PDF page-range controls. Use start/end to limit parsing (and OCR wh
|
|
31
27
|
#
|
|
32
28
|
# @param shorten_base64_images [Boolean] Query param: Shorten base64-encoded image data in the Markdown output
|
|
33
29
|
#
|
data/lib/context_dev/version.rb
CHANGED
|
@@ -14,29 +14,20 @@ module ContextDev
|
|
|
14
14
|
sig { returns(ContextDev::Internal::FileInput) }
|
|
15
15
|
attr_accessor :body
|
|
16
16
|
|
|
17
|
-
# Optional
|
|
18
|
-
#
|
|
19
|
-
sig
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
sig { params(base_url: String).void }
|
|
23
|
-
attr_writer :base_url
|
|
24
|
-
|
|
25
|
-
# Optional file extension hint, such as pdf, docx, xlsx, pptx, html, json, csv,
|
|
26
|
-
# md, py, rtf, jpg, png, or txt.
|
|
27
|
-
sig { returns(T.nilable(String)) }
|
|
17
|
+
# Optional file extension hint. Case-insensitive; a leading dot is accepted (e.g.
|
|
18
|
+
# ".pdf").
|
|
19
|
+
sig do
|
|
20
|
+
returns(T.nilable(ContextDev::ParseHandleParams::Extension::OrSymbol))
|
|
21
|
+
end
|
|
28
22
|
attr_reader :extension
|
|
29
23
|
|
|
30
|
-
sig
|
|
24
|
+
sig do
|
|
25
|
+
params(
|
|
26
|
+
extension: ContextDev::ParseHandleParams::Extension::OrSymbol
|
|
27
|
+
).void
|
|
28
|
+
end
|
|
31
29
|
attr_writer :extension
|
|
32
30
|
|
|
33
|
-
# Optional filename hint used to infer the extension when extension is omitted.
|
|
34
|
-
sig { returns(T.nilable(String)) }
|
|
35
|
-
attr_reader :filename
|
|
36
|
-
|
|
37
|
-
sig { params(filename: String).void }
|
|
38
|
-
attr_writer :filename
|
|
39
|
-
|
|
40
31
|
# Include image references in Markdown output
|
|
41
32
|
sig { returns(T.nilable(T::Boolean)) }
|
|
42
33
|
attr_reader :include_images
|
|
@@ -51,30 +42,26 @@ module ContextDev
|
|
|
51
42
|
sig { params(include_links: T::Boolean).void }
|
|
52
43
|
attr_writer :include_links
|
|
53
44
|
|
|
54
|
-
#
|
|
55
|
-
#
|
|
56
|
-
#
|
|
57
|
-
#
|
|
45
|
+
# Gates all OCR. When true, PDFs get embedded-image OCR (recognized text inserted
|
|
46
|
+
# at each image's position in page reading order, preserving the text layer;
|
|
47
|
+
# pdf.start/pdf.end limit the page range), scanned PDFs with no text layer get
|
|
48
|
+
# full-document OCR, and raster images get their visible text transcribed. When
|
|
49
|
+
# false, no OCR runs: scanned PDFs may yield no content and images return only
|
|
50
|
+
# format/dimension metadata. Calls where OCR actually runs cost 5 credits instead
|
|
51
|
+
# of 1.
|
|
58
52
|
sig { returns(T.nilable(T::Boolean)) }
|
|
59
53
|
attr_reader :ocr
|
|
60
54
|
|
|
61
55
|
sig { params(ocr: T::Boolean).void }
|
|
62
56
|
attr_writer :ocr
|
|
63
57
|
|
|
64
|
-
#
|
|
65
|
-
#
|
|
66
|
-
sig { returns(T.nilable(
|
|
67
|
-
attr_reader :
|
|
58
|
+
# PDF page-range controls. Use start/end to limit parsing (and OCR when ocr=true)
|
|
59
|
+
# to an inclusive 1-based page range.
|
|
60
|
+
sig { returns(T.nilable(ContextDev::ParseHandleParams::Pdf)) }
|
|
61
|
+
attr_reader :pdf
|
|
68
62
|
|
|
69
|
-
sig { params(
|
|
70
|
-
attr_writer :
|
|
71
|
-
|
|
72
|
-
# First 1-based PDF page to parse. When omitted, parsing starts at the first page.
|
|
73
|
-
sig { returns(T.nilable(Integer)) }
|
|
74
|
-
attr_reader :pdf_start
|
|
75
|
-
|
|
76
|
-
sig { params(pdf_start: Integer).void }
|
|
77
|
-
attr_writer :pdf_start
|
|
63
|
+
sig { params(pdf: ContextDev::ParseHandleParams::Pdf::OrHash).void }
|
|
64
|
+
attr_writer :pdf
|
|
78
65
|
|
|
79
66
|
# Shorten base64-encoded image data in the Markdown output
|
|
80
67
|
sig { returns(T.nilable(T::Boolean)) }
|
|
@@ -93,14 +80,11 @@ module ContextDev
|
|
|
93
80
|
sig do
|
|
94
81
|
params(
|
|
95
82
|
body: ContextDev::Internal::FileInput,
|
|
96
|
-
|
|
97
|
-
extension: String,
|
|
98
|
-
filename: String,
|
|
83
|
+
extension: ContextDev::ParseHandleParams::Extension::OrSymbol,
|
|
99
84
|
include_images: T::Boolean,
|
|
100
85
|
include_links: T::Boolean,
|
|
101
86
|
ocr: T::Boolean,
|
|
102
|
-
|
|
103
|
-
pdf_start: Integer,
|
|
87
|
+
pdf: ContextDev::ParseHandleParams::Pdf::OrHash,
|
|
104
88
|
shorten_base64_images: T::Boolean,
|
|
105
89
|
use_main_content_only: T::Boolean,
|
|
106
90
|
request_options: ContextDev::RequestOptions::OrHash
|
|
@@ -108,28 +92,24 @@ module ContextDev
|
|
|
108
92
|
end
|
|
109
93
|
def self.new(
|
|
110
94
|
body:,
|
|
111
|
-
# Optional
|
|
112
|
-
#
|
|
113
|
-
base_url: nil,
|
|
114
|
-
# Optional file extension hint, such as pdf, docx, xlsx, pptx, html, json, csv,
|
|
115
|
-
# md, py, rtf, jpg, png, or txt.
|
|
95
|
+
# Optional file extension hint. Case-insensitive; a leading dot is accepted (e.g.
|
|
96
|
+
# ".pdf").
|
|
116
97
|
extension: nil,
|
|
117
|
-
# Optional filename hint used to infer the extension when extension is omitted.
|
|
118
|
-
filename: nil,
|
|
119
98
|
# Include image references in Markdown output
|
|
120
99
|
include_images: nil,
|
|
121
100
|
# Preserve hyperlinks in Markdown output
|
|
122
101
|
include_links: nil,
|
|
123
|
-
#
|
|
124
|
-
#
|
|
125
|
-
#
|
|
126
|
-
#
|
|
102
|
+
# Gates all OCR. When true, PDFs get embedded-image OCR (recognized text inserted
|
|
103
|
+
# at each image's position in page reading order, preserving the text layer;
|
|
104
|
+
# pdf.start/pdf.end limit the page range), scanned PDFs with no text layer get
|
|
105
|
+
# full-document OCR, and raster images get their visible text transcribed. When
|
|
106
|
+
# false, no OCR runs: scanned PDFs may yield no content and images return only
|
|
107
|
+
# format/dimension metadata. Calls where OCR actually runs cost 5 credits instead
|
|
108
|
+
# of 1.
|
|
127
109
|
ocr: nil,
|
|
128
|
-
#
|
|
129
|
-
#
|
|
130
|
-
|
|
131
|
-
# First 1-based PDF page to parse. When omitted, parsing starts at the first page.
|
|
132
|
-
pdf_start: nil,
|
|
110
|
+
# PDF page-range controls. Use start/end to limit parsing (and OCR when ocr=true)
|
|
111
|
+
# to an inclusive 1-based page range.
|
|
112
|
+
pdf: nil,
|
|
133
113
|
# Shorten base64-encoded image data in the Markdown output
|
|
134
114
|
shorten_base64_images: nil,
|
|
135
115
|
# Extract only the main content from HTML-like inputs
|
|
@@ -142,14 +122,11 @@ module ContextDev
|
|
|
142
122
|
override.returns(
|
|
143
123
|
{
|
|
144
124
|
body: ContextDev::Internal::FileInput,
|
|
145
|
-
|
|
146
|
-
extension: String,
|
|
147
|
-
filename: String,
|
|
125
|
+
extension: ContextDev::ParseHandleParams::Extension::OrSymbol,
|
|
148
126
|
include_images: T::Boolean,
|
|
149
127
|
include_links: T::Boolean,
|
|
150
128
|
ocr: T::Boolean,
|
|
151
|
-
|
|
152
|
-
pdf_start: Integer,
|
|
129
|
+
pdf: ContextDev::ParseHandleParams::Pdf,
|
|
153
130
|
shorten_base64_images: T::Boolean,
|
|
154
131
|
use_main_content_only: T::Boolean,
|
|
155
132
|
request_options: ContextDev::RequestOptions
|
|
@@ -158,6 +135,205 @@ module ContextDev
|
|
|
158
135
|
end
|
|
159
136
|
def to_hash
|
|
160
137
|
end
|
|
138
|
+
|
|
139
|
+
# Optional file extension hint. Case-insensitive; a leading dot is accepted (e.g.
|
|
140
|
+
# ".pdf").
|
|
141
|
+
module Extension
|
|
142
|
+
extend ContextDev::Internal::Type::Enum
|
|
143
|
+
|
|
144
|
+
TaggedSymbol =
|
|
145
|
+
T.type_alias do
|
|
146
|
+
T.all(Symbol, ContextDev::ParseHandleParams::Extension)
|
|
147
|
+
end
|
|
148
|
+
OrSymbol = T.type_alias { T.any(Symbol, String) }
|
|
149
|
+
|
|
150
|
+
TXT =
|
|
151
|
+
T.let(:txt, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
152
|
+
TEXT =
|
|
153
|
+
T.let(:text, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
154
|
+
MD = T.let(:md, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
155
|
+
MARKDOWN =
|
|
156
|
+
T.let(
|
|
157
|
+
:markdown,
|
|
158
|
+
ContextDev::ParseHandleParams::Extension::TaggedSymbol
|
|
159
|
+
)
|
|
160
|
+
HTML =
|
|
161
|
+
T.let(:html, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
162
|
+
HTM =
|
|
163
|
+
T.let(:htm, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
164
|
+
XHTML =
|
|
165
|
+
T.let(:xhtml, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
166
|
+
XML =
|
|
167
|
+
T.let(:xml, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
168
|
+
RSS =
|
|
169
|
+
T.let(:rss, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
170
|
+
ATOM =
|
|
171
|
+
T.let(:atom, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
172
|
+
CSV =
|
|
173
|
+
T.let(:csv, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
174
|
+
TSV =
|
|
175
|
+
T.let(:tsv, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
176
|
+
YAML =
|
|
177
|
+
T.let(:yaml, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
178
|
+
YML =
|
|
179
|
+
T.let(:yml, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
180
|
+
PY = T.let(:py, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
181
|
+
JAVA =
|
|
182
|
+
T.let(:java, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
183
|
+
JS = T.let(:js, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
184
|
+
JSX =
|
|
185
|
+
T.let(:jsx, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
186
|
+
MJS =
|
|
187
|
+
T.let(:mjs, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
188
|
+
CJS =
|
|
189
|
+
T.let(:cjs, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
190
|
+
JSON =
|
|
191
|
+
T.let(:json, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
192
|
+
JSONL =
|
|
193
|
+
T.let(:jsonl, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
194
|
+
NDJSON =
|
|
195
|
+
T.let(:ndjson, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
196
|
+
PHP =
|
|
197
|
+
T.let(:php, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
198
|
+
SH = T.let(:sh, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
199
|
+
BASH =
|
|
200
|
+
T.let(:bash, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
201
|
+
ZSH =
|
|
202
|
+
T.let(:zsh, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
203
|
+
FISH =
|
|
204
|
+
T.let(:fish, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
205
|
+
RB = T.let(:rb, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
206
|
+
TS = T.let(:ts, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
207
|
+
TSX =
|
|
208
|
+
T.let(:tsx, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
209
|
+
RTF =
|
|
210
|
+
T.let(:rtf, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
211
|
+
SRT =
|
|
212
|
+
T.let(:srt, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
213
|
+
CSS =
|
|
214
|
+
T.let(:css, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
215
|
+
SCSS =
|
|
216
|
+
T.let(:scss, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
217
|
+
LESS =
|
|
218
|
+
T.let(:less, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
219
|
+
STYL =
|
|
220
|
+
T.let(:styl, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
221
|
+
SASS =
|
|
222
|
+
T.let(:sass, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
223
|
+
SVG =
|
|
224
|
+
T.let(:svg, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
225
|
+
PDF =
|
|
226
|
+
T.let(:pdf, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
227
|
+
DOCX =
|
|
228
|
+
T.let(:docx, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
229
|
+
DOC =
|
|
230
|
+
T.let(:doc, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
231
|
+
XLSX =
|
|
232
|
+
T.let(:xlsx, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
233
|
+
XLSM =
|
|
234
|
+
T.let(:xlsm, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
235
|
+
XLSB =
|
|
236
|
+
T.let(:xlsb, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
237
|
+
XLTX =
|
|
238
|
+
T.let(:xltx, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
239
|
+
XLTM =
|
|
240
|
+
T.let(:xltm, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
241
|
+
XLS =
|
|
242
|
+
T.let(:xls, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
243
|
+
PPTX =
|
|
244
|
+
T.let(:pptx, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
245
|
+
PPTM =
|
|
246
|
+
T.let(:pptm, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
247
|
+
PPSX =
|
|
248
|
+
T.let(:ppsx, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
249
|
+
PPSM =
|
|
250
|
+
T.let(:ppsm, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
251
|
+
POTX =
|
|
252
|
+
T.let(:potx, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
253
|
+
POTM =
|
|
254
|
+
T.let(:potm, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
255
|
+
PPT =
|
|
256
|
+
T.let(:ppt, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
257
|
+
PPS =
|
|
258
|
+
T.let(:pps, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
259
|
+
POT =
|
|
260
|
+
T.let(:pot, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
261
|
+
JPG =
|
|
262
|
+
T.let(:jpg, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
263
|
+
JPEG =
|
|
264
|
+
T.let(:jpeg, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
265
|
+
JPE =
|
|
266
|
+
T.let(:jpe, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
267
|
+
PNG =
|
|
268
|
+
T.let(:png, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
269
|
+
GIF =
|
|
270
|
+
T.let(:gif, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
271
|
+
BMP =
|
|
272
|
+
T.let(:bmp, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
273
|
+
TIFF =
|
|
274
|
+
T.let(:tiff, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
275
|
+
TIF =
|
|
276
|
+
T.let(:tif, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
277
|
+
WEBP =
|
|
278
|
+
T.let(:webp, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
279
|
+
PPM =
|
|
280
|
+
T.let(:ppm, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
281
|
+
PBM =
|
|
282
|
+
T.let(:pbm, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
283
|
+
PGM =
|
|
284
|
+
T.let(:pgm, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
285
|
+
PNM =
|
|
286
|
+
T.let(:pnm, ContextDev::ParseHandleParams::Extension::TaggedSymbol)
|
|
287
|
+
|
|
288
|
+
sig do
|
|
289
|
+
override.returns(
|
|
290
|
+
T::Array[ContextDev::ParseHandleParams::Extension::TaggedSymbol]
|
|
291
|
+
)
|
|
292
|
+
end
|
|
293
|
+
def self.values
|
|
294
|
+
end
|
|
295
|
+
end
|
|
296
|
+
|
|
297
|
+
class Pdf < ContextDev::Internal::Type::BaseModel
|
|
298
|
+
OrHash =
|
|
299
|
+
T.type_alias do
|
|
300
|
+
T.any(
|
|
301
|
+
ContextDev::ParseHandleParams::Pdf,
|
|
302
|
+
ContextDev::Internal::AnyHash
|
|
303
|
+
)
|
|
304
|
+
end
|
|
305
|
+
|
|
306
|
+
# Last 1-based PDF page to parse. When omitted, parsing ends at the last page.
|
|
307
|
+
# Must be greater than or equal to start when both are provided.
|
|
308
|
+
sig { returns(T.nilable(Integer)) }
|
|
309
|
+
attr_reader :end_
|
|
310
|
+
|
|
311
|
+
sig { params(end_: Integer).void }
|
|
312
|
+
attr_writer :end_
|
|
313
|
+
|
|
314
|
+
# First 1-based PDF page to parse. When omitted, parsing starts at the first page.
|
|
315
|
+
sig { returns(T.nilable(Integer)) }
|
|
316
|
+
attr_reader :start
|
|
317
|
+
|
|
318
|
+
sig { params(start: Integer).void }
|
|
319
|
+
attr_writer :start
|
|
320
|
+
|
|
321
|
+
# PDF page-range controls. Use start/end to limit parsing (and OCR when ocr=true)
|
|
322
|
+
# to an inclusive 1-based page range.
|
|
323
|
+
sig { params(end_: Integer, start: Integer).returns(T.attached_class) }
|
|
324
|
+
def self.new(
|
|
325
|
+
# Last 1-based PDF page to parse. When omitted, parsing ends at the last page.
|
|
326
|
+
# Must be greater than or equal to start when both are provided.
|
|
327
|
+
end_: nil,
|
|
328
|
+
# First 1-based PDF page to parse. When omitted, parsing starts at the first page.
|
|
329
|
+
start: nil
|
|
330
|
+
)
|
|
331
|
+
end
|
|
332
|
+
|
|
333
|
+
sig { override.returns({ end_: Integer, start: Integer }) }
|
|
334
|
+
def to_hash
|
|
335
|
+
end
|
|
336
|
+
end
|
|
161
337
|
end
|
|
162
338
|
end
|
|
163
339
|
end
|
|
@@ -4,18 +4,17 @@ module ContextDev
|
|
|
4
4
|
module Resources
|
|
5
5
|
class Parse
|
|
6
6
|
# Converts raw text, source code, web/data, PDF, Microsoft Office, and image bytes
|
|
7
|
-
# into LLM-usable Markdown.
|
|
7
|
+
# into LLM-usable Markdown. The base request costs 1 credit. When OCR runs
|
|
8
|
+
# (requires ocr=true), the entire call costs 5 credits; ocr=true requests where no
|
|
9
|
+
# OCR ends up running still cost 1 credit.
|
|
8
10
|
sig do
|
|
9
11
|
params(
|
|
10
12
|
body: ContextDev::Internal::FileInput,
|
|
11
|
-
|
|
12
|
-
extension: String,
|
|
13
|
-
filename: String,
|
|
13
|
+
extension: ContextDev::ParseHandleParams::Extension::OrSymbol,
|
|
14
14
|
include_images: T::Boolean,
|
|
15
15
|
include_links: T::Boolean,
|
|
16
16
|
ocr: T::Boolean,
|
|
17
|
-
|
|
18
|
-
pdf_start: Integer,
|
|
17
|
+
pdf: ContextDev::ParseHandleParams::Pdf::OrHash,
|
|
19
18
|
shorten_base64_images: T::Boolean,
|
|
20
19
|
use_main_content_only: T::Boolean,
|
|
21
20
|
request_options: ContextDev::RequestOptions::OrHash
|
|
@@ -24,30 +23,24 @@ module ContextDev
|
|
|
24
23
|
def handle(
|
|
25
24
|
# Body param
|
|
26
25
|
body:,
|
|
27
|
-
# Query param: Optional
|
|
28
|
-
#
|
|
29
|
-
base_url: nil,
|
|
30
|
-
# Query param: Optional file extension hint, such as pdf, docx, xlsx, pptx, html,
|
|
31
|
-
# json, csv, md, py, rtf, jpg, png, or txt.
|
|
26
|
+
# Query param: Optional file extension hint. Case-insensitive; a leading dot is
|
|
27
|
+
# accepted (e.g. ".pdf").
|
|
32
28
|
extension: nil,
|
|
33
|
-
# Query param: Optional filename hint used to infer the extension when extension
|
|
34
|
-
# is omitted.
|
|
35
|
-
filename: nil,
|
|
36
29
|
# Query param: Include image references in Markdown output
|
|
37
30
|
include_images: nil,
|
|
38
31
|
# Query param: Preserve hyperlinks in Markdown output
|
|
39
32
|
include_links: nil,
|
|
40
|
-
# Query param:
|
|
41
|
-
#
|
|
42
|
-
#
|
|
43
|
-
#
|
|
33
|
+
# Query param: Gates all OCR. When true, PDFs get embedded-image OCR (recognized
|
|
34
|
+
# text inserted at each image's position in page reading order, preserving the
|
|
35
|
+
# text layer; pdf.start/pdf.end limit the page range), scanned PDFs with no text
|
|
36
|
+
# layer get full-document OCR, and raster images get their visible text
|
|
37
|
+
# transcribed. When false, no OCR runs: scanned PDFs may yield no content and
|
|
38
|
+
# images return only format/dimension metadata. Calls where OCR actually runs cost
|
|
39
|
+
# 5 credits instead of 1.
|
|
44
40
|
ocr: nil,
|
|
45
|
-
# Query param:
|
|
46
|
-
#
|
|
47
|
-
|
|
48
|
-
# Query param: First 1-based PDF page to parse. When omitted, parsing starts at
|
|
49
|
-
# the first page.
|
|
50
|
-
pdf_start: nil,
|
|
41
|
+
# Query param: PDF page-range controls. Use start/end to limit parsing (and OCR
|
|
42
|
+
# when ocr=true) to an inclusive 1-based page range.
|
|
43
|
+
pdf: nil,
|
|
51
44
|
# Query param: Shorten base64-encoded image data in the Markdown output
|
|
52
45
|
shorten_base64_images: nil,
|
|
53
46
|
# Query param: Extract only the main content from HTML-like inputs
|
|
@@ -3,14 +3,11 @@ module ContextDev
|
|
|
3
3
|
type parse_handle_params =
|
|
4
4
|
{
|
|
5
5
|
body: ContextDev::Internal::file_input,
|
|
6
|
-
|
|
7
|
-
extension: String,
|
|
8
|
-
filename: String,
|
|
6
|
+
extension: ContextDev::Models::ParseHandleParams::extension,
|
|
9
7
|
include_images: bool,
|
|
10
8
|
include_links: bool,
|
|
11
9
|
ocr: bool,
|
|
12
|
-
|
|
13
|
-
pdf_start: Integer,
|
|
10
|
+
pdf: ContextDev::ParseHandleParams::Pdf,
|
|
14
11
|
:shorten_base64_images => bool,
|
|
15
12
|
use_main_content_only: bool
|
|
16
13
|
}
|
|
@@ -22,17 +19,11 @@ module ContextDev
|
|
|
22
19
|
|
|
23
20
|
attr_accessor body: ContextDev::Internal::file_input
|
|
24
21
|
|
|
25
|
-
attr_reader
|
|
22
|
+
attr_reader extension: ContextDev::Models::ParseHandleParams::extension?
|
|
26
23
|
|
|
27
|
-
def
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
def extension=: (String) -> String
|
|
32
|
-
|
|
33
|
-
attr_reader filename: String?
|
|
34
|
-
|
|
35
|
-
def filename=: (String) -> String
|
|
24
|
+
def extension=: (
|
|
25
|
+
ContextDev::Models::ParseHandleParams::extension
|
|
26
|
+
) -> ContextDev::Models::ParseHandleParams::extension
|
|
36
27
|
|
|
37
28
|
attr_reader include_images: bool?
|
|
38
29
|
|
|
@@ -46,13 +37,11 @@ module ContextDev
|
|
|
46
37
|
|
|
47
38
|
def ocr=: (bool) -> bool
|
|
48
39
|
|
|
49
|
-
attr_reader
|
|
40
|
+
attr_reader pdf: ContextDev::ParseHandleParams::Pdf?
|
|
50
41
|
|
|
51
|
-
def
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
def pdf_start=: (Integer) -> Integer
|
|
42
|
+
def pdf=: (
|
|
43
|
+
ContextDev::ParseHandleParams::Pdf
|
|
44
|
+
) -> ContextDev::ParseHandleParams::Pdf
|
|
56
45
|
|
|
57
46
|
attr_reader shorten_base64_images: bool?
|
|
58
47
|
|
|
@@ -64,14 +53,11 @@ module ContextDev
|
|
|
64
53
|
|
|
65
54
|
def initialize: (
|
|
66
55
|
body: ContextDev::Internal::file_input,
|
|
67
|
-
?
|
|
68
|
-
?extension: String,
|
|
69
|
-
?filename: String,
|
|
56
|
+
?extension: ContextDev::Models::ParseHandleParams::extension,
|
|
70
57
|
?include_images: bool,
|
|
71
58
|
?include_links: bool,
|
|
72
59
|
?ocr: bool,
|
|
73
|
-
?
|
|
74
|
-
?pdf_start: Integer,
|
|
60
|
+
?pdf: ContextDev::ParseHandleParams::Pdf,
|
|
75
61
|
?shorten_base64_images: bool,
|
|
76
62
|
?use_main_content_only: bool,
|
|
77
63
|
?request_options: ContextDev::request_opts
|
|
@@ -79,18 +65,180 @@ module ContextDev
|
|
|
79
65
|
|
|
80
66
|
def to_hash: -> {
|
|
81
67
|
body: ContextDev::Internal::file_input,
|
|
82
|
-
|
|
83
|
-
extension: String,
|
|
84
|
-
filename: String,
|
|
68
|
+
extension: ContextDev::Models::ParseHandleParams::extension,
|
|
85
69
|
include_images: bool,
|
|
86
70
|
include_links: bool,
|
|
87
71
|
ocr: bool,
|
|
88
|
-
|
|
89
|
-
pdf_start: Integer,
|
|
72
|
+
pdf: ContextDev::ParseHandleParams::Pdf,
|
|
90
73
|
:shorten_base64_images => bool,
|
|
91
74
|
use_main_content_only: bool,
|
|
92
75
|
request_options: ContextDev::RequestOptions
|
|
93
76
|
}
|
|
77
|
+
|
|
78
|
+
type extension =
|
|
79
|
+
:txt
|
|
80
|
+
| :text
|
|
81
|
+
| :md
|
|
82
|
+
| :markdown
|
|
83
|
+
| :html
|
|
84
|
+
| :htm
|
|
85
|
+
| :xhtml
|
|
86
|
+
| :xml
|
|
87
|
+
| :rss
|
|
88
|
+
| :atom
|
|
89
|
+
| :csv
|
|
90
|
+
| :tsv
|
|
91
|
+
| :yaml
|
|
92
|
+
| :yml
|
|
93
|
+
| :py
|
|
94
|
+
| :java
|
|
95
|
+
| :js
|
|
96
|
+
| :jsx
|
|
97
|
+
| :mjs
|
|
98
|
+
| :cjs
|
|
99
|
+
| :json
|
|
100
|
+
| :jsonl
|
|
101
|
+
| :ndjson
|
|
102
|
+
| :php
|
|
103
|
+
| :sh
|
|
104
|
+
| :bash
|
|
105
|
+
| :zsh
|
|
106
|
+
| :fish
|
|
107
|
+
| :rb
|
|
108
|
+
| :ts
|
|
109
|
+
| :tsx
|
|
110
|
+
| :rtf
|
|
111
|
+
| :srt
|
|
112
|
+
| :css
|
|
113
|
+
| :scss
|
|
114
|
+
| :less
|
|
115
|
+
| :styl
|
|
116
|
+
| :sass
|
|
117
|
+
| :svg
|
|
118
|
+
| :pdf
|
|
119
|
+
| :docx
|
|
120
|
+
| :doc
|
|
121
|
+
| :xlsx
|
|
122
|
+
| :xlsm
|
|
123
|
+
| :xlsb
|
|
124
|
+
| :xltx
|
|
125
|
+
| :xltm
|
|
126
|
+
| :xls
|
|
127
|
+
| :pptx
|
|
128
|
+
| :pptm
|
|
129
|
+
| :ppsx
|
|
130
|
+
| :ppsm
|
|
131
|
+
| :potx
|
|
132
|
+
| :potm
|
|
133
|
+
| :ppt
|
|
134
|
+
| :pps
|
|
135
|
+
| :pot
|
|
136
|
+
| :jpg
|
|
137
|
+
| :jpeg
|
|
138
|
+
| :jpe
|
|
139
|
+
| :png
|
|
140
|
+
| :gif
|
|
141
|
+
| :bmp
|
|
142
|
+
| :tiff
|
|
143
|
+
| :tif
|
|
144
|
+
| :webp
|
|
145
|
+
| :ppm
|
|
146
|
+
| :pbm
|
|
147
|
+
| :pgm
|
|
148
|
+
| :pnm
|
|
149
|
+
|
|
150
|
+
module Extension
|
|
151
|
+
extend ContextDev::Internal::Type::Enum
|
|
152
|
+
|
|
153
|
+
TXT: :txt
|
|
154
|
+
TEXT: :text
|
|
155
|
+
MD: :md
|
|
156
|
+
MARKDOWN: :markdown
|
|
157
|
+
HTML: :html
|
|
158
|
+
HTM: :htm
|
|
159
|
+
XHTML: :xhtml
|
|
160
|
+
XML: :xml
|
|
161
|
+
RSS: :rss
|
|
162
|
+
ATOM: :atom
|
|
163
|
+
CSV: :csv
|
|
164
|
+
TSV: :tsv
|
|
165
|
+
YAML: :yaml
|
|
166
|
+
YML: :yml
|
|
167
|
+
PY: :py
|
|
168
|
+
JAVA: :java
|
|
169
|
+
JS: :js
|
|
170
|
+
JSX: :jsx
|
|
171
|
+
MJS: :mjs
|
|
172
|
+
CJS: :cjs
|
|
173
|
+
JSON: :json
|
|
174
|
+
JSONL: :jsonl
|
|
175
|
+
NDJSON: :ndjson
|
|
176
|
+
PHP: :php
|
|
177
|
+
SH: :sh
|
|
178
|
+
BASH: :bash
|
|
179
|
+
ZSH: :zsh
|
|
180
|
+
FISH: :fish
|
|
181
|
+
RB: :rb
|
|
182
|
+
TS: :ts
|
|
183
|
+
TSX: :tsx
|
|
184
|
+
RTF: :rtf
|
|
185
|
+
SRT: :srt
|
|
186
|
+
CSS: :css
|
|
187
|
+
SCSS: :scss
|
|
188
|
+
LESS: :less
|
|
189
|
+
STYL: :styl
|
|
190
|
+
SASS: :sass
|
|
191
|
+
SVG: :svg
|
|
192
|
+
PDF: :pdf
|
|
193
|
+
DOCX: :docx
|
|
194
|
+
DOC: :doc
|
|
195
|
+
XLSX: :xlsx
|
|
196
|
+
XLSM: :xlsm
|
|
197
|
+
XLSB: :xlsb
|
|
198
|
+
XLTX: :xltx
|
|
199
|
+
XLTM: :xltm
|
|
200
|
+
XLS: :xls
|
|
201
|
+
PPTX: :pptx
|
|
202
|
+
PPTM: :pptm
|
|
203
|
+
PPSX: :ppsx
|
|
204
|
+
PPSM: :ppsm
|
|
205
|
+
POTX: :potx
|
|
206
|
+
POTM: :potm
|
|
207
|
+
PPT: :ppt
|
|
208
|
+
PPS: :pps
|
|
209
|
+
POT: :pot
|
|
210
|
+
JPG: :jpg
|
|
211
|
+
JPEG: :jpeg
|
|
212
|
+
JPE: :jpe
|
|
213
|
+
PNG: :png
|
|
214
|
+
GIF: :gif
|
|
215
|
+
BMP: :bmp
|
|
216
|
+
TIFF: :tiff
|
|
217
|
+
TIF: :tif
|
|
218
|
+
WEBP: :webp
|
|
219
|
+
PPM: :ppm
|
|
220
|
+
PBM: :pbm
|
|
221
|
+
PGM: :pgm
|
|
222
|
+
PNM: :pnm
|
|
223
|
+
|
|
224
|
+
def self?.values: -> ::Array[ContextDev::Models::ParseHandleParams::extension]
|
|
225
|
+
end
|
|
226
|
+
|
|
227
|
+
type pdf = { end_: Integer, start: Integer }
|
|
228
|
+
|
|
229
|
+
class Pdf < ContextDev::Internal::Type::BaseModel
|
|
230
|
+
attr_reader end_: Integer?
|
|
231
|
+
|
|
232
|
+
def end_=: (Integer) -> Integer
|
|
233
|
+
|
|
234
|
+
attr_reader start: Integer?
|
|
235
|
+
|
|
236
|
+
def start=: (Integer) -> Integer
|
|
237
|
+
|
|
238
|
+
def initialize: (?end_: Integer, ?start: Integer) -> void
|
|
239
|
+
|
|
240
|
+
def to_hash: -> { end_: Integer, start: Integer }
|
|
241
|
+
end
|
|
94
242
|
end
|
|
95
243
|
end
|
|
96
244
|
end
|
|
@@ -3,14 +3,11 @@ module ContextDev
|
|
|
3
3
|
class Parse
|
|
4
4
|
def handle: (
|
|
5
5
|
body: ContextDev::Internal::file_input,
|
|
6
|
-
?
|
|
7
|
-
?extension: String,
|
|
8
|
-
?filename: String,
|
|
6
|
+
?extension: ContextDev::Models::ParseHandleParams::extension,
|
|
9
7
|
?include_images: bool,
|
|
10
8
|
?include_links: bool,
|
|
11
9
|
?ocr: bool,
|
|
12
|
-
?
|
|
13
|
-
?pdf_start: Integer,
|
|
10
|
+
?pdf: ContextDev::ParseHandleParams::Pdf,
|
|
14
11
|
?shorten_base64_images: bool,
|
|
15
12
|
?use_main_content_only: bool,
|
|
16
13
|
?request_options: ContextDev::request_opts
|