sphinx_learn 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sphinx_ai/__init__.py +17 -0
- sphinx_ai/_environment.py +6 -0
- sphinx_ai/_file.py +193 -0
- sphinx_ai/_ingestion.py +12 -0
- sphinx_ai/_knowledge_base.py +206 -0
- sphinx_ai/gen/__init__.py +1 -0
- sphinx_ai/gen/object_storage_pb.py +115 -0
- sphinx_ai/gen/ra/__init__.py +1 -0
- sphinx_ai/gen/ra/jobs_connect.py +148 -0
- sphinx_ai/gen/ra/jobs_pb.py +581 -0
- sphinx_ai/gen/sessions/__init__.py +1 -0
- sphinx_ai/gen/sessions/compute_pb.py +90 -0
- sphinx_ai/gen/sphinx_connect.py +148 -0
- sphinx_ai/gen/sphinx_pb.py +32 -0
- sphinx_learn-0.1.0.dist-info/METADATA +12 -0
- sphinx_learn-0.1.0.dist-info/RECORD +18 -0
- sphinx_learn-0.1.0.dist-info/WHEEL +5 -0
- sphinx_learn-0.1.0.dist-info/top_level.txt +1 -0
sphinx_ai/__init__.py
ADDED
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
"""Public interface for the sphinx_ai package."""
|
|
2
|
+
|
|
3
|
+
from ._ingestion import _Ingestion as Ingestion
|
|
4
|
+
from ._file import _File as File
|
|
5
|
+
from ._knowledge_base import (
|
|
6
|
+
_Document as Document,
|
|
7
|
+
_KnowledgeBaseClient as KnowledgeBaseClient,
|
|
8
|
+
_createKnowledgeBaseClient as createKnowledgeBaseClient,
|
|
9
|
+
)
|
|
10
|
+
|
|
11
|
+
__all__ = [
|
|
12
|
+
"createKnowledgeBaseClient",
|
|
13
|
+
"Document",
|
|
14
|
+
"File",
|
|
15
|
+
"Ingestion",
|
|
16
|
+
"KnowledgeBaseClient",
|
|
17
|
+
]
|
sphinx_ai/_file.py
ADDED
|
@@ -0,0 +1,193 @@
|
|
|
1
|
+
"""Validated in-memory files for knowledge-base ingestion."""
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
import re
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
_MIN_SOURCE_PATH_CHARACTERS = 1
|
|
9
|
+
_MAX_SOURCE_PATH_CHARACTERS = 512
|
|
10
|
+
|
|
11
|
+
# Keep aligned with common/src/contracts/ingestContracts.json.
|
|
12
|
+
_BYTES_PER_MIB = 1024**2
|
|
13
|
+
_MAX_TEXT_FILE_BYTES = 4 * _BYTES_PER_MIB
|
|
14
|
+
_MAX_BINARY_DOCUMENT_BYTES = 100 * _BYTES_PER_MIB
|
|
15
|
+
_MAX_DOCUMENT_ESTIMATED_TOKENS = 250_000
|
|
16
|
+
_BINARY_DOCUMENT_EXTENSIONS = frozenset(
|
|
17
|
+
{
|
|
18
|
+
".pdf",
|
|
19
|
+
".docx",
|
|
20
|
+
".doc",
|
|
21
|
+
".pptx",
|
|
22
|
+
".ppt",
|
|
23
|
+
".png",
|
|
24
|
+
".jpg",
|
|
25
|
+
".jpeg",
|
|
26
|
+
".xlsx",
|
|
27
|
+
".ods",
|
|
28
|
+
}
|
|
29
|
+
)
|
|
30
|
+
_JSON_LINES_EXTENSIONS = frozenset({".jsonl", ".ndjson"})
|
|
31
|
+
_STRUCTURED_TEXT_EXTENSIONS = _JSON_LINES_EXTENSIONS | {".json", ".ipynb"}
|
|
32
|
+
|
|
33
|
+
# Match Thoth's text decoding and token estimate.
|
|
34
|
+
_TEXT_ENCODING = "utf-8-sig"
|
|
35
|
+
_CHARACTERS_PER_ESTIMATED_TOKEN = 4
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
@dataclass(frozen=True)
|
|
39
|
+
class _File:
|
|
40
|
+
"""An immutable in-memory document validated on creation.
|
|
41
|
+
|
|
42
|
+
Args:
|
|
43
|
+
path: A string of 1–512 characters. Allowed characters are ASCII
|
|
44
|
+
letters, digits, ! - _ . * ' ( ), and / as a path separator.
|
|
45
|
+
Empty, '.' and '..' path segments are forbidden.
|
|
46
|
+
content: Raw document bytes, subject to the constraints below.
|
|
47
|
+
|
|
48
|
+
Content constraints (all size limits are inclusive):
|
|
49
|
+
- Supported binary documents: at most 100 MiB (104,857,600 bytes).
|
|
50
|
+
Extensions are .pdf, .doc/.docx, .ppt/.pptx, .png, .jpg/.jpeg,
|
|
51
|
+
.xlsx and .ods, matched case-insensitively. Binary document
|
|
52
|
+
internals are not parsed or validated by this constructor.
|
|
53
|
+
- All other files: at most 4 MiB (4,194,304 bytes), valid UTF-8,
|
|
54
|
+
and no null bytes. A leading UTF-8 BOM is allowed and excluded
|
|
55
|
+
from the decoded character count.
|
|
56
|
+
- Text: at most 1,000,000 decoded characters, equivalent to the
|
|
57
|
+
250,000-token estimate computed as ceil(character_count / 4).
|
|
58
|
+
- .json: must parse as JSON.
|
|
59
|
+
- .jsonl/.ndjson: each nonblank line must parse as JSON.
|
|
60
|
+
- .ipynb: must parse as a JSON object containing a 'cells' list;
|
|
61
|
+
every cell must be an object.
|
|
62
|
+
- JSON parsing rejects NaN, Infinity and -Infinity.
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
Raises:
|
|
66
|
+
TypeError: path is not a string or content is not bytes.
|
|
67
|
+
ValueError: The path or content violates a constraint above.
|
|
68
|
+
"""
|
|
69
|
+
|
|
70
|
+
path: str
|
|
71
|
+
content: bytes
|
|
72
|
+
|
|
73
|
+
def __post_init__(self) -> None:
|
|
74
|
+
_checkPathIsGood(self)
|
|
75
|
+
_checkContentIsGood(self)
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def _checkPathIsGood(file: _File) -> None:
|
|
79
|
+
if not isinstance(file.path, str):
|
|
80
|
+
raise TypeError(f"File.path must be a string; got {type(file.path).__name__}")
|
|
81
|
+
if not _MIN_SOURCE_PATH_CHARACTERS <= len(file.path) <= _MAX_SOURCE_PATH_CHARACTERS:
|
|
82
|
+
raise ValueError(
|
|
83
|
+
f"File.path must contain {_MIN_SOURCE_PATH_CHARACTERS} to "
|
|
84
|
+
f"{_MAX_SOURCE_PATH_CHARACTERS} characters; got {len(file.path):,} in {file.path!r}"
|
|
85
|
+
)
|
|
86
|
+
if re.fullmatch(r"[a-zA-Z0-9!_.*'()/\-]+", file.path) is None:
|
|
87
|
+
unsupported_characters = sorted(set(re.findall(r"[^a-zA-Z0-9!_.*'()/\-]", file.path)))
|
|
88
|
+
raise ValueError(
|
|
89
|
+
"File.path must use only S3-safe characters: "
|
|
90
|
+
"letters A-Z/a-z, digits, ! - _ . * ' ( ), and / as a path separator; "
|
|
91
|
+
f"got path {file.path!r} with unsupported characters "
|
|
92
|
+
f"{unsupported_characters!r}"
|
|
93
|
+
)
|
|
94
|
+
if any(part in ("", ".", "..") for part in file.path.split("/")):
|
|
95
|
+
raise ValueError(f"File.path must not contain empty, '.' or '..' segments; got {file.path!r}")
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def _checkContentIsGood(file: _File) -> None:
|
|
99
|
+
if not isinstance(file.content, bytes):
|
|
100
|
+
raise TypeError(f"File.content must be bytes; got {type(file.content).__name__}")
|
|
101
|
+
|
|
102
|
+
basename = file.path.rsplit("/", 1)[-1]
|
|
103
|
+
stem, separator, suffix = basename.rpartition(".")
|
|
104
|
+
extension = f".{suffix.lower()}" if separator and stem else ""
|
|
105
|
+
is_binary = extension in _BINARY_DOCUMENT_EXTENSIONS
|
|
106
|
+
|
|
107
|
+
if is_binary:
|
|
108
|
+
_checkBinaryFileIsGood(file)
|
|
109
|
+
else:
|
|
110
|
+
_checkTextFileIsGood(file, extension)
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def _checkBinaryFileIsGood(file: _File) -> None:
|
|
114
|
+
if len(file.content) > _MAX_BINARY_DOCUMENT_BYTES:
|
|
115
|
+
raise ValueError(
|
|
116
|
+
f"File ({file.path!r}) must be at most {_MAX_BINARY_DOCUMENT_BYTES / _BYTES_PER_MIB:g} MiB "
|
|
117
|
+
f"({_MAX_BINARY_DOCUMENT_BYTES:,} bytes); got {len(file.content):,} bytes"
|
|
118
|
+
)
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def _checkTextFileIsGood(file: _File, extension: str) -> None:
|
|
122
|
+
_checkTextFileIsAtMost4MiB(file)
|
|
123
|
+
text = _tryParseTextAsUTF(file, extension)
|
|
124
|
+
_checkTextIsAtMost250kEstimatedTokens(text, file.path)
|
|
125
|
+
if extension == ".ipynb":
|
|
126
|
+
_checkIpynbIsValid(text, file.path)
|
|
127
|
+
elif extension in _STRUCTURED_TEXT_EXTENSIONS:
|
|
128
|
+
_checkJsonIsValid(text, file.path, extension)
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def _checkTextFileIsAtMost4MiB(file: _File) -> None:
|
|
132
|
+
if len(file.content) > _MAX_TEXT_FILE_BYTES:
|
|
133
|
+
raise ValueError(
|
|
134
|
+
f"File ({file.path!r}) must be at most {_MAX_TEXT_FILE_BYTES / _BYTES_PER_MIB:g} MiB "
|
|
135
|
+
f"({_MAX_TEXT_FILE_BYTES:,} bytes); got {len(file.content):,} bytes"
|
|
136
|
+
)
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def _tryParseTextAsUTF(file: _File, extension: str) -> str:
|
|
140
|
+
"""Return decoded UTF-8 text without a leading BOM, or raise ValueError."""
|
|
141
|
+
# The caller checks the byte limit before allocating the decoded string.
|
|
142
|
+
try:
|
|
143
|
+
if b"\x00" in file.content:
|
|
144
|
+
raise ValueError(f"text contains a null byte at byte offset {file.content.index(0):,}")
|
|
145
|
+
return file.content.decode(_TEXT_ENCODING)
|
|
146
|
+
except ValueError as error:
|
|
147
|
+
raise ValueError(
|
|
148
|
+
f"File ({file.path!r}) must be a supported binary document "
|
|
149
|
+
f"({', '.join(sorted(_BINARY_DOCUMENT_EXTENSIONS))}) or UTF-8 text without null bytes; "
|
|
150
|
+
f"got extension {extension!r}, {len(file.content):,} bytes: {error}"
|
|
151
|
+
) from error
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def _checkTextIsAtMost250kEstimatedTokens(text: str, path: str) -> None:
|
|
155
|
+
decoded_characters = len(text)
|
|
156
|
+
estimated_tokens = (decoded_characters + _CHARACTERS_PER_ESTIMATED_TOKEN - 1) // _CHARACTERS_PER_ESTIMATED_TOKEN
|
|
157
|
+
if estimated_tokens > _MAX_DOCUMENT_ESTIMATED_TOKENS:
|
|
158
|
+
raise ValueError(
|
|
159
|
+
f"File ({path!r}) exceeds the {_MAX_DOCUMENT_ESTIMATED_TOKENS:,} estimated-token limit; "
|
|
160
|
+
f"got {estimated_tokens:,} estimated tokens ({decoded_characters:,} decoded characters)"
|
|
161
|
+
)
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
def _checkJsonIsValid(text: str, path: str, extension: str) -> None:
|
|
165
|
+
try:
|
|
166
|
+
if extension in _JSON_LINES_EXTENSIONS:
|
|
167
|
+
for line_number, line in enumerate(text.split("\n"), start=1):
|
|
168
|
+
if not line.strip(" \t\r"):
|
|
169
|
+
continue
|
|
170
|
+
try:
|
|
171
|
+
json.loads(line, parse_constant=_rejectNonFiniteJsonConstant)
|
|
172
|
+
except ValueError as error:
|
|
173
|
+
raise ValueError(f"invalid JSON on line {line_number}: {error}") from error
|
|
174
|
+
else:
|
|
175
|
+
json.loads(text, parse_constant=_rejectNonFiniteJsonConstant)
|
|
176
|
+
except ValueError as error:
|
|
177
|
+
raise ValueError(f"File ({path!r}) contains invalid {extension} content: {error}") from error
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
def _checkIpynbIsValid(text: str, path: str) -> None:
|
|
181
|
+
try:
|
|
182
|
+
parsed = json.loads(text, parse_constant=_rejectNonFiniteJsonConstant)
|
|
183
|
+
if not isinstance(parsed, dict) or not isinstance(parsed.get("cells"), list):
|
|
184
|
+
raise ValueError("notebook must be an object with a cells list")
|
|
185
|
+
if any(not isinstance(cell, dict) for cell in parsed["cells"]):
|
|
186
|
+
raise ValueError("notebook cells must be objects")
|
|
187
|
+
except ValueError as error:
|
|
188
|
+
raise ValueError(f"File ({path!r}) contains invalid .ipynb content: {error}") from error
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
def _rejectNonFiniteJsonConstant(value: str) -> None:
|
|
192
|
+
# Match Thoth: Python's JSON decoder otherwise accepts NaN and Infinity.
|
|
193
|
+
raise ValueError(f"Invalid JSON constant: {value}")
|
sphinx_ai/_ingestion.py
ADDED
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
"""Scaffolding for knowledge-base ingestion jobs."""
|
|
2
|
+
|
|
3
|
+
|
|
4
|
+
class _Ingestion:
|
|
5
|
+
"""Handle returned by a knowledge-base ingestion request."""
|
|
6
|
+
|
|
7
|
+
def __init__(self, url: str) -> None:
|
|
8
|
+
self._url = url
|
|
9
|
+
|
|
10
|
+
def job_url(self) -> str:
|
|
11
|
+
"""Return the URL for this ingestion job."""
|
|
12
|
+
return self._url
|
|
@@ -0,0 +1,206 @@
|
|
|
1
|
+
"""Client helpers for creating and populating Sphinx knowledge bases."""
|
|
2
|
+
|
|
3
|
+
import base64
|
|
4
|
+
import hashlib
|
|
5
|
+
import json
|
|
6
|
+
import mimetypes
|
|
7
|
+
from typing import TypeAlias
|
|
8
|
+
|
|
9
|
+
import urllib3
|
|
10
|
+
|
|
11
|
+
from .gen.object_storage_pb import CreatePresignedPutRequest
|
|
12
|
+
from .gen.ra.jobs_connect import RaJobServiceClientSync
|
|
13
|
+
from .gen.ra.jobs_pb import SubmitJobRequest
|
|
14
|
+
from .gen.sessions.compute_pb import ComputeEnvironmentConfiguration, ComputeEnvironmentType
|
|
15
|
+
from .gen.sphinx_connect import SphinxRuntimeServiceClientSync
|
|
16
|
+
from ._environment import _SPHINX_API_URL, _SPHINX_UI_URL
|
|
17
|
+
from ._ingestion import _Ingestion
|
|
18
|
+
from ._file import _File
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
# SDK input policies.
|
|
22
|
+
_MIN_DOCUMENT_COUNT = 1
|
|
23
|
+
_MAX_DOCUMENT_COUNT = 20_000
|
|
24
|
+
_MAX_PROMPT_CHARACTERS = 10_000
|
|
25
|
+
_BYTES_PER_GIB = 1024**3
|
|
26
|
+
_MAX_COLLECTION_BYTES = 2_684_354_560
|
|
27
|
+
|
|
28
|
+
_HTTP = urllib3.PoolManager()
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
_Document: TypeAlias = _File
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
class _KnowledgeBaseClient:
|
|
35
|
+
"""Client for fitting a Sphinx knowledge base to your data"""
|
|
36
|
+
|
|
37
|
+
def __init__(self, api_key: str, project_id: str) -> None:
|
|
38
|
+
self.api_key = api_key
|
|
39
|
+
self.project_id = project_id
|
|
40
|
+
self._runtime_client = SphinxRuntimeServiceClientSync(_SPHINX_API_URL)
|
|
41
|
+
self._ra_jobs_client = RaJobServiceClientSync(_SPHINX_API_URL)
|
|
42
|
+
|
|
43
|
+
def fit(
|
|
44
|
+
self,
|
|
45
|
+
prompt: str,
|
|
46
|
+
docs: list[_Document],
|
|
47
|
+
) -> _Ingestion:
|
|
48
|
+
"""Fits the Sphinx Knowledge base to your documents
|
|
49
|
+
by creating a new ingest job
|
|
50
|
+
|
|
51
|
+
Args:
|
|
52
|
+
prompt: Instructions that guide how the documents are ingested.
|
|
53
|
+
docs: Documents to upload and ingest.
|
|
54
|
+
|
|
55
|
+
Returns:
|
|
56
|
+
A handle containing the URL of the newly created ingestion job.
|
|
57
|
+
|
|
58
|
+
Raises:
|
|
59
|
+
TypeError: The prompt or documents have the wrong type.
|
|
60
|
+
ValueError: Prompt or batch limits documented by _checkInputIsGood are exceeded.
|
|
61
|
+
"""
|
|
62
|
+
_checkInputIsGood(prompt, docs)
|
|
63
|
+
refs = [self._upload_document(doc) for doc in docs]
|
|
64
|
+
id = self._submit_job(prompt, docs, refs)
|
|
65
|
+
|
|
66
|
+
return _Ingestion(f"{_SPHINX_UI_URL}/knowledge-base/projects/{self.project_id}/ingest/{id}")
|
|
67
|
+
|
|
68
|
+
def _upload_document(self, doc: _Document) -> str:
|
|
69
|
+
content_type = self._guess_mimetype(doc.path)
|
|
70
|
+
presigned = self._runtime_client.create_presigned_put(
|
|
71
|
+
CreatePresignedPutRequest(
|
|
72
|
+
project_id=self.project_id,
|
|
73
|
+
content_type=content_type,
|
|
74
|
+
filename=doc.path,
|
|
75
|
+
),
|
|
76
|
+
headers=self._auth_headers(),
|
|
77
|
+
)
|
|
78
|
+
|
|
79
|
+
upload_response = _HTTP.request(
|
|
80
|
+
"PUT",
|
|
81
|
+
presigned.presigned_url,
|
|
82
|
+
body=doc.content,
|
|
83
|
+
headers=dict(presigned.required_headers),
|
|
84
|
+
)
|
|
85
|
+
|
|
86
|
+
if not 200 <= upload_response.status < 300:
|
|
87
|
+
raise RuntimeError(f'Failed to upload "{doc.path}": {upload_response.status} {upload_response.reason}')
|
|
88
|
+
|
|
89
|
+
return presigned.ref
|
|
90
|
+
|
|
91
|
+
def _submit_job(
|
|
92
|
+
self,
|
|
93
|
+
prompt: str,
|
|
94
|
+
doc: list[_Document],
|
|
95
|
+
refs: list[str],
|
|
96
|
+
) -> str:
|
|
97
|
+
# Omitting target domains lets the server scope the job to the root domain.
|
|
98
|
+
res = self._ra_jobs_client.submit_job(
|
|
99
|
+
SubmitJobRequest(
|
|
100
|
+
job_type="agentic_ingest",
|
|
101
|
+
project_id=self.project_id,
|
|
102
|
+
args=json.dumps(
|
|
103
|
+
{
|
|
104
|
+
"prompt": prompt.strip(),
|
|
105
|
+
"files": [
|
|
106
|
+
{
|
|
107
|
+
"ref": ref,
|
|
108
|
+
"filename": doc.path,
|
|
109
|
+
"content_type": self._guess_mimetype(doc.path),
|
|
110
|
+
"size_bytes": len(doc.content),
|
|
111
|
+
"sha256": hashlib.sha256(doc.content).hexdigest(),
|
|
112
|
+
}
|
|
113
|
+
for doc, ref in zip(doc, refs, strict=True)
|
|
114
|
+
],
|
|
115
|
+
}
|
|
116
|
+
),
|
|
117
|
+
compute_environment_configuration=ComputeEnvironmentConfiguration(
|
|
118
|
+
name="Sphinx Cloud",
|
|
119
|
+
type=ComputeEnvironmentType.SPHINX_CLOUD,
|
|
120
|
+
),
|
|
121
|
+
),
|
|
122
|
+
headers=self._auth_headers(),
|
|
123
|
+
)
|
|
124
|
+
|
|
125
|
+
return res.job.id
|
|
126
|
+
|
|
127
|
+
def _auth_headers(self) -> dict[str, str]:
|
|
128
|
+
"""Build the HTTP headers used to authenticate Sphinx API requests."""
|
|
129
|
+
credentials = f":{self.api_key}".encode("utf-8")
|
|
130
|
+
encoded_api_key = base64.b64encode(credentials).decode("ascii")
|
|
131
|
+
return {
|
|
132
|
+
"Authorization": f"Basic {encoded_api_key}",
|
|
133
|
+
"X-Sphinx-Version": "1.0.0",
|
|
134
|
+
}
|
|
135
|
+
|
|
136
|
+
def _guess_mimetype(self, path: str) -> str:
|
|
137
|
+
"""Infer a MIME type from a path, defaulting to binary content."""
|
|
138
|
+
return mimetypes.guess_type(path)[0] or "application/octet-stream"
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
def _createKnowledgeBaseClient(api_key: str, project_id: str) -> _KnowledgeBaseClient:
|
|
142
|
+
"""Create a client for a Sphinx knowledge base.
|
|
143
|
+
|
|
144
|
+
Args:
|
|
145
|
+
api_key: PAT Token key used to authenticate requests to Sphinx. The corresponding
|
|
146
|
+
user must be an admin.
|
|
147
|
+
project_id: Identifier of the Sphinx project that owns the knowledge base.
|
|
148
|
+
|
|
149
|
+
Returns:
|
|
150
|
+
A configured knowledge-base client.
|
|
151
|
+
"""
|
|
152
|
+
return _KnowledgeBaseClient(api_key, project_id)
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
def _checkInputIsGood(prompt: str, docs: list[_Document]) -> None:
|
|
156
|
+
"""Validate prompt and batch policies before uploads.
|
|
157
|
+
|
|
158
|
+
Requires 1–20,000 validated File objects, at most 2.5 GiB in total, and
|
|
159
|
+
a nonblank prompt of at most 10,000 characters. File construction owns
|
|
160
|
+
path and content validation. Wrong types raise TypeError; invalid values
|
|
161
|
+
raise ValueError.
|
|
162
|
+
"""
|
|
163
|
+
|
|
164
|
+
def checkInputTypesAreCorrect() -> None:
|
|
165
|
+
if not isinstance(prompt, str):
|
|
166
|
+
raise TypeError(f"prompt must be a string; got {type(prompt).__name__}")
|
|
167
|
+
if not isinstance(docs, list):
|
|
168
|
+
raise TypeError(f"docs must be a list of File objects; got {type(docs).__name__}")
|
|
169
|
+
for index, doc in enumerate(docs):
|
|
170
|
+
if not isinstance(doc, _File):
|
|
171
|
+
raise TypeError(f"docs[{index}] must be a File; got {type(doc).__name__}")
|
|
172
|
+
|
|
173
|
+
def checkDocumentCountIsAtMost20k() -> None:
|
|
174
|
+
if len(docs) > _MAX_DOCUMENT_COUNT:
|
|
175
|
+
raise ValueError(f"docs must contain at most {_MAX_DOCUMENT_COUNT:,} documents; got {len(docs):,}")
|
|
176
|
+
|
|
177
|
+
def checkAtLeastOneDocument() -> None:
|
|
178
|
+
if len(docs) < _MIN_DOCUMENT_COUNT:
|
|
179
|
+
raise ValueError(f"docs must contain at least {_MIN_DOCUMENT_COUNT} document; got {len(docs)}")
|
|
180
|
+
|
|
181
|
+
def checkPromptSizeAtMost10kCharacters() -> None:
|
|
182
|
+
if len(prompt) > _MAX_PROMPT_CHARACTERS:
|
|
183
|
+
raise ValueError(f"prompt must contain at most {_MAX_PROMPT_CHARACTERS:,} characters; got {len(prompt):,}")
|
|
184
|
+
|
|
185
|
+
def checkPromptIsNotEmpty() -> None:
|
|
186
|
+
if not prompt.strip():
|
|
187
|
+
raise ValueError(
|
|
188
|
+
"prompt must not be empty or contain only whitespace; "
|
|
189
|
+
f"got {len(prompt):,} characters, 0 after stripping whitespace"
|
|
190
|
+
)
|
|
191
|
+
|
|
192
|
+
def checkTotalDocSizeIsAtMost2p5Gb() -> None:
|
|
193
|
+
total_bytes = sum(len(doc.content) for doc in docs)
|
|
194
|
+
if total_bytes > _MAX_COLLECTION_BYTES:
|
|
195
|
+
raise ValueError(
|
|
196
|
+
f"docs must total at most {_MAX_COLLECTION_BYTES / _BYTES_PER_GIB:g} GiB "
|
|
197
|
+
f"({_MAX_COLLECTION_BYTES:,} bytes); "
|
|
198
|
+
f"got {total_bytes:,} bytes across {len(docs):,} documents"
|
|
199
|
+
)
|
|
200
|
+
|
|
201
|
+
checkInputTypesAreCorrect()
|
|
202
|
+
checkPromptSizeAtMost10kCharacters()
|
|
203
|
+
checkPromptIsNotEmpty()
|
|
204
|
+
checkDocumentCountIsAtMost20k()
|
|
205
|
+
checkAtLeastOneDocument()
|
|
206
|
+
checkTotalDocSizeIsAtMost2p5Gb()
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
@@ -0,0 +1,115 @@
|
|
|
1
|
+
# Generated from object_storage.proto. DO NOT EDIT.
|
|
2
|
+
# Generated by protoc-gen-py v0.3.0 with parameter "".
|
|
3
|
+
# ruff: noqa: PGH004
|
|
4
|
+
# ruff: noqa
|
|
5
|
+
# fmt: off
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from typing import Literal, TYPE_CHECKING, TypeAlias
|
|
10
|
+
|
|
11
|
+
from protobuf import Message
|
|
12
|
+
from protobuf._codegen import file_desc
|
|
13
|
+
|
|
14
|
+
if TYPE_CHECKING:
|
|
15
|
+
from protobuf import DescFile
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
_CreatePresignedPutRequestFields: TypeAlias = Literal["project_id", "content_type", "filename"]
|
|
19
|
+
|
|
20
|
+
class CreatePresignedPutRequest(Message[_CreatePresignedPutRequestFields]):
|
|
21
|
+
"""
|
|
22
|
+
```proto
|
|
23
|
+
message sphinx.CreatePresignedPutRequest
|
|
24
|
+
```
|
|
25
|
+
|
|
26
|
+
Attributes:
|
|
27
|
+
project_id:
|
|
28
|
+
```proto
|
|
29
|
+
string project_id = 1;
|
|
30
|
+
```
|
|
31
|
+
content_type:
|
|
32
|
+
```proto
|
|
33
|
+
optional string content_type = 2;
|
|
34
|
+
```
|
|
35
|
+
filename:
|
|
36
|
+
```proto
|
|
37
|
+
optional string filename = 3;
|
|
38
|
+
```
|
|
39
|
+
"""
|
|
40
|
+
|
|
41
|
+
__slots__ = ("project_id", "content_type", "filename")
|
|
42
|
+
|
|
43
|
+
if TYPE_CHECKING:
|
|
44
|
+
|
|
45
|
+
def __init__(
|
|
46
|
+
self,
|
|
47
|
+
*,
|
|
48
|
+
project_id: str = "",
|
|
49
|
+
content_type: str | None = None,
|
|
50
|
+
filename: str | None = None,
|
|
51
|
+
) -> None:
|
|
52
|
+
pass
|
|
53
|
+
|
|
54
|
+
project_id: str
|
|
55
|
+
content_type: str
|
|
56
|
+
filename: str
|
|
57
|
+
|
|
58
|
+
_CreatePresignedPutResponseFields: TypeAlias = Literal["presigned_url", "ref", "required_headers"]
|
|
59
|
+
|
|
60
|
+
class CreatePresignedPutResponse(Message[_CreatePresignedPutResponseFields]):
|
|
61
|
+
"""
|
|
62
|
+
```proto
|
|
63
|
+
message sphinx.CreatePresignedPutResponse
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
Attributes:
|
|
67
|
+
presigned_url:
|
|
68
|
+
```proto
|
|
69
|
+
string presigned_url = 1;
|
|
70
|
+
```
|
|
71
|
+
ref:
|
|
72
|
+
```proto
|
|
73
|
+
string ref = 2;
|
|
74
|
+
```
|
|
75
|
+
required_headers:
|
|
76
|
+
Headers the client MUST include on the PUT request. The presigned URL
|
|
77
|
+
is signed against these exact header name/value pairs; omitting any of
|
|
78
|
+
them causes S3 to reject the upload with SignatureDoesNotMatch.
|
|
79
|
+
|
|
80
|
+
```proto
|
|
81
|
+
map<string, string> required_headers = 3;
|
|
82
|
+
```
|
|
83
|
+
"""
|
|
84
|
+
|
|
85
|
+
__slots__ = ("presigned_url", "ref", "required_headers")
|
|
86
|
+
|
|
87
|
+
if TYPE_CHECKING:
|
|
88
|
+
|
|
89
|
+
def __init__(
|
|
90
|
+
self,
|
|
91
|
+
*,
|
|
92
|
+
presigned_url: str = "",
|
|
93
|
+
ref: str = "",
|
|
94
|
+
required_headers: dict[str, str] | None = None,
|
|
95
|
+
) -> None:
|
|
96
|
+
pass
|
|
97
|
+
|
|
98
|
+
presigned_url: str
|
|
99
|
+
ref: str
|
|
100
|
+
required_headers: dict[str, str]
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
_DESC = file_desc(
|
|
104
|
+
b'\n\x14object_storage.proto\x12\x06sphinx"\xa1\x01\n\x19CreatePresignedPutRequest\x12\x1d\n\nproject_id\x18\x01 \x01(\tR\tprojectId\x12&\n\x0ccontent_type\x18\x02 \x01(\tH\x00R\x0bcontentType\x88\x01\x01\x12\x1f\n\x08filename\x18\x03 \x01(\tH\x01R\x08filename\x88\x01\x01B\x0f\n\r_content_typeB\x0b\n\t_filename"\xfb\x01\n\x1aCreatePresignedPutResponse\x12#\n\rpresigned_url\x18\x01 \x01(\tR\x0cpresignedUrl\x12\x10\n\x03ref\x18\x02 \x01(\tR\x03ref\x12b\n\x10required_headers\x18\x03 \x03(\x0b27.sphinx.CreatePresignedPutResponse.RequiredHeadersEntryR\x0frequiredHeaders\x1aB\n\x14RequiredHeadersEntry\x12\x10\n\x03key\x18\x01 \x01(\tR\x03key\x12\x14\n\x05value\x18\x02 \x01(\tR\x05value:\x028\x01b\x06proto3',
|
|
105
|
+
[],
|
|
106
|
+
{
|
|
107
|
+
"CreatePresignedPutRequest": CreatePresignedPutRequest,
|
|
108
|
+
"CreatePresignedPutResponse": CreatePresignedPutResponse,
|
|
109
|
+
},
|
|
110
|
+
)
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def desc() -> DescFile:
|
|
114
|
+
"""Returns the descriptor for the file `object_storage.proto`."""
|
|
115
|
+
return _DESC
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
from __future__ import annotations
|