needlesearchai 2.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- needlesearchai/__init__.py +78 -0
- needlesearchai/_http.py +138 -0
- needlesearchai/chat.py +69 -0
- needlesearchai/client.py +114 -0
- needlesearchai/exceptions.py +47 -0
- needlesearchai/files.py +187 -0
- needlesearchai/folders.py +30 -0
- needlesearchai/items.py +103 -0
- needlesearchai/jobs.py +99 -0
- needlesearchai/research.py +130 -0
- needlesearchai/search.py +54 -0
- needlesearchai/types.py +212 -0
- needlesearchai/uploads.py +70 -0
- needlesearchai/webhooks.py +186 -0
- needlesearchai-2.1.0.dist-info/METADATA +121 -0
- needlesearchai-2.1.0.dist-info/RECORD +18 -0
- needlesearchai-2.1.0.dist-info/WHEEL +5 -0
- needlesearchai-2.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
"""NeedleSearch Python SDK.
|
|
2
|
+
|
|
3
|
+
Quick start:
|
|
4
|
+
from needlesearchai import NeedleSearch
|
|
5
|
+
|
|
6
|
+
ns = NeedleSearch(api_key="nsk_...")
|
|
7
|
+
results = ns.search("termination clauses") # vector search
|
|
8
|
+
overview = ns.search("notice period?", include_overview=True) # search + AI answer
|
|
9
|
+
for chunk in ns.chat.stream("Explain promissory estoppel"): # plain chat (no docs)
|
|
10
|
+
print(chunk.text, end="", flush=True)
|
|
11
|
+
|
|
12
|
+
for item in ns.items.iterate(): # browse, cursor-paged
|
|
13
|
+
print(item.name, item.status)
|
|
14
|
+
|
|
15
|
+
job = ns.jobs.create("Summarise the risks", item_ids=[doc_id]) # async research
|
|
16
|
+
print(ns.jobs.wait(job.id).answer)
|
|
17
|
+
"""
|
|
18
|
+
from needlesearchai.client import NeedleSearch
|
|
19
|
+
from needlesearchai.exceptions import (
|
|
20
|
+
AuthError,
|
|
21
|
+
DocumentNotReadyError,
|
|
22
|
+
InsufficientScopeError,
|
|
23
|
+
NeedleSearchError,
|
|
24
|
+
NotFoundError,
|
|
25
|
+
QuotaError,
|
|
26
|
+
RateLimitError,
|
|
27
|
+
ServerError,
|
|
28
|
+
)
|
|
29
|
+
from needlesearchai.types import (
|
|
30
|
+
ChatChunk,
|
|
31
|
+
FileRecord,
|
|
32
|
+
Item,
|
|
33
|
+
ItemPage,
|
|
34
|
+
MeInfo,
|
|
35
|
+
QuotaInfo,
|
|
36
|
+
ResearchJob,
|
|
37
|
+
ResearchResult,
|
|
38
|
+
SearchResponse,
|
|
39
|
+
SearchResult,
|
|
40
|
+
ServiceStatus,
|
|
41
|
+
UploadJob,
|
|
42
|
+
UploadJobPage,
|
|
43
|
+
UploadResult,
|
|
44
|
+
Webhook,
|
|
45
|
+
WebhookDelivery,
|
|
46
|
+
)
|
|
47
|
+
from needlesearchai.webhooks import verify_signature
|
|
48
|
+
|
|
49
|
+
__all__ = [
|
|
50
|
+
"NeedleSearch",
|
|
51
|
+
"NeedleSearchError",
|
|
52
|
+
"AuthError",
|
|
53
|
+
"InsufficientScopeError",
|
|
54
|
+
"RateLimitError",
|
|
55
|
+
"QuotaError",
|
|
56
|
+
"NotFoundError",
|
|
57
|
+
"ServerError",
|
|
58
|
+
"DocumentNotReadyError",
|
|
59
|
+
"SearchResult",
|
|
60
|
+
"SearchResponse",
|
|
61
|
+
"ChatChunk",
|
|
62
|
+
"FileRecord",
|
|
63
|
+
"UploadResult",
|
|
64
|
+
"QuotaInfo",
|
|
65
|
+
"MeInfo",
|
|
66
|
+
"ResearchResult",
|
|
67
|
+
"Item",
|
|
68
|
+
"ItemPage",
|
|
69
|
+
"ResearchJob",
|
|
70
|
+
"ServiceStatus",
|
|
71
|
+
"UploadJob",
|
|
72
|
+
"UploadJobPage",
|
|
73
|
+
"Webhook",
|
|
74
|
+
"WebhookDelivery",
|
|
75
|
+
"verify_signature",
|
|
76
|
+
]
|
|
77
|
+
|
|
78
|
+
__version__ = "2.1.0"
|
needlesearchai/_http.py
ADDED
|
@@ -0,0 +1,138 @@
|
|
|
1
|
+
"""HTTP client wrapper with auth, retry, and error handling."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import time
|
|
5
|
+
from typing import Any, Generator
|
|
6
|
+
|
|
7
|
+
try:
|
|
8
|
+
import httpx
|
|
9
|
+
except ImportError as e:
|
|
10
|
+
raise ImportError("needlesearchai requires httpx. Install with: pip install needlesearchai[http]") from e
|
|
11
|
+
|
|
12
|
+
from needlesearchai.exceptions import (
|
|
13
|
+
AuthError,
|
|
14
|
+
DocumentNotReadyError,
|
|
15
|
+
InsufficientScopeError,
|
|
16
|
+
NeedleSearchError,
|
|
17
|
+
NotFoundError,
|
|
18
|
+
QuotaError,
|
|
19
|
+
RateLimitError,
|
|
20
|
+
ServerError,
|
|
21
|
+
)
|
|
22
|
+
|
|
23
|
+
_DEFAULT_TIMEOUT = 30.0
|
|
24
|
+
_STREAM_TIMEOUT = 300.0
|
|
25
|
+
_MAX_RETRIES = 3
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _raise_for_status(response: httpx.Response) -> None:
|
|
29
|
+
if response.is_success:
|
|
30
|
+
return
|
|
31
|
+
|
|
32
|
+
try:
|
|
33
|
+
body = response.json()
|
|
34
|
+
except Exception:
|
|
35
|
+
body = {}
|
|
36
|
+
|
|
37
|
+
error_code = body.get("error", "")
|
|
38
|
+
try:
|
|
39
|
+
raw_text = response.text[:300]
|
|
40
|
+
except Exception:
|
|
41
|
+
raw_text = ""
|
|
42
|
+
message = body.get("message", body.get("detail", raw_text))
|
|
43
|
+
status = response.status_code
|
|
44
|
+
|
|
45
|
+
if status == 401:
|
|
46
|
+
raise AuthError(message or "Invalid or missing API key", status_code=status, error_code=error_code)
|
|
47
|
+
if status == 403:
|
|
48
|
+
if error_code == "insufficient_scope":
|
|
49
|
+
raise InsufficientScopeError(message, status_code=status, error_code=error_code)
|
|
50
|
+
raise AuthError(message, status_code=status, error_code=error_code)
|
|
51
|
+
if status == 404:
|
|
52
|
+
raise NotFoundError(message, status_code=status, error_code=error_code)
|
|
53
|
+
if status == 429:
|
|
54
|
+
if error_code == "monthly_quota_exceeded":
|
|
55
|
+
raise QuotaError(message, used=body.get("used"), limit=body.get("limit"))
|
|
56
|
+
retry_after = int(response.headers.get("Retry-After", 60))
|
|
57
|
+
raise RateLimitError(message, retry_after=retry_after)
|
|
58
|
+
if status == 400:
|
|
59
|
+
if error_code == "document_not_ready":
|
|
60
|
+
raise DocumentNotReadyError(message, status_code=status, error_code=error_code)
|
|
61
|
+
raise NeedleSearchError(message, status_code=status, error_code=error_code)
|
|
62
|
+
if status >= 500:
|
|
63
|
+
raise ServerError(message or f"Server error {status}", status_code=status)
|
|
64
|
+
|
|
65
|
+
raise NeedleSearchError(message or f"HTTP {status}", status_code=status, error_code=error_code)
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
class HttpClient:
|
|
69
|
+
"""Synchronous HTTP client with exponential backoff retry."""
|
|
70
|
+
|
|
71
|
+
def __init__(self, api_key: str, base_url: str, timeout: float = _DEFAULT_TIMEOUT):
|
|
72
|
+
self._client = httpx.Client(
|
|
73
|
+
base_url=base_url,
|
|
74
|
+
headers={"Authorization": f"Bearer {api_key}", "User-Agent": "needlesearchai-python/1.0"},
|
|
75
|
+
timeout=timeout,
|
|
76
|
+
)
|
|
77
|
+
|
|
78
|
+
def get(self, path: str, **params) -> Any:
|
|
79
|
+
return self._request("GET", path, params=params or None)
|
|
80
|
+
|
|
81
|
+
def post(self, path: str, json: Any = None, *, idempotency_key: str | None = None) -> Any:
|
|
82
|
+
headers = {"Idempotency-Key": idempotency_key} if idempotency_key else None
|
|
83
|
+
return self._request("POST", path, json=json, headers=headers)
|
|
84
|
+
|
|
85
|
+
def patch(self, path: str, json: Any = None) -> Any:
|
|
86
|
+
return self._request("PATCH", path, json=json)
|
|
87
|
+
|
|
88
|
+
def delete(self, path: str) -> Any:
|
|
89
|
+
return self._request("DELETE", path)
|
|
90
|
+
|
|
91
|
+
def _request(self, method: str, path: str, **kwargs) -> Any:
|
|
92
|
+
for attempt in range(_MAX_RETRIES):
|
|
93
|
+
try:
|
|
94
|
+
# Drop unset optionals so httpx does not send empty headers.
|
|
95
|
+
call = {k: v for k, v in kwargs.items() if v is not None}
|
|
96
|
+
r = self._client.request(method, path, **call)
|
|
97
|
+
_raise_for_status(r)
|
|
98
|
+
return r.json() if r.content else None
|
|
99
|
+
except RateLimitError as e:
|
|
100
|
+
if attempt == _MAX_RETRIES - 1:
|
|
101
|
+
raise
|
|
102
|
+
wait = min(e.retry_after, 30)
|
|
103
|
+
time.sleep(wait)
|
|
104
|
+
except (httpx.TimeoutException, httpx.NetworkError) as e:
|
|
105
|
+
if attempt == _MAX_RETRIES - 1:
|
|
106
|
+
raise NeedleSearchError(f"Network error: {e}") from e
|
|
107
|
+
time.sleep(2 ** attempt)
|
|
108
|
+
|
|
109
|
+
def raw(self, method: str, path: str, **kwargs):
|
|
110
|
+
"""A checked response with its body intact — for endpoints that return
|
|
111
|
+
bytes rather than JSON (file downloads)."""
|
|
112
|
+
r = self._client.request(method, path, **kwargs)
|
|
113
|
+
_raise_for_status(r)
|
|
114
|
+
return r
|
|
115
|
+
|
|
116
|
+
def stream(self, path: str, json: Any = None) -> Generator[str, None, None]:
|
|
117
|
+
"""Yield raw SSE data lines from a streaming endpoint."""
|
|
118
|
+
with self._client.stream("POST", path, json=json, timeout=_STREAM_TIMEOUT) as r:
|
|
119
|
+
_raise_for_status(r)
|
|
120
|
+
parts: list[str] = []
|
|
121
|
+
for line in r.iter_lines():
|
|
122
|
+
if line.startswith("data:"):
|
|
123
|
+
raw = line[5:]
|
|
124
|
+
parts.append(raw[1:] if raw.startswith(" ") else raw)
|
|
125
|
+
elif line == "" and parts:
|
|
126
|
+
yield "\n".join(parts)
|
|
127
|
+
parts = []
|
|
128
|
+
if parts:
|
|
129
|
+
yield "\n".join(parts)
|
|
130
|
+
|
|
131
|
+
def close(self):
|
|
132
|
+
self._client.close()
|
|
133
|
+
|
|
134
|
+
def __enter__(self):
|
|
135
|
+
return self
|
|
136
|
+
|
|
137
|
+
def __exit__(self, *args):
|
|
138
|
+
self.close()
|
needlesearchai/chat.py
ADDED
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
"""NeedleSearch SDK — Chat API with streaming support."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
import json
|
|
4
|
+
from typing import Generator
|
|
5
|
+
|
|
6
|
+
from needlesearchai.types import ChatChunk
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class ChatAPI:
|
|
10
|
+
"""Wrapper for POST /v1/chat."""
|
|
11
|
+
|
|
12
|
+
def __init__(self, http):
|
|
13
|
+
self._http = http
|
|
14
|
+
|
|
15
|
+
def stream(
|
|
16
|
+
self,
|
|
17
|
+
message: str,
|
|
18
|
+
*,
|
|
19
|
+
provider: str | None = None,
|
|
20
|
+
) -> Generator[ChatChunk, None, None]:
|
|
21
|
+
"""Stream a plain chat response from the AI.
|
|
22
|
+
|
|
23
|
+
Conversational chat only — it does NOT read your documents. For
|
|
24
|
+
document-grounded answers use ``client.search(..., include_overview=True)``
|
|
25
|
+
or the research API.
|
|
26
|
+
|
|
27
|
+
Args:
|
|
28
|
+
message: User question or instruction
|
|
29
|
+
provider: LLM provider override ('deepseek', 'openai', 'anthropic')
|
|
30
|
+
|
|
31
|
+
Yields:
|
|
32
|
+
ChatChunk with .text containing the streaming text
|
|
33
|
+
|
|
34
|
+
Example:
|
|
35
|
+
for chunk in ns.chat.stream("Explain promissory estoppel"):
|
|
36
|
+
print(chunk.text, end="", flush=True)
|
|
37
|
+
"""
|
|
38
|
+
body = {"message": message}
|
|
39
|
+
if provider:
|
|
40
|
+
body["provider"] = provider
|
|
41
|
+
|
|
42
|
+
for data_line in self._http.stream("/chat", json=body):
|
|
43
|
+
if not data_line:
|
|
44
|
+
continue
|
|
45
|
+
try:
|
|
46
|
+
payload = json.loads(data_line)
|
|
47
|
+
# Only yield actual text delta events
|
|
48
|
+
if isinstance(payload, str):
|
|
49
|
+
yield ChatChunk(text=payload)
|
|
50
|
+
elif isinstance(payload, dict) and "text" in payload:
|
|
51
|
+
yield ChatChunk(text=payload["text"])
|
|
52
|
+
elif isinstance(payload, dict) and payload.get("event") == "done":
|
|
53
|
+
break
|
|
54
|
+
except (json.JSONDecodeError, TypeError):
|
|
55
|
+
# Plain text delta
|
|
56
|
+
if data_line and not data_line.startswith("{"):
|
|
57
|
+
yield ChatChunk(text=data_line)
|
|
58
|
+
|
|
59
|
+
def complete(self, message: str, **kwargs) -> str:
|
|
60
|
+
"""Non-streaming chat — collects all chunks and returns the full text.
|
|
61
|
+
|
|
62
|
+
Args:
|
|
63
|
+
message: User question or instruction
|
|
64
|
+
**kwargs: Same keyword arguments as stream()
|
|
65
|
+
|
|
66
|
+
Returns:
|
|
67
|
+
Complete AI response as a string
|
|
68
|
+
"""
|
|
69
|
+
return "".join(chunk.text for chunk in self.stream(message, **kwargs))
|
needlesearchai/client.py
ADDED
|
@@ -0,0 +1,114 @@
|
|
|
1
|
+
"""NeedleSearch Python SDK — main client."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
from needlesearchai._http import HttpClient
|
|
5
|
+
from needlesearchai.chat import ChatAPI
|
|
6
|
+
from needlesearchai.files import FilesAPI
|
|
7
|
+
from needlesearchai.folders import FoldersAPI
|
|
8
|
+
from needlesearchai.items import ItemsAPI
|
|
9
|
+
from needlesearchai.jobs import ResearchJobsAPI
|
|
10
|
+
from needlesearchai.research import ResearchAPI
|
|
11
|
+
from needlesearchai.search import SearchAPI
|
|
12
|
+
from needlesearchai.types import MeInfo, QuotaInfo, ServiceStatus
|
|
13
|
+
from needlesearchai.uploads import UploadsAPI
|
|
14
|
+
from needlesearchai.webhooks import WebhooksAPI
|
|
15
|
+
|
|
16
|
+
_DEFAULT_BASE_URL = "https://your-instance.com/v1"
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class NeedleSearch:
|
|
20
|
+
"""NeedleSearch API client.
|
|
21
|
+
|
|
22
|
+
Usage:
|
|
23
|
+
from needlesearchai import NeedleSearch
|
|
24
|
+
|
|
25
|
+
ns = NeedleSearch(api_key="nsk_...")
|
|
26
|
+
|
|
27
|
+
# Semantic search (optionally with an AI overview over your documents)
|
|
28
|
+
results = ns.search("What are the termination clauses?")
|
|
29
|
+
overview = ns.search("How long is the notice period?", include_overview=True)
|
|
30
|
+
|
|
31
|
+
# Streaming chat — plain conversation, does NOT read your documents
|
|
32
|
+
for chunk in ns.chat.stream("Explain promissory estoppel"):
|
|
33
|
+
print(chunk.text, end="", flush=True)
|
|
34
|
+
|
|
35
|
+
# Deep research over your documents (Starter+ plan)
|
|
36
|
+
answer = ns.research.complete("What are the risks in this contract?", document_ids=["..."])
|
|
37
|
+
"""
|
|
38
|
+
|
|
39
|
+
def __init__(self, api_key: str, base_url: str = _DEFAULT_BASE_URL, timeout: float = 30.0):
|
|
40
|
+
"""Create a NeedleSearch client.
|
|
41
|
+
|
|
42
|
+
Args:
|
|
43
|
+
api_key: Your API key (starts with 'nsk_')
|
|
44
|
+
base_url: API base URL (default: https://your-instance.com/v1)
|
|
45
|
+
timeout: Request timeout in seconds
|
|
46
|
+
"""
|
|
47
|
+
if not api_key:
|
|
48
|
+
raise ValueError("api_key is required")
|
|
49
|
+
if not api_key.startswith("nsk_"):
|
|
50
|
+
raise ValueError("api_key must start with 'nsk_'")
|
|
51
|
+
|
|
52
|
+
self._http = HttpClient(api_key=api_key, base_url=base_url.rstrip("/"), timeout=timeout)
|
|
53
|
+
|
|
54
|
+
self.search = SearchAPI(self._http)
|
|
55
|
+
self.chat = ChatAPI(self._http)
|
|
56
|
+
self.research = ResearchAPI(self._http)
|
|
57
|
+
self.files = FilesAPI(self._http)
|
|
58
|
+
self.items = ItemsAPI(self._http)
|
|
59
|
+
self.folders = FoldersAPI(self._http)
|
|
60
|
+
self.webhooks = WebhooksAPI(self._http)
|
|
61
|
+
self.jobs = ResearchJobsAPI(self._http)
|
|
62
|
+
self.uploads = UploadsAPI(self._http)
|
|
63
|
+
|
|
64
|
+
def status(self) -> ServiceStatus:
|
|
65
|
+
"""Whether the API can currently serve each class of operation.
|
|
66
|
+
|
|
67
|
+
The one call that needs no key — so it still answers when the key itself
|
|
68
|
+
is the suspect. `down` means retry with backoff rather than debugging
|
|
69
|
+
your integration.
|
|
70
|
+
"""
|
|
71
|
+
data = self._http.get("/status")
|
|
72
|
+
return ServiceStatus(
|
|
73
|
+
status=data.get("status", "unknown"),
|
|
74
|
+
services=data.get("services", {}),
|
|
75
|
+
api_version=data.get("api_version", "v1"),
|
|
76
|
+
checked_at=data.get("checked_at"),
|
|
77
|
+
)
|
|
78
|
+
|
|
79
|
+
def me(self) -> MeInfo:
|
|
80
|
+
"""Return identity and plan information for this API key."""
|
|
81
|
+
data = self._http.get("/me")
|
|
82
|
+
return MeInfo(
|
|
83
|
+
key_prefix=data.get("key_prefix", ""),
|
|
84
|
+
name=data.get("name", ""),
|
|
85
|
+
plan=data.get("plan", "free"),
|
|
86
|
+
scopes=data.get("scopes", []),
|
|
87
|
+
organization_id=data.get("organization_id", ""),
|
|
88
|
+
features=data.get("features", {}),
|
|
89
|
+
created_at=data.get("created_at"),
|
|
90
|
+
)
|
|
91
|
+
|
|
92
|
+
def quota(self) -> QuotaInfo:
|
|
93
|
+
"""Return current quota and plan status."""
|
|
94
|
+
data = self._http.get("/quota")
|
|
95
|
+
monthly = data.get("monthly_requests", {})
|
|
96
|
+
return QuotaInfo(
|
|
97
|
+
plan=data.get("plan", "free"),
|
|
98
|
+
monthly_limit=monthly.get("limit"),
|
|
99
|
+
monthly_used=monthly.get("used", 0),
|
|
100
|
+
monthly_remaining=monthly.get("remaining"),
|
|
101
|
+
rate_limit_rpm=data.get("rate_limit_rpm", 0),
|
|
102
|
+
features=data.get("features", {}),
|
|
103
|
+
reset_at=data.get("reset_at"),
|
|
104
|
+
)
|
|
105
|
+
|
|
106
|
+
def close(self) -> None:
|
|
107
|
+
"""Close the underlying HTTP client."""
|
|
108
|
+
self._http.close()
|
|
109
|
+
|
|
110
|
+
def __enter__(self):
|
|
111
|
+
return self
|
|
112
|
+
|
|
113
|
+
def __exit__(self, *args):
|
|
114
|
+
self.close()
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
"""NeedleSearch SDK exceptions."""
|
|
2
|
+
|
|
3
|
+
|
|
4
|
+
class NeedleSearchError(Exception):
|
|
5
|
+
"""Base exception for all NeedleSearch errors."""
|
|
6
|
+
def __init__(self, message: str, status_code: int | None = None, error_code: str | None = None):
|
|
7
|
+
super().__init__(message)
|
|
8
|
+
self.status_code = status_code
|
|
9
|
+
self.error_code = error_code
|
|
10
|
+
|
|
11
|
+
def __repr__(self):
|
|
12
|
+
return f"{self.__class__.__name__}({self.args[0]!r}, status_code={self.status_code})"
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class AuthError(NeedleSearchError):
|
|
16
|
+
"""API key missing, invalid, revoked, or expired."""
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class InsufficientScopeError(NeedleSearchError):
|
|
20
|
+
"""API key lacks the required scope for this operation."""
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class RateLimitError(NeedleSearchError):
|
|
24
|
+
"""Rate limit exceeded. Check retry_after for seconds to wait."""
|
|
25
|
+
def __init__(self, message: str, retry_after: int = 60):
|
|
26
|
+
super().__init__(message, status_code=429, error_code="rate_limit_exceeded")
|
|
27
|
+
self.retry_after = retry_after
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
class QuotaError(NeedleSearchError):
|
|
31
|
+
"""Monthly request quota exceeded."""
|
|
32
|
+
def __init__(self, message: str, used: int | None = None, limit: int | None = None):
|
|
33
|
+
super().__init__(message, status_code=429, error_code="monthly_quota_exceeded")
|
|
34
|
+
self.used = used
|
|
35
|
+
self.limit = limit
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class NotFoundError(NeedleSearchError):
|
|
39
|
+
"""Requested resource not found."""
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
class ServerError(NeedleSearchError):
|
|
43
|
+
"""Internal server error."""
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
class DocumentNotReadyError(NeedleSearchError):
|
|
47
|
+
"""Document is still being processed."""
|
needlesearchai/files.py
ADDED
|
@@ -0,0 +1,187 @@
|
|
|
1
|
+
"""NeedleSearch SDK — Files API."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
from typing import BinaryIO
|
|
5
|
+
|
|
6
|
+
from needlesearchai.exceptions import NeedleSearchError
|
|
7
|
+
from needlesearchai.types import FileRecord, UploadResult
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
class FilesAPI:
|
|
11
|
+
"""Wrapper for /v1/files endpoints.
|
|
12
|
+
|
|
13
|
+
Requires Pro plan or higher.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
def __init__(self, http):
|
|
17
|
+
self._http = http
|
|
18
|
+
|
|
19
|
+
def list(self, *, limit: int = 50, offset: int = 0) -> list[FileRecord]:
|
|
20
|
+
"""Files at the workspace root.
|
|
21
|
+
|
|
22
|
+
Deprecated in favour of `client.items.list()`, which returns folders too
|
|
23
|
+
and pages by cursor. This called `GET /v1/files`, an endpoint that no
|
|
24
|
+
longer exists — every call raised NotFoundError — so it is routed
|
|
25
|
+
through /v1/items and filtered to files.
|
|
26
|
+
"""
|
|
27
|
+
data = self._http.get("/items", limit=limit, offset=offset)
|
|
28
|
+
return [
|
|
29
|
+
_parse_file(i) for i in data.get("items", []) if i.get("type") == "file"
|
|
30
|
+
]
|
|
31
|
+
|
|
32
|
+
def get(self, file_id: str) -> FileRecord:
|
|
33
|
+
"""Metadata for one file.
|
|
34
|
+
|
|
35
|
+
Deprecated in favour of `client.items.get()`. Also pointed at a removed
|
|
36
|
+
endpoint (`GET /v1/files/{id}`); it uses /v1/items/{id} now.
|
|
37
|
+
"""
|
|
38
|
+
return _parse_file(self._http.get(f"/items/{file_id}"))
|
|
39
|
+
|
|
40
|
+
def delete(self, *file_ids: str, force: bool = False) -> dict:
|
|
41
|
+
"""Delete files and/or folders by id — the type is detected per id.
|
|
42
|
+
|
|
43
|
+
Returns `{"deleted": [...], "failed": [...]}`. A document published in a
|
|
44
|
+
market dataset comes back under `failed` unless `force=True`.
|
|
45
|
+
"""
|
|
46
|
+
if not file_ids:
|
|
47
|
+
raise ValueError("at least one id is required")
|
|
48
|
+
return self._http.post(
|
|
49
|
+
"/files/delete", json={"ids": list(file_ids), "force": force}
|
|
50
|
+
)
|
|
51
|
+
|
|
52
|
+
def reprocess(self, *file_ids: str) -> dict:
|
|
53
|
+
"""Retry failed documents in place — no re-upload, no duplicate.
|
|
54
|
+
|
|
55
|
+
Ids that are not in a failed state come back under `failed`.
|
|
56
|
+
"""
|
|
57
|
+
if not file_ids:
|
|
58
|
+
raise ValueError("at least one id is required")
|
|
59
|
+
return self._http.post("/files/reprocess", json={"ids": list(file_ids)})
|
|
60
|
+
|
|
61
|
+
def download(self, file_id: str) -> bytes:
|
|
62
|
+
"""The original bytes. The document must have finished processing."""
|
|
63
|
+
response = self._http.raw("GET", f"/files/{file_id}/download")
|
|
64
|
+
return response.content
|
|
65
|
+
|
|
66
|
+
def upload(
|
|
67
|
+
self,
|
|
68
|
+
file: Path | BinaryIO | bytes,
|
|
69
|
+
*,
|
|
70
|
+
filename: str | None = None,
|
|
71
|
+
folder_id: str | None = None,
|
|
72
|
+
timeout: float = 120.0,
|
|
73
|
+
) -> UploadResult:
|
|
74
|
+
"""Upload a document for async processing.
|
|
75
|
+
|
|
76
|
+
Uses the resumable tus protocol under the hood: it creates an upload job
|
|
77
|
+
(`POST /v1/uploads`), streams the bytes to the tus endpoint, seals the
|
|
78
|
+
job, then polls until the document row exists. ZIP/archive files are
|
|
79
|
+
expanded server-side into a folder of documents.
|
|
80
|
+
|
|
81
|
+
Args:
|
|
82
|
+
file: File path, file-like object, or bytes
|
|
83
|
+
filename: Override filename (required if file is bytes or BinaryIO)
|
|
84
|
+
folder_id: Place document in this folder
|
|
85
|
+
timeout: Seconds to wait for the byte upload to finalize
|
|
86
|
+
|
|
87
|
+
Returns:
|
|
88
|
+
UploadResult with the new document ID and status
|
|
89
|
+
"""
|
|
90
|
+
import base64
|
|
91
|
+
import hashlib
|
|
92
|
+
import time
|
|
93
|
+
import uuid
|
|
94
|
+
|
|
95
|
+
try:
|
|
96
|
+
import httpx as _httpx
|
|
97
|
+
except ImportError:
|
|
98
|
+
raise ImportError("needlesearchai[http] required for file uploads")
|
|
99
|
+
|
|
100
|
+
if isinstance(file, Path):
|
|
101
|
+
filename = filename or file.name
|
|
102
|
+
content = file.read_bytes()
|
|
103
|
+
elif isinstance(file, bytes):
|
|
104
|
+
content = file
|
|
105
|
+
else:
|
|
106
|
+
content = file.read()
|
|
107
|
+
|
|
108
|
+
if not filename:
|
|
109
|
+
raise ValueError("filename is required when uploading bytes or a file-like object")
|
|
110
|
+
|
|
111
|
+
client_file_id = uuid.uuid4().hex
|
|
112
|
+
digest = hashlib.sha256(content).hexdigest()
|
|
113
|
+
|
|
114
|
+
# 1. Create the job + a single file slot.
|
|
115
|
+
job = self._http.post("/uploads", json={
|
|
116
|
+
"client_request_id": uuid.uuid4().hex,
|
|
117
|
+
"files": [{
|
|
118
|
+
"client_file_id": client_file_id,
|
|
119
|
+
"name": filename,
|
|
120
|
+
"size": len(content),
|
|
121
|
+
"hash": digest,
|
|
122
|
+
}],
|
|
123
|
+
**({"folder_id": folder_id} if folder_id else {}),
|
|
124
|
+
})
|
|
125
|
+
slot = next((f for f in job["files"] if f["client_file_id"] == client_file_id), job["files"][0])
|
|
126
|
+
token = slot["upload_token"]
|
|
127
|
+
tus_endpoint = job["tus_endpoint"]
|
|
128
|
+
|
|
129
|
+
# 2. tus upload — the token in Upload-Metadata is the auth (no Bearer on
|
|
130
|
+
# the tus host). One create + one PATCH; resumable by design.
|
|
131
|
+
def _b64(s: str) -> str:
|
|
132
|
+
return base64.b64encode(s.encode()).decode()
|
|
133
|
+
|
|
134
|
+
meta = f"upload_token {_b64(token)},filename {_b64(filename)}"
|
|
135
|
+
with _httpx.Client(timeout=timeout) as tus:
|
|
136
|
+
cr = tus.post(tus_endpoint, headers={
|
|
137
|
+
"Tus-Resumable": "1.0.0",
|
|
138
|
+
"Upload-Length": str(len(content)),
|
|
139
|
+
"Upload-Metadata": meta,
|
|
140
|
+
})
|
|
141
|
+
cr.raise_for_status()
|
|
142
|
+
location = cr.headers.get("Location") or ""
|
|
143
|
+
if not location:
|
|
144
|
+
raise NeedleSearchError("tus create did not return a Location header")
|
|
145
|
+
if location.startswith("/"):
|
|
146
|
+
base = tus_endpoint.split("/", 3)
|
|
147
|
+
location = f"{base[0]}//{base[2]}{location}"
|
|
148
|
+
pr = tus.patch(location, content=content, headers={
|
|
149
|
+
"Tus-Resumable": "1.0.0",
|
|
150
|
+
"Upload-Offset": "0",
|
|
151
|
+
"Content-Type": "application/offset+octet-stream",
|
|
152
|
+
})
|
|
153
|
+
pr.raise_for_status()
|
|
154
|
+
|
|
155
|
+
# 3. Seal the job.
|
|
156
|
+
self._http.post(f"/uploads/{job['id']}/seal")
|
|
157
|
+
|
|
158
|
+
# 4. Poll until the document row exists (post-finish finalized the bytes).
|
|
159
|
+
deadline = time.time() + timeout
|
|
160
|
+
terminal = {"uploaded", "processing", "ready", "duplicate", "failed"}
|
|
161
|
+
last = slot
|
|
162
|
+
while time.time() < deadline:
|
|
163
|
+
prog = self._http.get(f"/uploads/{job['id']}")
|
|
164
|
+
last = next((f for f in prog.get("files", []) if f["client_file_id"] == client_file_id), last)
|
|
165
|
+
if last.get("document_id") or last.get("status") in terminal:
|
|
166
|
+
break
|
|
167
|
+
time.sleep(1.0)
|
|
168
|
+
|
|
169
|
+
if last.get("status") == "failed":
|
|
170
|
+
raise NeedleSearchError(f"upload failed for {filename}")
|
|
171
|
+
return UploadResult(
|
|
172
|
+
id=last.get("document_id") or "",
|
|
173
|
+
name=filename,
|
|
174
|
+
status=last.get("status", "pending"),
|
|
175
|
+
message="" if last.get("document_id") else "still finalizing; poll items API",
|
|
176
|
+
)
|
|
177
|
+
|
|
178
|
+
def _parse_file(d: dict) -> FileRecord:
|
|
179
|
+
return FileRecord(
|
|
180
|
+
id=d.get("id", ""),
|
|
181
|
+
name=d.get("name", ""),
|
|
182
|
+
status=d.get("status", ""),
|
|
183
|
+
size_bytes=d.get("size_bytes"),
|
|
184
|
+
folder_id=d.get("folder_id"),
|
|
185
|
+
created_at=d.get("created_at"),
|
|
186
|
+
updated_at=d.get("updated_at"),
|
|
187
|
+
)
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
"""NeedleSearch SDK — Folders API (`/v1/folders`)."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
from typing import Optional
|
|
5
|
+
|
|
6
|
+
from needlesearchai.types import Item
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class FoldersAPI:
|
|
10
|
+
"""Wrapper for /v1/folders. Requires the `files:write` scope."""
|
|
11
|
+
|
|
12
|
+
def __init__(self, http):
|
|
13
|
+
self._http = http
|
|
14
|
+
|
|
15
|
+
def create(self, name: str, *, parent_id: Optional[str] = None) -> Item:
|
|
16
|
+
"""Create a folder, or return the existing one with the same name in the
|
|
17
|
+
same place. Idempotent server-side, so a retry is safe without a key.
|
|
18
|
+
"""
|
|
19
|
+
body: dict = {"name": name}
|
|
20
|
+
if parent_id is not None:
|
|
21
|
+
body["parent_id"] = parent_id
|
|
22
|
+
data = self._http.post("/folders", json=body)
|
|
23
|
+
return Item(
|
|
24
|
+
id=data.get("id", ""),
|
|
25
|
+
type="folder",
|
|
26
|
+
name=data.get("name", ""),
|
|
27
|
+
parent_id=data.get("parent_id"),
|
|
28
|
+
item_count=data.get("item_count"),
|
|
29
|
+
created_at=data.get("created_at"),
|
|
30
|
+
)
|