krenzo 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- krenzo-0.1.0/.gitignore +49 -0
- krenzo-0.1.0/LICENSE +21 -0
- krenzo-0.1.0/PKG-INFO +178 -0
- krenzo-0.1.0/README.md +161 -0
- krenzo-0.1.0/krenzo/__init__.py +34 -0
- krenzo-0.1.0/krenzo/client.py +467 -0
- krenzo-0.1.0/pyproject.toml +29 -0
- krenzo-0.1.0/tests/test_client.py +143 -0
krenzo-0.1.0/.gitignore
ADDED
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
# See https://help.github.com/articles/ignoring-files/ for more about ignoring files.
|
|
2
|
+
|
|
3
|
+
# dependencies
|
|
4
|
+
/node_modules
|
|
5
|
+
/.pnp
|
|
6
|
+
.pnp.*
|
|
7
|
+
.yarn/*
|
|
8
|
+
!.yarn/patches
|
|
9
|
+
!.yarn/plugins
|
|
10
|
+
!.yarn/releases
|
|
11
|
+
!.yarn/versions
|
|
12
|
+
|
|
13
|
+
# testing
|
|
14
|
+
/coverage
|
|
15
|
+
|
|
16
|
+
# next.js
|
|
17
|
+
/.next/
|
|
18
|
+
/out/
|
|
19
|
+
|
|
20
|
+
# production
|
|
21
|
+
/build
|
|
22
|
+
|
|
23
|
+
# misc
|
|
24
|
+
.DS_Store
|
|
25
|
+
*.pem
|
|
26
|
+
|
|
27
|
+
# debug
|
|
28
|
+
npm-debug.log*
|
|
29
|
+
yarn-debug.log*
|
|
30
|
+
yarn-error.log*
|
|
31
|
+
.pnpm-debug.log*
|
|
32
|
+
|
|
33
|
+
# env files (can opt-in for committing if needed)
|
|
34
|
+
.env*
|
|
35
|
+
|
|
36
|
+
# vercel
|
|
37
|
+
.vercel
|
|
38
|
+
|
|
39
|
+
# typescript
|
|
40
|
+
*.tsbuildinfo
|
|
41
|
+
next-env.d.ts
|
|
42
|
+
|
|
43
|
+
# Owned by its own repository (report_program/.git), not by this one.
|
|
44
|
+
report_program/
|
|
45
|
+
|
|
46
|
+
# Python SDK build artifacts
|
|
47
|
+
sdk/python/dist/
|
|
48
|
+
sdk/python/**/__pycache__/
|
|
49
|
+
sdk/python/*.egg-info/
|
krenzo-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Krenzo
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
krenzo-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,178 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: krenzo
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Search and extract API for LLM agents — clean, token-efficient web context.
|
|
5
|
+
Project-URL: Homepage, https://krenzo.in
|
|
6
|
+
Project-URL: Documentation, https://krenzo.in/docs
|
|
7
|
+
License: MIT
|
|
8
|
+
License-File: LICENSE
|
|
9
|
+
Keywords: agents,llm,rag,scraping,search,web-extraction
|
|
10
|
+
Classifier: Development Status :: 4 - Beta
|
|
11
|
+
Classifier: Intended Audience :: Developers
|
|
12
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
13
|
+
Classifier: Programming Language :: Python :: 3
|
|
14
|
+
Classifier: Topic :: Internet :: WWW/HTTP :: Indexing/Search
|
|
15
|
+
Requires-Python: >=3.9
|
|
16
|
+
Description-Content-Type: text/markdown
|
|
17
|
+
|
|
18
|
+
# krenzo
|
|
19
|
+
|
|
20
|
+
Search and extract API for LLM agents. One endpoint for live web search, one
|
|
21
|
+
for turning any URL into clean markdown — returned as token-efficient context
|
|
22
|
+
rather than raw HTML.
|
|
23
|
+
|
|
24
|
+
No runtime dependencies.
|
|
25
|
+
|
|
26
|
+
```bash
|
|
27
|
+
pip install krenzo
|
|
28
|
+
```
|
|
29
|
+
|
|
30
|
+
## Quickstart
|
|
31
|
+
|
|
32
|
+
Get a key at [krenzo.in/dashboard/keys](https://krenzo.in/dashboard/keys), then:
|
|
33
|
+
|
|
34
|
+
```bash
|
|
35
|
+
export KRENZO_API_KEY=krz_live_...
|
|
36
|
+
```
|
|
37
|
+
|
|
38
|
+
```python
|
|
39
|
+
from krenzo import Krenzo
|
|
40
|
+
|
|
41
|
+
client = Krenzo()
|
|
42
|
+
|
|
43
|
+
for hit in client.search("nvidia q3 earnings", max_results=5):
|
|
44
|
+
print(hit.source, "—", hit.title)
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
## Search
|
|
48
|
+
|
|
49
|
+
```python
|
|
50
|
+
results = client.search(
|
|
51
|
+
"western digital analyst price target",
|
|
52
|
+
depth="deep", # also returns page contents for the top results
|
|
53
|
+
max_results=10,
|
|
54
|
+
include_domains=["stockanalysis.com"],
|
|
55
|
+
freshness="week",
|
|
56
|
+
)
|
|
57
|
+
|
|
58
|
+
for r in results:
|
|
59
|
+
print(r.title, r.url)
|
|
60
|
+
if r.content: # populated when depth="deep"
|
|
61
|
+
print(r.content[:500])
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
`depth="deep"` fetches and extracts the top results in the same call, which
|
|
65
|
+
saves a round trip per result when you were going to read them anyway.
|
|
66
|
+
|
|
67
|
+
## Extract
|
|
68
|
+
|
|
69
|
+
```python
|
|
70
|
+
pages = client.extract([
|
|
71
|
+
"https://example.com/a",
|
|
72
|
+
"https://example.com/b",
|
|
73
|
+
])
|
|
74
|
+
|
|
75
|
+
for p in pages:
|
|
76
|
+
if p.ok:
|
|
77
|
+
print(p.title, p.word_count)
|
|
78
|
+
print(p.content)
|
|
79
|
+
else:
|
|
80
|
+
print("failed:", p.url, p.error_type, p.error_message)
|
|
81
|
+
```
|
|
82
|
+
|
|
83
|
+
Each URL carries its own `ok`/`error`, so one dead link does not discard the
|
|
84
|
+
rest of the batch. Pass `format="text"` for plain text instead of markdown, and
|
|
85
|
+
`render=True` for pages that block plain clients or build their content in the
|
|
86
|
+
browser.
|
|
87
|
+
|
|
88
|
+
### When an empty page is not an empty page
|
|
89
|
+
|
|
90
|
+
A site that blocks you and a site with nothing on it look identical downstream
|
|
91
|
+
— and only one of them is a bug in your parser. `extract_one` makes the
|
|
92
|
+
difference explicit:
|
|
93
|
+
|
|
94
|
+
```python
|
|
95
|
+
from krenzo import Krenzo, BlockedError
|
|
96
|
+
|
|
97
|
+
client = Krenzo()
|
|
98
|
+
|
|
99
|
+
try:
|
|
100
|
+
text = client.extract_one("https://www.tipranks.com/stocks/wdc/forecast")
|
|
101
|
+
except BlockedError as e:
|
|
102
|
+
print("needs a browser or was refused:", e)
|
|
103
|
+
```
|
|
104
|
+
|
|
105
|
+
Batch `extract()` reports the same thing without raising: check
|
|
106
|
+
`p.needs_rendering` and `p.error_type`.
|
|
107
|
+
|
|
108
|
+
## Answer
|
|
109
|
+
|
|
110
|
+
A grounded answer in one call, with every quote checked against the page it
|
|
111
|
+
came from.
|
|
112
|
+
|
|
113
|
+
```python
|
|
114
|
+
res = client.answer("What did the RBI decide on the repo rate?", max_sources=4)
|
|
115
|
+
|
|
116
|
+
print(res.answer)
|
|
117
|
+
for c in res.citations:
|
|
118
|
+
print(f" — {c.quote}\n {c.url}")
|
|
119
|
+
```
|
|
120
|
+
|
|
121
|
+
The model proposes quotes; the service verifies each one against the fetched
|
|
122
|
+
source and drops anything it cannot find. A fabricated citation therefore
|
|
123
|
+
cannot reach you. Two fields are worth reading every time:
|
|
124
|
+
|
|
125
|
+
- `res.unverified` — quotes that were dropped. Empty means every claim was
|
|
126
|
+
grounded; non-empty means the answer drifted from its sources.
|
|
127
|
+
- `res.skipped` — pages found but not readable, with a reason. An answer built
|
|
128
|
+
on one of four sources is weaker than one built on four.
|
|
129
|
+
|
|
130
|
+
If the sources do not contain the answer, it says so and returns no citations
|
|
131
|
+
rather than guessing. Pass `render=True` to retry unreadable sources through a
|
|
132
|
+
browser; it is billed at the rendered rate, so it is off by default.
|
|
133
|
+
|
|
134
|
+
## Errors
|
|
135
|
+
|
|
136
|
+
| Exception | When |
|
|
137
|
+
|---|---|
|
|
138
|
+
| `AuthenticationError` | missing, malformed, or revoked key |
|
|
139
|
+
| `InsufficientCredits` | balance cannot cover the call — top up |
|
|
140
|
+
| `RateLimited` | too many requests; carries `retry_after` |
|
|
141
|
+
| `UpstreamError` | search index or origin failed; usually transient |
|
|
142
|
+
| `BlockedError` | page refused, or needs a browser (`extract_one` only) |
|
|
143
|
+
| `KrenzoError` | base class for all of the above |
|
|
144
|
+
|
|
145
|
+
Rate limits and upstream failures are retried automatically with backoff.
|
|
146
|
+
Authentication and credit errors are not — they fail identically on a retry.
|
|
147
|
+
|
|
148
|
+
## Usage and billing
|
|
149
|
+
|
|
150
|
+
Every call records what it cost:
|
|
151
|
+
|
|
152
|
+
```python
|
|
153
|
+
client.search("anything")
|
|
154
|
+
print(client.last_usage.balance_usd, client.last_usage.free_calls_remaining)
|
|
155
|
+
```
|
|
156
|
+
|
|
157
|
+
Calls are billed per call; `extract` is billed per URL, so a 20-URL batch is 20
|
|
158
|
+
calls. Failed requests are not billed, and a call that fails upstream is
|
|
159
|
+
refunded.
|
|
160
|
+
|
|
161
|
+
## Configuration
|
|
162
|
+
|
|
163
|
+
| | |
|
|
164
|
+
|---|---|
|
|
165
|
+
| `KRENZO_API_KEY` | your API key (or pass `api_key=`) |
|
|
166
|
+
| `KRENZO_API_URL` | override the base URL, for testing against a local server |
|
|
167
|
+
|
|
168
|
+
```python
|
|
169
|
+
client = Krenzo(api_key="krz_live_...", timeout=30, max_retries=3)
|
|
170
|
+
```
|
|
171
|
+
|
|
172
|
+
**Never hard-code a key in published source.** A key committed to a repository
|
|
173
|
+
is a key anyone can spend, and keys cannot be un-published. Read it from the
|
|
174
|
+
environment or a secret manager.
|
|
175
|
+
|
|
176
|
+
## License
|
|
177
|
+
|
|
178
|
+
MIT
|
krenzo-0.1.0/README.md
ADDED
|
@@ -0,0 +1,161 @@
|
|
|
1
|
+
# krenzo
|
|
2
|
+
|
|
3
|
+
Search and extract API for LLM agents. One endpoint for live web search, one
|
|
4
|
+
for turning any URL into clean markdown — returned as token-efficient context
|
|
5
|
+
rather than raw HTML.
|
|
6
|
+
|
|
7
|
+
No runtime dependencies.
|
|
8
|
+
|
|
9
|
+
```bash
|
|
10
|
+
pip install krenzo
|
|
11
|
+
```
|
|
12
|
+
|
|
13
|
+
## Quickstart
|
|
14
|
+
|
|
15
|
+
Get a key at [krenzo.in/dashboard/keys](https://krenzo.in/dashboard/keys), then:
|
|
16
|
+
|
|
17
|
+
```bash
|
|
18
|
+
export KRENZO_API_KEY=krz_live_...
|
|
19
|
+
```
|
|
20
|
+
|
|
21
|
+
```python
|
|
22
|
+
from krenzo import Krenzo
|
|
23
|
+
|
|
24
|
+
client = Krenzo()
|
|
25
|
+
|
|
26
|
+
for hit in client.search("nvidia q3 earnings", max_results=5):
|
|
27
|
+
print(hit.source, "—", hit.title)
|
|
28
|
+
```
|
|
29
|
+
|
|
30
|
+
## Search
|
|
31
|
+
|
|
32
|
+
```python
|
|
33
|
+
results = client.search(
|
|
34
|
+
"western digital analyst price target",
|
|
35
|
+
depth="deep", # also returns page contents for the top results
|
|
36
|
+
max_results=10,
|
|
37
|
+
include_domains=["stockanalysis.com"],
|
|
38
|
+
freshness="week",
|
|
39
|
+
)
|
|
40
|
+
|
|
41
|
+
for r in results:
|
|
42
|
+
print(r.title, r.url)
|
|
43
|
+
if r.content: # populated when depth="deep"
|
|
44
|
+
print(r.content[:500])
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
`depth="deep"` fetches and extracts the top results in the same call, which
|
|
48
|
+
saves a round trip per result when you were going to read them anyway.
|
|
49
|
+
|
|
50
|
+
## Extract
|
|
51
|
+
|
|
52
|
+
```python
|
|
53
|
+
pages = client.extract([
|
|
54
|
+
"https://example.com/a",
|
|
55
|
+
"https://example.com/b",
|
|
56
|
+
])
|
|
57
|
+
|
|
58
|
+
for p in pages:
|
|
59
|
+
if p.ok:
|
|
60
|
+
print(p.title, p.word_count)
|
|
61
|
+
print(p.content)
|
|
62
|
+
else:
|
|
63
|
+
print("failed:", p.url, p.error_type, p.error_message)
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
Each URL carries its own `ok`/`error`, so one dead link does not discard the
|
|
67
|
+
rest of the batch. Pass `format="text"` for plain text instead of markdown, and
|
|
68
|
+
`render=True` for pages that block plain clients or build their content in the
|
|
69
|
+
browser.
|
|
70
|
+
|
|
71
|
+
### When an empty page is not an empty page
|
|
72
|
+
|
|
73
|
+
A site that blocks you and a site with nothing on it look identical downstream
|
|
74
|
+
— and only one of them is a bug in your parser. `extract_one` makes the
|
|
75
|
+
difference explicit:
|
|
76
|
+
|
|
77
|
+
```python
|
|
78
|
+
from krenzo import Krenzo, BlockedError
|
|
79
|
+
|
|
80
|
+
client = Krenzo()
|
|
81
|
+
|
|
82
|
+
try:
|
|
83
|
+
text = client.extract_one("https://www.tipranks.com/stocks/wdc/forecast")
|
|
84
|
+
except BlockedError as e:
|
|
85
|
+
print("needs a browser or was refused:", e)
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
Batch `extract()` reports the same thing without raising: check
|
|
89
|
+
`p.needs_rendering` and `p.error_type`.
|
|
90
|
+
|
|
91
|
+
## Answer
|
|
92
|
+
|
|
93
|
+
A grounded answer in one call, with every quote checked against the page it
|
|
94
|
+
came from.
|
|
95
|
+
|
|
96
|
+
```python
|
|
97
|
+
res = client.answer("What did the RBI decide on the repo rate?", max_sources=4)
|
|
98
|
+
|
|
99
|
+
print(res.answer)
|
|
100
|
+
for c in res.citations:
|
|
101
|
+
print(f" — {c.quote}\n {c.url}")
|
|
102
|
+
```
|
|
103
|
+
|
|
104
|
+
The model proposes quotes; the service verifies each one against the fetched
|
|
105
|
+
source and drops anything it cannot find. A fabricated citation therefore
|
|
106
|
+
cannot reach you. Two fields are worth reading every time:
|
|
107
|
+
|
|
108
|
+
- `res.unverified` — quotes that were dropped. Empty means every claim was
|
|
109
|
+
grounded; non-empty means the answer drifted from its sources.
|
|
110
|
+
- `res.skipped` — pages found but not readable, with a reason. An answer built
|
|
111
|
+
on one of four sources is weaker than one built on four.
|
|
112
|
+
|
|
113
|
+
If the sources do not contain the answer, it says so and returns no citations
|
|
114
|
+
rather than guessing. Pass `render=True` to retry unreadable sources through a
|
|
115
|
+
browser; it is billed at the rendered rate, so it is off by default.
|
|
116
|
+
|
|
117
|
+
## Errors
|
|
118
|
+
|
|
119
|
+
| Exception | When |
|
|
120
|
+
|---|---|
|
|
121
|
+
| `AuthenticationError` | missing, malformed, or revoked key |
|
|
122
|
+
| `InsufficientCredits` | balance cannot cover the call — top up |
|
|
123
|
+
| `RateLimited` | too many requests; carries `retry_after` |
|
|
124
|
+
| `UpstreamError` | search index or origin failed; usually transient |
|
|
125
|
+
| `BlockedError` | page refused, or needs a browser (`extract_one` only) |
|
|
126
|
+
| `KrenzoError` | base class for all of the above |
|
|
127
|
+
|
|
128
|
+
Rate limits and upstream failures are retried automatically with backoff.
|
|
129
|
+
Authentication and credit errors are not — they fail identically on a retry.
|
|
130
|
+
|
|
131
|
+
## Usage and billing
|
|
132
|
+
|
|
133
|
+
Every call records what it cost:
|
|
134
|
+
|
|
135
|
+
```python
|
|
136
|
+
client.search("anything")
|
|
137
|
+
print(client.last_usage.balance_usd, client.last_usage.free_calls_remaining)
|
|
138
|
+
```
|
|
139
|
+
|
|
140
|
+
Calls are billed per call; `extract` is billed per URL, so a 20-URL batch is 20
|
|
141
|
+
calls. Failed requests are not billed, and a call that fails upstream is
|
|
142
|
+
refunded.
|
|
143
|
+
|
|
144
|
+
## Configuration
|
|
145
|
+
|
|
146
|
+
| | |
|
|
147
|
+
|---|---|
|
|
148
|
+
| `KRENZO_API_KEY` | your API key (or pass `api_key=`) |
|
|
149
|
+
| `KRENZO_API_URL` | override the base URL, for testing against a local server |
|
|
150
|
+
|
|
151
|
+
```python
|
|
152
|
+
client = Krenzo(api_key="krz_live_...", timeout=30, max_retries=3)
|
|
153
|
+
```
|
|
154
|
+
|
|
155
|
+
**Never hard-code a key in published source.** A key committed to a repository
|
|
156
|
+
is a key anyone can spend, and keys cannot be un-published. Read it from the
|
|
157
|
+
environment or a secret manager.
|
|
158
|
+
|
|
159
|
+
## License
|
|
160
|
+
|
|
161
|
+
MIT
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
"""Krenzo — search and extract API for LLM agents."""
|
|
2
|
+
|
|
3
|
+
from .client import (
|
|
4
|
+
Answer,
|
|
5
|
+
AuthenticationError,
|
|
6
|
+
BlockedError,
|
|
7
|
+
Citation,
|
|
8
|
+
ExtractResult,
|
|
9
|
+
InsufficientCredits,
|
|
10
|
+
Krenzo,
|
|
11
|
+
KrenzoError,
|
|
12
|
+
RateLimited,
|
|
13
|
+
SearchResult,
|
|
14
|
+
UpstreamError,
|
|
15
|
+
Usage,
|
|
16
|
+
)
|
|
17
|
+
|
|
18
|
+
__version__ = "0.1.0"
|
|
19
|
+
|
|
20
|
+
__all__ = [
|
|
21
|
+
"Krenzo",
|
|
22
|
+
"Answer",
|
|
23
|
+
"Citation",
|
|
24
|
+
"KrenzoError",
|
|
25
|
+
"AuthenticationError",
|
|
26
|
+
"InsufficientCredits",
|
|
27
|
+
"RateLimited",
|
|
28
|
+
"UpstreamError",
|
|
29
|
+
"BlockedError",
|
|
30
|
+
"SearchResult",
|
|
31
|
+
"ExtractResult",
|
|
32
|
+
"Usage",
|
|
33
|
+
"__version__",
|
|
34
|
+
]
|
|
@@ -0,0 +1,467 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Krenzo API client.
|
|
3
|
+
|
|
4
|
+
Stdlib only, on purpose. An SDK that pulls in `requests` forces its version on
|
|
5
|
+
every project that installs it, and this one does nothing `urllib` can't.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import json
|
|
11
|
+
import os
|
|
12
|
+
import time
|
|
13
|
+
import urllib.error
|
|
14
|
+
import urllib.parse
|
|
15
|
+
import urllib.request
|
|
16
|
+
from dataclasses import dataclass, field
|
|
17
|
+
from typing import Any, Iterable, Literal
|
|
18
|
+
|
|
19
|
+
__all__ = [
|
|
20
|
+
"Krenzo",
|
|
21
|
+
"Answer",
|
|
22
|
+
"Citation",
|
|
23
|
+
"KrenzoError",
|
|
24
|
+
"AuthenticationError",
|
|
25
|
+
"InsufficientCredits",
|
|
26
|
+
"RateLimited",
|
|
27
|
+
"UpstreamError",
|
|
28
|
+
"BlockedError",
|
|
29
|
+
"SearchResult",
|
|
30
|
+
"ExtractResult",
|
|
31
|
+
"Usage",
|
|
32
|
+
]
|
|
33
|
+
|
|
34
|
+
DEFAULT_BASE_URL = "https://krenzo.in/api/v1"
|
|
35
|
+
DEFAULT_TIMEOUT = 60.0
|
|
36
|
+
USER_AGENT = "krenzo-python/0.1.0"
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
# --------------------------------------------------------------- exceptions
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
class KrenzoError(Exception):
|
|
43
|
+
"""Base for every error this client raises."""
|
|
44
|
+
|
|
45
|
+
def __init__(self, message: str, *, status: int | None = None, type: str | None = None):
|
|
46
|
+
super().__init__(message)
|
|
47
|
+
self.message = message
|
|
48
|
+
self.status = status
|
|
49
|
+
self.type = type
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
class AuthenticationError(KrenzoError):
|
|
53
|
+
"""Missing, malformed, or revoked API key."""
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
class InsufficientCredits(KrenzoError):
|
|
57
|
+
"""
|
|
58
|
+
The account balance cannot cover the call.
|
|
59
|
+
|
|
60
|
+
Separate from other 4xx because it is the one error a caller can usually
|
|
61
|
+
resolve without a code change — by topping up.
|
|
62
|
+
"""
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
class RateLimited(KrenzoError):
|
|
66
|
+
"""Too many requests. `retry_after` is seconds, when the server says."""
|
|
67
|
+
|
|
68
|
+
def __init__(self, message: str, *, retry_after: float | None = None, **kw: Any):
|
|
69
|
+
super().__init__(message, **kw)
|
|
70
|
+
self.retry_after = retry_after
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
class UpstreamError(KrenzoError):
|
|
74
|
+
"""The search index or an origin failed. Usually transient."""
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
class BlockedError(KrenzoError):
|
|
78
|
+
"""
|
|
79
|
+
A page was refused, or came back needing a browser to render.
|
|
80
|
+
|
|
81
|
+
Raised only by `extract_one`. It exists so that "the origin blocked us" is
|
|
82
|
+
never silently indistinguishable from "the page had nothing on it" — the
|
|
83
|
+
two call for completely different responses, and only one is a bug.
|
|
84
|
+
"""
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
# ------------------------------------------------------------------ models
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
@dataclass(frozen=True)
|
|
91
|
+
class Usage:
|
|
92
|
+
charged_paise: int
|
|
93
|
+
free_call: bool
|
|
94
|
+
free_calls_remaining: int
|
|
95
|
+
balance_paise: int
|
|
96
|
+
balance_inr: str
|
|
97
|
+
balance_usd: str
|
|
98
|
+
#: Calls this balance can still pay for, at the current rate.
|
|
99
|
+
calls_remaining: int = 0
|
|
100
|
+
#: True once the balance is nearly spent — act before calls start failing.
|
|
101
|
+
low_balance: bool = False
|
|
102
|
+
|
|
103
|
+
@classmethod
|
|
104
|
+
def _from(cls, d: dict[str, Any]) -> "Usage":
|
|
105
|
+
return cls(
|
|
106
|
+
charged_paise=int(d.get("charged_paise", 0)),
|
|
107
|
+
free_call=bool(d.get("free_call", False)),
|
|
108
|
+
free_calls_remaining=int(d.get("free_calls_remaining", 0)),
|
|
109
|
+
balance_paise=int(d.get("balance_paise", 0)),
|
|
110
|
+
balance_inr=str(d.get("balance_inr", "")),
|
|
111
|
+
balance_usd=str(d.get("balance_usd", "")),
|
|
112
|
+
calls_remaining=int(d.get("calls_remaining", 0)),
|
|
113
|
+
low_balance=bool(d.get("low_balance", False)),
|
|
114
|
+
)
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
@dataclass(frozen=True)
|
|
118
|
+
class SearchResult:
|
|
119
|
+
title: str
|
|
120
|
+
url: str
|
|
121
|
+
snippet: str
|
|
122
|
+
source: str
|
|
123
|
+
published: str | None = None
|
|
124
|
+
#: Present only for depth="deep".
|
|
125
|
+
content: str | None = None
|
|
126
|
+
word_count: int | None = None
|
|
127
|
+
needs_rendering: bool | None = None
|
|
128
|
+
content_error: str | None = None
|
|
129
|
+
|
|
130
|
+
@classmethod
|
|
131
|
+
def _from(cls, d: dict[str, Any]) -> "SearchResult":
|
|
132
|
+
return cls(
|
|
133
|
+
title=d.get("title", ""),
|
|
134
|
+
url=d.get("url", ""),
|
|
135
|
+
snippet=d.get("snippet", ""),
|
|
136
|
+
source=d.get("source", ""),
|
|
137
|
+
published=d.get("published"),
|
|
138
|
+
content=d.get("content"),
|
|
139
|
+
word_count=d.get("word_count"),
|
|
140
|
+
needs_rendering=d.get("needs_rendering"),
|
|
141
|
+
content_error=d.get("content_error"),
|
|
142
|
+
)
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
@dataclass(frozen=True)
|
|
146
|
+
class Citation:
|
|
147
|
+
"""A quote that was verified against the source it is attributed to."""
|
|
148
|
+
|
|
149
|
+
#: The supporting text, exactly as it appears in the source — not as the
|
|
150
|
+
#: model wrote it.
|
|
151
|
+
quote: str
|
|
152
|
+
url: str
|
|
153
|
+
title: str | None = None
|
|
154
|
+
#: Character offsets into the fetched document.
|
|
155
|
+
start: int | None = None
|
|
156
|
+
end: int | None = None
|
|
157
|
+
|
|
158
|
+
@classmethod
|
|
159
|
+
def _from(cls, d: dict[str, Any]) -> "Citation":
|
|
160
|
+
return cls(
|
|
161
|
+
quote=d.get("quote", ""),
|
|
162
|
+
url=d.get("url", ""),
|
|
163
|
+
title=d.get("title"),
|
|
164
|
+
start=d.get("start"),
|
|
165
|
+
end=d.get("end"),
|
|
166
|
+
)
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
@dataclass(frozen=True)
|
|
170
|
+
class Answer:
|
|
171
|
+
answer: str
|
|
172
|
+
citations: list[Citation] = field(default_factory=list)
|
|
173
|
+
sources: list[dict[str, Any]] = field(default_factory=list)
|
|
174
|
+
#: Pages found but not readable, with a reason. An answer built on one of
|
|
175
|
+
#: four sources is weaker than one built on four, and this is how you tell.
|
|
176
|
+
skipped: list[dict[str, Any]] = field(default_factory=list)
|
|
177
|
+
#: Quotes the model claimed that could NOT be found in any source. Empty
|
|
178
|
+
#: means every claim was grounded; non-empty means the answer drifted.
|
|
179
|
+
unverified: list[dict[str, Any]] = field(default_factory=list)
|
|
180
|
+
rendered: int = 0
|
|
181
|
+
model: str = ""
|
|
182
|
+
|
|
183
|
+
@classmethod
|
|
184
|
+
def _from(cls, d: dict[str, Any]) -> "Answer":
|
|
185
|
+
return cls(
|
|
186
|
+
answer=d.get("answer", ""),
|
|
187
|
+
citations=[Citation._from(c) for c in d.get("citations", [])],
|
|
188
|
+
sources=d.get("sources") or [],
|
|
189
|
+
skipped=d.get("skipped") or [],
|
|
190
|
+
unverified=d.get("unverified") or [],
|
|
191
|
+
rendered=int(d.get("rendered", 0)),
|
|
192
|
+
model=d.get("model", ""),
|
|
193
|
+
)
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
@dataclass(frozen=True)
|
|
197
|
+
class ExtractResult:
|
|
198
|
+
url: str
|
|
199
|
+
ok: bool
|
|
200
|
+
title: str | None = None
|
|
201
|
+
description: str | None = None
|
|
202
|
+
content: str | None = None
|
|
203
|
+
word_count: int | None = None
|
|
204
|
+
#: True when the page was fetched through a real browser.
|
|
205
|
+
rendered: bool = False
|
|
206
|
+
#: True when the page returned almost no text for its size — the signature
|
|
207
|
+
#: of a client-rendered app. Treating this as an empty page is a bug.
|
|
208
|
+
needs_rendering: bool = False
|
|
209
|
+
status: int | None = None
|
|
210
|
+
error_type: str | None = None
|
|
211
|
+
error_message: str | None = None
|
|
212
|
+
links: list[dict[str, str]] = field(default_factory=list)
|
|
213
|
+
|
|
214
|
+
@classmethod
|
|
215
|
+
def _from(cls, d: dict[str, Any]) -> "ExtractResult":
|
|
216
|
+
err = d.get("error") or {}
|
|
217
|
+
return cls(
|
|
218
|
+
url=d.get("url", ""),
|
|
219
|
+
ok=bool(d.get("ok")),
|
|
220
|
+
title=d.get("title"),
|
|
221
|
+
description=d.get("description"),
|
|
222
|
+
content=d.get("content"),
|
|
223
|
+
word_count=d.get("word_count"),
|
|
224
|
+
rendered=bool(d.get("rendered", False)),
|
|
225
|
+
needs_rendering=bool(d.get("needs_rendering", False)),
|
|
226
|
+
status=d.get("status"),
|
|
227
|
+
error_type=err.get("type"),
|
|
228
|
+
error_message=err.get("message"),
|
|
229
|
+
links=d.get("links") or [],
|
|
230
|
+
)
|
|
231
|
+
|
|
232
|
+
|
|
233
|
+
# ------------------------------------------------------------------ client
|
|
234
|
+
|
|
235
|
+
|
|
236
|
+
class Krenzo:
|
|
237
|
+
"""
|
|
238
|
+
Client for the Krenzo search and extract API.
|
|
239
|
+
|
|
240
|
+
from krenzo import Krenzo
|
|
241
|
+
|
|
242
|
+
client = Krenzo() # reads KRENZO_API_KEY
|
|
243
|
+
for hit in client.search("nvidia earnings date"):
|
|
244
|
+
print(hit.source, hit.title)
|
|
245
|
+
|
|
246
|
+
The key is read from the environment by default. Do not hard-code one in
|
|
247
|
+
source you publish — a key in a repository is a key anyone can spend.
|
|
248
|
+
"""
|
|
249
|
+
|
|
250
|
+
def __init__(
|
|
251
|
+
self,
|
|
252
|
+
api_key: str | None = None,
|
|
253
|
+
*,
|
|
254
|
+
base_url: str | None = None,
|
|
255
|
+
timeout: float = DEFAULT_TIMEOUT,
|
|
256
|
+
max_retries: int = 2,
|
|
257
|
+
):
|
|
258
|
+
self.api_key = api_key or os.environ.get("KRENZO_API_KEY", "").strip()
|
|
259
|
+
if not self.api_key:
|
|
260
|
+
raise AuthenticationError(
|
|
261
|
+
"No API key. Pass api_key= or set KRENZO_API_KEY. "
|
|
262
|
+
"Create one at https://krenzo.in/dashboard/keys"
|
|
263
|
+
)
|
|
264
|
+
self.base_url = (base_url or os.environ.get("KRENZO_API_URL") or DEFAULT_BASE_URL).rstrip("/")
|
|
265
|
+
self.timeout = timeout
|
|
266
|
+
self.max_retries = max_retries
|
|
267
|
+
|
|
268
|
+
# -- transport ---------------------------------------------------------
|
|
269
|
+
|
|
270
|
+
def _post(self, path: str, payload: dict[str, Any]) -> dict[str, Any]:
|
|
271
|
+
body = json.dumps(payload).encode()
|
|
272
|
+
last: Exception | None = None
|
|
273
|
+
|
|
274
|
+
for attempt in range(self.max_retries + 1):
|
|
275
|
+
req = urllib.request.Request(
|
|
276
|
+
f"{self.base_url}{path}",
|
|
277
|
+
data=body,
|
|
278
|
+
headers={
|
|
279
|
+
"Authorization": f"Bearer {self.api_key}",
|
|
280
|
+
"Content-Type": "application/json",
|
|
281
|
+
"User-Agent": USER_AGENT,
|
|
282
|
+
},
|
|
283
|
+
method="POST",
|
|
284
|
+
)
|
|
285
|
+
try:
|
|
286
|
+
with urllib.request.urlopen(req, timeout=self.timeout) as resp:
|
|
287
|
+
return json.loads(resp.read().decode())
|
|
288
|
+
except urllib.error.HTTPError as exc:
|
|
289
|
+
err = self._error_from(exc)
|
|
290
|
+
# Only retry what retrying can fix. A 402 or 401 will fail
|
|
291
|
+
# identically forever, and retrying a 429 without honouring
|
|
292
|
+
# Retry-After just deepens the hole.
|
|
293
|
+
if isinstance(err, (RateLimited, UpstreamError)) and attempt < self.max_retries:
|
|
294
|
+
delay = getattr(err, "retry_after", None) or (2 ** attempt)
|
|
295
|
+
time.sleep(min(delay, 30))
|
|
296
|
+
last = err
|
|
297
|
+
continue
|
|
298
|
+
raise err from exc
|
|
299
|
+
except urllib.error.URLError as exc:
|
|
300
|
+
last = KrenzoError(f"Could not reach {self.base_url}: {exc.reason}")
|
|
301
|
+
if attempt < self.max_retries:
|
|
302
|
+
time.sleep(2 ** attempt)
|
|
303
|
+
continue
|
|
304
|
+
raise last from exc
|
|
305
|
+
|
|
306
|
+
raise last or KrenzoError("Request failed.")
|
|
307
|
+
|
|
308
|
+
@staticmethod
|
|
309
|
+
def _error_from(exc: urllib.error.HTTPError) -> KrenzoError:
|
|
310
|
+
raw = exc.read().decode(errors="replace")
|
|
311
|
+
message, kind = raw, None
|
|
312
|
+
try:
|
|
313
|
+
err = json.loads(raw).get("error") or {}
|
|
314
|
+
message = err.get("message") or raw
|
|
315
|
+
kind = err.get("type")
|
|
316
|
+
except (ValueError, AttributeError):
|
|
317
|
+
pass
|
|
318
|
+
|
|
319
|
+
if exc.code == 401:
|
|
320
|
+
return AuthenticationError(message, status=401, type=kind)
|
|
321
|
+
if exc.code == 402:
|
|
322
|
+
return InsufficientCredits(message, status=402, type=kind)
|
|
323
|
+
if exc.code == 429:
|
|
324
|
+
retry = exc.headers.get("Retry-After") if exc.headers else None
|
|
325
|
+
return RateLimited(
|
|
326
|
+
message,
|
|
327
|
+
retry_after=float(retry) if retry and retry.isdigit() else None,
|
|
328
|
+
status=429,
|
|
329
|
+
type=kind,
|
|
330
|
+
)
|
|
331
|
+
if exc.code in (502, 503, 504):
|
|
332
|
+
return UpstreamError(message, status=exc.code, type=kind)
|
|
333
|
+
return KrenzoError(message, status=exc.code, type=kind)
|
|
334
|
+
|
|
335
|
+
# -- endpoints ---------------------------------------------------------
|
|
336
|
+
|
|
337
|
+
def search(
|
|
338
|
+
self,
|
|
339
|
+
query: str,
|
|
340
|
+
*,
|
|
341
|
+
depth: Literal["quick", "deep"] = "quick",
|
|
342
|
+
max_results: int = 10,
|
|
343
|
+
include_domains: Iterable[str] | None = None,
|
|
344
|
+
exclude_domains: Iterable[str] | None = None,
|
|
345
|
+
freshness: Literal["day", "week", "month", "year"] | None = None,
|
|
346
|
+
) -> list[SearchResult]:
|
|
347
|
+
"""
|
|
348
|
+
Web search.
|
|
349
|
+
|
|
350
|
+
`depth="deep"` also returns the extracted contents of the top results,
|
|
351
|
+
which saves a round trip per result when you were going to fetch them
|
|
352
|
+
anyway.
|
|
353
|
+
"""
|
|
354
|
+
payload: dict[str, Any] = {
|
|
355
|
+
"query": query,
|
|
356
|
+
"depth": depth,
|
|
357
|
+
"max_results": max_results,
|
|
358
|
+
}
|
|
359
|
+
if include_domains:
|
|
360
|
+
payload["include_domains"] = list(include_domains)
|
|
361
|
+
if exclude_domains:
|
|
362
|
+
payload["exclude_domains"] = list(exclude_domains)
|
|
363
|
+
if freshness:
|
|
364
|
+
payload["freshness"] = freshness
|
|
365
|
+
|
|
366
|
+
body = self._post("/search", payload)
|
|
367
|
+
self.last_usage = Usage._from(body.get("usage") or {})
|
|
368
|
+
return [SearchResult._from(r) for r in body.get("results", [])]
|
|
369
|
+
|
|
370
|
+
def extract(
|
|
371
|
+
self,
|
|
372
|
+
urls: str | Iterable[str],
|
|
373
|
+
*,
|
|
374
|
+
format: Literal["markdown", "text"] = "markdown",
|
|
375
|
+
render: bool = False,
|
|
376
|
+
respect_robots: bool = True,
|
|
377
|
+
include_links: bool = False,
|
|
378
|
+
) -> list[ExtractResult]:
|
|
379
|
+
"""
|
|
380
|
+
Fetch and clean one or more URLs.
|
|
381
|
+
|
|
382
|
+
Returns one result per URL, each with its own ok/error — a dead link
|
|
383
|
+
among ten does not discard the nine that worked.
|
|
384
|
+
"""
|
|
385
|
+
if isinstance(urls, str):
|
|
386
|
+
urls = [urls]
|
|
387
|
+
body = self._post(
|
|
388
|
+
"/extract",
|
|
389
|
+
{
|
|
390
|
+
"urls": list(urls),
|
|
391
|
+
"format": format,
|
|
392
|
+
"render": render,
|
|
393
|
+
"respect_robots": respect_robots,
|
|
394
|
+
"include_links": include_links,
|
|
395
|
+
},
|
|
396
|
+
)
|
|
397
|
+
self.last_usage = Usage._from(body.get("usage") or {})
|
|
398
|
+
return [ExtractResult._from(r) for r in body.get("results", [])]
|
|
399
|
+
|
|
400
|
+
def answer(
|
|
401
|
+
self,
|
|
402
|
+
query: str,
|
|
403
|
+
*,
|
|
404
|
+
max_sources: int = 4,
|
|
405
|
+
urls: Iterable[str] | None = None,
|
|
406
|
+
render: bool = False,
|
|
407
|
+
include_domains: Iterable[str] | None = None,
|
|
408
|
+
exclude_domains: Iterable[str] | None = None,
|
|
409
|
+
freshness: Literal["day", "week", "month", "year"] | None = None,
|
|
410
|
+
) -> Answer:
|
|
411
|
+
"""
|
|
412
|
+
A grounded answer, with every quote verified against its source.
|
|
413
|
+
|
|
414
|
+
The model proposes quotes; the service checks each one against the page
|
|
415
|
+
it was fetched from and drops anything not found. A fabricated citation
|
|
416
|
+
therefore cannot reach you — but read `unverified`, which lists what was
|
|
417
|
+
dropped, and `skipped`, which lists pages that could not be read.
|
|
418
|
+
|
|
419
|
+
`render=True` retries unreadable sources through a browser; it is billed
|
|
420
|
+
at the rendered rate, so it is off by default.
|
|
421
|
+
"""
|
|
422
|
+
payload: dict[str, Any] = {
|
|
423
|
+
"query": query,
|
|
424
|
+
"max_sources": max_sources,
|
|
425
|
+
"render": render,
|
|
426
|
+
}
|
|
427
|
+
if urls:
|
|
428
|
+
payload["urls"] = list(urls)
|
|
429
|
+
if include_domains:
|
|
430
|
+
payload["include_domains"] = list(include_domains)
|
|
431
|
+
if exclude_domains:
|
|
432
|
+
payload["exclude_domains"] = list(exclude_domains)
|
|
433
|
+
if freshness:
|
|
434
|
+
payload["freshness"] = freshness
|
|
435
|
+
|
|
436
|
+
body = self._post("/answer", payload)
|
|
437
|
+
self.last_usage = Usage._from(body.get("usage") or {})
|
|
438
|
+
return Answer._from(body)
|
|
439
|
+
|
|
440
|
+
def extract_one(self, url: str, **kw: Any) -> str:
|
|
441
|
+
"""
|
|
442
|
+
Content for a single URL, or an exception.
|
|
443
|
+
|
|
444
|
+
Raises `BlockedError` when the origin refused the fetch or returned a
|
|
445
|
+
page that needs a browser. Use this when an empty result would be
|
|
446
|
+
indistinguishable from a real absence and that distinction matters.
|
|
447
|
+
"""
|
|
448
|
+
results = self.extract(url, **kw)
|
|
449
|
+
if not results:
|
|
450
|
+
raise KrenzoError(f"No result returned for {url}")
|
|
451
|
+
|
|
452
|
+
r = results[0]
|
|
453
|
+
if not r.ok:
|
|
454
|
+
message = f"{r.error_type} for {url}: {r.error_message}"
|
|
455
|
+
if r.error_type in ("http_error", "robots_disallowed", "timeout"):
|
|
456
|
+
raise BlockedError(message, type=r.error_type, status=r.status)
|
|
457
|
+
raise KrenzoError(message, type=r.error_type, status=r.status)
|
|
458
|
+
|
|
459
|
+
if r.needs_rendering:
|
|
460
|
+
hint = "" if kw.get("render") else " Retry with render=True."
|
|
461
|
+
raise BlockedError(
|
|
462
|
+
f"{url} returned {r.word_count} words — the page renders its "
|
|
463
|
+
f"content with JavaScript.{hint}",
|
|
464
|
+
type="needs_rendering",
|
|
465
|
+
)
|
|
466
|
+
|
|
467
|
+
return r.content or ""
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["hatchling"]
|
|
3
|
+
build-backend = "hatchling.build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "krenzo"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Search and extract API for LLM agents — clean, token-efficient web context."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.9"
|
|
11
|
+
license = { text = "MIT" }
|
|
12
|
+
keywords = ["search", "scraping", "llm", "agents", "rag", "web-extraction"]
|
|
13
|
+
classifiers = [
|
|
14
|
+
"Development Status :: 4 - Beta",
|
|
15
|
+
"Intended Audience :: Developers",
|
|
16
|
+
"License :: OSI Approved :: MIT License",
|
|
17
|
+
"Programming Language :: Python :: 3",
|
|
18
|
+
"Topic :: Internet :: WWW/HTTP :: Indexing/Search",
|
|
19
|
+
]
|
|
20
|
+
# No runtime dependencies on purpose: an SDK that pulls in `requests` forces
|
|
21
|
+
# its version onto every project that installs it, and urllib is sufficient.
|
|
22
|
+
dependencies = []
|
|
23
|
+
|
|
24
|
+
[project.urls]
|
|
25
|
+
Homepage = "https://krenzo.in"
|
|
26
|
+
Documentation = "https://krenzo.in/docs"
|
|
27
|
+
|
|
28
|
+
[tool.hatch.build.targets.wheel]
|
|
29
|
+
packages = ["krenzo"]
|
|
@@ -0,0 +1,143 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Offline tests. No network, no API key, no credits spent.
|
|
3
|
+
|
|
4
|
+
Everything here exercises the parts that are easy to get wrong without
|
|
5
|
+
noticing: error mapping by status code, and the parsing that turns a response
|
|
6
|
+
into the objects callers actually touch.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
import json
|
|
10
|
+
import sys
|
|
11
|
+
import unittest
|
|
12
|
+
import urllib.error
|
|
13
|
+
from io import BytesIO
|
|
14
|
+
from pathlib import Path
|
|
15
|
+
|
|
16
|
+
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
|
|
17
|
+
|
|
18
|
+
from krenzo import ( # noqa: E402
|
|
19
|
+
AuthenticationError,
|
|
20
|
+
BlockedError,
|
|
21
|
+
ExtractResult,
|
|
22
|
+
InsufficientCredits,
|
|
23
|
+
Krenzo,
|
|
24
|
+
KrenzoError,
|
|
25
|
+
RateLimited,
|
|
26
|
+
SearchResult,
|
|
27
|
+
UpstreamError,
|
|
28
|
+
)
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def http_error(code: int, payload: dict, headers: dict | None = None):
|
|
32
|
+
body = BytesIO(json.dumps(payload).encode())
|
|
33
|
+
return urllib.error.HTTPError(
|
|
34
|
+
"https://krenzo.in/api/v1/search", code, "err", headers or {}, body
|
|
35
|
+
)
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class ErrorMapping(unittest.TestCase):
|
|
39
|
+
def map(self, code, type_="x", headers=None):
|
|
40
|
+
exc = http_error(code, {"error": {"type": type_, "message": "boom"}}, headers)
|
|
41
|
+
return Krenzo._error_from(exc)
|
|
42
|
+
|
|
43
|
+
def test_401_is_authentication(self):
|
|
44
|
+
self.assertIsInstance(self.map(401), AuthenticationError)
|
|
45
|
+
|
|
46
|
+
def test_402_is_insufficient_credits(self):
|
|
47
|
+
err = self.map(402, "insufficient_credits")
|
|
48
|
+
self.assertIsInstance(err, InsufficientCredits)
|
|
49
|
+
self.assertEqual(err.status, 402)
|
|
50
|
+
|
|
51
|
+
def test_429_carries_retry_after(self):
|
|
52
|
+
err = self.map(429, headers={"Retry-After": "7"})
|
|
53
|
+
self.assertIsInstance(err, RateLimited)
|
|
54
|
+
self.assertEqual(err.retry_after, 7.0)
|
|
55
|
+
|
|
56
|
+
def test_5xx_is_upstream(self):
|
|
57
|
+
for code in (502, 503, 504):
|
|
58
|
+
self.assertIsInstance(self.map(code), UpstreamError)
|
|
59
|
+
|
|
60
|
+
def test_unknown_code_falls_back_to_base(self):
|
|
61
|
+
err = self.map(418)
|
|
62
|
+
self.assertIsInstance(err, KrenzoError)
|
|
63
|
+
self.assertNotIsInstance(err, UpstreamError)
|
|
64
|
+
|
|
65
|
+
def test_non_json_body_still_produces_a_message(self):
|
|
66
|
+
exc = urllib.error.HTTPError("u", 500, "e", {}, BytesIO(b"<html>502</html>"))
|
|
67
|
+
self.assertIn("502", Krenzo._error_from(exc).message)
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
class Parsing(unittest.TestCase):
|
|
71
|
+
def test_search_result_tolerates_missing_fields(self):
|
|
72
|
+
r = SearchResult._from({"url": "https://a.test"})
|
|
73
|
+
self.assertEqual(r.url, "https://a.test")
|
|
74
|
+
self.assertEqual(r.title, "")
|
|
75
|
+
self.assertIsNone(r.content)
|
|
76
|
+
|
|
77
|
+
def test_extract_result_lifts_nested_error(self):
|
|
78
|
+
r = ExtractResult._from(
|
|
79
|
+
{"url": "u", "ok": False, "error": {"type": "http_error", "message": "403"}}
|
|
80
|
+
)
|
|
81
|
+
self.assertFalse(r.ok)
|
|
82
|
+
self.assertEqual(r.error_type, "http_error")
|
|
83
|
+
self.assertEqual(r.error_message, "403")
|
|
84
|
+
|
|
85
|
+
def test_extract_result_defaults_are_safe(self):
|
|
86
|
+
r = ExtractResult._from({"url": "u", "ok": True})
|
|
87
|
+
self.assertFalse(r.needs_rendering)
|
|
88
|
+
self.assertFalse(r.rendered)
|
|
89
|
+
self.assertEqual(r.links, [])
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
class ExtractOneContract(unittest.TestCase):
|
|
93
|
+
"""The distinction this SDK exists to preserve: blocked != empty."""
|
|
94
|
+
|
|
95
|
+
def client(self, results):
|
|
96
|
+
c = Krenzo.__new__(Krenzo) # bypass __init__'s key requirement
|
|
97
|
+
c.extract = lambda *a, **k: [ExtractResult._from(r) for r in results]
|
|
98
|
+
return c
|
|
99
|
+
|
|
100
|
+
def test_returns_content_when_ok(self):
|
|
101
|
+
c = self.client([{"url": "u", "ok": True, "content": "hello"}])
|
|
102
|
+
self.assertEqual(c.extract_one("u"), "hello")
|
|
103
|
+
|
|
104
|
+
def test_raises_blocked_on_http_error(self):
|
|
105
|
+
c = self.client(
|
|
106
|
+
[{"url": "u", "ok": False, "error": {"type": "http_error", "message": "403"}}]
|
|
107
|
+
)
|
|
108
|
+
with self.assertRaises(BlockedError):
|
|
109
|
+
c.extract_one("u")
|
|
110
|
+
|
|
111
|
+
def test_raises_blocked_on_needs_rendering(self):
|
|
112
|
+
# The important one: a 200 with almost no text must not look like a
|
|
113
|
+
# page that genuinely had nothing on it.
|
|
114
|
+
c = self.client(
|
|
115
|
+
[{"url": "u", "ok": True, "content": "", "word_count": 12,
|
|
116
|
+
"needs_rendering": True}]
|
|
117
|
+
)
|
|
118
|
+
with self.assertRaises(BlockedError) as ctx:
|
|
119
|
+
c.extract_one("u")
|
|
120
|
+
self.assertIn("render=True", str(ctx.exception))
|
|
121
|
+
|
|
122
|
+
def test_non_blocking_failure_is_plain_error(self):
|
|
123
|
+
c = self.client(
|
|
124
|
+
[{"url": "u", "ok": False, "error": {"type": "invalid_url", "message": "bad"}}]
|
|
125
|
+
)
|
|
126
|
+
with self.assertRaises(KrenzoError) as ctx:
|
|
127
|
+
c.extract_one("u")
|
|
128
|
+
self.assertNotIsInstance(ctx.exception, BlockedError)
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
class Construction(unittest.TestCase):
|
|
132
|
+
def test_missing_key_raises_with_a_useful_message(self):
|
|
133
|
+
with self.assertRaises(AuthenticationError) as ctx:
|
|
134
|
+
Krenzo(api_key="")
|
|
135
|
+
self.assertIn("KRENZO_API_KEY", str(ctx.exception))
|
|
136
|
+
|
|
137
|
+
def test_base_url_trailing_slash_is_normalised(self):
|
|
138
|
+
c = Krenzo(api_key="k", base_url="https://example.test/api/v1/")
|
|
139
|
+
self.assertEqual(c.base_url, "https://example.test/api/v1")
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
if __name__ == "__main__":
|
|
143
|
+
unittest.main(verbosity=2)
|