jevrake 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- jevrake-0.1.0/LICENSE +21 -0
- jevrake-0.1.0/PKG-INFO +93 -0
- jevrake-0.1.0/README.md +67 -0
- jevrake-0.1.0/pyproject.toml +79 -0
- jevrake-0.1.0/pyproject.toml.orig +49 -0
- jevrake-0.1.0/src/jevrake/__init__.py +13 -0
- jevrake-0.1.0/src/jevrake/_internal/__init__.py +1 -0
- jevrake-0.1.0/src/jevrake/_internal/api.py +138 -0
- jevrake-0.1.0/src/jevrake/_internal/client.py +81 -0
- jevrake-0.1.0/src/jevrake/_internal/collect.py +210 -0
- jevrake-0.1.0/src/jevrake/_internal/convert.py +78 -0
- jevrake-0.1.0/src/jevrake/_internal/dom.py +200 -0
- jevrake-0.1.0/src/jevrake/_internal/exceptions.py +25 -0
- jevrake-0.1.0/src/jevrake/_internal/fetch.py +75 -0
- jevrake-0.1.0/src/jevrake/_internal/jev.py +181 -0
- jevrake-0.1.0/src/jevrake/_internal/models.py +131 -0
- jevrake-0.1.0/src/jevrake/_internal/schema.py +105 -0
- jevrake-0.1.0/src/jevrake/_internal/version.py +2 -0
- jevrake-0.1.0/src/jevrake/cli.py +127 -0
- jevrake-0.1.0/src/jevrake/py.typed +0 -0
jevrake-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 RKasai127
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
jevrake-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,93 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: jevrake
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Selector-free web scraping powered by TypeSafe AI's Jev
|
|
5
|
+
Keywords: scraping,web-scraping,llm,pydantic,typesafe,jev
|
|
6
|
+
Author: RKasai127
|
|
7
|
+
License-Expression: MIT
|
|
8
|
+
License-File: LICENSE
|
|
9
|
+
Classifier: Development Status :: 3 - Alpha
|
|
10
|
+
Classifier: Environment :: Console
|
|
11
|
+
Classifier: Intended Audience :: Developers
|
|
12
|
+
Classifier: Operating System :: OS Independent
|
|
13
|
+
Classifier: Programming Language :: Python :: 3
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
15
|
+
Classifier: Topic :: Internet :: WWW/HTTP :: Indexing/Search
|
|
16
|
+
Classifier: Typing :: Typed
|
|
17
|
+
Requires-Dist: pydantic>=2
|
|
18
|
+
Requires-Dist: httpx
|
|
19
|
+
Requires-Dist: selectolax
|
|
20
|
+
Requires-Dist: typesafe-sdk
|
|
21
|
+
Requires-Python: >=3.12
|
|
22
|
+
Project-URL: Homepage, https://github.com/RKasai127/jevrake
|
|
23
|
+
Project-URL: Repository, https://github.com/RKasai127/jevrake
|
|
24
|
+
Project-URL: Issues, https://github.com/RKasai127/jevrake/issues
|
|
25
|
+
Description-Content-Type: text/markdown
|
|
26
|
+
|
|
27
|
+
<p align="right">English | <a href="https://github.com/RKasai127/jevrake/blob/main/README.ja.md">日本語</a></p>
|
|
28
|
+
|
|
29
|
+
# jevrake — a scraping tool powered by Jev
|
|
30
|
+
|
|
31
|
+
Example on [books.toscrape.com](https://books.toscrape.com/), a site made for scraping practice:
|
|
32
|
+
|
|
33
|
+
```sh
|
|
34
|
+
jevrake https://books.toscrape.com/ --fields '{"title": "Book title", "price": {"description": "Price", "type": "decimal"}, "availability": "Availability"}'
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
Output (1 of the 20 books on the page):
|
|
38
|
+
|
|
39
|
+
```json
|
|
40
|
+
{
|
|
41
|
+
"url": "https://books.toscrape.com/",
|
|
42
|
+
"data": {
|
|
43
|
+
"title": "A Light in the ...",
|
|
44
|
+
"price": "51.77",
|
|
45
|
+
"availability": "In stock"
|
|
46
|
+
},
|
|
47
|
+
"fields": {
|
|
48
|
+
"title": {"value": "A Light in the ...", "confidence": 1.0, "probability": 1.0, "source": "A Light in the ..."},
|
|
49
|
+
"price": {"value": "51.77", "confidence": 0.98, "probability": 0.99, "source": "£51.77"},
|
|
50
|
+
"availability": {"value": "In stock", "confidence": 1.0, "probability": 1.0, "source": "In stock"}
|
|
51
|
+
}
|
|
52
|
+
}
|
|
53
|
+
```
|
|
54
|
+
|
|
55
|
+
## Install
|
|
56
|
+
|
|
57
|
+
```sh
|
|
58
|
+
pipx install git+https://github.com/RKasai127/jevrake
|
|
59
|
+
export TYPESAFE_API_KEY=your-api-key
|
|
60
|
+
```
|
|
61
|
+
|
|
62
|
+
## Usage
|
|
63
|
+
|
|
64
|
+
```sh
|
|
65
|
+
jevrake https://example.com/item/1 --fields '{"title": "Product name", "price": {"description": "Price", "type": "int"}}'
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
Or from Python:
|
|
69
|
+
|
|
70
|
+
```python
|
|
71
|
+
from pydantic import BaseModel, Field
|
|
72
|
+
from jevrake import rake
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
class Product(BaseModel):
|
|
76
|
+
title: str = Field(description="Product name")
|
|
77
|
+
price: int = Field(description="Price")
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
results = rake(["https://example.com/item/1"], Product)
|
|
81
|
+
```
|
|
82
|
+
|
|
83
|
+
jevrake collects every value on the page that matches each field's type, and Jev picks the one
|
|
84
|
+
that fits the description. Values always come from the page.
|
|
85
|
+
|
|
86
|
+
- Works on list pages too: each repeated item (e.g. each shop in a list) becomes one record
|
|
87
|
+
- Types: `str`, `int`, `float`, `decimal`
|
|
88
|
+
- A value that is not found, or found with confidence below 0.9, becomes `null`
|
|
89
|
+
- robots.txt is not checked. Please make sure you follow each site's rules.
|
|
90
|
+
|
|
91
|
+
## License
|
|
92
|
+
|
|
93
|
+
MIT
|
jevrake-0.1.0/README.md
ADDED
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
<p align="right">English | <a href="https://github.com/RKasai127/jevrake/blob/main/README.ja.md">日本語</a></p>
|
|
2
|
+
|
|
3
|
+
# jevrake — a scraping tool powered by Jev
|
|
4
|
+
|
|
5
|
+
Example on [books.toscrape.com](https://books.toscrape.com/), a site made for scraping practice:
|
|
6
|
+
|
|
7
|
+
```sh
|
|
8
|
+
jevrake https://books.toscrape.com/ --fields '{"title": "Book title", "price": {"description": "Price", "type": "decimal"}, "availability": "Availability"}'
|
|
9
|
+
```
|
|
10
|
+
|
|
11
|
+
Output (1 of the 20 books on the page):
|
|
12
|
+
|
|
13
|
+
```json
|
|
14
|
+
{
|
|
15
|
+
"url": "https://books.toscrape.com/",
|
|
16
|
+
"data": {
|
|
17
|
+
"title": "A Light in the ...",
|
|
18
|
+
"price": "51.77",
|
|
19
|
+
"availability": "In stock"
|
|
20
|
+
},
|
|
21
|
+
"fields": {
|
|
22
|
+
"title": {"value": "A Light in the ...", "confidence": 1.0, "probability": 1.0, "source": "A Light in the ..."},
|
|
23
|
+
"price": {"value": "51.77", "confidence": 0.98, "probability": 0.99, "source": "£51.77"},
|
|
24
|
+
"availability": {"value": "In stock", "confidence": 1.0, "probability": 1.0, "source": "In stock"}
|
|
25
|
+
}
|
|
26
|
+
}
|
|
27
|
+
```
|
|
28
|
+
|
|
29
|
+
## Install
|
|
30
|
+
|
|
31
|
+
```sh
|
|
32
|
+
pipx install git+https://github.com/RKasai127/jevrake
|
|
33
|
+
export TYPESAFE_API_KEY=your-api-key
|
|
34
|
+
```
|
|
35
|
+
|
|
36
|
+
## Usage
|
|
37
|
+
|
|
38
|
+
```sh
|
|
39
|
+
jevrake https://example.com/item/1 --fields '{"title": "Product name", "price": {"description": "Price", "type": "int"}}'
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
Or from Python:
|
|
43
|
+
|
|
44
|
+
```python
|
|
45
|
+
from pydantic import BaseModel, Field
|
|
46
|
+
from jevrake import rake
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
class Product(BaseModel):
|
|
50
|
+
title: str = Field(description="Product name")
|
|
51
|
+
price: int = Field(description="Price")
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
results = rake(["https://example.com/item/1"], Product)
|
|
55
|
+
```
|
|
56
|
+
|
|
57
|
+
jevrake collects every value on the page that matches each field's type, and Jev picks the one
|
|
58
|
+
that fits the description. Values always come from the page.
|
|
59
|
+
|
|
60
|
+
- Works on list pages too: each repeated item (e.g. each shop in a list) becomes one record
|
|
61
|
+
- Types: `str`, `int`, `float`, `decimal`
|
|
62
|
+
- A value that is not found, or found with confidence below 0.9, becomes `null`
|
|
63
|
+
- robots.txt is not checked. Please make sure you follow each site's rules.
|
|
64
|
+
|
|
65
|
+
## License
|
|
66
|
+
|
|
67
|
+
MIT
|
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "jevrake"
|
|
3
|
+
version = "0.1.0"
|
|
4
|
+
description = "Selector-free web scraping powered by TypeSafe AI's Jev"
|
|
5
|
+
readme = "README.md"
|
|
6
|
+
license = "MIT"
|
|
7
|
+
license-files = ["LICENSE"]
|
|
8
|
+
keywords = [
|
|
9
|
+
"scraping",
|
|
10
|
+
"web-scraping",
|
|
11
|
+
"llm",
|
|
12
|
+
"pydantic",
|
|
13
|
+
"typesafe",
|
|
14
|
+
"jev",
|
|
15
|
+
]
|
|
16
|
+
classifiers = [
|
|
17
|
+
"Development Status :: 3 - Alpha",
|
|
18
|
+
"Environment :: Console",
|
|
19
|
+
"Intended Audience :: Developers",
|
|
20
|
+
"Operating System :: OS Independent",
|
|
21
|
+
"Programming Language :: Python :: 3",
|
|
22
|
+
"Programming Language :: Python :: 3.12",
|
|
23
|
+
"Topic :: Internet :: WWW/HTTP :: Indexing/Search",
|
|
24
|
+
"Typing :: Typed",
|
|
25
|
+
]
|
|
26
|
+
requires-python = ">=3.12"
|
|
27
|
+
dependencies = [
|
|
28
|
+
"pydantic>=2",
|
|
29
|
+
"httpx",
|
|
30
|
+
"selectolax",
|
|
31
|
+
"typesafe-sdk",
|
|
32
|
+
]
|
|
33
|
+
|
|
34
|
+
[[project.authors]]
|
|
35
|
+
name = "RKasai127"
|
|
36
|
+
|
|
37
|
+
[project.urls]
|
|
38
|
+
Homepage = "https://github.com/RKasai127/jevrake"
|
|
39
|
+
Repository = "https://github.com/RKasai127/jevrake"
|
|
40
|
+
Issues = "https://github.com/RKasai127/jevrake/issues"
|
|
41
|
+
|
|
42
|
+
[project.scripts]
|
|
43
|
+
jevrake = "jevrake.cli:main"
|
|
44
|
+
|
|
45
|
+
[build-system]
|
|
46
|
+
requires = ["uv_build"]
|
|
47
|
+
build-backend = "uv_build"
|
|
48
|
+
|
|
49
|
+
[dependency-groups]
|
|
50
|
+
dev = [
|
|
51
|
+
"pytest",
|
|
52
|
+
"ruff",
|
|
53
|
+
]
|
|
54
|
+
|
|
55
|
+
[tool.ruff]
|
|
56
|
+
line-length = 100
|
|
57
|
+
target-version = "py312"
|
|
58
|
+
|
|
59
|
+
[tool.ruff.lint]
|
|
60
|
+
select = [
|
|
61
|
+
"E",
|
|
62
|
+
"F",
|
|
63
|
+
"I",
|
|
64
|
+
"UP",
|
|
65
|
+
"B",
|
|
66
|
+
"SIM",
|
|
67
|
+
"RUF",
|
|
68
|
+
"ASYNC",
|
|
69
|
+
"PT",
|
|
70
|
+
]
|
|
71
|
+
allowed-confusables = [
|
|
72
|
+
"(",
|
|
73
|
+
")",
|
|
74
|
+
":",
|
|
75
|
+
",",
|
|
76
|
+
]
|
|
77
|
+
|
|
78
|
+
[tool.ruff.lint.per-file-ignores]
|
|
79
|
+
"tests/**" = ["RUF001"]
|
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "jevrake"
|
|
3
|
+
version = "0.1.0"
|
|
4
|
+
description = "Selector-free web scraping powered by TypeSafe AI's Jev"
|
|
5
|
+
readme = "README.md"
|
|
6
|
+
license = "MIT"
|
|
7
|
+
license-files = ["LICENSE"]
|
|
8
|
+
authors = [{ name = "RKasai127" }]
|
|
9
|
+
keywords = ["scraping", "web-scraping", "llm", "pydantic", "typesafe", "jev"]
|
|
10
|
+
classifiers = [
|
|
11
|
+
"Development Status :: 3 - Alpha",
|
|
12
|
+
"Environment :: Console",
|
|
13
|
+
"Intended Audience :: Developers",
|
|
14
|
+
"Operating System :: OS Independent",
|
|
15
|
+
"Programming Language :: Python :: 3",
|
|
16
|
+
"Programming Language :: Python :: 3.12",
|
|
17
|
+
"Topic :: Internet :: WWW/HTTP :: Indexing/Search",
|
|
18
|
+
"Typing :: Typed",
|
|
19
|
+
]
|
|
20
|
+
requires-python = ">=3.12"
|
|
21
|
+
dependencies = ["pydantic>=2", "httpx", "selectolax", "typesafe-sdk"]
|
|
22
|
+
|
|
23
|
+
[project.urls]
|
|
24
|
+
Homepage = "https://github.com/RKasai127/jevrake"
|
|
25
|
+
Repository = "https://github.com/RKasai127/jevrake"
|
|
26
|
+
Issues = "https://github.com/RKasai127/jevrake/issues"
|
|
27
|
+
|
|
28
|
+
[project.scripts]
|
|
29
|
+
jevrake = "jevrake.cli:main"
|
|
30
|
+
|
|
31
|
+
[build-system]
|
|
32
|
+
requires = ["uv_build"]
|
|
33
|
+
build-backend = "uv_build"
|
|
34
|
+
|
|
35
|
+
[dependency-groups]
|
|
36
|
+
dev = ["pytest", "ruff"]
|
|
37
|
+
|
|
38
|
+
[tool.ruff]
|
|
39
|
+
line-length = 100
|
|
40
|
+
target-version = "py312"
|
|
41
|
+
|
|
42
|
+
[tool.ruff.lint]
|
|
43
|
+
select = ["E", "F", "I", "UP", "B", "SIM", "RUF", "ASYNC", "PT"]
|
|
44
|
+
# Full-width punctuation is normal in Japanese messages.
|
|
45
|
+
allowed-confusables = ["(", ")", ":", ","]
|
|
46
|
+
|
|
47
|
+
[tool.ruff.lint.per-file-ignores]
|
|
48
|
+
# Tests use full-width digits on purpose (Japanese price / date inputs).
|
|
49
|
+
"tests/**" = ["RUF001"]
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
"""jevrake: selector-free web scraping with TypeSafe AI's Jev."""
|
|
2
|
+
|
|
3
|
+
from ._internal.api import rake
|
|
4
|
+
from ._internal.models import ExtractedValue, RakeItem, RakeResult
|
|
5
|
+
from ._internal.version import VERSION as __version__
|
|
6
|
+
|
|
7
|
+
__all__ = [
|
|
8
|
+
"ExtractedValue",
|
|
9
|
+
"RakeItem",
|
|
10
|
+
"RakeResult",
|
|
11
|
+
"__version__",
|
|
12
|
+
"rake",
|
|
13
|
+
]
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""Internal implementation. Not part of the public API; may change at any time."""
|
|
@@ -0,0 +1,138 @@
|
|
|
1
|
+
"""Extract records from each page with ``rake()``.
|
|
2
|
+
|
|
3
|
+
For each URL: download the page, split it into items (one for a single-record page), collect
|
|
4
|
+
candidate values, ask Jev to choose one per field, and validate each item with the user's model.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from collections.abc import Callable, Sequence
|
|
10
|
+
from typing import Any
|
|
11
|
+
|
|
12
|
+
from pydantic import BaseModel, ValidationError
|
|
13
|
+
|
|
14
|
+
from . import collect, dom, jev
|
|
15
|
+
from .client import MODEL, JevClient, TypeSafeJevClient
|
|
16
|
+
from .exceptions import FetchError, JevAuthError, JevError
|
|
17
|
+
from .fetch import Fetcher, HttpxFetcher
|
|
18
|
+
from .models import ExtractionTarget, ItemBlock, JevResponse, RakeItem, RakeResult
|
|
19
|
+
from .schema import build_targets_from_model
|
|
20
|
+
|
|
21
|
+
DEFAULT_MIN_CONFIDENCE = 0.9
|
|
22
|
+
MAX_ITEMS_PER_REQUEST = 20
|
|
23
|
+
|
|
24
|
+
RequestHook = Callable[[str, dict[str, Any]], None]
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def _build_failed_result[T: BaseModel](model: type[T], url: str, error: str) -> RakeResult[T]:
|
|
28
|
+
return RakeResult[model](url=url, items=[], request_ids=[], errors=[error]) # type: ignore[valid-type]
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def _format_validation_errors(error: ValidationError) -> list[str]:
|
|
32
|
+
return [f"{'.'.join(str(p) for p in err['loc'])}: {err['msg']}" for err in error.errors()]
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _rake_page[T: BaseModel](
|
|
36
|
+
url: str,
|
|
37
|
+
model: type[T],
|
|
38
|
+
targets: list[ExtractionTarget],
|
|
39
|
+
fetcher: Fetcher,
|
|
40
|
+
client: JevClient | None,
|
|
41
|
+
min_confidence: float,
|
|
42
|
+
on_request: RequestHook | None,
|
|
43
|
+
) -> RakeResult[T]:
|
|
44
|
+
try:
|
|
45
|
+
html = fetcher.fetch(url)
|
|
46
|
+
except FetchError as e:
|
|
47
|
+
return _build_failed_result(model, url, str(e))
|
|
48
|
+
|
|
49
|
+
page = dom.parse_page(html)
|
|
50
|
+
blocks = collect.find_item_blocks(page, targets)
|
|
51
|
+
numbered: list[tuple[int | None, ItemBlock]] = (
|
|
52
|
+
[(None, blocks[0])] if blocks[0].is_whole_page else list(enumerate(blocks))
|
|
53
|
+
)
|
|
54
|
+
|
|
55
|
+
items: list[RakeItem[T]] = []
|
|
56
|
+
request_ids: list[str] = []
|
|
57
|
+
errors: list[str] = []
|
|
58
|
+
for start in range(0, len(numbered), MAX_ITEMS_PER_REQUEST):
|
|
59
|
+
chunk = numbered[start : start + MAX_ITEMS_PER_REQUEST]
|
|
60
|
+
candidates_by_item = {item: block.candidates for item, block in chunk}
|
|
61
|
+
request = jev.build_jev_request(url, page, targets, candidates_by_item)
|
|
62
|
+
if on_request is not None:
|
|
63
|
+
on_request(url, jev.build_jev_payload(request, MODEL))
|
|
64
|
+
if client is None:
|
|
65
|
+
continue
|
|
66
|
+
|
|
67
|
+
response = JevResponse()
|
|
68
|
+
if request.questions:
|
|
69
|
+
try:
|
|
70
|
+
response = client.ask(request.state, request.questions)
|
|
71
|
+
except JevAuthError:
|
|
72
|
+
raise
|
|
73
|
+
except JevError as e:
|
|
74
|
+
first, last = chunk[0][0], chunk[-1][0]
|
|
75
|
+
errors.append(str(e) if first is None else f"items {first}-{last}: {e}")
|
|
76
|
+
continue
|
|
77
|
+
if response.request_id:
|
|
78
|
+
request_ids.append(response.request_id)
|
|
79
|
+
|
|
80
|
+
for item, _ in chunk:
|
|
81
|
+
prefix = "" if item is None else f"item {item}: "
|
|
82
|
+
outcome = jev.parse_jev_response(request, response, targets, min_confidence, item)
|
|
83
|
+
if outcome.errors:
|
|
84
|
+
errors.extend(prefix + e for e in outcome.errors)
|
|
85
|
+
continue
|
|
86
|
+
try:
|
|
87
|
+
data = model.model_validate(outcome.values)
|
|
88
|
+
except ValidationError as e:
|
|
89
|
+
errors.extend(prefix + m for m in _format_validation_errors(e))
|
|
90
|
+
continue
|
|
91
|
+
items.append(RakeItem[model](data=data, fields=outcome.fields)) # type: ignore[valid-type]
|
|
92
|
+
|
|
93
|
+
return RakeResult[model](url=url, items=items, request_ids=request_ids, errors=errors) # type: ignore[valid-type]
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def _rake_page_safely[T: BaseModel](url: str, model: type[T], *args: Any) -> RakeResult[T]:
|
|
97
|
+
try:
|
|
98
|
+
return _rake_page(url, model, *args)
|
|
99
|
+
except JevAuthError:
|
|
100
|
+
raise
|
|
101
|
+
except Exception as e: # one bad page must not stop the others
|
|
102
|
+
return _build_failed_result(model, url, f"{type(e).__name__}: {e}")
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def rake[T: BaseModel](
|
|
106
|
+
urls: Sequence[str],
|
|
107
|
+
model: type[T],
|
|
108
|
+
*,
|
|
109
|
+
min_confidence: float = DEFAULT_MIN_CONFIDENCE,
|
|
110
|
+
fetcher: Fetcher | None = None,
|
|
111
|
+
client: JevClient | None = None,
|
|
112
|
+
dry_run: bool = False,
|
|
113
|
+
on_request: RequestHook | None = None,
|
|
114
|
+
) -> list[RakeResult[T]]:
|
|
115
|
+
"""Extract records of ``model`` from each URL, one page at a time, in the order of ``urls``.
|
|
116
|
+
|
|
117
|
+
Raises :class:`JevAuthError` if the API key is missing or rejected.
|
|
118
|
+
"""
|
|
119
|
+
targets = build_targets_from_model(model)
|
|
120
|
+
|
|
121
|
+
created: list[HttpxFetcher | TypeSafeJevClient] = []
|
|
122
|
+
if fetcher is None:
|
|
123
|
+
fetcher = HttpxFetcher()
|
|
124
|
+
created.append(fetcher)
|
|
125
|
+
if dry_run:
|
|
126
|
+
client = None
|
|
127
|
+
elif client is None:
|
|
128
|
+
client = TypeSafeJevClient()
|
|
129
|
+
created.append(client)
|
|
130
|
+
|
|
131
|
+
try:
|
|
132
|
+
return [
|
|
133
|
+
_rake_page_safely(url, model, targets, fetcher, client, min_confidence, on_request)
|
|
134
|
+
for url in urls
|
|
135
|
+
]
|
|
136
|
+
finally:
|
|
137
|
+
for resource in created:
|
|
138
|
+
resource.close()
|
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
"""Send the question to Jev (TypeSafe AI API) and return its answer.
|
|
2
|
+
|
|
3
|
+
Wraps the official SDK so the rest of jevrake does not depend on SDK types or exceptions.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
import os
|
|
9
|
+
from collections.abc import Mapping
|
|
10
|
+
from typing import TYPE_CHECKING, Any, Protocol
|
|
11
|
+
|
|
12
|
+
import typesafe_sdk as ts
|
|
13
|
+
|
|
14
|
+
from .exceptions import JevAuthError, JevError, MissingAPIKeyError
|
|
15
|
+
from .models import JevChoiceAnswer, JevResponse
|
|
16
|
+
|
|
17
|
+
if TYPE_CHECKING:
|
|
18
|
+
import httpx2
|
|
19
|
+
|
|
20
|
+
API_KEY_ENV = "TYPESAFE_API_KEY"
|
|
21
|
+
MODEL = "jev-latest"
|
|
22
|
+
TIMEOUT = 30.0
|
|
23
|
+
MAX_RETRIES = 3
|
|
24
|
+
|
|
25
|
+
MISSING_API_KEY_MESSAGE = f"""\
|
|
26
|
+
{API_KEY_ENV} is not set.
|
|
27
|
+
Create an API key at https://console.typesafe.ai and set it with one of:
|
|
28
|
+
export {API_KEY_ENV}=<your API key>
|
|
29
|
+
cp .envrc.example .envrc # fill in the key, then run `direnv allow`"""
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
class JevClient(Protocol):
|
|
33
|
+
def ask(self, state: Mapping[str, Any], questions: Mapping[str, Any]) -> JevResponse: ...
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
class TypeSafeJevClient:
|
|
37
|
+
""":class:`JevClient` backed by ``typesafe_sdk.TypeSafeClient``.
|
|
38
|
+
|
|
39
|
+
429 / 5xx responses are retried with exponential backoff (up to ``MAX_RETRIES`` times) by
|
|
40
|
+
the SDK's retry policy.
|
|
41
|
+
"""
|
|
42
|
+
|
|
43
|
+
def __init__(self, *, transport: httpx2.BaseTransport | None = None) -> None:
|
|
44
|
+
api_key = os.environ.get(API_KEY_ENV)
|
|
45
|
+
|
|
46
|
+
if not api_key:
|
|
47
|
+
raise MissingAPIKeyError(MISSING_API_KEY_MESSAGE)
|
|
48
|
+
|
|
49
|
+
self._client = ts.TypeSafeClient(
|
|
50
|
+
api_key=api_key,
|
|
51
|
+
retry=ts.RetryPolicy(max_retries=MAX_RETRIES, timeout=None),
|
|
52
|
+
timeout=TIMEOUT,
|
|
53
|
+
transport=transport,
|
|
54
|
+
)
|
|
55
|
+
|
|
56
|
+
def ask(self, state: Mapping[str, Any], questions: Mapping[str, Any]) -> JevResponse:
|
|
57
|
+
try:
|
|
58
|
+
response = self._client.system_one(state=dict(state), questions=questions, model=MODEL)
|
|
59
|
+
except (ts.TypeSafeAuthenticationError, ts.TypeSafePermissionDeniedError) as e:
|
|
60
|
+
raise JevAuthError(
|
|
61
|
+
f"Jev API rejected the API key (HTTP {e.status}). Check {API_KEY_ENV}."
|
|
62
|
+
) from None
|
|
63
|
+
except ts.TypeSafeAPITimeoutError:
|
|
64
|
+
raise JevError("Jev API request timed out") from None
|
|
65
|
+
except ts.TypeSafeAPIError as e:
|
|
66
|
+
raise JevError(f"Jev API error (HTTP {e.status}): {e.body}") from None
|
|
67
|
+
except ts.TypeSafeError as e:
|
|
68
|
+
raise JevError(f"Jev API error: {e}") from None
|
|
69
|
+
|
|
70
|
+
return JevResponse(
|
|
71
|
+
choices={
|
|
72
|
+
name: JevChoiceAnswer(
|
|
73
|
+
choice=a.choice, confidence=a.confidence, probabilities=dict(a.probabilities)
|
|
74
|
+
)
|
|
75
|
+
for name, a in response.choices.items()
|
|
76
|
+
},
|
|
77
|
+
request_id=response.request_id,
|
|
78
|
+
)
|
|
79
|
+
|
|
80
|
+
def close(self) -> None:
|
|
81
|
+
self._client.close()
|