jevrake 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
jevrake-0.1.0/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 RKasai127
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
jevrake-0.1.0/PKG-INFO ADDED
@@ -0,0 +1,93 @@
1
+ Metadata-Version: 2.4
2
+ Name: jevrake
3
+ Version: 0.1.0
4
+ Summary: Selector-free web scraping powered by TypeSafe AI's Jev
5
+ Keywords: scraping,web-scraping,llm,pydantic,typesafe,jev
6
+ Author: RKasai127
7
+ License-Expression: MIT
8
+ License-File: LICENSE
9
+ Classifier: Development Status :: 3 - Alpha
10
+ Classifier: Environment :: Console
11
+ Classifier: Intended Audience :: Developers
12
+ Classifier: Operating System :: OS Independent
13
+ Classifier: Programming Language :: Python :: 3
14
+ Classifier: Programming Language :: Python :: 3.12
15
+ Classifier: Topic :: Internet :: WWW/HTTP :: Indexing/Search
16
+ Classifier: Typing :: Typed
17
+ Requires-Dist: pydantic>=2
18
+ Requires-Dist: httpx
19
+ Requires-Dist: selectolax
20
+ Requires-Dist: typesafe-sdk
21
+ Requires-Python: >=3.12
22
+ Project-URL: Homepage, https://github.com/RKasai127/jevrake
23
+ Project-URL: Repository, https://github.com/RKasai127/jevrake
24
+ Project-URL: Issues, https://github.com/RKasai127/jevrake/issues
25
+ Description-Content-Type: text/markdown
26
+
27
+ <p align="right">English | <a href="https://github.com/RKasai127/jevrake/blob/main/README.ja.md">日本語</a></p>
28
+
29
+ # jevrake — a scraping tool powered by Jev
30
+
31
+ Example on [books.toscrape.com](https://books.toscrape.com/), a site made for scraping practice:
32
+
33
+ ```sh
34
+ jevrake https://books.toscrape.com/ --fields '{"title": "Book title", "price": {"description": "Price", "type": "decimal"}, "availability": "Availability"}'
35
+ ```
36
+
37
+ Output (1 of the 20 books on the page):
38
+
39
+ ```json
40
+ {
41
+ "url": "https://books.toscrape.com/",
42
+ "data": {
43
+ "title": "A Light in the ...",
44
+ "price": "51.77",
45
+ "availability": "In stock"
46
+ },
47
+ "fields": {
48
+ "title": {"value": "A Light in the ...", "confidence": 1.0, "probability": 1.0, "source": "A Light in the ..."},
49
+ "price": {"value": "51.77", "confidence": 0.98, "probability": 0.99, "source": "£51.77"},
50
+ "availability": {"value": "In stock", "confidence": 1.0, "probability": 1.0, "source": "In stock"}
51
+ }
52
+ }
53
+ ```
54
+
55
+ ## Install
56
+
57
+ ```sh
58
+ pipx install git+https://github.com/RKasai127/jevrake
59
+ export TYPESAFE_API_KEY=your-api-key
60
+ ```
61
+
62
+ ## Usage
63
+
64
+ ```sh
65
+ jevrake https://example.com/item/1 --fields '{"title": "Product name", "price": {"description": "Price", "type": "int"}}'
66
+ ```
67
+
68
+ Or from Python:
69
+
70
+ ```python
71
+ from pydantic import BaseModel, Field
72
+ from jevrake import rake
73
+
74
+
75
+ class Product(BaseModel):
76
+ title: str = Field(description="Product name")
77
+ price: int = Field(description="Price")
78
+
79
+
80
+ results = rake(["https://example.com/item/1"], Product)
81
+ ```
82
+
83
+ jevrake collects every value on the page that matches each field's type, and Jev picks the one
84
+ that fits the description. Values always come from the page.
85
+
86
+ - Works on list pages too: each repeated item (e.g. each shop in a list) becomes one record
87
+ - Types: `str`, `int`, `float`, `decimal`
88
+ - A value that is not found, or found with confidence below 0.9, becomes `null`
89
+ - robots.txt is not checked. Please make sure you follow each site's rules.
90
+
91
+ ## License
92
+
93
+ MIT
@@ -0,0 +1,67 @@
1
+ <p align="right">English | <a href="https://github.com/RKasai127/jevrake/blob/main/README.ja.md">日本語</a></p>
2
+
3
+ # jevrake — a scraping tool powered by Jev
4
+
5
+ Example on [books.toscrape.com](https://books.toscrape.com/), a site made for scraping practice:
6
+
7
+ ```sh
8
+ jevrake https://books.toscrape.com/ --fields '{"title": "Book title", "price": {"description": "Price", "type": "decimal"}, "availability": "Availability"}'
9
+ ```
10
+
11
+ Output (1 of the 20 books on the page):
12
+
13
+ ```json
14
+ {
15
+ "url": "https://books.toscrape.com/",
16
+ "data": {
17
+ "title": "A Light in the ...",
18
+ "price": "51.77",
19
+ "availability": "In stock"
20
+ },
21
+ "fields": {
22
+ "title": {"value": "A Light in the ...", "confidence": 1.0, "probability": 1.0, "source": "A Light in the ..."},
23
+ "price": {"value": "51.77", "confidence": 0.98, "probability": 0.99, "source": "£51.77"},
24
+ "availability": {"value": "In stock", "confidence": 1.0, "probability": 1.0, "source": "In stock"}
25
+ }
26
+ }
27
+ ```
28
+
29
+ ## Install
30
+
31
+ ```sh
32
+ pipx install git+https://github.com/RKasai127/jevrake
33
+ export TYPESAFE_API_KEY=your-api-key
34
+ ```
35
+
36
+ ## Usage
37
+
38
+ ```sh
39
+ jevrake https://example.com/item/1 --fields '{"title": "Product name", "price": {"description": "Price", "type": "int"}}'
40
+ ```
41
+
42
+ Or from Python:
43
+
44
+ ```python
45
+ from pydantic import BaseModel, Field
46
+ from jevrake import rake
47
+
48
+
49
+ class Product(BaseModel):
50
+ title: str = Field(description="Product name")
51
+ price: int = Field(description="Price")
52
+
53
+
54
+ results = rake(["https://example.com/item/1"], Product)
55
+ ```
56
+
57
+ jevrake collects every value on the page that matches each field's type, and Jev picks the one
58
+ that fits the description. Values always come from the page.
59
+
60
+ - Works on list pages too: each repeated item (e.g. each shop in a list) becomes one record
61
+ - Types: `str`, `int`, `float`, `decimal`
62
+ - A value that is not found, or found with confidence below 0.9, becomes `null`
63
+ - robots.txt is not checked. Please make sure you follow each site's rules.
64
+
65
+ ## License
66
+
67
+ MIT
@@ -0,0 +1,79 @@
1
+ [project]
2
+ name = "jevrake"
3
+ version = "0.1.0"
4
+ description = "Selector-free web scraping powered by TypeSafe AI's Jev"
5
+ readme = "README.md"
6
+ license = "MIT"
7
+ license-files = ["LICENSE"]
8
+ keywords = [
9
+ "scraping",
10
+ "web-scraping",
11
+ "llm",
12
+ "pydantic",
13
+ "typesafe",
14
+ "jev",
15
+ ]
16
+ classifiers = [
17
+ "Development Status :: 3 - Alpha",
18
+ "Environment :: Console",
19
+ "Intended Audience :: Developers",
20
+ "Operating System :: OS Independent",
21
+ "Programming Language :: Python :: 3",
22
+ "Programming Language :: Python :: 3.12",
23
+ "Topic :: Internet :: WWW/HTTP :: Indexing/Search",
24
+ "Typing :: Typed",
25
+ ]
26
+ requires-python = ">=3.12"
27
+ dependencies = [
28
+ "pydantic>=2",
29
+ "httpx",
30
+ "selectolax",
31
+ "typesafe-sdk",
32
+ ]
33
+
34
+ [[project.authors]]
35
+ name = "RKasai127"
36
+
37
+ [project.urls]
38
+ Homepage = "https://github.com/RKasai127/jevrake"
39
+ Repository = "https://github.com/RKasai127/jevrake"
40
+ Issues = "https://github.com/RKasai127/jevrake/issues"
41
+
42
+ [project.scripts]
43
+ jevrake = "jevrake.cli:main"
44
+
45
+ [build-system]
46
+ requires = ["uv_build"]
47
+ build-backend = "uv_build"
48
+
49
+ [dependency-groups]
50
+ dev = [
51
+ "pytest",
52
+ "ruff",
53
+ ]
54
+
55
+ [tool.ruff]
56
+ line-length = 100
57
+ target-version = "py312"
58
+
59
+ [tool.ruff.lint]
60
+ select = [
61
+ "E",
62
+ "F",
63
+ "I",
64
+ "UP",
65
+ "B",
66
+ "SIM",
67
+ "RUF",
68
+ "ASYNC",
69
+ "PT",
70
+ ]
71
+ allowed-confusables = [
72
+ "(",
73
+ ")",
74
+ ":",
75
+ ",",
76
+ ]
77
+
78
+ [tool.ruff.lint.per-file-ignores]
79
+ "tests/**" = ["RUF001"]
@@ -0,0 +1,49 @@
1
+ [project]
2
+ name = "jevrake"
3
+ version = "0.1.0"
4
+ description = "Selector-free web scraping powered by TypeSafe AI's Jev"
5
+ readme = "README.md"
6
+ license = "MIT"
7
+ license-files = ["LICENSE"]
8
+ authors = [{ name = "RKasai127" }]
9
+ keywords = ["scraping", "web-scraping", "llm", "pydantic", "typesafe", "jev"]
10
+ classifiers = [
11
+ "Development Status :: 3 - Alpha",
12
+ "Environment :: Console",
13
+ "Intended Audience :: Developers",
14
+ "Operating System :: OS Independent",
15
+ "Programming Language :: Python :: 3",
16
+ "Programming Language :: Python :: 3.12",
17
+ "Topic :: Internet :: WWW/HTTP :: Indexing/Search",
18
+ "Typing :: Typed",
19
+ ]
20
+ requires-python = ">=3.12"
21
+ dependencies = ["pydantic>=2", "httpx", "selectolax", "typesafe-sdk"]
22
+
23
+ [project.urls]
24
+ Homepage = "https://github.com/RKasai127/jevrake"
25
+ Repository = "https://github.com/RKasai127/jevrake"
26
+ Issues = "https://github.com/RKasai127/jevrake/issues"
27
+
28
+ [project.scripts]
29
+ jevrake = "jevrake.cli:main"
30
+
31
+ [build-system]
32
+ requires = ["uv_build"]
33
+ build-backend = "uv_build"
34
+
35
+ [dependency-groups]
36
+ dev = ["pytest", "ruff"]
37
+
38
+ [tool.ruff]
39
+ line-length = 100
40
+ target-version = "py312"
41
+
42
+ [tool.ruff.lint]
43
+ select = ["E", "F", "I", "UP", "B", "SIM", "RUF", "ASYNC", "PT"]
44
+ # Full-width punctuation is normal in Japanese messages.
45
+ allowed-confusables = ["(", ")", ":", ","]
46
+
47
+ [tool.ruff.lint.per-file-ignores]
48
+ # Tests use full-width digits on purpose (Japanese price / date inputs).
49
+ "tests/**" = ["RUF001"]
@@ -0,0 +1,13 @@
1
+ """jevrake: selector-free web scraping with TypeSafe AI's Jev."""
2
+
3
+ from ._internal.api import rake
4
+ from ._internal.models import ExtractedValue, RakeItem, RakeResult
5
+ from ._internal.version import VERSION as __version__
6
+
7
+ __all__ = [
8
+ "ExtractedValue",
9
+ "RakeItem",
10
+ "RakeResult",
11
+ "__version__",
12
+ "rake",
13
+ ]
@@ -0,0 +1 @@
1
+ """Internal implementation. Not part of the public API; may change at any time."""
@@ -0,0 +1,138 @@
1
+ """Extract records from each page with ``rake()``.
2
+
3
+ For each URL: download the page, split it into items (one for a single-record page), collect
4
+ candidate values, ask Jev to choose one per field, and validate each item with the user's model.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ from collections.abc import Callable, Sequence
10
+ from typing import Any
11
+
12
+ from pydantic import BaseModel, ValidationError
13
+
14
+ from . import collect, dom, jev
15
+ from .client import MODEL, JevClient, TypeSafeJevClient
16
+ from .exceptions import FetchError, JevAuthError, JevError
17
+ from .fetch import Fetcher, HttpxFetcher
18
+ from .models import ExtractionTarget, ItemBlock, JevResponse, RakeItem, RakeResult
19
+ from .schema import build_targets_from_model
20
+
21
+ DEFAULT_MIN_CONFIDENCE = 0.9
22
+ MAX_ITEMS_PER_REQUEST = 20
23
+
24
+ RequestHook = Callable[[str, dict[str, Any]], None]
25
+
26
+
27
+ def _build_failed_result[T: BaseModel](model: type[T], url: str, error: str) -> RakeResult[T]:
28
+ return RakeResult[model](url=url, items=[], request_ids=[], errors=[error]) # type: ignore[valid-type]
29
+
30
+
31
+ def _format_validation_errors(error: ValidationError) -> list[str]:
32
+ return [f"{'.'.join(str(p) for p in err['loc'])}: {err['msg']}" for err in error.errors()]
33
+
34
+
35
+ def _rake_page[T: BaseModel](
36
+ url: str,
37
+ model: type[T],
38
+ targets: list[ExtractionTarget],
39
+ fetcher: Fetcher,
40
+ client: JevClient | None,
41
+ min_confidence: float,
42
+ on_request: RequestHook | None,
43
+ ) -> RakeResult[T]:
44
+ try:
45
+ html = fetcher.fetch(url)
46
+ except FetchError as e:
47
+ return _build_failed_result(model, url, str(e))
48
+
49
+ page = dom.parse_page(html)
50
+ blocks = collect.find_item_blocks(page, targets)
51
+ numbered: list[tuple[int | None, ItemBlock]] = (
52
+ [(None, blocks[0])] if blocks[0].is_whole_page else list(enumerate(blocks))
53
+ )
54
+
55
+ items: list[RakeItem[T]] = []
56
+ request_ids: list[str] = []
57
+ errors: list[str] = []
58
+ for start in range(0, len(numbered), MAX_ITEMS_PER_REQUEST):
59
+ chunk = numbered[start : start + MAX_ITEMS_PER_REQUEST]
60
+ candidates_by_item = {item: block.candidates for item, block in chunk}
61
+ request = jev.build_jev_request(url, page, targets, candidates_by_item)
62
+ if on_request is not None:
63
+ on_request(url, jev.build_jev_payload(request, MODEL))
64
+ if client is None:
65
+ continue
66
+
67
+ response = JevResponse()
68
+ if request.questions:
69
+ try:
70
+ response = client.ask(request.state, request.questions)
71
+ except JevAuthError:
72
+ raise
73
+ except JevError as e:
74
+ first, last = chunk[0][0], chunk[-1][0]
75
+ errors.append(str(e) if first is None else f"items {first}-{last}: {e}")
76
+ continue
77
+ if response.request_id:
78
+ request_ids.append(response.request_id)
79
+
80
+ for item, _ in chunk:
81
+ prefix = "" if item is None else f"item {item}: "
82
+ outcome = jev.parse_jev_response(request, response, targets, min_confidence, item)
83
+ if outcome.errors:
84
+ errors.extend(prefix + e for e in outcome.errors)
85
+ continue
86
+ try:
87
+ data = model.model_validate(outcome.values)
88
+ except ValidationError as e:
89
+ errors.extend(prefix + m for m in _format_validation_errors(e))
90
+ continue
91
+ items.append(RakeItem[model](data=data, fields=outcome.fields)) # type: ignore[valid-type]
92
+
93
+ return RakeResult[model](url=url, items=items, request_ids=request_ids, errors=errors) # type: ignore[valid-type]
94
+
95
+
96
+ def _rake_page_safely[T: BaseModel](url: str, model: type[T], *args: Any) -> RakeResult[T]:
97
+ try:
98
+ return _rake_page(url, model, *args)
99
+ except JevAuthError:
100
+ raise
101
+ except Exception as e: # one bad page must not stop the others
102
+ return _build_failed_result(model, url, f"{type(e).__name__}: {e}")
103
+
104
+
105
+ def rake[T: BaseModel](
106
+ urls: Sequence[str],
107
+ model: type[T],
108
+ *,
109
+ min_confidence: float = DEFAULT_MIN_CONFIDENCE,
110
+ fetcher: Fetcher | None = None,
111
+ client: JevClient | None = None,
112
+ dry_run: bool = False,
113
+ on_request: RequestHook | None = None,
114
+ ) -> list[RakeResult[T]]:
115
+ """Extract records of ``model`` from each URL, one page at a time, in the order of ``urls``.
116
+
117
+ Raises :class:`JevAuthError` if the API key is missing or rejected.
118
+ """
119
+ targets = build_targets_from_model(model)
120
+
121
+ created: list[HttpxFetcher | TypeSafeJevClient] = []
122
+ if fetcher is None:
123
+ fetcher = HttpxFetcher()
124
+ created.append(fetcher)
125
+ if dry_run:
126
+ client = None
127
+ elif client is None:
128
+ client = TypeSafeJevClient()
129
+ created.append(client)
130
+
131
+ try:
132
+ return [
133
+ _rake_page_safely(url, model, targets, fetcher, client, min_confidence, on_request)
134
+ for url in urls
135
+ ]
136
+ finally:
137
+ for resource in created:
138
+ resource.close()
@@ -0,0 +1,81 @@
1
+ """Send the question to Jev (TypeSafe AI API) and return its answer.
2
+
3
+ Wraps the official SDK so the rest of jevrake does not depend on SDK types or exceptions.
4
+ """
5
+
6
+ from __future__ import annotations
7
+
8
+ import os
9
+ from collections.abc import Mapping
10
+ from typing import TYPE_CHECKING, Any, Protocol
11
+
12
+ import typesafe_sdk as ts
13
+
14
+ from .exceptions import JevAuthError, JevError, MissingAPIKeyError
15
+ from .models import JevChoiceAnswer, JevResponse
16
+
17
+ if TYPE_CHECKING:
18
+ import httpx2
19
+
20
+ API_KEY_ENV = "TYPESAFE_API_KEY"
21
+ MODEL = "jev-latest"
22
+ TIMEOUT = 30.0
23
+ MAX_RETRIES = 3
24
+
25
+ MISSING_API_KEY_MESSAGE = f"""\
26
+ {API_KEY_ENV} is not set.
27
+ Create an API key at https://console.typesafe.ai and set it with one of:
28
+ export {API_KEY_ENV}=<your API key>
29
+ cp .envrc.example .envrc # fill in the key, then run `direnv allow`"""
30
+
31
+
32
+ class JevClient(Protocol):
33
+ def ask(self, state: Mapping[str, Any], questions: Mapping[str, Any]) -> JevResponse: ...
34
+
35
+
36
+ class TypeSafeJevClient:
37
+ """:class:`JevClient` backed by ``typesafe_sdk.TypeSafeClient``.
38
+
39
+ 429 / 5xx responses are retried with exponential backoff (up to ``MAX_RETRIES`` times) by
40
+ the SDK's retry policy.
41
+ """
42
+
43
+ def __init__(self, *, transport: httpx2.BaseTransport | None = None) -> None:
44
+ api_key = os.environ.get(API_KEY_ENV)
45
+
46
+ if not api_key:
47
+ raise MissingAPIKeyError(MISSING_API_KEY_MESSAGE)
48
+
49
+ self._client = ts.TypeSafeClient(
50
+ api_key=api_key,
51
+ retry=ts.RetryPolicy(max_retries=MAX_RETRIES, timeout=None),
52
+ timeout=TIMEOUT,
53
+ transport=transport,
54
+ )
55
+
56
+ def ask(self, state: Mapping[str, Any], questions: Mapping[str, Any]) -> JevResponse:
57
+ try:
58
+ response = self._client.system_one(state=dict(state), questions=questions, model=MODEL)
59
+ except (ts.TypeSafeAuthenticationError, ts.TypeSafePermissionDeniedError) as e:
60
+ raise JevAuthError(
61
+ f"Jev API rejected the API key (HTTP {e.status}). Check {API_KEY_ENV}."
62
+ ) from None
63
+ except ts.TypeSafeAPITimeoutError:
64
+ raise JevError("Jev API request timed out") from None
65
+ except ts.TypeSafeAPIError as e:
66
+ raise JevError(f"Jev API error (HTTP {e.status}): {e.body}") from None
67
+ except ts.TypeSafeError as e:
68
+ raise JevError(f"Jev API error: {e}") from None
69
+
70
+ return JevResponse(
71
+ choices={
72
+ name: JevChoiceAnswer(
73
+ choice=a.choice, confidence=a.confidence, probabilities=dict(a.probabilities)
74
+ )
75
+ for name, a in response.choices.items()
76
+ },
77
+ request_id=response.request_id,
78
+ )
79
+
80
+ def close(self) -> None:
81
+ self._client.close()