jevrake 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
jevrake/__init__.py ADDED
@@ -0,0 +1,13 @@
1
+ """jevrake: selector-free web scraping with TypeSafe AI's Jev."""
2
+
3
+ from ._internal.api import rake
4
+ from ._internal.models import ExtractedValue, RakeItem, RakeResult
5
+ from ._internal.version import VERSION as __version__
6
+
7
+ __all__ = [
8
+ "ExtractedValue",
9
+ "RakeItem",
10
+ "RakeResult",
11
+ "__version__",
12
+ "rake",
13
+ ]
@@ -0,0 +1 @@
1
+ """Internal implementation. Not part of the public API; may change at any time."""
@@ -0,0 +1,138 @@
1
+ """Extract records from each page with ``rake()``.
2
+
3
+ For each URL: download the page, split it into items (one for a single-record page), collect
4
+ candidate values, ask Jev to choose one per field, and validate each item with the user's model.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ from collections.abc import Callable, Sequence
10
+ from typing import Any
11
+
12
+ from pydantic import BaseModel, ValidationError
13
+
14
+ from . import collect, dom, jev
15
+ from .client import MODEL, JevClient, TypeSafeJevClient
16
+ from .exceptions import FetchError, JevAuthError, JevError
17
+ from .fetch import Fetcher, HttpxFetcher
18
+ from .models import ExtractionTarget, ItemBlock, JevResponse, RakeItem, RakeResult
19
+ from .schema import build_targets_from_model
20
+
21
+ DEFAULT_MIN_CONFIDENCE = 0.9
22
+ MAX_ITEMS_PER_REQUEST = 20
23
+
24
+ RequestHook = Callable[[str, dict[str, Any]], None]
25
+
26
+
27
+ def _build_failed_result[T: BaseModel](model: type[T], url: str, error: str) -> RakeResult[T]:
28
+ return RakeResult[model](url=url, items=[], request_ids=[], errors=[error]) # type: ignore[valid-type]
29
+
30
+
31
+ def _format_validation_errors(error: ValidationError) -> list[str]:
32
+ return [f"{'.'.join(str(p) for p in err['loc'])}: {err['msg']}" for err in error.errors()]
33
+
34
+
35
+ def _rake_page[T: BaseModel](
36
+ url: str,
37
+ model: type[T],
38
+ targets: list[ExtractionTarget],
39
+ fetcher: Fetcher,
40
+ client: JevClient | None,
41
+ min_confidence: float,
42
+ on_request: RequestHook | None,
43
+ ) -> RakeResult[T]:
44
+ try:
45
+ html = fetcher.fetch(url)
46
+ except FetchError as e:
47
+ return _build_failed_result(model, url, str(e))
48
+
49
+ page = dom.parse_page(html)
50
+ blocks = collect.find_item_blocks(page, targets)
51
+ numbered: list[tuple[int | None, ItemBlock]] = (
52
+ [(None, blocks[0])] if blocks[0].is_whole_page else list(enumerate(blocks))
53
+ )
54
+
55
+ items: list[RakeItem[T]] = []
56
+ request_ids: list[str] = []
57
+ errors: list[str] = []
58
+ for start in range(0, len(numbered), MAX_ITEMS_PER_REQUEST):
59
+ chunk = numbered[start : start + MAX_ITEMS_PER_REQUEST]
60
+ candidates_by_item = {item: block.candidates for item, block in chunk}
61
+ request = jev.build_jev_request(url, page, targets, candidates_by_item)
62
+ if on_request is not None:
63
+ on_request(url, jev.build_jev_payload(request, MODEL))
64
+ if client is None:
65
+ continue
66
+
67
+ response = JevResponse()
68
+ if request.questions:
69
+ try:
70
+ response = client.ask(request.state, request.questions)
71
+ except JevAuthError:
72
+ raise
73
+ except JevError as e:
74
+ first, last = chunk[0][0], chunk[-1][0]
75
+ errors.append(str(e) if first is None else f"items {first}-{last}: {e}")
76
+ continue
77
+ if response.request_id:
78
+ request_ids.append(response.request_id)
79
+
80
+ for item, _ in chunk:
81
+ prefix = "" if item is None else f"item {item}: "
82
+ outcome = jev.parse_jev_response(request, response, targets, min_confidence, item)
83
+ if outcome.errors:
84
+ errors.extend(prefix + e for e in outcome.errors)
85
+ continue
86
+ try:
87
+ data = model.model_validate(outcome.values)
88
+ except ValidationError as e:
89
+ errors.extend(prefix + m for m in _format_validation_errors(e))
90
+ continue
91
+ items.append(RakeItem[model](data=data, fields=outcome.fields)) # type: ignore[valid-type]
92
+
93
+ return RakeResult[model](url=url, items=items, request_ids=request_ids, errors=errors) # type: ignore[valid-type]
94
+
95
+
96
+ def _rake_page_safely[T: BaseModel](url: str, model: type[T], *args: Any) -> RakeResult[T]:
97
+ try:
98
+ return _rake_page(url, model, *args)
99
+ except JevAuthError:
100
+ raise
101
+ except Exception as e: # one bad page must not stop the others
102
+ return _build_failed_result(model, url, f"{type(e).__name__}: {e}")
103
+
104
+
105
+ def rake[T: BaseModel](
106
+ urls: Sequence[str],
107
+ model: type[T],
108
+ *,
109
+ min_confidence: float = DEFAULT_MIN_CONFIDENCE,
110
+ fetcher: Fetcher | None = None,
111
+ client: JevClient | None = None,
112
+ dry_run: bool = False,
113
+ on_request: RequestHook | None = None,
114
+ ) -> list[RakeResult[T]]:
115
+ """Extract records of ``model`` from each URL, one page at a time, in the order of ``urls``.
116
+
117
+ Raises :class:`JevAuthError` if the API key is missing or rejected.
118
+ """
119
+ targets = build_targets_from_model(model)
120
+
121
+ created: list[HttpxFetcher | TypeSafeJevClient] = []
122
+ if fetcher is None:
123
+ fetcher = HttpxFetcher()
124
+ created.append(fetcher)
125
+ if dry_run:
126
+ client = None
127
+ elif client is None:
128
+ client = TypeSafeJevClient()
129
+ created.append(client)
130
+
131
+ try:
132
+ return [
133
+ _rake_page_safely(url, model, targets, fetcher, client, min_confidence, on_request)
134
+ for url in urls
135
+ ]
136
+ finally:
137
+ for resource in created:
138
+ resource.close()
@@ -0,0 +1,81 @@
1
+ """Send the question to Jev (TypeSafe AI API) and return its answer.
2
+
3
+ Wraps the official SDK so the rest of jevrake does not depend on SDK types or exceptions.
4
+ """
5
+
6
+ from __future__ import annotations
7
+
8
+ import os
9
+ from collections.abc import Mapping
10
+ from typing import TYPE_CHECKING, Any, Protocol
11
+
12
+ import typesafe_sdk as ts
13
+
14
+ from .exceptions import JevAuthError, JevError, MissingAPIKeyError
15
+ from .models import JevChoiceAnswer, JevResponse
16
+
17
+ if TYPE_CHECKING:
18
+ import httpx2
19
+
20
+ API_KEY_ENV = "TYPESAFE_API_KEY"
21
+ MODEL = "jev-latest"
22
+ TIMEOUT = 30.0
23
+ MAX_RETRIES = 3
24
+
25
+ MISSING_API_KEY_MESSAGE = f"""\
26
+ {API_KEY_ENV} is not set.
27
+ Create an API key at https://console.typesafe.ai and set it with one of:
28
+ export {API_KEY_ENV}=<your API key>
29
+ cp .envrc.example .envrc # fill in the key, then run `direnv allow`"""
30
+
31
+
32
+ class JevClient(Protocol):
33
+ def ask(self, state: Mapping[str, Any], questions: Mapping[str, Any]) -> JevResponse: ...
34
+
35
+
36
+ class TypeSafeJevClient:
37
+ """:class:`JevClient` backed by ``typesafe_sdk.TypeSafeClient``.
38
+
39
+ 429 / 5xx responses are retried with exponential backoff (up to ``MAX_RETRIES`` times) by
40
+ the SDK's retry policy.
41
+ """
42
+
43
+ def __init__(self, *, transport: httpx2.BaseTransport | None = None) -> None:
44
+ api_key = os.environ.get(API_KEY_ENV)
45
+
46
+ if not api_key:
47
+ raise MissingAPIKeyError(MISSING_API_KEY_MESSAGE)
48
+
49
+ self._client = ts.TypeSafeClient(
50
+ api_key=api_key,
51
+ retry=ts.RetryPolicy(max_retries=MAX_RETRIES, timeout=None),
52
+ timeout=TIMEOUT,
53
+ transport=transport,
54
+ )
55
+
56
+ def ask(self, state: Mapping[str, Any], questions: Mapping[str, Any]) -> JevResponse:
57
+ try:
58
+ response = self._client.system_one(state=dict(state), questions=questions, model=MODEL)
59
+ except (ts.TypeSafeAuthenticationError, ts.TypeSafePermissionDeniedError) as e:
60
+ raise JevAuthError(
61
+ f"Jev API rejected the API key (HTTP {e.status}). Check {API_KEY_ENV}."
62
+ ) from None
63
+ except ts.TypeSafeAPITimeoutError:
64
+ raise JevError("Jev API request timed out") from None
65
+ except ts.TypeSafeAPIError as e:
66
+ raise JevError(f"Jev API error (HTTP {e.status}): {e.body}") from None
67
+ except ts.TypeSafeError as e:
68
+ raise JevError(f"Jev API error: {e}") from None
69
+
70
+ return JevResponse(
71
+ choices={
72
+ name: JevChoiceAnswer(
73
+ choice=a.choice, confidence=a.confidence, probabilities=dict(a.probabilities)
74
+ )
75
+ for name, a in response.choices.items()
76
+ },
77
+ request_id=response.request_id,
78
+ )
79
+
80
+ def close(self) -> None:
81
+ self._client.close()
@@ -0,0 +1,210 @@
1
+ """Collect possible values (candidates) for each target from the page.
2
+
3
+ A list page is split into item blocks (one per record); any other page is one block. Text targets
4
+ take titles and text elements; number targets take prices and other numbers. Candidates are
5
+ de-duplicated and, if there are too many, only the highest-scoring ones are kept.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import re
11
+ from collections import Counter
12
+ from collections.abc import Iterator
13
+ from typing import Any
14
+
15
+ from .convert import NUMBER_RE, convert_text, normalize_text
16
+ from .exceptions import CoercionError
17
+ from .models import ExtractionCandidate, ExtractionTarget, ItemBlock, ParsedPage, TextElement
18
+
19
+ MAX_CANDIDATES = 254 # plus "__none__" = 255 choices
20
+ MAX_TEXT_CHARS = 200
21
+ _PREFIX_LABEL_CHARS = 20
22
+ MIN_LIST_ITEMS = 3
23
+ MIN_LIST_TEXT_SHARE = 0.5
24
+ MAX_SHORT_TEXT_CHARS = 40
25
+
26
+
27
+ def _format_tag(element: TextElement) -> str:
28
+ attrs = "".join(
29
+ f" {key}={element.attrs[key]}"
30
+ for key in ("class", "id", "itemprop")
31
+ if element.attrs.get(key)
32
+ )
33
+ return f"<{element.tag}{attrs}>"
34
+
35
+
36
+ def _build_context(element: TextElement, label: str | None = None) -> dict[str, str]:
37
+ context = {"tag": _format_tag(element)}
38
+ label = label or element.label
39
+ if label and label != element.text:
40
+ context["label"] = label
41
+ if element.heading_level:
42
+ context["heading"] = f"h{element.heading_level}"
43
+ return context
44
+
45
+
46
+ def _find_prefix_label(text: str, start: int) -> str | None:
47
+ prefix = text[:start].strip().rstrip("::").strip()
48
+ if 0 < len(prefix) <= _PREFIX_LABEL_CHARS and not re.search(r"\d", prefix[-3:]):
49
+ return prefix
50
+ return None
51
+
52
+
53
+ def _find_text_candidates(page: ParsedPage, block: ItemBlock) -> Iterator[ExtractionCandidate]:
54
+ if block.is_whole_page and page.title:
55
+ yield ExtractionCandidate(
56
+ text=page.title,
57
+ normalized_value=None,
58
+ context={"tag": "<title>", "source": "page title"},
59
+ )
60
+ if block.is_whole_page and page.og_title:
61
+ yield ExtractionCandidate(
62
+ text=page.og_title,
63
+ normalized_value=None,
64
+ context={"tag": "<meta og:title>", "source": "og:title"},
65
+ )
66
+ for element in block.elements:
67
+ if 2 <= len(element.text) <= MAX_TEXT_CHARS:
68
+ yield ExtractionCandidate(
69
+ text=element.text,
70
+ normalized_value=None,
71
+ context=_build_context(element),
72
+ )
73
+
74
+
75
+ def _find_number_candidates(block: ItemBlock) -> Iterator[ExtractionCandidate]:
76
+ for element in block.elements:
77
+ text = normalize_text(element.text)
78
+ for m in NUMBER_RE.finditer(text):
79
+ context = _build_context(element, _find_prefix_label(text, m.start()))
80
+ if m.group("prefix") or m.group("suffix"):
81
+ context["unit"] = (m.group("prefix") or m.group("suffix")).strip()
82
+ yield ExtractionCandidate(
83
+ text=m.group(0).strip(),
84
+ normalized_value=None,
85
+ context=context,
86
+ )
87
+
88
+
89
+ def _split_terms(text: str) -> set[str]:
90
+ """Matching terms: whole words for Latin text, character bigrams for CJK text."""
91
+ s = normalize_text(text).lower()
92
+ terms = {w for w in re.findall(r"[a-z]{3,}", s)}
93
+ for run in re.findall(r"[^\x00-\x7f\s\W]+", s):
94
+ terms |= {run[i : i + 2] for i in range(len(run) - 1)}
95
+ return terms
96
+
97
+
98
+ def _calculate_score(
99
+ candidate: ExtractionCandidate, target: ExtractionTarget, frequency: int
100
+ ) -> float:
101
+ ctx = candidate.context
102
+ score = 0.0
103
+ name_terms = _split_terms(target.name.replace("_", " "))
104
+ wanted = _split_terms(target.description) | name_terms
105
+ label = ctx.get("label")
106
+ if label:
107
+ score += 1.0
108
+ if _split_terms(label) & wanted:
109
+ score += 2.0
110
+ if _split_terms(ctx.get("tag", "").replace("-", " ").replace("_", " ")) & name_terms:
111
+ score += 1.5
112
+ heading = ctx.get("heading")
113
+ if heading and target.kind == "str":
114
+ score += {"h1": 3.0, "h2": 2.0}.get(heading, 1.0)
115
+ if ctx.get("source"):
116
+ score += 3.0
117
+ score += 0.5 * min(frequency - 1, 3)
118
+ if target.kind != "str" and ctx.get("unit"):
119
+ score += 1.0
120
+ if target.kind == "str" and len(candidate.text) > 80:
121
+ score -= 1.0
122
+ return score
123
+
124
+
125
+ def remove_duplicates(candidates: list[ExtractionCandidate]) -> list[ExtractionCandidate]:
126
+ """Keep one candidate per ``normalized_value``: the one with the most context."""
127
+ best: dict[Any, int] = {}
128
+ for i, candidate in enumerate(candidates):
129
+ key = candidate.normalized_value
130
+ if key not in best or len(candidate.context) > len(candidates[best[key]].context):
131
+ best[key] = i
132
+ return [candidates[i] for i in sorted(best.values())]
133
+
134
+
135
+ def keep_top_candidates(candidates: list[ExtractionCandidate]) -> list[ExtractionCandidate]:
136
+ """Keep the ``MAX_CANDIDATES`` highest-scoring candidates, preserving document order."""
137
+ if len(candidates) <= MAX_CANDIDATES:
138
+ return candidates
139
+ ranked = sorted(range(len(candidates)), key=lambda i: -candidates[i].score)
140
+ keep = sorted(ranked[:MAX_CANDIDATES])
141
+ return [candidates[i] for i in keep]
142
+
143
+
144
+ def generate_candidates(
145
+ page: ParsedPage, block: ItemBlock, target: ExtractionTarget
146
+ ) -> list[ExtractionCandidate]:
147
+ converted: list[ExtractionCandidate] = []
148
+ raw_candidates = (
149
+ _find_text_candidates(page, block)
150
+ if target.kind == "str"
151
+ else _find_number_candidates(block)
152
+ )
153
+ for raw in raw_candidates:
154
+ try:
155
+ value = convert_text(raw.text, target.kind)
156
+ except CoercionError:
157
+ continue
158
+ converted.append(raw.model_copy(update={"normalized_value": value}))
159
+
160
+ frequency = Counter(c.normalized_value for c in converted)
161
+ unique = remove_duplicates(converted)
162
+ scored = [
163
+ c.model_copy(update={"score": _calculate_score(c, target, frequency[c.normalized_value])})
164
+ for c in unique
165
+ ]
166
+ return keep_top_candidates(scored)
167
+
168
+
169
+ def _count_text(elements: list[TextElement]) -> int:
170
+ return sum(len(e.text) for e in elements)
171
+
172
+
173
+ def _build_block(
174
+ page: ParsedPage,
175
+ elements: list[TextElement],
176
+ is_whole_page: bool,
177
+ targets: list[ExtractionTarget],
178
+ ) -> ItemBlock:
179
+ """Make a block and collect the candidates of every target in it."""
180
+ block = ItemBlock(elements=elements, is_whole_page=is_whole_page)
181
+ candidates = {t.name: generate_candidates(page, block, t) for t in targets}
182
+ return block.model_copy(update={"candidates": candidates})
183
+
184
+
185
+ def find_item_blocks(page: ParsedPage, targets: list[ExtractionTarget]) -> list[ItemBlock]:
186
+ """Return one block per item on a list page, or the whole page as a single block.
187
+
188
+ Each block comes with the candidates of every target, so they are collected only once.
189
+
190
+ Every block of the chosen group is returned, so an item that misses a value is reported later
191
+ instead of silently disappearing.
192
+ """
193
+ required = [t for t in targets if not t.is_optional]
194
+ page_text = _count_text(page.elements)
195
+ best: list[ItemBlock] = []
196
+ best_items = 0
197
+ for group in page.repeated_groups:
198
+ blocks = [_build_block(page, elements, False, targets) for elements in group]
199
+ items = sum(
200
+ 1
201
+ for block in blocks
202
+ if len(block.elements) >= 2
203
+ and any(len(e.text) <= MAX_SHORT_TEXT_CHARS for e in block.elements)
204
+ and all(block.candidates[t.name] for t in required)
205
+ )
206
+ text_share = _count_text([e for b in blocks for e in b.elements]) / max(page_text, 1)
207
+ is_list = items >= MIN_LIST_ITEMS and text_share >= MIN_LIST_TEXT_SHARE
208
+ if is_list and items > best_items:
209
+ best, best_items = blocks, items
210
+ return best or [_build_block(page, page.elements, True, targets)]
@@ -0,0 +1,78 @@
1
+ """Convert candidate text into the target's type, e.g. ``"¥3,980"`` -> ``3980``.
2
+
3
+ Used to turn page text into values and to drop text that cannot be converted.
4
+ """
5
+
6
+ from __future__ import annotations
7
+
8
+ import re
9
+ import unicodedata
10
+ from collections.abc import Callable
11
+ from decimal import Decimal, InvalidOperation
12
+ from typing import Any
13
+
14
+ from .exceptions import CoercionError
15
+
16
+
17
+ def normalize_text(text: str) -> str:
18
+ """NFKC-normalize (full-width -> half-width) and collapse whitespace."""
19
+ return " ".join(unicodedata.normalize("NFKC", text).split())
20
+
21
+
22
+ CURRENCY_PREFIXES = "¥$€£"
23
+ CURRENCY_SUFFIXES = ("円", "ドル", "ユーロ", "USD", "JPY", "EUR", "yen")
24
+
25
+ NUMBER_RE = re.compile(
26
+ r"(?<![\w.,/:\-])"
27
+ r"(?P<prefix>[¥$€£]\s?)?"
28
+ r"(?P<number>-?(?:\d{1,3}(?:,\d{3})+|\d+)(?:\.\d+)?)"
29
+ r"(?P<suffix>\s?(?:円|ドル|ユーロ|USD|JPY|EUR|yen))?"
30
+ r"(?!\d|,\d|[.:/\-]\d|[年月日時分秒])"
31
+ )
32
+
33
+
34
+ _PLAIN_NUMBER_RE = re.compile(r"-?(?:\d{1,3}(?:,\d{3})+|\d+)(?:\.\d+)?")
35
+
36
+
37
+ def parse_number(text: str) -> Decimal:
38
+ s = normalize_text(text).replace("\u2212", "-")
39
+ s = s.strip(CURRENCY_PREFIXES + " ")
40
+ for suffix in CURRENCY_SUFFIXES:
41
+ s = s.removesuffix(suffix).strip()
42
+ if not _PLAIN_NUMBER_RE.fullmatch(s):
43
+ raise CoercionError(f"not a number: {text!r}")
44
+ try:
45
+ return Decimal(s.replace(",", ""))
46
+ except InvalidOperation:
47
+ raise CoercionError(f"not a number: {text!r}") from None
48
+
49
+
50
+ def _convert_to_int(text: str) -> int:
51
+ value = parse_number(text)
52
+ if value != value.to_integral_value():
53
+ raise CoercionError(f"not an integer: {text!r}")
54
+ return int(value)
55
+
56
+
57
+ def _convert_to_str(text: str) -> str:
58
+ s = " ".join(text.split())
59
+ if not s:
60
+ raise CoercionError("empty string")
61
+ return s
62
+
63
+
64
+ _CONVERTERS: dict[str, Callable[[str], Any]] = {
65
+ "str": _convert_to_str,
66
+ "int": _convert_to_int,
67
+ "float": lambda t: float(parse_number(t)),
68
+ "decimal": parse_number,
69
+ }
70
+
71
+
72
+ def convert_text(text: str, kind: str) -> Any:
73
+ """Convert ``text`` to the Python value for ``kind``. Raises :class:`CoercionError`."""
74
+ try:
75
+ converter = _CONVERTERS[kind]
76
+ except KeyError:
77
+ raise CoercionError(f"unsupported type: {kind}") from None
78
+ return converter(text)