jevrake 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- jevrake/__init__.py +13 -0
- jevrake/_internal/__init__.py +1 -0
- jevrake/_internal/api.py +138 -0
- jevrake/_internal/client.py +81 -0
- jevrake/_internal/collect.py +210 -0
- jevrake/_internal/convert.py +78 -0
- jevrake/_internal/dom.py +200 -0
- jevrake/_internal/exceptions.py +25 -0
- jevrake/_internal/fetch.py +75 -0
- jevrake/_internal/jev.py +181 -0
- jevrake/_internal/models.py +131 -0
- jevrake/_internal/schema.py +105 -0
- jevrake/_internal/version.py +2 -0
- jevrake/cli.py +127 -0
- jevrake/py.typed +0 -0
- jevrake-0.1.0.dist-info/METADATA +93 -0
- jevrake-0.1.0.dist-info/RECORD +20 -0
- jevrake-0.1.0.dist-info/WHEEL +4 -0
- jevrake-0.1.0.dist-info/entry_points.txt +3 -0
- jevrake-0.1.0.dist-info/licenses/LICENSE +21 -0
jevrake/__init__.py
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
"""jevrake: selector-free web scraping with TypeSafe AI's Jev."""
|
|
2
|
+
|
|
3
|
+
from ._internal.api import rake
|
|
4
|
+
from ._internal.models import ExtractedValue, RakeItem, RakeResult
|
|
5
|
+
from ._internal.version import VERSION as __version__
|
|
6
|
+
|
|
7
|
+
__all__ = [
|
|
8
|
+
"ExtractedValue",
|
|
9
|
+
"RakeItem",
|
|
10
|
+
"RakeResult",
|
|
11
|
+
"__version__",
|
|
12
|
+
"rake",
|
|
13
|
+
]
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""Internal implementation. Not part of the public API; may change at any time."""
|
jevrake/_internal/api.py
ADDED
|
@@ -0,0 +1,138 @@
|
|
|
1
|
+
"""Extract records from each page with ``rake()``.
|
|
2
|
+
|
|
3
|
+
For each URL: download the page, split it into items (one for a single-record page), collect
|
|
4
|
+
candidate values, ask Jev to choose one per field, and validate each item with the user's model.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from collections.abc import Callable, Sequence
|
|
10
|
+
from typing import Any
|
|
11
|
+
|
|
12
|
+
from pydantic import BaseModel, ValidationError
|
|
13
|
+
|
|
14
|
+
from . import collect, dom, jev
|
|
15
|
+
from .client import MODEL, JevClient, TypeSafeJevClient
|
|
16
|
+
from .exceptions import FetchError, JevAuthError, JevError
|
|
17
|
+
from .fetch import Fetcher, HttpxFetcher
|
|
18
|
+
from .models import ExtractionTarget, ItemBlock, JevResponse, RakeItem, RakeResult
|
|
19
|
+
from .schema import build_targets_from_model
|
|
20
|
+
|
|
21
|
+
DEFAULT_MIN_CONFIDENCE = 0.9
|
|
22
|
+
MAX_ITEMS_PER_REQUEST = 20
|
|
23
|
+
|
|
24
|
+
RequestHook = Callable[[str, dict[str, Any]], None]
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def _build_failed_result[T: BaseModel](model: type[T], url: str, error: str) -> RakeResult[T]:
|
|
28
|
+
return RakeResult[model](url=url, items=[], request_ids=[], errors=[error]) # type: ignore[valid-type]
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def _format_validation_errors(error: ValidationError) -> list[str]:
|
|
32
|
+
return [f"{'.'.join(str(p) for p in err['loc'])}: {err['msg']}" for err in error.errors()]
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _rake_page[T: BaseModel](
|
|
36
|
+
url: str,
|
|
37
|
+
model: type[T],
|
|
38
|
+
targets: list[ExtractionTarget],
|
|
39
|
+
fetcher: Fetcher,
|
|
40
|
+
client: JevClient | None,
|
|
41
|
+
min_confidence: float,
|
|
42
|
+
on_request: RequestHook | None,
|
|
43
|
+
) -> RakeResult[T]:
|
|
44
|
+
try:
|
|
45
|
+
html = fetcher.fetch(url)
|
|
46
|
+
except FetchError as e:
|
|
47
|
+
return _build_failed_result(model, url, str(e))
|
|
48
|
+
|
|
49
|
+
page = dom.parse_page(html)
|
|
50
|
+
blocks = collect.find_item_blocks(page, targets)
|
|
51
|
+
numbered: list[tuple[int | None, ItemBlock]] = (
|
|
52
|
+
[(None, blocks[0])] if blocks[0].is_whole_page else list(enumerate(blocks))
|
|
53
|
+
)
|
|
54
|
+
|
|
55
|
+
items: list[RakeItem[T]] = []
|
|
56
|
+
request_ids: list[str] = []
|
|
57
|
+
errors: list[str] = []
|
|
58
|
+
for start in range(0, len(numbered), MAX_ITEMS_PER_REQUEST):
|
|
59
|
+
chunk = numbered[start : start + MAX_ITEMS_PER_REQUEST]
|
|
60
|
+
candidates_by_item = {item: block.candidates for item, block in chunk}
|
|
61
|
+
request = jev.build_jev_request(url, page, targets, candidates_by_item)
|
|
62
|
+
if on_request is not None:
|
|
63
|
+
on_request(url, jev.build_jev_payload(request, MODEL))
|
|
64
|
+
if client is None:
|
|
65
|
+
continue
|
|
66
|
+
|
|
67
|
+
response = JevResponse()
|
|
68
|
+
if request.questions:
|
|
69
|
+
try:
|
|
70
|
+
response = client.ask(request.state, request.questions)
|
|
71
|
+
except JevAuthError:
|
|
72
|
+
raise
|
|
73
|
+
except JevError as e:
|
|
74
|
+
first, last = chunk[0][0], chunk[-1][0]
|
|
75
|
+
errors.append(str(e) if first is None else f"items {first}-{last}: {e}")
|
|
76
|
+
continue
|
|
77
|
+
if response.request_id:
|
|
78
|
+
request_ids.append(response.request_id)
|
|
79
|
+
|
|
80
|
+
for item, _ in chunk:
|
|
81
|
+
prefix = "" if item is None else f"item {item}: "
|
|
82
|
+
outcome = jev.parse_jev_response(request, response, targets, min_confidence, item)
|
|
83
|
+
if outcome.errors:
|
|
84
|
+
errors.extend(prefix + e for e in outcome.errors)
|
|
85
|
+
continue
|
|
86
|
+
try:
|
|
87
|
+
data = model.model_validate(outcome.values)
|
|
88
|
+
except ValidationError as e:
|
|
89
|
+
errors.extend(prefix + m for m in _format_validation_errors(e))
|
|
90
|
+
continue
|
|
91
|
+
items.append(RakeItem[model](data=data, fields=outcome.fields)) # type: ignore[valid-type]
|
|
92
|
+
|
|
93
|
+
return RakeResult[model](url=url, items=items, request_ids=request_ids, errors=errors) # type: ignore[valid-type]
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def _rake_page_safely[T: BaseModel](url: str, model: type[T], *args: Any) -> RakeResult[T]:
|
|
97
|
+
try:
|
|
98
|
+
return _rake_page(url, model, *args)
|
|
99
|
+
except JevAuthError:
|
|
100
|
+
raise
|
|
101
|
+
except Exception as e: # one bad page must not stop the others
|
|
102
|
+
return _build_failed_result(model, url, f"{type(e).__name__}: {e}")
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def rake[T: BaseModel](
|
|
106
|
+
urls: Sequence[str],
|
|
107
|
+
model: type[T],
|
|
108
|
+
*,
|
|
109
|
+
min_confidence: float = DEFAULT_MIN_CONFIDENCE,
|
|
110
|
+
fetcher: Fetcher | None = None,
|
|
111
|
+
client: JevClient | None = None,
|
|
112
|
+
dry_run: bool = False,
|
|
113
|
+
on_request: RequestHook | None = None,
|
|
114
|
+
) -> list[RakeResult[T]]:
|
|
115
|
+
"""Extract records of ``model`` from each URL, one page at a time, in the order of ``urls``.
|
|
116
|
+
|
|
117
|
+
Raises :class:`JevAuthError` if the API key is missing or rejected.
|
|
118
|
+
"""
|
|
119
|
+
targets = build_targets_from_model(model)
|
|
120
|
+
|
|
121
|
+
created: list[HttpxFetcher | TypeSafeJevClient] = []
|
|
122
|
+
if fetcher is None:
|
|
123
|
+
fetcher = HttpxFetcher()
|
|
124
|
+
created.append(fetcher)
|
|
125
|
+
if dry_run:
|
|
126
|
+
client = None
|
|
127
|
+
elif client is None:
|
|
128
|
+
client = TypeSafeJevClient()
|
|
129
|
+
created.append(client)
|
|
130
|
+
|
|
131
|
+
try:
|
|
132
|
+
return [
|
|
133
|
+
_rake_page_safely(url, model, targets, fetcher, client, min_confidence, on_request)
|
|
134
|
+
for url in urls
|
|
135
|
+
]
|
|
136
|
+
finally:
|
|
137
|
+
for resource in created:
|
|
138
|
+
resource.close()
|
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
"""Send the question to Jev (TypeSafe AI API) and return its answer.
|
|
2
|
+
|
|
3
|
+
Wraps the official SDK so the rest of jevrake does not depend on SDK types or exceptions.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
import os
|
|
9
|
+
from collections.abc import Mapping
|
|
10
|
+
from typing import TYPE_CHECKING, Any, Protocol
|
|
11
|
+
|
|
12
|
+
import typesafe_sdk as ts
|
|
13
|
+
|
|
14
|
+
from .exceptions import JevAuthError, JevError, MissingAPIKeyError
|
|
15
|
+
from .models import JevChoiceAnswer, JevResponse
|
|
16
|
+
|
|
17
|
+
if TYPE_CHECKING:
|
|
18
|
+
import httpx2
|
|
19
|
+
|
|
20
|
+
API_KEY_ENV = "TYPESAFE_API_KEY"
|
|
21
|
+
MODEL = "jev-latest"
|
|
22
|
+
TIMEOUT = 30.0
|
|
23
|
+
MAX_RETRIES = 3
|
|
24
|
+
|
|
25
|
+
MISSING_API_KEY_MESSAGE = f"""\
|
|
26
|
+
{API_KEY_ENV} is not set.
|
|
27
|
+
Create an API key at https://console.typesafe.ai and set it with one of:
|
|
28
|
+
export {API_KEY_ENV}=<your API key>
|
|
29
|
+
cp .envrc.example .envrc # fill in the key, then run `direnv allow`"""
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
class JevClient(Protocol):
|
|
33
|
+
def ask(self, state: Mapping[str, Any], questions: Mapping[str, Any]) -> JevResponse: ...
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
class TypeSafeJevClient:
|
|
37
|
+
""":class:`JevClient` backed by ``typesafe_sdk.TypeSafeClient``.
|
|
38
|
+
|
|
39
|
+
429 / 5xx responses are retried with exponential backoff (up to ``MAX_RETRIES`` times) by
|
|
40
|
+
the SDK's retry policy.
|
|
41
|
+
"""
|
|
42
|
+
|
|
43
|
+
def __init__(self, *, transport: httpx2.BaseTransport | None = None) -> None:
|
|
44
|
+
api_key = os.environ.get(API_KEY_ENV)
|
|
45
|
+
|
|
46
|
+
if not api_key:
|
|
47
|
+
raise MissingAPIKeyError(MISSING_API_KEY_MESSAGE)
|
|
48
|
+
|
|
49
|
+
self._client = ts.TypeSafeClient(
|
|
50
|
+
api_key=api_key,
|
|
51
|
+
retry=ts.RetryPolicy(max_retries=MAX_RETRIES, timeout=None),
|
|
52
|
+
timeout=TIMEOUT,
|
|
53
|
+
transport=transport,
|
|
54
|
+
)
|
|
55
|
+
|
|
56
|
+
def ask(self, state: Mapping[str, Any], questions: Mapping[str, Any]) -> JevResponse:
|
|
57
|
+
try:
|
|
58
|
+
response = self._client.system_one(state=dict(state), questions=questions, model=MODEL)
|
|
59
|
+
except (ts.TypeSafeAuthenticationError, ts.TypeSafePermissionDeniedError) as e:
|
|
60
|
+
raise JevAuthError(
|
|
61
|
+
f"Jev API rejected the API key (HTTP {e.status}). Check {API_KEY_ENV}."
|
|
62
|
+
) from None
|
|
63
|
+
except ts.TypeSafeAPITimeoutError:
|
|
64
|
+
raise JevError("Jev API request timed out") from None
|
|
65
|
+
except ts.TypeSafeAPIError as e:
|
|
66
|
+
raise JevError(f"Jev API error (HTTP {e.status}): {e.body}") from None
|
|
67
|
+
except ts.TypeSafeError as e:
|
|
68
|
+
raise JevError(f"Jev API error: {e}") from None
|
|
69
|
+
|
|
70
|
+
return JevResponse(
|
|
71
|
+
choices={
|
|
72
|
+
name: JevChoiceAnswer(
|
|
73
|
+
choice=a.choice, confidence=a.confidence, probabilities=dict(a.probabilities)
|
|
74
|
+
)
|
|
75
|
+
for name, a in response.choices.items()
|
|
76
|
+
},
|
|
77
|
+
request_id=response.request_id,
|
|
78
|
+
)
|
|
79
|
+
|
|
80
|
+
def close(self) -> None:
|
|
81
|
+
self._client.close()
|
|
@@ -0,0 +1,210 @@
|
|
|
1
|
+
"""Collect possible values (candidates) for each target from the page.
|
|
2
|
+
|
|
3
|
+
A list page is split into item blocks (one per record); any other page is one block. Text targets
|
|
4
|
+
take titles and text elements; number targets take prices and other numbers. Candidates are
|
|
5
|
+
de-duplicated and, if there are too many, only the highest-scoring ones are kept.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import re
|
|
11
|
+
from collections import Counter
|
|
12
|
+
from collections.abc import Iterator
|
|
13
|
+
from typing import Any
|
|
14
|
+
|
|
15
|
+
from .convert import NUMBER_RE, convert_text, normalize_text
|
|
16
|
+
from .exceptions import CoercionError
|
|
17
|
+
from .models import ExtractionCandidate, ExtractionTarget, ItemBlock, ParsedPage, TextElement
|
|
18
|
+
|
|
19
|
+
MAX_CANDIDATES = 254 # plus "__none__" = 255 choices
|
|
20
|
+
MAX_TEXT_CHARS = 200
|
|
21
|
+
_PREFIX_LABEL_CHARS = 20
|
|
22
|
+
MIN_LIST_ITEMS = 3
|
|
23
|
+
MIN_LIST_TEXT_SHARE = 0.5
|
|
24
|
+
MAX_SHORT_TEXT_CHARS = 40
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def _format_tag(element: TextElement) -> str:
|
|
28
|
+
attrs = "".join(
|
|
29
|
+
f" {key}={element.attrs[key]}"
|
|
30
|
+
for key in ("class", "id", "itemprop")
|
|
31
|
+
if element.attrs.get(key)
|
|
32
|
+
)
|
|
33
|
+
return f"<{element.tag}{attrs}>"
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def _build_context(element: TextElement, label: str | None = None) -> dict[str, str]:
|
|
37
|
+
context = {"tag": _format_tag(element)}
|
|
38
|
+
label = label or element.label
|
|
39
|
+
if label and label != element.text:
|
|
40
|
+
context["label"] = label
|
|
41
|
+
if element.heading_level:
|
|
42
|
+
context["heading"] = f"h{element.heading_level}"
|
|
43
|
+
return context
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def _find_prefix_label(text: str, start: int) -> str | None:
|
|
47
|
+
prefix = text[:start].strip().rstrip("::").strip()
|
|
48
|
+
if 0 < len(prefix) <= _PREFIX_LABEL_CHARS and not re.search(r"\d", prefix[-3:]):
|
|
49
|
+
return prefix
|
|
50
|
+
return None
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def _find_text_candidates(page: ParsedPage, block: ItemBlock) -> Iterator[ExtractionCandidate]:
|
|
54
|
+
if block.is_whole_page and page.title:
|
|
55
|
+
yield ExtractionCandidate(
|
|
56
|
+
text=page.title,
|
|
57
|
+
normalized_value=None,
|
|
58
|
+
context={"tag": "<title>", "source": "page title"},
|
|
59
|
+
)
|
|
60
|
+
if block.is_whole_page and page.og_title:
|
|
61
|
+
yield ExtractionCandidate(
|
|
62
|
+
text=page.og_title,
|
|
63
|
+
normalized_value=None,
|
|
64
|
+
context={"tag": "<meta og:title>", "source": "og:title"},
|
|
65
|
+
)
|
|
66
|
+
for element in block.elements:
|
|
67
|
+
if 2 <= len(element.text) <= MAX_TEXT_CHARS:
|
|
68
|
+
yield ExtractionCandidate(
|
|
69
|
+
text=element.text,
|
|
70
|
+
normalized_value=None,
|
|
71
|
+
context=_build_context(element),
|
|
72
|
+
)
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def _find_number_candidates(block: ItemBlock) -> Iterator[ExtractionCandidate]:
|
|
76
|
+
for element in block.elements:
|
|
77
|
+
text = normalize_text(element.text)
|
|
78
|
+
for m in NUMBER_RE.finditer(text):
|
|
79
|
+
context = _build_context(element, _find_prefix_label(text, m.start()))
|
|
80
|
+
if m.group("prefix") or m.group("suffix"):
|
|
81
|
+
context["unit"] = (m.group("prefix") or m.group("suffix")).strip()
|
|
82
|
+
yield ExtractionCandidate(
|
|
83
|
+
text=m.group(0).strip(),
|
|
84
|
+
normalized_value=None,
|
|
85
|
+
context=context,
|
|
86
|
+
)
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def _split_terms(text: str) -> set[str]:
|
|
90
|
+
"""Matching terms: whole words for Latin text, character bigrams for CJK text."""
|
|
91
|
+
s = normalize_text(text).lower()
|
|
92
|
+
terms = {w for w in re.findall(r"[a-z]{3,}", s)}
|
|
93
|
+
for run in re.findall(r"[^\x00-\x7f\s\W]+", s):
|
|
94
|
+
terms |= {run[i : i + 2] for i in range(len(run) - 1)}
|
|
95
|
+
return terms
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def _calculate_score(
|
|
99
|
+
candidate: ExtractionCandidate, target: ExtractionTarget, frequency: int
|
|
100
|
+
) -> float:
|
|
101
|
+
ctx = candidate.context
|
|
102
|
+
score = 0.0
|
|
103
|
+
name_terms = _split_terms(target.name.replace("_", " "))
|
|
104
|
+
wanted = _split_terms(target.description) | name_terms
|
|
105
|
+
label = ctx.get("label")
|
|
106
|
+
if label:
|
|
107
|
+
score += 1.0
|
|
108
|
+
if _split_terms(label) & wanted:
|
|
109
|
+
score += 2.0
|
|
110
|
+
if _split_terms(ctx.get("tag", "").replace("-", " ").replace("_", " ")) & name_terms:
|
|
111
|
+
score += 1.5
|
|
112
|
+
heading = ctx.get("heading")
|
|
113
|
+
if heading and target.kind == "str":
|
|
114
|
+
score += {"h1": 3.0, "h2": 2.0}.get(heading, 1.0)
|
|
115
|
+
if ctx.get("source"):
|
|
116
|
+
score += 3.0
|
|
117
|
+
score += 0.5 * min(frequency - 1, 3)
|
|
118
|
+
if target.kind != "str" and ctx.get("unit"):
|
|
119
|
+
score += 1.0
|
|
120
|
+
if target.kind == "str" and len(candidate.text) > 80:
|
|
121
|
+
score -= 1.0
|
|
122
|
+
return score
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def remove_duplicates(candidates: list[ExtractionCandidate]) -> list[ExtractionCandidate]:
|
|
126
|
+
"""Keep one candidate per ``normalized_value``: the one with the most context."""
|
|
127
|
+
best: dict[Any, int] = {}
|
|
128
|
+
for i, candidate in enumerate(candidates):
|
|
129
|
+
key = candidate.normalized_value
|
|
130
|
+
if key not in best or len(candidate.context) > len(candidates[best[key]].context):
|
|
131
|
+
best[key] = i
|
|
132
|
+
return [candidates[i] for i in sorted(best.values())]
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def keep_top_candidates(candidates: list[ExtractionCandidate]) -> list[ExtractionCandidate]:
|
|
136
|
+
"""Keep the ``MAX_CANDIDATES`` highest-scoring candidates, preserving document order."""
|
|
137
|
+
if len(candidates) <= MAX_CANDIDATES:
|
|
138
|
+
return candidates
|
|
139
|
+
ranked = sorted(range(len(candidates)), key=lambda i: -candidates[i].score)
|
|
140
|
+
keep = sorted(ranked[:MAX_CANDIDATES])
|
|
141
|
+
return [candidates[i] for i in keep]
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
def generate_candidates(
|
|
145
|
+
page: ParsedPage, block: ItemBlock, target: ExtractionTarget
|
|
146
|
+
) -> list[ExtractionCandidate]:
|
|
147
|
+
converted: list[ExtractionCandidate] = []
|
|
148
|
+
raw_candidates = (
|
|
149
|
+
_find_text_candidates(page, block)
|
|
150
|
+
if target.kind == "str"
|
|
151
|
+
else _find_number_candidates(block)
|
|
152
|
+
)
|
|
153
|
+
for raw in raw_candidates:
|
|
154
|
+
try:
|
|
155
|
+
value = convert_text(raw.text, target.kind)
|
|
156
|
+
except CoercionError:
|
|
157
|
+
continue
|
|
158
|
+
converted.append(raw.model_copy(update={"normalized_value": value}))
|
|
159
|
+
|
|
160
|
+
frequency = Counter(c.normalized_value for c in converted)
|
|
161
|
+
unique = remove_duplicates(converted)
|
|
162
|
+
scored = [
|
|
163
|
+
c.model_copy(update={"score": _calculate_score(c, target, frequency[c.normalized_value])})
|
|
164
|
+
for c in unique
|
|
165
|
+
]
|
|
166
|
+
return keep_top_candidates(scored)
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
def _count_text(elements: list[TextElement]) -> int:
|
|
170
|
+
return sum(len(e.text) for e in elements)
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
def _build_block(
|
|
174
|
+
page: ParsedPage,
|
|
175
|
+
elements: list[TextElement],
|
|
176
|
+
is_whole_page: bool,
|
|
177
|
+
targets: list[ExtractionTarget],
|
|
178
|
+
) -> ItemBlock:
|
|
179
|
+
"""Make a block and collect the candidates of every target in it."""
|
|
180
|
+
block = ItemBlock(elements=elements, is_whole_page=is_whole_page)
|
|
181
|
+
candidates = {t.name: generate_candidates(page, block, t) for t in targets}
|
|
182
|
+
return block.model_copy(update={"candidates": candidates})
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
def find_item_blocks(page: ParsedPage, targets: list[ExtractionTarget]) -> list[ItemBlock]:
|
|
186
|
+
"""Return one block per item on a list page, or the whole page as a single block.
|
|
187
|
+
|
|
188
|
+
Each block comes with the candidates of every target, so they are collected only once.
|
|
189
|
+
|
|
190
|
+
Every block of the chosen group is returned, so an item that misses a value is reported later
|
|
191
|
+
instead of silently disappearing.
|
|
192
|
+
"""
|
|
193
|
+
required = [t for t in targets if not t.is_optional]
|
|
194
|
+
page_text = _count_text(page.elements)
|
|
195
|
+
best: list[ItemBlock] = []
|
|
196
|
+
best_items = 0
|
|
197
|
+
for group in page.repeated_groups:
|
|
198
|
+
blocks = [_build_block(page, elements, False, targets) for elements in group]
|
|
199
|
+
items = sum(
|
|
200
|
+
1
|
|
201
|
+
for block in blocks
|
|
202
|
+
if len(block.elements) >= 2
|
|
203
|
+
and any(len(e.text) <= MAX_SHORT_TEXT_CHARS for e in block.elements)
|
|
204
|
+
and all(block.candidates[t.name] for t in required)
|
|
205
|
+
)
|
|
206
|
+
text_share = _count_text([e for b in blocks for e in b.elements]) / max(page_text, 1)
|
|
207
|
+
is_list = items >= MIN_LIST_ITEMS and text_share >= MIN_LIST_TEXT_SHARE
|
|
208
|
+
if is_list and items > best_items:
|
|
209
|
+
best, best_items = blocks, items
|
|
210
|
+
return best or [_build_block(page, page.elements, True, targets)]
|
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
"""Convert candidate text into the target's type, e.g. ``"¥3,980"`` -> ``3980``.
|
|
2
|
+
|
|
3
|
+
Used to turn page text into values and to drop text that cannot be converted.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
import re
|
|
9
|
+
import unicodedata
|
|
10
|
+
from collections.abc import Callable
|
|
11
|
+
from decimal import Decimal, InvalidOperation
|
|
12
|
+
from typing import Any
|
|
13
|
+
|
|
14
|
+
from .exceptions import CoercionError
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def normalize_text(text: str) -> str:
|
|
18
|
+
"""NFKC-normalize (full-width -> half-width) and collapse whitespace."""
|
|
19
|
+
return " ".join(unicodedata.normalize("NFKC", text).split())
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
CURRENCY_PREFIXES = "¥$€£"
|
|
23
|
+
CURRENCY_SUFFIXES = ("円", "ドル", "ユーロ", "USD", "JPY", "EUR", "yen")
|
|
24
|
+
|
|
25
|
+
NUMBER_RE = re.compile(
|
|
26
|
+
r"(?<![\w.,/:\-])"
|
|
27
|
+
r"(?P<prefix>[¥$€£]\s?)?"
|
|
28
|
+
r"(?P<number>-?(?:\d{1,3}(?:,\d{3})+|\d+)(?:\.\d+)?)"
|
|
29
|
+
r"(?P<suffix>\s?(?:円|ドル|ユーロ|USD|JPY|EUR|yen))?"
|
|
30
|
+
r"(?!\d|,\d|[.:/\-]\d|[年月日時分秒])"
|
|
31
|
+
)
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
_PLAIN_NUMBER_RE = re.compile(r"-?(?:\d{1,3}(?:,\d{3})+|\d+)(?:\.\d+)?")
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def parse_number(text: str) -> Decimal:
|
|
38
|
+
s = normalize_text(text).replace("\u2212", "-")
|
|
39
|
+
s = s.strip(CURRENCY_PREFIXES + " ")
|
|
40
|
+
for suffix in CURRENCY_SUFFIXES:
|
|
41
|
+
s = s.removesuffix(suffix).strip()
|
|
42
|
+
if not _PLAIN_NUMBER_RE.fullmatch(s):
|
|
43
|
+
raise CoercionError(f"not a number: {text!r}")
|
|
44
|
+
try:
|
|
45
|
+
return Decimal(s.replace(",", ""))
|
|
46
|
+
except InvalidOperation:
|
|
47
|
+
raise CoercionError(f"not a number: {text!r}") from None
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def _convert_to_int(text: str) -> int:
|
|
51
|
+
value = parse_number(text)
|
|
52
|
+
if value != value.to_integral_value():
|
|
53
|
+
raise CoercionError(f"not an integer: {text!r}")
|
|
54
|
+
return int(value)
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def _convert_to_str(text: str) -> str:
|
|
58
|
+
s = " ".join(text.split())
|
|
59
|
+
if not s:
|
|
60
|
+
raise CoercionError("empty string")
|
|
61
|
+
return s
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
_CONVERTERS: dict[str, Callable[[str], Any]] = {
|
|
65
|
+
"str": _convert_to_str,
|
|
66
|
+
"int": _convert_to_int,
|
|
67
|
+
"float": lambda t: float(parse_number(t)),
|
|
68
|
+
"decimal": parse_number,
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def convert_text(text: str, kind: str) -> Any:
|
|
73
|
+
"""Convert ``text`` to the Python value for ``kind``. Raises :class:`CoercionError`."""
|
|
74
|
+
try:
|
|
75
|
+
converter = _CONVERTERS[kind]
|
|
76
|
+
except KeyError:
|
|
77
|
+
raise CoercionError(f"unsupported type: {kind}") from None
|
|
78
|
+
return converter(text)
|