modelspec-dev 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- api/__init__.py +0 -0
- api/class_fit.py +334 -0
- api/classes.py +557 -0
- api/ranking/__init__.py +12 -0
- api/ranking/engine.py +1943 -0
- cli/__init__.py +0 -0
- cli/modelspec/__init__.py +0 -0
- cli/modelspec/cli.py +1819 -0
- cli/modelspec/commands/__init__.py +0 -0
- cli/modelspec/decide_cmd.py +333 -0
- cli/modelspec/offline.py +623 -0
- cli/modelspec/snapshot.py +698 -0
- cli/modelspec/snapshot_build_cmd.py +49 -0
- cli/modelspec/verify_cmd.py +125 -0
- cli/modelspec/vocab_cmd.py +204 -0
- cli/modelspec/vocabulary_cache.py +54 -0
- decision/__init__.py +13 -0
- decision/capability.py +872 -0
- decision/computed.py +125 -0
- decision/contract.py +1575 -0
- decision/engine.py +238 -0
- decision/excluded.py +34 -0
- decision/explain.py +908 -0
- decision/filter.py +796 -0
- decision/model.py +438 -0
- decision/normalise.py +604 -0
- decision/optimise.py +320 -0
- decision/registry.py +717 -0
- decision/relax.py +132 -0
- decision/resolve.py +111 -0
- decision/schema.py +21 -0
- decision/snapshot.py +1483 -0
- decision/sources.py +544 -0
- decision/templates.py +134 -0
- decision/verify.py +1745 -0
- decision/vocabulary.py +433 -0
- modelspec_dev-0.1.0.dist-info/METADATA +101 -0
- modelspec_dev-0.1.0.dist-info/RECORD +63 -0
- modelspec_dev-0.1.0.dist-info/WHEEL +4 -0
- modelspec_dev-0.1.0.dist-info/entry_points.txt +2 -0
- modelspec_dev-0.1.0.dist-info/licenses/LICENSE +43 -0
- modelspec_dev-0.1.0.dist-info/licenses/LICENSE-DATA +428 -0
- pipeline/__init__.py +0 -0
- pipeline/class_export.py +172 -0
- pipeline/hardware.py +434 -0
- pipeline/hosts.py +247 -0
- pipeline/load.py +224 -0
- pipeline/ranking.py +551 -0
- registry/domains.yaml +130 -0
- registry/facets.yaml +888 -0
- registry/harnesses.yaml +79 -0
- registry/providers.yaml +354 -0
- registry/sources.yaml +3059 -0
- registry/templates.yaml +166 -0
- schema/__init__.py +0 -0
- schema/applicability.py +147 -0
- schema/benchmark.py +175 -0
- schema/benchmark_eligibility.py +304 -0
- schema/card.py +1463 -0
- schema/enrichment.py +162 -0
- schema/enums.py +327 -0
- schema/graph.py +406 -0
- schema/suppliers.py +72 -0
decision/normalise.py
ADDED
|
@@ -0,0 +1,604 @@
|
|
|
1
|
+
"""Normalise a fetched page into stable text, and select its cited regions.
|
|
2
|
+
|
|
3
|
+
Change detection compares fingerprints of normalised text, so everything a page
|
|
4
|
+
changes without changing what it says must normalise away: scripts, styles,
|
|
5
|
+
navigation, the site header and footer, cookie banners, ads, timestamps and
|
|
6
|
+
relative dates ("3 days ago"), tracking query strings, attribute order and
|
|
7
|
+
whitespace. The output is text only, one block per line and table cells joined
|
|
8
|
+
by `` | ``, so markup churn never reaches a fingerprint.
|
|
9
|
+
|
|
10
|
+
A rule set is chosen by name on each source (``Source.normaliser``). HTML and
|
|
11
|
+
plain text are handled; PDF is not yet (see ``UnsupportedContentError``).
|
|
12
|
+
|
|
13
|
+
Deterministic and offline: the standard library only, no network.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import hashlib
|
|
19
|
+
import re
|
|
20
|
+
import unicodedata
|
|
21
|
+
from collections.abc import Iterator
|
|
22
|
+
from dataclasses import dataclass, field
|
|
23
|
+
from html.parser import HTMLParser
|
|
24
|
+
from typing import Literal
|
|
25
|
+
from urllib.parse import parse_qsl, urlencode, urlsplit, urlunsplit
|
|
26
|
+
|
|
27
|
+
# --- rule sets -----------------------------------------------------------------------------------
|
|
28
|
+
|
|
29
|
+
#: Query parameters that identify a click, never a page.
|
|
30
|
+
TRACKING_PARAMS = frozenset(
|
|
31
|
+
{
|
|
32
|
+
"gclid",
|
|
33
|
+
"dclid",
|
|
34
|
+
"gbraid",
|
|
35
|
+
"wbraid",
|
|
36
|
+
"fbclid",
|
|
37
|
+
"msclkid",
|
|
38
|
+
"yclid",
|
|
39
|
+
"igshid",
|
|
40
|
+
"twclid",
|
|
41
|
+
"mc_cid",
|
|
42
|
+
"mc_eid",
|
|
43
|
+
"_ga",
|
|
44
|
+
"_gl",
|
|
45
|
+
"_hsenc",
|
|
46
|
+
"_hsmi",
|
|
47
|
+
"hsctatracking",
|
|
48
|
+
"mkt_tok",
|
|
49
|
+
"ref",
|
|
50
|
+
"ref_src",
|
|
51
|
+
"ref_url",
|
|
52
|
+
"spm",
|
|
53
|
+
"si",
|
|
54
|
+
"cb",
|
|
55
|
+
"cachebust",
|
|
56
|
+
}
|
|
57
|
+
)
|
|
58
|
+
TRACKING_PREFIXES = ("utm_", "pk_", "mtm_", "oly_")
|
|
59
|
+
|
|
60
|
+
_MONTH = (
|
|
61
|
+
r"(?:jan(?:uary)?|feb(?:ruary)?|mar(?:ch)?|apr(?:il)?|may|june?|july?|aug(?:ust)?"
|
|
62
|
+
r"|sep(?:t(?:ember)?)?|oct(?:ober)?|nov(?:ember)?|dec(?:ember)?)\.?"
|
|
63
|
+
)
|
|
64
|
+
_DATE = (
|
|
65
|
+
rf"(?:\d{{4}}-\d{{2}}-\d{{2}}|{_MONTH}\s+\d{{1,2}}(?:st|nd|rd|th)?,?\s+\d{{4}}"
|
|
66
|
+
rf"|\d{{1,2}}(?:st|nd|rd|th)?\s+{_MONTH},?\s+\d{{4}}|\d{{1,2}}/\d{{1,2}}/\d{{2,4}})"
|
|
67
|
+
)
|
|
68
|
+
_CLOCK = r"\d{1,2}:\d{2}(?::\d{2}(?:\.\d+)?)?\s*(?:[ap]\.?m\.?)?"
|
|
69
|
+
_ZONE = r"(?:z|utc|gmt|bst|cest|cet|ist|jst|aest|[ecmp][sd]t|[+-]\d{2}:?\d{2})"
|
|
70
|
+
|
|
71
|
+
#: Volatile time text. Deliberately narrow: a bare date can be a fact (a release
|
|
72
|
+
#: date), so only a date that a page labels as its own update time is removed.
|
|
73
|
+
VOLATILE_PATTERNS = tuple(
|
|
74
|
+
re.compile(p, re.IGNORECASE)
|
|
75
|
+
for p in (
|
|
76
|
+
# "Last updated: September 20, 2026 14:02 UTC", "Generated 2026-09-20T14:02:11Z"
|
|
77
|
+
rf"\b(?:last\s+)?(?:updated|modified|edited|generated|refreshed|retrieved|checked|synced|built)"
|
|
78
|
+
rf"(?:\s+(?:on|at))?\s*:?\s*{_DATE}(?:[t,\s]+{_CLOCK})?(?:\s*{_ZONE})?\b",
|
|
79
|
+
# Full ISO date-times are page timestamps, not facts.
|
|
80
|
+
rf"\b\d{{4}}-\d{{2}}-\d{{2}}t\d{{2}}:\d{{2}}(?::\d{{2}}(?:\.\d+)?)?(?:{_ZONE})?",
|
|
81
|
+
# "3 days ago", "an hour ago", "just now"
|
|
82
|
+
r"\b(?:\d+|a|an|one|a\s+few|few)\s+"
|
|
83
|
+
r"(?:second|sec|minute|min|hour|hr|day|week|month|year)s?\s+ago\b",
|
|
84
|
+
r"\bjust\s+now\b",
|
|
85
|
+
# A clock time with a zone or am/pm: "14:02 UTC", "9:47 pm"
|
|
86
|
+
rf"\b\d{{1,2}}:\d{{2}}(?::\d{{2}})?\s*(?:[ap]\.?m\.?|{_ZONE})(?![a-z])",
|
|
87
|
+
)
|
|
88
|
+
)
|
|
89
|
+
|
|
90
|
+
_URL = re.compile(r"https?://[^\s<>\"')\]]+", re.IGNORECASE)
|
|
91
|
+
_SPACE = re.compile(r"\s+")
|
|
92
|
+
#: A line left with nothing but separators once volatile text is gone.
|
|
93
|
+
_EMPTY_LINE = re.compile(r"^[\W_]*$")
|
|
94
|
+
|
|
95
|
+
#: Class or id tokens of page furniture that carries no facts.
|
|
96
|
+
FURNITURE_TOKEN = re.compile(
|
|
97
|
+
r"^(?:ad|ads|advert\w*|advertisement|sponsor\w*|promo\w*|banner-ad|"
|
|
98
|
+
r"[\w-]*cookie[\w-]*|[\w-]*consent[\w-]*|gdpr|onetrust\w*|cc-banner|newsletter\w*|skip-link)$",
|
|
99
|
+
re.IGNORECASE,
|
|
100
|
+
)
|
|
101
|
+
FURNITURE_ROLES = frozenset({"navigation", "banner", "contentinfo", "complementary", "search"})
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
@dataclass(frozen=True)
|
|
105
|
+
class RuleSet:
|
|
106
|
+
"""A named normalisation recipe. ``content`` says which parser applies."""
|
|
107
|
+
|
|
108
|
+
name: str
|
|
109
|
+
content: Literal["html", "text"]
|
|
110
|
+
drop_tags: frozenset[str] = frozenset()
|
|
111
|
+
#: Drop ``<header>``/``<footer>`` unless they sit inside an article, main or section.
|
|
112
|
+
drop_page_chrome: bool = True
|
|
113
|
+
drop_furniture: bool = True
|
|
114
|
+
strip_volatile: bool = True
|
|
115
|
+
strip_tracking: bool = True
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
NORMALISERS: dict[str, RuleSet] = {
|
|
119
|
+
rules.name: rules
|
|
120
|
+
for rules in (
|
|
121
|
+
RuleSet(
|
|
122
|
+
"html-default",
|
|
123
|
+
"html",
|
|
124
|
+
drop_tags=frozenset(
|
|
125
|
+
{
|
|
126
|
+
"head",
|
|
127
|
+
"script",
|
|
128
|
+
"style",
|
|
129
|
+
"noscript",
|
|
130
|
+
"template",
|
|
131
|
+
"iframe",
|
|
132
|
+
"svg",
|
|
133
|
+
"canvas",
|
|
134
|
+
"nav",
|
|
135
|
+
"aside",
|
|
136
|
+
"form",
|
|
137
|
+
"button",
|
|
138
|
+
"dialog",
|
|
139
|
+
"link",
|
|
140
|
+
"meta",
|
|
141
|
+
"object",
|
|
142
|
+
"embed",
|
|
143
|
+
}
|
|
144
|
+
),
|
|
145
|
+
),
|
|
146
|
+
RuleSet("text-default", "text"),
|
|
147
|
+
)
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
class UnsupportedContentError(ValueError):
|
|
152
|
+
"""The content type has no normaliser yet (PDF, for now)."""
|
|
153
|
+
|
|
154
|
+
def __init__(self, kind: str) -> None:
|
|
155
|
+
super().__init__(f"unsupported content: {kind}")
|
|
156
|
+
self.kind = kind
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
# --- locators ------------------------------------------------------------------------------------
|
|
160
|
+
|
|
161
|
+
LocatorKind = Literal["page", "css", "heading", "table"]
|
|
162
|
+
|
|
163
|
+
_COMPOUND = re.compile(
|
|
164
|
+
r"^(?P<tag>[a-zA-Z][\w-]*|\*)?(?P<rest>(?:#[\w-]+|\.[\w-]+|\[[\w-]+(?:=(?:\"[^\"]*\"|'[^']*'|[^\]]*))?\])*)$"
|
|
165
|
+
)
|
|
166
|
+
_PART = re.compile(r"#([\w-]+)|\.([\w-]+)|\[([\w-]+)(?:=(\"[^\"]*\"|'[^']*'|[^\]]*))?\]")
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
@dataclass(frozen=True)
|
|
170
|
+
class _Compound:
|
|
171
|
+
tag: str | None
|
|
172
|
+
id: str | None
|
|
173
|
+
classes: tuple[str, ...]
|
|
174
|
+
attrs: tuple[tuple[str, str | None], ...]
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
@dataclass(frozen=True)
|
|
178
|
+
class Locator:
|
|
179
|
+
"""Where a cited region sits in a page.
|
|
180
|
+
|
|
181
|
+
``page`` is the whole normalised page; ``css`` is a simple selector (tags, ``#id``,
|
|
182
|
+
``.class``, ``[attr]`` or ``[attr=value]``, joined by descendant or ``>``
|
|
183
|
+
combinators); ``heading`` is a heading's id or text, and the region runs to the
|
|
184
|
+
next heading of the same or a higher level; ``table`` is the n-th table (0-based).
|
|
185
|
+
"""
|
|
186
|
+
|
|
187
|
+
kind: LocatorKind
|
|
188
|
+
value: str = ""
|
|
189
|
+
|
|
190
|
+
def __post_init__(self) -> None:
|
|
191
|
+
if self.kind == "css":
|
|
192
|
+
_parse_selector(self.value)
|
|
193
|
+
elif self.kind == "table":
|
|
194
|
+
if not self.value.isdigit():
|
|
195
|
+
raise ValueError(f"table locator needs a non-negative index, got {self.value!r}")
|
|
196
|
+
elif self.kind == "heading":
|
|
197
|
+
if not self.value.strip():
|
|
198
|
+
raise ValueError("heading locator needs an id or heading text")
|
|
199
|
+
elif self.kind != "page":
|
|
200
|
+
raise ValueError(f"unknown locator kind {self.kind!r}")
|
|
201
|
+
|
|
202
|
+
@classmethod
|
|
203
|
+
def page(cls) -> Locator:
|
|
204
|
+
return cls("page")
|
|
205
|
+
|
|
206
|
+
@classmethod
|
|
207
|
+
def css(cls, selector: str) -> Locator:
|
|
208
|
+
return cls("css", selector)
|
|
209
|
+
|
|
210
|
+
@classmethod
|
|
211
|
+
def heading(cls, anchor: str) -> Locator:
|
|
212
|
+
return cls("heading", anchor)
|
|
213
|
+
|
|
214
|
+
@classmethod
|
|
215
|
+
def table(cls, index: int) -> Locator:
|
|
216
|
+
return cls("table", str(index))
|
|
217
|
+
|
|
218
|
+
|
|
219
|
+
def _parse_selector(selector: str) -> tuple[tuple[str, _Compound], ...]:
|
|
220
|
+
"""Parse into ``((combinator, compound), ...)``; combinator is ``" "`` or ``">"``."""
|
|
221
|
+
tokens = re.sub(r"\s*>\s*", " > ", selector.strip()).split()
|
|
222
|
+
steps: list[tuple[str, _Compound]] = []
|
|
223
|
+
combinator = " "
|
|
224
|
+
for token in tokens:
|
|
225
|
+
if token == ">":
|
|
226
|
+
if not steps or combinator == ">":
|
|
227
|
+
raise ValueError(f"bad CSS selector {selector!r}")
|
|
228
|
+
combinator = ">"
|
|
229
|
+
continue
|
|
230
|
+
m = _COMPOUND.match(token)
|
|
231
|
+
if not m or not token:
|
|
232
|
+
raise ValueError(f"unsupported CSS selector {selector!r}")
|
|
233
|
+
tag = m["tag"] if m["tag"] not in (None, "*") else None
|
|
234
|
+
ident, classes, attrs = None, [], []
|
|
235
|
+
for part in _PART.finditer(m["rest"]):
|
|
236
|
+
if part[1]:
|
|
237
|
+
ident = part[1]
|
|
238
|
+
elif part[2]:
|
|
239
|
+
classes.append(part[2])
|
|
240
|
+
else:
|
|
241
|
+
raw = part[4]
|
|
242
|
+
attrs.append((part[3].lower(), raw.strip("\"'") if raw is not None else None))
|
|
243
|
+
steps.append(
|
|
244
|
+
(combinator, _Compound(tag and tag.lower(), ident, tuple(classes), tuple(attrs)))
|
|
245
|
+
)
|
|
246
|
+
combinator = " "
|
|
247
|
+
if not steps or combinator == ">":
|
|
248
|
+
raise ValueError(f"bad CSS selector {selector!r}")
|
|
249
|
+
return tuple(steps)
|
|
250
|
+
|
|
251
|
+
|
|
252
|
+
# --- a small DOM ---------------------------------------------------------------------------------
|
|
253
|
+
|
|
254
|
+
VOID = frozenset(
|
|
255
|
+
{
|
|
256
|
+
"area",
|
|
257
|
+
"base",
|
|
258
|
+
"br",
|
|
259
|
+
"col",
|
|
260
|
+
"embed",
|
|
261
|
+
"hr",
|
|
262
|
+
"img",
|
|
263
|
+
"input",
|
|
264
|
+
"link",
|
|
265
|
+
"meta",
|
|
266
|
+
"source",
|
|
267
|
+
"track",
|
|
268
|
+
"wbr",
|
|
269
|
+
}
|
|
270
|
+
)
|
|
271
|
+
BLOCK = frozenset(
|
|
272
|
+
{
|
|
273
|
+
"address",
|
|
274
|
+
"article",
|
|
275
|
+
"blockquote",
|
|
276
|
+
"body",
|
|
277
|
+
"caption",
|
|
278
|
+
"dd",
|
|
279
|
+
"details",
|
|
280
|
+
"div",
|
|
281
|
+
"dl",
|
|
282
|
+
"dt",
|
|
283
|
+
"figcaption",
|
|
284
|
+
"figure",
|
|
285
|
+
"footer",
|
|
286
|
+
"h1",
|
|
287
|
+
"h2",
|
|
288
|
+
"h3",
|
|
289
|
+
"h4",
|
|
290
|
+
"h5",
|
|
291
|
+
"h6",
|
|
292
|
+
"header",
|
|
293
|
+
"hr",
|
|
294
|
+
"html",
|
|
295
|
+
"li",
|
|
296
|
+
"main",
|
|
297
|
+
"ol",
|
|
298
|
+
"p",
|
|
299
|
+
"pre",
|
|
300
|
+
"section",
|
|
301
|
+
"summary",
|
|
302
|
+
"table",
|
|
303
|
+
"tbody",
|
|
304
|
+
"tfoot",
|
|
305
|
+
"thead",
|
|
306
|
+
"tr",
|
|
307
|
+
"ul",
|
|
308
|
+
}
|
|
309
|
+
)
|
|
310
|
+
HEADINGS = {f"h{n}": n for n in range(1, 7)}
|
|
311
|
+
#: Opening one of these closes an open element of the same group, as browsers do.
|
|
312
|
+
_IMPLIED_CLOSE = {
|
|
313
|
+
"p": ({"p"}, {"div", "section", "article", "main", "body", "td", "th", "li"}),
|
|
314
|
+
"li": ({"li"}, {"ul", "ol"}),
|
|
315
|
+
"dt": ({"dt", "dd"}, {"dl"}),
|
|
316
|
+
"dd": ({"dt", "dd"}, {"dl"}),
|
|
317
|
+
"tr": ({"tr", "td", "th"}, {"table", "thead", "tbody", "tfoot"}),
|
|
318
|
+
"td": ({"td", "th"}, {"tr", "table"}),
|
|
319
|
+
"th": ({"td", "th"}, {"tr", "table"}),
|
|
320
|
+
"thead": ({"thead", "tbody", "tr", "td", "th"}, {"table"}),
|
|
321
|
+
"tbody": ({"thead", "tbody", "tr", "td", "th"}, {"table"}),
|
|
322
|
+
"tfoot": ({"thead", "tbody", "tr", "td", "th"}, {"table"}),
|
|
323
|
+
}
|
|
324
|
+
|
|
325
|
+
|
|
326
|
+
@dataclass(eq=False)
|
|
327
|
+
class Node:
|
|
328
|
+
tag: str
|
|
329
|
+
attrs: dict[str, str] = field(default_factory=dict)
|
|
330
|
+
children: list[Node | str] = field(default_factory=list)
|
|
331
|
+
parent: Node | None = None
|
|
332
|
+
|
|
333
|
+
@property
|
|
334
|
+
def classes(self) -> tuple[str, ...]:
|
|
335
|
+
return tuple(self.attrs.get("class", "").split())
|
|
336
|
+
|
|
337
|
+
def elements(self) -> Iterator[Node]:
|
|
338
|
+
"""This node's descendant elements, in document order."""
|
|
339
|
+
for child in self.children:
|
|
340
|
+
if isinstance(child, Node):
|
|
341
|
+
yield child
|
|
342
|
+
yield from child.elements()
|
|
343
|
+
|
|
344
|
+
|
|
345
|
+
class _TreeBuilder(HTMLParser):
|
|
346
|
+
def __init__(self) -> None:
|
|
347
|
+
super().__init__(convert_charrefs=True)
|
|
348
|
+
self.root = Node("#document")
|
|
349
|
+
self.stack = [self.root]
|
|
350
|
+
|
|
351
|
+
def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:
|
|
352
|
+
if tag in _IMPLIED_CLOSE:
|
|
353
|
+
closes, scope = _IMPLIED_CLOSE[tag]
|
|
354
|
+
for i in range(len(self.stack) - 1, 0, -1):
|
|
355
|
+
open_tag = self.stack[i].tag
|
|
356
|
+
if open_tag in closes:
|
|
357
|
+
del self.stack[i:]
|
|
358
|
+
break
|
|
359
|
+
if open_tag in scope:
|
|
360
|
+
break
|
|
361
|
+
node = Node(tag, {k.lower(): v or "" for k, v in attrs}, parent=self.stack[-1])
|
|
362
|
+
self.stack[-1].children.append(node)
|
|
363
|
+
if tag not in VOID:
|
|
364
|
+
self.stack.append(node)
|
|
365
|
+
|
|
366
|
+
def handle_startendtag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:
|
|
367
|
+
self.handle_starttag(tag, attrs)
|
|
368
|
+
if tag not in VOID and self.stack[-1].tag == tag:
|
|
369
|
+
self.stack.pop()
|
|
370
|
+
|
|
371
|
+
def handle_endtag(self, tag: str) -> None:
|
|
372
|
+
for i in range(len(self.stack) - 1, 0, -1):
|
|
373
|
+
if self.stack[i].tag == tag:
|
|
374
|
+
del self.stack[i:]
|
|
375
|
+
return
|
|
376
|
+
|
|
377
|
+
def handle_data(self, data: str) -> None:
|
|
378
|
+
self.stack[-1].children.append(data)
|
|
379
|
+
|
|
380
|
+
|
|
381
|
+
def _is_furniture(node: Node, rules: RuleSet) -> bool:
|
|
382
|
+
if node.tag in rules.drop_tags:
|
|
383
|
+
return True
|
|
384
|
+
if "hidden" in node.attrs or node.attrs.get("aria-hidden") == "true":
|
|
385
|
+
return True
|
|
386
|
+
if rules.drop_page_chrome and node.tag in ("header", "footer"):
|
|
387
|
+
ancestor = node.parent
|
|
388
|
+
while ancestor is not None:
|
|
389
|
+
if ancestor.tag in ("article", "main", "section"):
|
|
390
|
+
break
|
|
391
|
+
ancestor = ancestor.parent
|
|
392
|
+
else:
|
|
393
|
+
return True
|
|
394
|
+
if rules.drop_furniture:
|
|
395
|
+
if node.attrs.get("role", "").lower() in FURNITURE_ROLES:
|
|
396
|
+
return True
|
|
397
|
+
tokens = (*node.classes, *node.attrs.get("id", "").split())
|
|
398
|
+
if any(FURNITURE_TOKEN.match(t) for t in tokens):
|
|
399
|
+
return True
|
|
400
|
+
return False
|
|
401
|
+
|
|
402
|
+
|
|
403
|
+
def _prune(node: Node, rules: RuleSet) -> None:
|
|
404
|
+
kept: list[Node | str] = []
|
|
405
|
+
for child in node.children:
|
|
406
|
+
if isinstance(child, Node):
|
|
407
|
+
if _is_furniture(child, rules):
|
|
408
|
+
continue
|
|
409
|
+
_prune(child, rules)
|
|
410
|
+
kept.append(child)
|
|
411
|
+
node.children = kept
|
|
412
|
+
|
|
413
|
+
|
|
414
|
+
# --- rendering to text ---------------------------------------------------------------------------
|
|
415
|
+
|
|
416
|
+
|
|
417
|
+
def _inline(node: Node | str) -> str:
|
|
418
|
+
if isinstance(node, str):
|
|
419
|
+
return node
|
|
420
|
+
if node.tag == "br":
|
|
421
|
+
return " "
|
|
422
|
+
return " ".join(_inline(c) for c in node.children)
|
|
423
|
+
|
|
424
|
+
|
|
425
|
+
def _render(node: Node | str, out: list[str]) -> None:
|
|
426
|
+
"""Append text to ``out``; a ``"\\n"`` entry is a block boundary."""
|
|
427
|
+
if isinstance(node, str):
|
|
428
|
+
out.append(_SPACE.sub(" ", node))
|
|
429
|
+
return
|
|
430
|
+
if node.tag == "tr":
|
|
431
|
+
cells = [c for c in node.children if isinstance(c, Node) and c.tag in ("td", "th")]
|
|
432
|
+
out += ["\n", " | ".join(_SPACE.sub(" ", _inline(c)).strip() for c in cells), "\n"]
|
|
433
|
+
return
|
|
434
|
+
if node.tag == "br":
|
|
435
|
+
out.append("\n")
|
|
436
|
+
return
|
|
437
|
+
block = node.tag in BLOCK
|
|
438
|
+
if block:
|
|
439
|
+
out.append("\n")
|
|
440
|
+
for child in node.children:
|
|
441
|
+
_render(child, out)
|
|
442
|
+
if block:
|
|
443
|
+
out.append("\n")
|
|
444
|
+
|
|
445
|
+
|
|
446
|
+
def _clean_line(line: str, rules: RuleSet) -> str:
|
|
447
|
+
if rules.strip_tracking:
|
|
448
|
+
line = _URL.sub(lambda m: canonical_url(m.group(0)), line)
|
|
449
|
+
if rules.strip_volatile:
|
|
450
|
+
for pattern in VOLATILE_PATTERNS:
|
|
451
|
+
line = pattern.sub(" ", line)
|
|
452
|
+
line = _SPACE.sub(" ", line).strip()
|
|
453
|
+
return "" if _EMPTY_LINE.match(line) else line
|
|
454
|
+
|
|
455
|
+
|
|
456
|
+
def _to_text(parts: list[str], rules: RuleSet) -> str:
|
|
457
|
+
lines = (_clean_line(line, rules) for line in "".join(parts).split("\n"))
|
|
458
|
+
return "\n".join(line for line in lines if line)
|
|
459
|
+
|
|
460
|
+
|
|
461
|
+
def _nodes_text(nodes: list[Node], rules: RuleSet) -> str:
|
|
462
|
+
parts: list[str] = []
|
|
463
|
+
for node in nodes:
|
|
464
|
+
_render(node, parts)
|
|
465
|
+
parts.append("\n")
|
|
466
|
+
return _to_text(parts, rules)
|
|
467
|
+
|
|
468
|
+
|
|
469
|
+
# --- documents -----------------------------------------------------------------------------------
|
|
470
|
+
|
|
471
|
+
|
|
472
|
+
@dataclass
|
|
473
|
+
class Document:
|
|
474
|
+
"""A normalised page: its full text, and (for HTML) the cleaned tree for locators."""
|
|
475
|
+
|
|
476
|
+
rules: RuleSet
|
|
477
|
+
text: str
|
|
478
|
+
root: Node | None = None
|
|
479
|
+
|
|
480
|
+
|
|
481
|
+
def normalise_document(body: bytes, rules: RuleSet, *, charset: str | None = None) -> Document:
|
|
482
|
+
if body.lstrip().startswith(b"%PDF"):
|
|
483
|
+
raise UnsupportedContentError("pdf")
|
|
484
|
+
raw = body.decode(charset or "utf-8", errors="replace")
|
|
485
|
+
raw = unicodedata.normalize("NFKC", raw).replace("\r\n", "\n").replace("\r", "\n")
|
|
486
|
+
if rules.content == "text":
|
|
487
|
+
return Document(rules, _to_text([raw], rules))
|
|
488
|
+
builder = _TreeBuilder()
|
|
489
|
+
builder.feed(raw)
|
|
490
|
+
builder.close()
|
|
491
|
+
root = builder.root
|
|
492
|
+
_prune(root, rules)
|
|
493
|
+
return Document(rules, _nodes_text([root], rules), root)
|
|
494
|
+
|
|
495
|
+
|
|
496
|
+
def select_region(doc: Document, locator: Locator) -> str | None:
|
|
497
|
+
"""The normalised text of a cited region, or ``None`` when the locator matches nothing."""
|
|
498
|
+
if locator.kind == "page":
|
|
499
|
+
return doc.text or None
|
|
500
|
+
if doc.root is None:
|
|
501
|
+
raise ValueError(f"{locator.kind} locators need an HTML document")
|
|
502
|
+
if locator.kind == "table":
|
|
503
|
+
tables = [n for n in doc.root.elements() if n.tag == "table"]
|
|
504
|
+
index = int(locator.value)
|
|
505
|
+
nodes = [tables[index]] if index < len(tables) else []
|
|
506
|
+
elif locator.kind == "css":
|
|
507
|
+
steps = _parse_selector(locator.value)
|
|
508
|
+
nodes = [n for n in doc.root.elements() if _matches(n, steps)]
|
|
509
|
+
else:
|
|
510
|
+
nodes = _heading_section(doc.root, locator.value)
|
|
511
|
+
return (_nodes_text(nodes, doc.rules) or None) if nodes else None
|
|
512
|
+
|
|
513
|
+
|
|
514
|
+
def _matches_compound(node: Node, c: _Compound) -> bool:
|
|
515
|
+
if c.tag and node.tag != c.tag:
|
|
516
|
+
return False
|
|
517
|
+
if c.id and node.attrs.get("id") != c.id:
|
|
518
|
+
return False
|
|
519
|
+
if any(cls not in node.classes for cls in c.classes):
|
|
520
|
+
return False
|
|
521
|
+
for name, value in c.attrs:
|
|
522
|
+
if name not in node.attrs or (value is not None and node.attrs[name] != value):
|
|
523
|
+
return False
|
|
524
|
+
return True
|
|
525
|
+
|
|
526
|
+
|
|
527
|
+
def _matches(node: Node, steps: tuple[tuple[str, _Compound], ...]) -> bool:
|
|
528
|
+
combinator, last = steps[-1]
|
|
529
|
+
if not _matches_compound(node, last):
|
|
530
|
+
return False
|
|
531
|
+
if len(steps) == 1:
|
|
532
|
+
return True
|
|
533
|
+
rest = steps[:-1]
|
|
534
|
+
ancestor = node.parent
|
|
535
|
+
while ancestor is not None and ancestor.tag != "#document":
|
|
536
|
+
if _matches(ancestor, rest):
|
|
537
|
+
return True
|
|
538
|
+
if combinator == ">":
|
|
539
|
+
return False
|
|
540
|
+
ancestor = ancestor.parent
|
|
541
|
+
return False
|
|
542
|
+
|
|
543
|
+
|
|
544
|
+
def _heading_level(node: Node) -> int | None:
|
|
545
|
+
if node.tag in HEADINGS:
|
|
546
|
+
return HEADINGS[node.tag]
|
|
547
|
+
for descendant in node.elements():
|
|
548
|
+
if descendant.tag in HEADINGS:
|
|
549
|
+
return HEADINGS[descendant.tag]
|
|
550
|
+
return None
|
|
551
|
+
|
|
552
|
+
|
|
553
|
+
def _heading_section(root: Node, anchor: str) -> list[Node]:
|
|
554
|
+
wanted = anchor.removeprefix("#")
|
|
555
|
+
folded = _SPACE.sub(" ", wanted).strip().casefold()
|
|
556
|
+
heading = None
|
|
557
|
+
for node in root.elements():
|
|
558
|
+
if node.tag not in HEADINGS:
|
|
559
|
+
continue
|
|
560
|
+
ids = {node.attrs.get("id")} | {
|
|
561
|
+
d.attrs.get("id") or d.attrs.get("name") for d in node.elements()
|
|
562
|
+
}
|
|
563
|
+
if wanted in ids or _SPACE.sub(" ", _inline(node)).strip().casefold() == folded:
|
|
564
|
+
heading = node
|
|
565
|
+
break
|
|
566
|
+
if heading is None:
|
|
567
|
+
return []
|
|
568
|
+
level = HEADINGS[heading.tag]
|
|
569
|
+
# A heading wrapped alone in a container (``<div><h2/></div>``) sections by the wrapper.
|
|
570
|
+
start = heading
|
|
571
|
+
for _ in range(2):
|
|
572
|
+
parent = start.parent
|
|
573
|
+
if parent is None or parent.tag in ("#document", "body", "main", "article", "section"):
|
|
574
|
+
break
|
|
575
|
+
if any(isinstance(c, Node) and c is not start for c in parent.children):
|
|
576
|
+
break
|
|
577
|
+
start = parent
|
|
578
|
+
assert start.parent is not None
|
|
579
|
+
siblings = [c for c in start.parent.children if isinstance(c, Node)]
|
|
580
|
+
section = [start]
|
|
581
|
+
for sibling in siblings[siblings.index(start) + 1 :]:
|
|
582
|
+
other = _heading_level(sibling)
|
|
583
|
+
if other is not None and other <= level:
|
|
584
|
+
break
|
|
585
|
+
section.append(sibling)
|
|
586
|
+
return section
|
|
587
|
+
|
|
588
|
+
|
|
589
|
+
# --- urls and fingerprints -----------------------------------------------------------------------
|
|
590
|
+
|
|
591
|
+
|
|
592
|
+
def canonical_url(url: str) -> str:
|
|
593
|
+
"""``url`` without tracking query parameters or a fragment."""
|
|
594
|
+
parts = urlsplit(url)
|
|
595
|
+
query = [
|
|
596
|
+
(k, v)
|
|
597
|
+
for k, v in parse_qsl(parts.query, keep_blank_values=True)
|
|
598
|
+
if k.lower() not in TRACKING_PARAMS and not k.lower().startswith(TRACKING_PREFIXES)
|
|
599
|
+
]
|
|
600
|
+
return urlunsplit((parts.scheme, parts.netloc, parts.path, urlencode(query), ""))
|
|
601
|
+
|
|
602
|
+
|
|
603
|
+
def fingerprint(text: str) -> str:
|
|
604
|
+
return "sha256:" + hashlib.sha256(text.encode("utf-8")).hexdigest()
|