modelspec-dev 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (63) hide show
  1. api/__init__.py +0 -0
  2. api/class_fit.py +334 -0
  3. api/classes.py +557 -0
  4. api/ranking/__init__.py +12 -0
  5. api/ranking/engine.py +1943 -0
  6. cli/__init__.py +0 -0
  7. cli/modelspec/__init__.py +0 -0
  8. cli/modelspec/cli.py +1819 -0
  9. cli/modelspec/commands/__init__.py +0 -0
  10. cli/modelspec/decide_cmd.py +333 -0
  11. cli/modelspec/offline.py +623 -0
  12. cli/modelspec/snapshot.py +698 -0
  13. cli/modelspec/snapshot_build_cmd.py +49 -0
  14. cli/modelspec/verify_cmd.py +125 -0
  15. cli/modelspec/vocab_cmd.py +204 -0
  16. cli/modelspec/vocabulary_cache.py +54 -0
  17. decision/__init__.py +13 -0
  18. decision/capability.py +872 -0
  19. decision/computed.py +125 -0
  20. decision/contract.py +1575 -0
  21. decision/engine.py +238 -0
  22. decision/excluded.py +34 -0
  23. decision/explain.py +908 -0
  24. decision/filter.py +796 -0
  25. decision/model.py +438 -0
  26. decision/normalise.py +604 -0
  27. decision/optimise.py +320 -0
  28. decision/registry.py +717 -0
  29. decision/relax.py +132 -0
  30. decision/resolve.py +111 -0
  31. decision/schema.py +21 -0
  32. decision/snapshot.py +1483 -0
  33. decision/sources.py +544 -0
  34. decision/templates.py +134 -0
  35. decision/verify.py +1745 -0
  36. decision/vocabulary.py +433 -0
  37. modelspec_dev-0.1.0.dist-info/METADATA +101 -0
  38. modelspec_dev-0.1.0.dist-info/RECORD +63 -0
  39. modelspec_dev-0.1.0.dist-info/WHEEL +4 -0
  40. modelspec_dev-0.1.0.dist-info/entry_points.txt +2 -0
  41. modelspec_dev-0.1.0.dist-info/licenses/LICENSE +43 -0
  42. modelspec_dev-0.1.0.dist-info/licenses/LICENSE-DATA +428 -0
  43. pipeline/__init__.py +0 -0
  44. pipeline/class_export.py +172 -0
  45. pipeline/hardware.py +434 -0
  46. pipeline/hosts.py +247 -0
  47. pipeline/load.py +224 -0
  48. pipeline/ranking.py +551 -0
  49. registry/domains.yaml +130 -0
  50. registry/facets.yaml +888 -0
  51. registry/harnesses.yaml +79 -0
  52. registry/providers.yaml +354 -0
  53. registry/sources.yaml +3059 -0
  54. registry/templates.yaml +166 -0
  55. schema/__init__.py +0 -0
  56. schema/applicability.py +147 -0
  57. schema/benchmark.py +175 -0
  58. schema/benchmark_eligibility.py +304 -0
  59. schema/card.py +1463 -0
  60. schema/enrichment.py +162 -0
  61. schema/enums.py +327 -0
  62. schema/graph.py +406 -0
  63. schema/suppliers.py +72 -0
decision/normalise.py ADDED
@@ -0,0 +1,604 @@
1
+ """Normalise a fetched page into stable text, and select its cited regions.
2
+
3
+ Change detection compares fingerprints of normalised text, so everything a page
4
+ changes without changing what it says must normalise away: scripts, styles,
5
+ navigation, the site header and footer, cookie banners, ads, timestamps and
6
+ relative dates ("3 days ago"), tracking query strings, attribute order and
7
+ whitespace. The output is text only, one block per line and table cells joined
8
+ by `` | ``, so markup churn never reaches a fingerprint.
9
+
10
+ A rule set is chosen by name on each source (``Source.normaliser``). HTML and
11
+ plain text are handled; PDF is not yet (see ``UnsupportedContentError``).
12
+
13
+ Deterministic and offline: the standard library only, no network.
14
+ """
15
+
16
+ from __future__ import annotations
17
+
18
+ import hashlib
19
+ import re
20
+ import unicodedata
21
+ from collections.abc import Iterator
22
+ from dataclasses import dataclass, field
23
+ from html.parser import HTMLParser
24
+ from typing import Literal
25
+ from urllib.parse import parse_qsl, urlencode, urlsplit, urlunsplit
26
+
27
+ # --- rule sets -----------------------------------------------------------------------------------
28
+
29
+ #: Query parameters that identify a click, never a page.
30
+ TRACKING_PARAMS = frozenset(
31
+ {
32
+ "gclid",
33
+ "dclid",
34
+ "gbraid",
35
+ "wbraid",
36
+ "fbclid",
37
+ "msclkid",
38
+ "yclid",
39
+ "igshid",
40
+ "twclid",
41
+ "mc_cid",
42
+ "mc_eid",
43
+ "_ga",
44
+ "_gl",
45
+ "_hsenc",
46
+ "_hsmi",
47
+ "hsctatracking",
48
+ "mkt_tok",
49
+ "ref",
50
+ "ref_src",
51
+ "ref_url",
52
+ "spm",
53
+ "si",
54
+ "cb",
55
+ "cachebust",
56
+ }
57
+ )
58
+ TRACKING_PREFIXES = ("utm_", "pk_", "mtm_", "oly_")
59
+
60
+ _MONTH = (
61
+ r"(?:jan(?:uary)?|feb(?:ruary)?|mar(?:ch)?|apr(?:il)?|may|june?|july?|aug(?:ust)?"
62
+ r"|sep(?:t(?:ember)?)?|oct(?:ober)?|nov(?:ember)?|dec(?:ember)?)\.?"
63
+ )
64
+ _DATE = (
65
+ rf"(?:\d{{4}}-\d{{2}}-\d{{2}}|{_MONTH}\s+\d{{1,2}}(?:st|nd|rd|th)?,?\s+\d{{4}}"
66
+ rf"|\d{{1,2}}(?:st|nd|rd|th)?\s+{_MONTH},?\s+\d{{4}}|\d{{1,2}}/\d{{1,2}}/\d{{2,4}})"
67
+ )
68
+ _CLOCK = r"\d{1,2}:\d{2}(?::\d{2}(?:\.\d+)?)?\s*(?:[ap]\.?m\.?)?"
69
+ _ZONE = r"(?:z|utc|gmt|bst|cest|cet|ist|jst|aest|[ecmp][sd]t|[+-]\d{2}:?\d{2})"
70
+
71
+ #: Volatile time text. Deliberately narrow: a bare date can be a fact (a release
72
+ #: date), so only a date that a page labels as its own update time is removed.
73
+ VOLATILE_PATTERNS = tuple(
74
+ re.compile(p, re.IGNORECASE)
75
+ for p in (
76
+ # "Last updated: September 20, 2026 14:02 UTC", "Generated 2026-09-20T14:02:11Z"
77
+ rf"\b(?:last\s+)?(?:updated|modified|edited|generated|refreshed|retrieved|checked|synced|built)"
78
+ rf"(?:\s+(?:on|at))?\s*:?\s*{_DATE}(?:[t,\s]+{_CLOCK})?(?:\s*{_ZONE})?\b",
79
+ # Full ISO date-times are page timestamps, not facts.
80
+ rf"\b\d{{4}}-\d{{2}}-\d{{2}}t\d{{2}}:\d{{2}}(?::\d{{2}}(?:\.\d+)?)?(?:{_ZONE})?",
81
+ # "3 days ago", "an hour ago", "just now"
82
+ r"\b(?:\d+|a|an|one|a\s+few|few)\s+"
83
+ r"(?:second|sec|minute|min|hour|hr|day|week|month|year)s?\s+ago\b",
84
+ r"\bjust\s+now\b",
85
+ # A clock time with a zone or am/pm: "14:02 UTC", "9:47 pm"
86
+ rf"\b\d{{1,2}}:\d{{2}}(?::\d{{2}})?\s*(?:[ap]\.?m\.?|{_ZONE})(?![a-z])",
87
+ )
88
+ )
89
+
90
+ _URL = re.compile(r"https?://[^\s<>\"')\]]+", re.IGNORECASE)
91
+ _SPACE = re.compile(r"\s+")
92
+ #: A line left with nothing but separators once volatile text is gone.
93
+ _EMPTY_LINE = re.compile(r"^[\W_]*$")
94
+
95
+ #: Class or id tokens of page furniture that carries no facts.
96
+ FURNITURE_TOKEN = re.compile(
97
+ r"^(?:ad|ads|advert\w*|advertisement|sponsor\w*|promo\w*|banner-ad|"
98
+ r"[\w-]*cookie[\w-]*|[\w-]*consent[\w-]*|gdpr|onetrust\w*|cc-banner|newsletter\w*|skip-link)$",
99
+ re.IGNORECASE,
100
+ )
101
+ FURNITURE_ROLES = frozenset({"navigation", "banner", "contentinfo", "complementary", "search"})
102
+
103
+
104
+ @dataclass(frozen=True)
105
+ class RuleSet:
106
+ """A named normalisation recipe. ``content`` says which parser applies."""
107
+
108
+ name: str
109
+ content: Literal["html", "text"]
110
+ drop_tags: frozenset[str] = frozenset()
111
+ #: Drop ``<header>``/``<footer>`` unless they sit inside an article, main or section.
112
+ drop_page_chrome: bool = True
113
+ drop_furniture: bool = True
114
+ strip_volatile: bool = True
115
+ strip_tracking: bool = True
116
+
117
+
118
+ NORMALISERS: dict[str, RuleSet] = {
119
+ rules.name: rules
120
+ for rules in (
121
+ RuleSet(
122
+ "html-default",
123
+ "html",
124
+ drop_tags=frozenset(
125
+ {
126
+ "head",
127
+ "script",
128
+ "style",
129
+ "noscript",
130
+ "template",
131
+ "iframe",
132
+ "svg",
133
+ "canvas",
134
+ "nav",
135
+ "aside",
136
+ "form",
137
+ "button",
138
+ "dialog",
139
+ "link",
140
+ "meta",
141
+ "object",
142
+ "embed",
143
+ }
144
+ ),
145
+ ),
146
+ RuleSet("text-default", "text"),
147
+ )
148
+ }
149
+
150
+
151
+ class UnsupportedContentError(ValueError):
152
+ """The content type has no normaliser yet (PDF, for now)."""
153
+
154
+ def __init__(self, kind: str) -> None:
155
+ super().__init__(f"unsupported content: {kind}")
156
+ self.kind = kind
157
+
158
+
159
+ # --- locators ------------------------------------------------------------------------------------
160
+
161
+ LocatorKind = Literal["page", "css", "heading", "table"]
162
+
163
+ _COMPOUND = re.compile(
164
+ r"^(?P<tag>[a-zA-Z][\w-]*|\*)?(?P<rest>(?:#[\w-]+|\.[\w-]+|\[[\w-]+(?:=(?:\"[^\"]*\"|'[^']*'|[^\]]*))?\])*)$"
165
+ )
166
+ _PART = re.compile(r"#([\w-]+)|\.([\w-]+)|\[([\w-]+)(?:=(\"[^\"]*\"|'[^']*'|[^\]]*))?\]")
167
+
168
+
169
+ @dataclass(frozen=True)
170
+ class _Compound:
171
+ tag: str | None
172
+ id: str | None
173
+ classes: tuple[str, ...]
174
+ attrs: tuple[tuple[str, str | None], ...]
175
+
176
+
177
+ @dataclass(frozen=True)
178
+ class Locator:
179
+ """Where a cited region sits in a page.
180
+
181
+ ``page`` is the whole normalised page; ``css`` is a simple selector (tags, ``#id``,
182
+ ``.class``, ``[attr]`` or ``[attr=value]``, joined by descendant or ``>``
183
+ combinators); ``heading`` is a heading's id or text, and the region runs to the
184
+ next heading of the same or a higher level; ``table`` is the n-th table (0-based).
185
+ """
186
+
187
+ kind: LocatorKind
188
+ value: str = ""
189
+
190
+ def __post_init__(self) -> None:
191
+ if self.kind == "css":
192
+ _parse_selector(self.value)
193
+ elif self.kind == "table":
194
+ if not self.value.isdigit():
195
+ raise ValueError(f"table locator needs a non-negative index, got {self.value!r}")
196
+ elif self.kind == "heading":
197
+ if not self.value.strip():
198
+ raise ValueError("heading locator needs an id or heading text")
199
+ elif self.kind != "page":
200
+ raise ValueError(f"unknown locator kind {self.kind!r}")
201
+
202
+ @classmethod
203
+ def page(cls) -> Locator:
204
+ return cls("page")
205
+
206
+ @classmethod
207
+ def css(cls, selector: str) -> Locator:
208
+ return cls("css", selector)
209
+
210
+ @classmethod
211
+ def heading(cls, anchor: str) -> Locator:
212
+ return cls("heading", anchor)
213
+
214
+ @classmethod
215
+ def table(cls, index: int) -> Locator:
216
+ return cls("table", str(index))
217
+
218
+
219
+ def _parse_selector(selector: str) -> tuple[tuple[str, _Compound], ...]:
220
+ """Parse into ``((combinator, compound), ...)``; combinator is ``" "`` or ``">"``."""
221
+ tokens = re.sub(r"\s*>\s*", " > ", selector.strip()).split()
222
+ steps: list[tuple[str, _Compound]] = []
223
+ combinator = " "
224
+ for token in tokens:
225
+ if token == ">":
226
+ if not steps or combinator == ">":
227
+ raise ValueError(f"bad CSS selector {selector!r}")
228
+ combinator = ">"
229
+ continue
230
+ m = _COMPOUND.match(token)
231
+ if not m or not token:
232
+ raise ValueError(f"unsupported CSS selector {selector!r}")
233
+ tag = m["tag"] if m["tag"] not in (None, "*") else None
234
+ ident, classes, attrs = None, [], []
235
+ for part in _PART.finditer(m["rest"]):
236
+ if part[1]:
237
+ ident = part[1]
238
+ elif part[2]:
239
+ classes.append(part[2])
240
+ else:
241
+ raw = part[4]
242
+ attrs.append((part[3].lower(), raw.strip("\"'") if raw is not None else None))
243
+ steps.append(
244
+ (combinator, _Compound(tag and tag.lower(), ident, tuple(classes), tuple(attrs)))
245
+ )
246
+ combinator = " "
247
+ if not steps or combinator == ">":
248
+ raise ValueError(f"bad CSS selector {selector!r}")
249
+ return tuple(steps)
250
+
251
+
252
+ # --- a small DOM ---------------------------------------------------------------------------------
253
+
254
+ VOID = frozenset(
255
+ {
256
+ "area",
257
+ "base",
258
+ "br",
259
+ "col",
260
+ "embed",
261
+ "hr",
262
+ "img",
263
+ "input",
264
+ "link",
265
+ "meta",
266
+ "source",
267
+ "track",
268
+ "wbr",
269
+ }
270
+ )
271
+ BLOCK = frozenset(
272
+ {
273
+ "address",
274
+ "article",
275
+ "blockquote",
276
+ "body",
277
+ "caption",
278
+ "dd",
279
+ "details",
280
+ "div",
281
+ "dl",
282
+ "dt",
283
+ "figcaption",
284
+ "figure",
285
+ "footer",
286
+ "h1",
287
+ "h2",
288
+ "h3",
289
+ "h4",
290
+ "h5",
291
+ "h6",
292
+ "header",
293
+ "hr",
294
+ "html",
295
+ "li",
296
+ "main",
297
+ "ol",
298
+ "p",
299
+ "pre",
300
+ "section",
301
+ "summary",
302
+ "table",
303
+ "tbody",
304
+ "tfoot",
305
+ "thead",
306
+ "tr",
307
+ "ul",
308
+ }
309
+ )
310
+ HEADINGS = {f"h{n}": n for n in range(1, 7)}
311
+ #: Opening one of these closes an open element of the same group, as browsers do.
312
+ _IMPLIED_CLOSE = {
313
+ "p": ({"p"}, {"div", "section", "article", "main", "body", "td", "th", "li"}),
314
+ "li": ({"li"}, {"ul", "ol"}),
315
+ "dt": ({"dt", "dd"}, {"dl"}),
316
+ "dd": ({"dt", "dd"}, {"dl"}),
317
+ "tr": ({"tr", "td", "th"}, {"table", "thead", "tbody", "tfoot"}),
318
+ "td": ({"td", "th"}, {"tr", "table"}),
319
+ "th": ({"td", "th"}, {"tr", "table"}),
320
+ "thead": ({"thead", "tbody", "tr", "td", "th"}, {"table"}),
321
+ "tbody": ({"thead", "tbody", "tr", "td", "th"}, {"table"}),
322
+ "tfoot": ({"thead", "tbody", "tr", "td", "th"}, {"table"}),
323
+ }
324
+
325
+
326
+ @dataclass(eq=False)
327
+ class Node:
328
+ tag: str
329
+ attrs: dict[str, str] = field(default_factory=dict)
330
+ children: list[Node | str] = field(default_factory=list)
331
+ parent: Node | None = None
332
+
333
+ @property
334
+ def classes(self) -> tuple[str, ...]:
335
+ return tuple(self.attrs.get("class", "").split())
336
+
337
+ def elements(self) -> Iterator[Node]:
338
+ """This node's descendant elements, in document order."""
339
+ for child in self.children:
340
+ if isinstance(child, Node):
341
+ yield child
342
+ yield from child.elements()
343
+
344
+
345
+ class _TreeBuilder(HTMLParser):
346
+ def __init__(self) -> None:
347
+ super().__init__(convert_charrefs=True)
348
+ self.root = Node("#document")
349
+ self.stack = [self.root]
350
+
351
+ def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:
352
+ if tag in _IMPLIED_CLOSE:
353
+ closes, scope = _IMPLIED_CLOSE[tag]
354
+ for i in range(len(self.stack) - 1, 0, -1):
355
+ open_tag = self.stack[i].tag
356
+ if open_tag in closes:
357
+ del self.stack[i:]
358
+ break
359
+ if open_tag in scope:
360
+ break
361
+ node = Node(tag, {k.lower(): v or "" for k, v in attrs}, parent=self.stack[-1])
362
+ self.stack[-1].children.append(node)
363
+ if tag not in VOID:
364
+ self.stack.append(node)
365
+
366
+ def handle_startendtag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:
367
+ self.handle_starttag(tag, attrs)
368
+ if tag not in VOID and self.stack[-1].tag == tag:
369
+ self.stack.pop()
370
+
371
+ def handle_endtag(self, tag: str) -> None:
372
+ for i in range(len(self.stack) - 1, 0, -1):
373
+ if self.stack[i].tag == tag:
374
+ del self.stack[i:]
375
+ return
376
+
377
+ def handle_data(self, data: str) -> None:
378
+ self.stack[-1].children.append(data)
379
+
380
+
381
+ def _is_furniture(node: Node, rules: RuleSet) -> bool:
382
+ if node.tag in rules.drop_tags:
383
+ return True
384
+ if "hidden" in node.attrs or node.attrs.get("aria-hidden") == "true":
385
+ return True
386
+ if rules.drop_page_chrome and node.tag in ("header", "footer"):
387
+ ancestor = node.parent
388
+ while ancestor is not None:
389
+ if ancestor.tag in ("article", "main", "section"):
390
+ break
391
+ ancestor = ancestor.parent
392
+ else:
393
+ return True
394
+ if rules.drop_furniture:
395
+ if node.attrs.get("role", "").lower() in FURNITURE_ROLES:
396
+ return True
397
+ tokens = (*node.classes, *node.attrs.get("id", "").split())
398
+ if any(FURNITURE_TOKEN.match(t) for t in tokens):
399
+ return True
400
+ return False
401
+
402
+
403
+ def _prune(node: Node, rules: RuleSet) -> None:
404
+ kept: list[Node | str] = []
405
+ for child in node.children:
406
+ if isinstance(child, Node):
407
+ if _is_furniture(child, rules):
408
+ continue
409
+ _prune(child, rules)
410
+ kept.append(child)
411
+ node.children = kept
412
+
413
+
414
+ # --- rendering to text ---------------------------------------------------------------------------
415
+
416
+
417
+ def _inline(node: Node | str) -> str:
418
+ if isinstance(node, str):
419
+ return node
420
+ if node.tag == "br":
421
+ return " "
422
+ return " ".join(_inline(c) for c in node.children)
423
+
424
+
425
+ def _render(node: Node | str, out: list[str]) -> None:
426
+ """Append text to ``out``; a ``"\\n"`` entry is a block boundary."""
427
+ if isinstance(node, str):
428
+ out.append(_SPACE.sub(" ", node))
429
+ return
430
+ if node.tag == "tr":
431
+ cells = [c for c in node.children if isinstance(c, Node) and c.tag in ("td", "th")]
432
+ out += ["\n", " | ".join(_SPACE.sub(" ", _inline(c)).strip() for c in cells), "\n"]
433
+ return
434
+ if node.tag == "br":
435
+ out.append("\n")
436
+ return
437
+ block = node.tag in BLOCK
438
+ if block:
439
+ out.append("\n")
440
+ for child in node.children:
441
+ _render(child, out)
442
+ if block:
443
+ out.append("\n")
444
+
445
+
446
+ def _clean_line(line: str, rules: RuleSet) -> str:
447
+ if rules.strip_tracking:
448
+ line = _URL.sub(lambda m: canonical_url(m.group(0)), line)
449
+ if rules.strip_volatile:
450
+ for pattern in VOLATILE_PATTERNS:
451
+ line = pattern.sub(" ", line)
452
+ line = _SPACE.sub(" ", line).strip()
453
+ return "" if _EMPTY_LINE.match(line) else line
454
+
455
+
456
+ def _to_text(parts: list[str], rules: RuleSet) -> str:
457
+ lines = (_clean_line(line, rules) for line in "".join(parts).split("\n"))
458
+ return "\n".join(line for line in lines if line)
459
+
460
+
461
+ def _nodes_text(nodes: list[Node], rules: RuleSet) -> str:
462
+ parts: list[str] = []
463
+ for node in nodes:
464
+ _render(node, parts)
465
+ parts.append("\n")
466
+ return _to_text(parts, rules)
467
+
468
+
469
+ # --- documents -----------------------------------------------------------------------------------
470
+
471
+
472
+ @dataclass
473
+ class Document:
474
+ """A normalised page: its full text, and (for HTML) the cleaned tree for locators."""
475
+
476
+ rules: RuleSet
477
+ text: str
478
+ root: Node | None = None
479
+
480
+
481
+ def normalise_document(body: bytes, rules: RuleSet, *, charset: str | None = None) -> Document:
482
+ if body.lstrip().startswith(b"%PDF"):
483
+ raise UnsupportedContentError("pdf")
484
+ raw = body.decode(charset or "utf-8", errors="replace")
485
+ raw = unicodedata.normalize("NFKC", raw).replace("\r\n", "\n").replace("\r", "\n")
486
+ if rules.content == "text":
487
+ return Document(rules, _to_text([raw], rules))
488
+ builder = _TreeBuilder()
489
+ builder.feed(raw)
490
+ builder.close()
491
+ root = builder.root
492
+ _prune(root, rules)
493
+ return Document(rules, _nodes_text([root], rules), root)
494
+
495
+
496
+ def select_region(doc: Document, locator: Locator) -> str | None:
497
+ """The normalised text of a cited region, or ``None`` when the locator matches nothing."""
498
+ if locator.kind == "page":
499
+ return doc.text or None
500
+ if doc.root is None:
501
+ raise ValueError(f"{locator.kind} locators need an HTML document")
502
+ if locator.kind == "table":
503
+ tables = [n for n in doc.root.elements() if n.tag == "table"]
504
+ index = int(locator.value)
505
+ nodes = [tables[index]] if index < len(tables) else []
506
+ elif locator.kind == "css":
507
+ steps = _parse_selector(locator.value)
508
+ nodes = [n for n in doc.root.elements() if _matches(n, steps)]
509
+ else:
510
+ nodes = _heading_section(doc.root, locator.value)
511
+ return (_nodes_text(nodes, doc.rules) or None) if nodes else None
512
+
513
+
514
+ def _matches_compound(node: Node, c: _Compound) -> bool:
515
+ if c.tag and node.tag != c.tag:
516
+ return False
517
+ if c.id and node.attrs.get("id") != c.id:
518
+ return False
519
+ if any(cls not in node.classes for cls in c.classes):
520
+ return False
521
+ for name, value in c.attrs:
522
+ if name not in node.attrs or (value is not None and node.attrs[name] != value):
523
+ return False
524
+ return True
525
+
526
+
527
+ def _matches(node: Node, steps: tuple[tuple[str, _Compound], ...]) -> bool:
528
+ combinator, last = steps[-1]
529
+ if not _matches_compound(node, last):
530
+ return False
531
+ if len(steps) == 1:
532
+ return True
533
+ rest = steps[:-1]
534
+ ancestor = node.parent
535
+ while ancestor is not None and ancestor.tag != "#document":
536
+ if _matches(ancestor, rest):
537
+ return True
538
+ if combinator == ">":
539
+ return False
540
+ ancestor = ancestor.parent
541
+ return False
542
+
543
+
544
+ def _heading_level(node: Node) -> int | None:
545
+ if node.tag in HEADINGS:
546
+ return HEADINGS[node.tag]
547
+ for descendant in node.elements():
548
+ if descendant.tag in HEADINGS:
549
+ return HEADINGS[descendant.tag]
550
+ return None
551
+
552
+
553
+ def _heading_section(root: Node, anchor: str) -> list[Node]:
554
+ wanted = anchor.removeprefix("#")
555
+ folded = _SPACE.sub(" ", wanted).strip().casefold()
556
+ heading = None
557
+ for node in root.elements():
558
+ if node.tag not in HEADINGS:
559
+ continue
560
+ ids = {node.attrs.get("id")} | {
561
+ d.attrs.get("id") or d.attrs.get("name") for d in node.elements()
562
+ }
563
+ if wanted in ids or _SPACE.sub(" ", _inline(node)).strip().casefold() == folded:
564
+ heading = node
565
+ break
566
+ if heading is None:
567
+ return []
568
+ level = HEADINGS[heading.tag]
569
+ # A heading wrapped alone in a container (``<div><h2/></div>``) sections by the wrapper.
570
+ start = heading
571
+ for _ in range(2):
572
+ parent = start.parent
573
+ if parent is None or parent.tag in ("#document", "body", "main", "article", "section"):
574
+ break
575
+ if any(isinstance(c, Node) and c is not start for c in parent.children):
576
+ break
577
+ start = parent
578
+ assert start.parent is not None
579
+ siblings = [c for c in start.parent.children if isinstance(c, Node)]
580
+ section = [start]
581
+ for sibling in siblings[siblings.index(start) + 1 :]:
582
+ other = _heading_level(sibling)
583
+ if other is not None and other <= level:
584
+ break
585
+ section.append(sibling)
586
+ return section
587
+
588
+
589
+ # --- urls and fingerprints -----------------------------------------------------------------------
590
+
591
+
592
+ def canonical_url(url: str) -> str:
593
+ """``url`` without tracking query parameters or a fragment."""
594
+ parts = urlsplit(url)
595
+ query = [
596
+ (k, v)
597
+ for k, v in parse_qsl(parts.query, keep_blank_values=True)
598
+ if k.lower() not in TRACKING_PARAMS and not k.lower().startswith(TRACKING_PREFIXES)
599
+ ]
600
+ return urlunsplit((parts.scheme, parts.netloc, parts.path, urlencode(query), ""))
601
+
602
+
603
+ def fingerprint(text: str) -> str:
604
+ return "sha256:" + hashlib.sha256(text.encode("utf-8")).hexdigest()