sluicer 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
sluicer/__init__.py ADDED
@@ -0,0 +1,22 @@
1
+ """Deterministic extraction of the data a web page already declares."""
2
+
3
+ from sluicer.api import Extraction, extract
4
+ from sluicer.declared.merge import Field, Record
5
+ from sluicer.declared.microformats import MicroformatsExtraMissing
6
+ from sluicer.markdown import MarkdownExtraMissing, to_markdown
7
+ from sluicer.structure import induce
8
+ from sluicer.summary import SummaryField
9
+
10
+ __all__ = [
11
+ "Extraction",
12
+ "Field",
13
+ "MarkdownExtraMissing",
14
+ "MicroformatsExtraMissing",
15
+ "Record",
16
+ "SummaryField",
17
+ "__version__",
18
+ "extract",
19
+ "induce",
20
+ "to_markdown",
21
+ ]
22
+ __version__ = "0.2.0"
sluicer/api.py ADDED
@@ -0,0 +1,128 @@
1
+ """The one call most people will make."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from dataclasses import dataclass, field
6
+
7
+ from sluicer.declared.dublincore import read_dublincore
8
+ from sluicer.declared.htmlmeta import read_htmlmeta
9
+ from sluicer.declared.jsonld import read_jsonld
10
+ from sluicer.declared.merge import ABOUT_A_THING, Record, merge
11
+ from sluicer.declared.microdata import read_microdata
12
+ from sluicer.declared.microformats import read_microformats
13
+ from sluicer.declared.opengraph import read_opengraph
14
+ from sluicer.declared.rdfa import read_rdfa
15
+ from sluicer.declared.twitter import read_twitter
16
+ from sluicer.document import load
17
+ from sluicer.structure import induce as induce_records
18
+ from sluicer.summary import SummaryField, summarise
19
+
20
+
21
+ @dataclass
22
+ class Extraction:
23
+ """What Sluicer found in one page, and where it came from.
24
+
25
+ ``url`` is the address given to ``extract``. ``summary`` answers the
26
+ questions most callers ask -- title, author, date, price -- one value each,
27
+ chosen from the records by fixed rules, each naming its reader and key (see
28
+ ``sluicer.summary.FIELDS``). ``records`` is everything the page declared.
29
+ ``sources`` names every reader that found something.
30
+ """
31
+
32
+ url: str | None = None
33
+ summary: dict[str, SummaryField] = field(default_factory=dict)
34
+ records: list[Record] = field(default_factory=list)
35
+ sources: list[str] = field(default_factory=list)
36
+
37
+
38
+ def extract(
39
+ html: str | bytes,
40
+ url: str | None = None,
41
+ induce: bool = False,
42
+ microformats: bool = False,
43
+ ) -> Extraction:
44
+ """Read the structured data ``html`` declares, merged, with its provenance.
45
+
46
+ Args:
47
+ html: the page. Bytes are best: the page's own charset is then honoured
48
+ (see ``sluicer.document.load``).
49
+ url: the address the page came from, used to resolve its links.
50
+ induce: when the page declares nothing about the things on it, also
51
+ read the rows its markup repeats; those fields say
52
+ ``source="induced"``. Never fills a gap in a declared record.
53
+ microformats: also read microformats2. Off by default; needs
54
+ ``sluicer[microformats]``.
55
+
56
+ Returns:
57
+ An ``Extraction``: the ``summary``, the ``records`` (a record with no
58
+ field is never reported), and the ``sources`` that found something,
59
+ in the order of precedence -- JSON-LD, microdata, microformats, RDFa,
60
+ Dublin Core, OpenGraph, the Twitter card, HTML's own meta names.
61
+
62
+ Raises:
63
+ MicroformatsExtraMissing: ``microformats=True`` without the extra.
64
+ Nothing else: any input, however broken, is read or reported empty.
65
+
66
+ Why the order is what it is, and when induction runs, is in
67
+ ``docs/design-notes.md``.
68
+ """
69
+ doc = load(html, url=url)
70
+ jsonld = read_jsonld(doc)
71
+ microdata = read_microdata(doc)
72
+ # Not calling the reader is what keeps mf2py unimported on a base install.
73
+ found_microformats = read_microformats(doc) if microformats else []
74
+ rdfa = read_rdfa(doc)
75
+ dublincore = read_dublincore(doc)
76
+ opengraph = read_opengraph(doc)
77
+ twitter = read_twitter(doc)
78
+ htmlmeta = read_htmlmeta(doc)
79
+
80
+ sources = [
81
+ name
82
+ for name, found in (
83
+ ("jsonld", jsonld),
84
+ ("microdata", microdata),
85
+ ("microformats", found_microformats),
86
+ ("rdfa", rdfa),
87
+ ("dublincore", dublincore),
88
+ ("opengraph", opengraph),
89
+ ("twitter", twitter),
90
+ ("html", htmlmeta),
91
+ )
92
+ if found
93
+ ]
94
+ records = merge(
95
+ jsonld=jsonld,
96
+ microdata=microdata,
97
+ microformats=found_microformats,
98
+ rdfa=rdfa,
99
+ dublincore=dublincore,
100
+ opengraph=opengraph,
101
+ twitter=twitter,
102
+ htmlmeta=htmlmeta,
103
+ )
104
+ # A record carrying no field is a type and nothing else -- a ``WebPage``
105
+ # with only an ``@id``, a microformats root that was a CSS class -- and an
106
+ # empty value is not a value, whole records included.
107
+ records = [record for record in records if record.fields]
108
+ summary = summarise(doc, records, dublincore, opengraph, twitter, htmlmeta)
109
+ if induce and not _declared_about_its_things(records):
110
+ induced = induce_records(doc)
111
+ if induced:
112
+ records = [*records, *induced]
113
+ sources = [*sources, "induced"]
114
+ return Extraction(url=url, summary=summary, records=records, sources=sources)
115
+
116
+
117
+ def _declared_about_its_things(records: list[Record]) -> bool:
118
+ """True when a reader produced a field about a thing on the page.
119
+
120
+ A field, not a parse: an empty declaration produces a record with nothing in
121
+ it. And a field about a thing: OpenGraph describes the page it sits on, so
122
+ its fields answer "what is this page", never "what is in this list".
123
+ """
124
+ return any(
125
+ field.source in ABOUT_A_THING
126
+ for record in records
127
+ for field in record.fields.values()
128
+ )
sluicer/cli.py ADDED
@@ -0,0 +1,389 @@
1
+ """Command line front door.
2
+
3
+ Exit codes follow grep: 0 when something was found -- a record, or at least one
4
+ summary answer -- 1 when the page was read and gives nothing at all, 2 when it
5
+ could not be read. A script can tell "this page gives nothing" from "the fetch
6
+ failed" without parsing English. ``run`` and ``heal`` add 3: a page broke the
7
+ extractor's contract, or healing lost a field, and that is never a success.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import json
13
+ import logging
14
+ import sys
15
+ from dataclasses import asdict
16
+ from pathlib import Path
17
+ from typing import NoReturn
18
+
19
+ import click
20
+
21
+ from sluicer.api import extract as extract_html
22
+ from sluicer.declared.microformats import MicroformatsExtraMissing
23
+ from sluicer.extractor import (
24
+ LOSSES,
25
+ Extractor,
26
+ NothingToLearn,
27
+ compile_extractor,
28
+ heal as heal_extractor,
29
+ run_extractor,
30
+ )
31
+ from sluicer.fetch import FetchFailed, RobotsRefused, fetch as fetch_url
32
+ from sluicer.fetch.result import Fetched
33
+ from sluicer.fetch.scrapling_rungs import FetchExtraMissing
34
+ from sluicer.markdown import MarkdownExtraMissing, to_markdown
35
+
36
+ NOTHING_FOUND = 1
37
+ COULD_NOT_READ = 2
38
+ CONTRACT_BROKEN = 3
39
+
40
+
41
+ @click.group()
42
+ @click.version_option(package_name="sluicer")
43
+ def main() -> None:
44
+ """Turn a web page into structured data with no model in the loop.
45
+
46
+ SOURCE is a URL, a saved HTML file, or - for standard input.
47
+ """
48
+ _quiet_scrapling()
49
+
50
+
51
+ def _quiet_scrapling() -> None:
52
+ """Keep scrapling's own log lines out of this command's stderr.
53
+
54
+ Every one of them is said again here, better: a request is part of the
55
+ ladder this command reports, and a failure is the message it exits with.
56
+ Printed as well, a failed fetch read twice, once in scrapling's words.
57
+
58
+ A filter and not a level: scrapling sets its logger to INFO when it is
59
+ imported, which happens later, at the first fetch, and would undo a level
60
+ set here. A filter on the logger survives that.
61
+ """
62
+ logging.getLogger("scrapling").addFilter(
63
+ lambda record: record.levelno >= logging.CRITICAL
64
+ )
65
+
66
+
67
+ def _fail(message: str, cause: BaseException | None = None) -> NoReturn:
68
+ click.echo(message, err=True)
69
+ raise SystemExit(COULD_NOT_READ) from cause
70
+
71
+
72
+ _fetch_options = [
73
+ click.option(
74
+ "--stealth",
75
+ is_flag=True,
76
+ help="Allow the stealth rung, which does not announce itself.",
77
+ ),
78
+ click.option(
79
+ "--no-robots",
80
+ is_flag=True,
81
+ help="Fetch even where the site's robots.txt says no.",
82
+ ),
83
+ click.option(
84
+ "--url",
85
+ "base_url",
86
+ metavar="URL",
87
+ help="The address a file or stdin came from, to resolve its links.",
88
+ ),
89
+ ]
90
+
91
+
92
+ def _with_fetch_options(command: click.decorators.FC) -> click.decorators.FC:
93
+ for option in reversed(_fetch_options):
94
+ command = option(command)
95
+ return command
96
+
97
+
98
+ def _read_source(
99
+ source: str,
100
+ stealth: bool = False,
101
+ no_robots: bool = False,
102
+ base_url: str | None = None,
103
+ ) -> tuple[str | bytes, str | None, Fetched | None]:
104
+ """Return the HTML of ``source``, the URL to attribute it to, and the fetch record.
105
+
106
+ ``source`` is a URL, a path, or ``-`` for standard input. Every way of
107
+ failing to read it -- a missing extra, a refusal, a failed fetch, a missing,
108
+ empty or non-file path -- exits here with a message and ``COULD_NOT_READ``,
109
+ one place for both commands. The fetch record is None for a file or stdin.
110
+ """
111
+ if source.lower().startswith(("http://", "https://")):
112
+ # Only FetchExtraMissing, not ImportError: an import failure inside a
113
+ # working scrapling install is a bug and keeps its traceback.
114
+ try:
115
+ fetched = fetch_url(source, stealth=stealth, obey_robots=not no_robots)
116
+ except FetchExtraMissing as missing:
117
+ _fail(str(missing), missing)
118
+ except RobotsRefused as refused:
119
+ # The site told us no: an answer, not a malfunction.
120
+ _fail(str(refused), refused)
121
+ except FetchFailed as failed:
122
+ # Every rung failed, whatever library it was built on: a browser's
123
+ # timeout, for one, is not an OSError.
124
+ _fail(str(failed), failed)
125
+ except (OSError, ValueError) as failure:
126
+ # An operational failure is a message and a bug is a traceback.
127
+ # OSError covers down, unresolvable and timed out; ValueError is a
128
+ # rung that came back with no HTML. Anything else keeps its
129
+ # traceback, deliberately.
130
+ _fail(
131
+ f"Could not fetch {source}: {type(failure).__name__}: {failure}",
132
+ failure,
133
+ )
134
+ return fetched.html, fetched.url, fetched
135
+
136
+ if source == "-":
137
+ data = sys.stdin.buffer.read()
138
+ if not data.strip():
139
+ _fail("Standard input contains no HTML.")
140
+ return data, base_url, None
141
+
142
+ path = Path(source)
143
+
144
+ # By hand, not click.Path(exists=True): the same argument also takes a URL.
145
+ if not path.is_file():
146
+ _fail(
147
+ f"{source} is not a file." if path.exists() else f"{source} does not exist."
148
+ )
149
+
150
+ # Bytes, not text: decoding here would pick the process default before
151
+ # the page's own charset declaration is read, and destroy every byte that
152
+ # was not UTF-8. extract() and to_markdown() both decode bytes properly.
153
+ data = path.read_bytes()
154
+
155
+ # An empty file is a user mistake and gets a message about the file;
156
+ # load() would quietly read it as a page that declares nothing.
157
+ if not data.strip():
158
+ _fail("This file contains no HTML.")
159
+
160
+ # A path is not an address: handed to the readers as the page's URL it
161
+ # would resolve every relative link against the file name.
162
+ return data, base_url, None
163
+
164
+
165
+ @main.command()
166
+ @click.argument("source")
167
+ @click.option(
168
+ "--induce",
169
+ is_flag=True,
170
+ help="Also read the rows a page repeats when it declares nothing about them.",
171
+ )
172
+ @click.option(
173
+ "--microformats",
174
+ is_flag=True,
175
+ help="Also read microformats2 (needs sluicer[microformats]).",
176
+ )
177
+ @_with_fetch_options
178
+ def extract(
179
+ source: str,
180
+ induce: bool,
181
+ microformats: bool,
182
+ stealth: bool,
183
+ no_robots: bool,
184
+ base_url: str | None,
185
+ ) -> None:
186
+ """Read the structured data a URL, a file or stdin declares."""
187
+ html, url, fetched = _read_source(source, stealth, no_robots, base_url)
188
+ try:
189
+ result = extract_html(html, url=url, induce=induce, microformats=microformats)
190
+ except MicroformatsExtraMissing as missing:
191
+ _fail(str(missing), missing)
192
+ if not result.records and not result.summary:
193
+ click.echo("This page gives nothing: no record and no summary.", err=True)
194
+ if fetched is not None:
195
+ # What the page cost is reported even when it declared nothing.
196
+ click.echo(f"Fetch reached the '{fetched.rung}' rung.", err=True)
197
+ for climb in fetched.climbs:
198
+ click.echo(
199
+ f" {climb.from_rung} -> {climb.to_rung}: {climb.reason}",
200
+ err=True,
201
+ )
202
+ raise SystemExit(NOTHING_FOUND)
203
+ payload = asdict(result)
204
+ if fetched is not None:
205
+ payload["fetch"] = {
206
+ "rung": fetched.rung,
207
+ "status": fetched.status,
208
+ "climbs": [asdict(climb) for climb in fetched.climbs],
209
+ }
210
+ click.echo(json.dumps(payload, indent=2, ensure_ascii=False))
211
+
212
+
213
+ @main.command()
214
+ @click.argument("source")
215
+ @_with_fetch_options
216
+ def markdown(source: str, stealth: bool, no_robots: bool, base_url: str | None) -> None:
217
+ """Print the main content of a URL, a file or stdin as markdown."""
218
+ html, url, _fetched = _read_source(source, stealth, no_robots, base_url)
219
+ # MarkdownExtraMissing: trafilatura is not installed; the message says how.
220
+ try:
221
+ content = to_markdown(html, url=url)
222
+ except MarkdownExtraMissing as missing:
223
+ _fail(str(missing), missing)
224
+ if not content:
225
+ click.echo("This page has no main content.", err=True)
226
+ raise SystemExit(NOTHING_FOUND)
227
+ click.echo(content)
228
+
229
+
230
+ def _read_pages(
231
+ sources: tuple[str, ...], stealth: bool, no_robots: bool
232
+ ) -> list[tuple[str | bytes, str | None]]:
233
+ return [_read_source(source, stealth, no_robots, None)[:2] for source in sources]
234
+
235
+
236
+ def _load_extractor(path: str) -> Extractor:
237
+ try:
238
+ return Extractor.from_json(Path(path).read_text(encoding="utf-8"))
239
+ except (OSError, ValueError) as failure:
240
+ _fail(f"{path} is not an extractor: {failure}", failure)
241
+
242
+
243
+ @main.command("compile")
244
+ @click.argument("sources", nargs=-1, required=True)
245
+ @click.option("-o", "--output", required=True, help="Where to write the extractor.")
246
+ @click.option(
247
+ "--listing/--no-listing",
248
+ default=None,
249
+ help="Learn the rows the pages repeat (default: only if they declare no thing).",
250
+ )
251
+ @click.option("--stealth", is_flag=True, help="Allow the stealth rung.")
252
+ @click.option("--no-robots", is_flag=True, help="Fetch even where robots.txt says no.")
253
+ def compile_command(
254
+ sources: tuple[str, ...],
255
+ output: str,
256
+ listing: bool | None,
257
+ stealth: bool,
258
+ no_robots: bool,
259
+ ) -> None:
260
+ """Learn an extractor from pages of one template, and write it to a file."""
261
+ pages = _read_pages(sources, stealth, no_robots)
262
+ try:
263
+ extractor = compile_extractor(pages, listing=listing, names=list(sources))
264
+ except NothingToLearn as nothing:
265
+ click.echo(f"Learnt nothing: {nothing}.", err=True)
266
+ raise SystemExit(NOTHING_FOUND) from nothing
267
+ _write(output, extractor.to_json())
268
+ learnt = []
269
+ if extractor.listing is not None:
270
+ rows = extractor.listing.rows
271
+ learnt.append(
272
+ f"a listing at {extractor.listing.container}, {rows[0]}-{rows[1]} rows "
273
+ f"of {len(extractor.listing.fields)} fields"
274
+ )
275
+ if extractor.summary:
276
+ learnt.append(f"{len(extractor.summary)} summary answers")
277
+ if extractor.types:
278
+ learnt.append("declared " + ", ".join(extractor.types))
279
+ click.echo(f"Learnt {'; '.join(learnt)}. Wrote {output}.", err=True)
280
+ for note in extractor.notes:
281
+ click.echo(f"Note: {note}.", err=True)
282
+
283
+
284
+ def _write(path: str, text: str) -> None:
285
+ try:
286
+ Path(path).write_text(text, encoding="utf-8")
287
+ except OSError as failure:
288
+ _fail(f"Could not write {path}: {failure.strerror or failure}", failure)
289
+
290
+
291
+ @main.command("run")
292
+ @click.argument("extractor_file")
293
+ @click.argument("sources", nargs=-1, required=True)
294
+ @click.option("--stealth", is_flag=True, help="Allow the stealth rung.")
295
+ @click.option("--no-robots", is_flag=True, help="Fetch even where robots.txt says no.")
296
+ def run_command(
297
+ extractor_file: str, sources: tuple[str, ...], stealth: bool, no_robots: bool
298
+ ) -> None:
299
+ """Replay an extractor on pages, and exit 3 if any page broke its contract."""
300
+ extractor = _load_extractor(extractor_file)
301
+ pages = []
302
+ broken = False
303
+ for source, (html, url) in zip(
304
+ sources, _read_pages(sources, stealth, no_robots), strict=True
305
+ ):
306
+ run = run_extractor(extractor, html, url=url)
307
+ failed = [asdict(check) for check in run.checks if not check.ok]
308
+ pages.append(
309
+ {
310
+ "source": source,
311
+ "url": run.url,
312
+ "ok": run.ok,
313
+ "rows": run.rows,
314
+ "summary": run.summary,
315
+ "failed": failed,
316
+ }
317
+ )
318
+ for check in run.checks:
319
+ if not check.ok:
320
+ broken = True
321
+ click.echo(
322
+ f"FAILED {source}: expected {check.expected}, got {check.got}",
323
+ err=True,
324
+ )
325
+ click.echo(
326
+ json.dumps(
327
+ {"extractor": extractor_file, "pages": pages}, indent=2, ensure_ascii=False
328
+ )
329
+ )
330
+ if broken:
331
+ raise SystemExit(CONTRACT_BROKEN)
332
+
333
+
334
+ @main.command("heal")
335
+ @click.argument("extractor_file")
336
+ @click.argument("sources", nargs=-1, required=True)
337
+ @click.option("-o", "--output", help="Where to write the healed extractor.")
338
+ @click.option(
339
+ "--force",
340
+ is_flag=True,
341
+ help="Write the healed extractor even when healing lost something.",
342
+ )
343
+ @click.option("--stealth", is_flag=True, help="Allow the stealth rung.")
344
+ @click.option("--no-robots", is_flag=True, help="Fetch even where robots.txt says no.")
345
+ def heal_command(
346
+ extractor_file: str,
347
+ sources: tuple[str, ...],
348
+ output: str | None,
349
+ force: bool,
350
+ stealth: bool,
351
+ no_robots: bool,
352
+ ) -> None:
353
+ """Learn pages again and say what moved; write the result only with -o.
354
+
355
+ Exits 3 when a field, a summary answer, a type or the listing was lost for
356
+ good: healing moved what it could, and what it could not needs a person.
357
+ Nothing is written then without --force, so a lossy extractor never
358
+ quietly replaces the one that would have kept failing.
359
+ """
360
+ extractor = _load_extractor(extractor_file)
361
+ pages = _read_pages(sources, stealth, no_robots)
362
+ try:
363
+ healed, changes = heal_extractor(extractor, pages, names=list(sources))
364
+ except NothingToLearn as nothing:
365
+ click.echo(f"The pages hold nothing to heal from: {nothing}.", err=True)
366
+ raise SystemExit(CONTRACT_BROKEN) from nothing
367
+ for change in changes:
368
+ if change.kind == "kept":
369
+ continue
370
+ if change.before and change.after:
371
+ click.echo(f"{change.kind}: {change.before} -> {change.after}", err=True)
372
+ else:
373
+ click.echo(f"{change.kind}: {change.before or change.after}", err=True)
374
+ lost = any(c.kind in LOSSES for c in changes)
375
+ if output and (force or not lost):
376
+ _write(output, healed.to_json())
377
+ click.echo(f"Wrote {output}.", err=True)
378
+ elif output:
379
+ click.echo(
380
+ f"Did not write {output}: healing lost data. Pass --force to write it.",
381
+ err=True,
382
+ )
383
+ click.echo(
384
+ json.dumps(
385
+ {"changes": [asdict(c) for c in changes]}, indent=2, ensure_ascii=False
386
+ )
387
+ )
388
+ if lost:
389
+ raise SystemExit(CONTRACT_BROKEN)
@@ -0,0 +1 @@
1
+ """Readers for the structured data a page already declares."""
@@ -0,0 +1,37 @@
1
+ """Read Dublin Core meta tags.
2
+
3
+ Older than schema.org, and still the only structured data on many library,
4
+ university, repository and government pages: flat ``<meta name="DC.title">``
5
+ tags, sometimes with a ``scheme`` attribute naming the value's encoding rather
6
+ than adding a field. ``DC.`` (the fifteen original elements) and ``DCTERMS.``
7
+ (the refined set) turn up on the same page and answer the same questions, so
8
+ both are read into one flat mapping.
9
+
10
+ The whole name is matched case-insensitively and the key lowercased: the 1997
11
+ examples wrote ``DC.Title``, the HTML that copied them ``dc.title``, and
12
+ first-wins only means something across one spelling.
13
+
14
+ Dublin Core describes the document, so it is not in
15
+ ``sluicer.declared.merge.ABOUT_A_THING``.
16
+ """
17
+
18
+ from __future__ import annotations
19
+
20
+ from sluicer.document import Document
21
+
22
+ _PREFIXES = ("dc.", "dcterms.")
23
+
24
+
25
+ def read_dublincore(doc: Document) -> dict[str, str]:
26
+ """Return the Dublin Core meta tags, prefixes stripped, keys lowercased."""
27
+ found: dict[str, str] = {}
28
+ for meta in doc.tree.xpath("//meta[@name]"):
29
+ name = (meta.get("name") or "").strip().lower()
30
+ content = (meta.get("content") or "").strip()
31
+ if not content:
32
+ continue
33
+ for prefix in _PREFIXES:
34
+ if name.startswith(prefix):
35
+ found.setdefault(name[len(prefix) :], content)
36
+ break
37
+ return found
@@ -0,0 +1,44 @@
1
+ """Read the metadata names HTML defines for itself.
2
+
3
+ The oldest declaration on the web and still the most common: pages with no
4
+ JSON-LD, microdata or OpenGraph very often carry ``<meta name="description">``
5
+ and ``<meta name="author">``. Measured on 2026-09-22 across the 359 WCXB pages
6
+ Sluicer targets, ``description`` is on 87% of them and ``author`` on 29%.
7
+
8
+ It is a closed list: ``author``, ``description``, ``keywords``, ``generator``,
9
+ ``application-name`` and ``theme-color``. Every ``name=`` attribute would
10
+ report a page's ``csrf-token`` and build id as statements about it; browser
11
+ directives (``robots``, ``viewport``) are instructions to a client, not
12
+ statements about the page's subject.
13
+
14
+ These values arrive as ``source="html"``: no vocabulary, no schema, no type.
15
+ Extruct files ``<meta name="description">`` under Dublin Core, which never
16
+ claimed it. The reader runs last, so any real vocabulary wins and a bare tag
17
+ only fills a gap. It describes the document, so it is not in
18
+ ``sluicer.declared.merge.ABOUT_A_THING``.
19
+ """
20
+
21
+ from __future__ import annotations
22
+
23
+ from sluicer.declared.meta import read_named_meta
24
+ from sluicer.document import Document
25
+
26
+ # The metadata names the HTML standard defines, and nothing else. Written as
27
+ # a frozenset because it is a membership test and a closed list, and the
28
+ # closedness is the design: every name added here is a claim that the name
29
+ # says something about the page's subject rather than about its plumbing.
30
+ NAMES = frozenset(
31
+ {
32
+ "author",
33
+ "description",
34
+ "keywords",
35
+ "generator",
36
+ "application-name",
37
+ "theme-color",
38
+ }
39
+ )
40
+
41
+
42
+ def read_htmlmeta(doc: Document) -> dict[str, str]:
43
+ """Return the standard ``<meta name=...>`` metadata, keys lowercased."""
44
+ return read_named_meta(doc, NAMES)