sluicer 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sluicer/__init__.py +22 -0
- sluicer/api.py +128 -0
- sluicer/cli.py +389 -0
- sluicer/declared/__init__.py +1 -0
- sluicer/declared/dublincore.py +37 -0
- sluicer/declared/htmlmeta.py +44 -0
- sluicer/declared/jsonld.py +200 -0
- sluicer/declared/merge.py +241 -0
- sluicer/declared/meta.py +93 -0
- sluicer/declared/microdata.py +228 -0
- sluicer/declared/microformats.py +143 -0
- sluicer/declared/opengraph.py +53 -0
- sluicer/declared/rdfa.py +220 -0
- sluicer/declared/twitter.py +24 -0
- sluicer/declared/types.py +24 -0
- sluicer/document.py +249 -0
- sluicer/extractor.py +802 -0
- sluicer/extras.py +77 -0
- sluicer/fetch/__init__.py +19 -0
- sluicer/fetch/address.py +120 -0
- sluicer/fetch/identity.py +142 -0
- sluicer/fetch/ladder.py +254 -0
- sluicer/fetch/result.py +40 -0
- sluicer/fetch/rules.py +73 -0
- sluicer/fetch/scrapling_rungs.py +131 -0
- sluicer/markdown.py +54 -0
- sluicer/mcp_server.py +314 -0
- sluicer/py.typed +0 -0
- sluicer/structure/__init__.py +36 -0
- sluicer/structure/groups.py +97 -0
- sluicer/structure/records.py +213 -0
- sluicer/structure/shape.py +69 -0
- sluicer/summary.py +570 -0
- sluicer-0.2.0.dist-info/METADATA +340 -0
- sluicer-0.2.0.dist-info/RECORD +38 -0
- sluicer-0.2.0.dist-info/WHEEL +4 -0
- sluicer-0.2.0.dist-info/entry_points.txt +3 -0
- sluicer-0.2.0.dist-info/licenses/LICENSE +21 -0
sluicer/__init__.py
ADDED
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
"""Deterministic extraction of the data a web page already declares."""
|
|
2
|
+
|
|
3
|
+
from sluicer.api import Extraction, extract
|
|
4
|
+
from sluicer.declared.merge import Field, Record
|
|
5
|
+
from sluicer.declared.microformats import MicroformatsExtraMissing
|
|
6
|
+
from sluicer.markdown import MarkdownExtraMissing, to_markdown
|
|
7
|
+
from sluicer.structure import induce
|
|
8
|
+
from sluicer.summary import SummaryField
|
|
9
|
+
|
|
10
|
+
__all__ = [
|
|
11
|
+
"Extraction",
|
|
12
|
+
"Field",
|
|
13
|
+
"MarkdownExtraMissing",
|
|
14
|
+
"MicroformatsExtraMissing",
|
|
15
|
+
"Record",
|
|
16
|
+
"SummaryField",
|
|
17
|
+
"__version__",
|
|
18
|
+
"extract",
|
|
19
|
+
"induce",
|
|
20
|
+
"to_markdown",
|
|
21
|
+
]
|
|
22
|
+
__version__ = "0.2.0"
|
sluicer/api.py
ADDED
|
@@ -0,0 +1,128 @@
|
|
|
1
|
+
"""The one call most people will make."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from dataclasses import dataclass, field
|
|
6
|
+
|
|
7
|
+
from sluicer.declared.dublincore import read_dublincore
|
|
8
|
+
from sluicer.declared.htmlmeta import read_htmlmeta
|
|
9
|
+
from sluicer.declared.jsonld import read_jsonld
|
|
10
|
+
from sluicer.declared.merge import ABOUT_A_THING, Record, merge
|
|
11
|
+
from sluicer.declared.microdata import read_microdata
|
|
12
|
+
from sluicer.declared.microformats import read_microformats
|
|
13
|
+
from sluicer.declared.opengraph import read_opengraph
|
|
14
|
+
from sluicer.declared.rdfa import read_rdfa
|
|
15
|
+
from sluicer.declared.twitter import read_twitter
|
|
16
|
+
from sluicer.document import load
|
|
17
|
+
from sluicer.structure import induce as induce_records
|
|
18
|
+
from sluicer.summary import SummaryField, summarise
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
@dataclass
|
|
22
|
+
class Extraction:
|
|
23
|
+
"""What Sluicer found in one page, and where it came from.
|
|
24
|
+
|
|
25
|
+
``url`` is the address given to ``extract``. ``summary`` answers the
|
|
26
|
+
questions most callers ask -- title, author, date, price -- one value each,
|
|
27
|
+
chosen from the records by fixed rules, each naming its reader and key (see
|
|
28
|
+
``sluicer.summary.FIELDS``). ``records`` is everything the page declared.
|
|
29
|
+
``sources`` names every reader that found something.
|
|
30
|
+
"""
|
|
31
|
+
|
|
32
|
+
url: str | None = None
|
|
33
|
+
summary: dict[str, SummaryField] = field(default_factory=dict)
|
|
34
|
+
records: list[Record] = field(default_factory=list)
|
|
35
|
+
sources: list[str] = field(default_factory=list)
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def extract(
|
|
39
|
+
html: str | bytes,
|
|
40
|
+
url: str | None = None,
|
|
41
|
+
induce: bool = False,
|
|
42
|
+
microformats: bool = False,
|
|
43
|
+
) -> Extraction:
|
|
44
|
+
"""Read the structured data ``html`` declares, merged, with its provenance.
|
|
45
|
+
|
|
46
|
+
Args:
|
|
47
|
+
html: the page. Bytes are best: the page's own charset is then honoured
|
|
48
|
+
(see ``sluicer.document.load``).
|
|
49
|
+
url: the address the page came from, used to resolve its links.
|
|
50
|
+
induce: when the page declares nothing about the things on it, also
|
|
51
|
+
read the rows its markup repeats; those fields say
|
|
52
|
+
``source="induced"``. Never fills a gap in a declared record.
|
|
53
|
+
microformats: also read microformats2. Off by default; needs
|
|
54
|
+
``sluicer[microformats]``.
|
|
55
|
+
|
|
56
|
+
Returns:
|
|
57
|
+
An ``Extraction``: the ``summary``, the ``records`` (a record with no
|
|
58
|
+
field is never reported), and the ``sources`` that found something,
|
|
59
|
+
in the order of precedence -- JSON-LD, microdata, microformats, RDFa,
|
|
60
|
+
Dublin Core, OpenGraph, the Twitter card, HTML's own meta names.
|
|
61
|
+
|
|
62
|
+
Raises:
|
|
63
|
+
MicroformatsExtraMissing: ``microformats=True`` without the extra.
|
|
64
|
+
Nothing else: any input, however broken, is read or reported empty.
|
|
65
|
+
|
|
66
|
+
Why the order is what it is, and when induction runs, is in
|
|
67
|
+
``docs/design-notes.md``.
|
|
68
|
+
"""
|
|
69
|
+
doc = load(html, url=url)
|
|
70
|
+
jsonld = read_jsonld(doc)
|
|
71
|
+
microdata = read_microdata(doc)
|
|
72
|
+
# Not calling the reader is what keeps mf2py unimported on a base install.
|
|
73
|
+
found_microformats = read_microformats(doc) if microformats else []
|
|
74
|
+
rdfa = read_rdfa(doc)
|
|
75
|
+
dublincore = read_dublincore(doc)
|
|
76
|
+
opengraph = read_opengraph(doc)
|
|
77
|
+
twitter = read_twitter(doc)
|
|
78
|
+
htmlmeta = read_htmlmeta(doc)
|
|
79
|
+
|
|
80
|
+
sources = [
|
|
81
|
+
name
|
|
82
|
+
for name, found in (
|
|
83
|
+
("jsonld", jsonld),
|
|
84
|
+
("microdata", microdata),
|
|
85
|
+
("microformats", found_microformats),
|
|
86
|
+
("rdfa", rdfa),
|
|
87
|
+
("dublincore", dublincore),
|
|
88
|
+
("opengraph", opengraph),
|
|
89
|
+
("twitter", twitter),
|
|
90
|
+
("html", htmlmeta),
|
|
91
|
+
)
|
|
92
|
+
if found
|
|
93
|
+
]
|
|
94
|
+
records = merge(
|
|
95
|
+
jsonld=jsonld,
|
|
96
|
+
microdata=microdata,
|
|
97
|
+
microformats=found_microformats,
|
|
98
|
+
rdfa=rdfa,
|
|
99
|
+
dublincore=dublincore,
|
|
100
|
+
opengraph=opengraph,
|
|
101
|
+
twitter=twitter,
|
|
102
|
+
htmlmeta=htmlmeta,
|
|
103
|
+
)
|
|
104
|
+
# A record carrying no field is a type and nothing else -- a ``WebPage``
|
|
105
|
+
# with only an ``@id``, a microformats root that was a CSS class -- and an
|
|
106
|
+
# empty value is not a value, whole records included.
|
|
107
|
+
records = [record for record in records if record.fields]
|
|
108
|
+
summary = summarise(doc, records, dublincore, opengraph, twitter, htmlmeta)
|
|
109
|
+
if induce and not _declared_about_its_things(records):
|
|
110
|
+
induced = induce_records(doc)
|
|
111
|
+
if induced:
|
|
112
|
+
records = [*records, *induced]
|
|
113
|
+
sources = [*sources, "induced"]
|
|
114
|
+
return Extraction(url=url, summary=summary, records=records, sources=sources)
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def _declared_about_its_things(records: list[Record]) -> bool:
|
|
118
|
+
"""True when a reader produced a field about a thing on the page.
|
|
119
|
+
|
|
120
|
+
A field, not a parse: an empty declaration produces a record with nothing in
|
|
121
|
+
it. And a field about a thing: OpenGraph describes the page it sits on, so
|
|
122
|
+
its fields answer "what is this page", never "what is in this list".
|
|
123
|
+
"""
|
|
124
|
+
return any(
|
|
125
|
+
field.source in ABOUT_A_THING
|
|
126
|
+
for record in records
|
|
127
|
+
for field in record.fields.values()
|
|
128
|
+
)
|
sluicer/cli.py
ADDED
|
@@ -0,0 +1,389 @@
|
|
|
1
|
+
"""Command line front door.
|
|
2
|
+
|
|
3
|
+
Exit codes follow grep: 0 when something was found -- a record, or at least one
|
|
4
|
+
summary answer -- 1 when the page was read and gives nothing at all, 2 when it
|
|
5
|
+
could not be read. A script can tell "this page gives nothing" from "the fetch
|
|
6
|
+
failed" without parsing English. ``run`` and ``heal`` add 3: a page broke the
|
|
7
|
+
extractor's contract, or healing lost a field, and that is never a success.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import json
|
|
13
|
+
import logging
|
|
14
|
+
import sys
|
|
15
|
+
from dataclasses import asdict
|
|
16
|
+
from pathlib import Path
|
|
17
|
+
from typing import NoReturn
|
|
18
|
+
|
|
19
|
+
import click
|
|
20
|
+
|
|
21
|
+
from sluicer.api import extract as extract_html
|
|
22
|
+
from sluicer.declared.microformats import MicroformatsExtraMissing
|
|
23
|
+
from sluicer.extractor import (
|
|
24
|
+
LOSSES,
|
|
25
|
+
Extractor,
|
|
26
|
+
NothingToLearn,
|
|
27
|
+
compile_extractor,
|
|
28
|
+
heal as heal_extractor,
|
|
29
|
+
run_extractor,
|
|
30
|
+
)
|
|
31
|
+
from sluicer.fetch import FetchFailed, RobotsRefused, fetch as fetch_url
|
|
32
|
+
from sluicer.fetch.result import Fetched
|
|
33
|
+
from sluicer.fetch.scrapling_rungs import FetchExtraMissing
|
|
34
|
+
from sluicer.markdown import MarkdownExtraMissing, to_markdown
|
|
35
|
+
|
|
36
|
+
NOTHING_FOUND = 1
|
|
37
|
+
COULD_NOT_READ = 2
|
|
38
|
+
CONTRACT_BROKEN = 3
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
@click.group()
|
|
42
|
+
@click.version_option(package_name="sluicer")
|
|
43
|
+
def main() -> None:
|
|
44
|
+
"""Turn a web page into structured data with no model in the loop.
|
|
45
|
+
|
|
46
|
+
SOURCE is a URL, a saved HTML file, or - for standard input.
|
|
47
|
+
"""
|
|
48
|
+
_quiet_scrapling()
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def _quiet_scrapling() -> None:
|
|
52
|
+
"""Keep scrapling's own log lines out of this command's stderr.
|
|
53
|
+
|
|
54
|
+
Every one of them is said again here, better: a request is part of the
|
|
55
|
+
ladder this command reports, and a failure is the message it exits with.
|
|
56
|
+
Printed as well, a failed fetch read twice, once in scrapling's words.
|
|
57
|
+
|
|
58
|
+
A filter and not a level: scrapling sets its logger to INFO when it is
|
|
59
|
+
imported, which happens later, at the first fetch, and would undo a level
|
|
60
|
+
set here. A filter on the logger survives that.
|
|
61
|
+
"""
|
|
62
|
+
logging.getLogger("scrapling").addFilter(
|
|
63
|
+
lambda record: record.levelno >= logging.CRITICAL
|
|
64
|
+
)
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def _fail(message: str, cause: BaseException | None = None) -> NoReturn:
|
|
68
|
+
click.echo(message, err=True)
|
|
69
|
+
raise SystemExit(COULD_NOT_READ) from cause
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
_fetch_options = [
|
|
73
|
+
click.option(
|
|
74
|
+
"--stealth",
|
|
75
|
+
is_flag=True,
|
|
76
|
+
help="Allow the stealth rung, which does not announce itself.",
|
|
77
|
+
),
|
|
78
|
+
click.option(
|
|
79
|
+
"--no-robots",
|
|
80
|
+
is_flag=True,
|
|
81
|
+
help="Fetch even where the site's robots.txt says no.",
|
|
82
|
+
),
|
|
83
|
+
click.option(
|
|
84
|
+
"--url",
|
|
85
|
+
"base_url",
|
|
86
|
+
metavar="URL",
|
|
87
|
+
help="The address a file or stdin came from, to resolve its links.",
|
|
88
|
+
),
|
|
89
|
+
]
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def _with_fetch_options(command: click.decorators.FC) -> click.decorators.FC:
|
|
93
|
+
for option in reversed(_fetch_options):
|
|
94
|
+
command = option(command)
|
|
95
|
+
return command
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def _read_source(
|
|
99
|
+
source: str,
|
|
100
|
+
stealth: bool = False,
|
|
101
|
+
no_robots: bool = False,
|
|
102
|
+
base_url: str | None = None,
|
|
103
|
+
) -> tuple[str | bytes, str | None, Fetched | None]:
|
|
104
|
+
"""Return the HTML of ``source``, the URL to attribute it to, and the fetch record.
|
|
105
|
+
|
|
106
|
+
``source`` is a URL, a path, or ``-`` for standard input. Every way of
|
|
107
|
+
failing to read it -- a missing extra, a refusal, a failed fetch, a missing,
|
|
108
|
+
empty or non-file path -- exits here with a message and ``COULD_NOT_READ``,
|
|
109
|
+
one place for both commands. The fetch record is None for a file or stdin.
|
|
110
|
+
"""
|
|
111
|
+
if source.lower().startswith(("http://", "https://")):
|
|
112
|
+
# Only FetchExtraMissing, not ImportError: an import failure inside a
|
|
113
|
+
# working scrapling install is a bug and keeps its traceback.
|
|
114
|
+
try:
|
|
115
|
+
fetched = fetch_url(source, stealth=stealth, obey_robots=not no_robots)
|
|
116
|
+
except FetchExtraMissing as missing:
|
|
117
|
+
_fail(str(missing), missing)
|
|
118
|
+
except RobotsRefused as refused:
|
|
119
|
+
# The site told us no: an answer, not a malfunction.
|
|
120
|
+
_fail(str(refused), refused)
|
|
121
|
+
except FetchFailed as failed:
|
|
122
|
+
# Every rung failed, whatever library it was built on: a browser's
|
|
123
|
+
# timeout, for one, is not an OSError.
|
|
124
|
+
_fail(str(failed), failed)
|
|
125
|
+
except (OSError, ValueError) as failure:
|
|
126
|
+
# An operational failure is a message and a bug is a traceback.
|
|
127
|
+
# OSError covers down, unresolvable and timed out; ValueError is a
|
|
128
|
+
# rung that came back with no HTML. Anything else keeps its
|
|
129
|
+
# traceback, deliberately.
|
|
130
|
+
_fail(
|
|
131
|
+
f"Could not fetch {source}: {type(failure).__name__}: {failure}",
|
|
132
|
+
failure,
|
|
133
|
+
)
|
|
134
|
+
return fetched.html, fetched.url, fetched
|
|
135
|
+
|
|
136
|
+
if source == "-":
|
|
137
|
+
data = sys.stdin.buffer.read()
|
|
138
|
+
if not data.strip():
|
|
139
|
+
_fail("Standard input contains no HTML.")
|
|
140
|
+
return data, base_url, None
|
|
141
|
+
|
|
142
|
+
path = Path(source)
|
|
143
|
+
|
|
144
|
+
# By hand, not click.Path(exists=True): the same argument also takes a URL.
|
|
145
|
+
if not path.is_file():
|
|
146
|
+
_fail(
|
|
147
|
+
f"{source} is not a file." if path.exists() else f"{source} does not exist."
|
|
148
|
+
)
|
|
149
|
+
|
|
150
|
+
# Bytes, not text: decoding here would pick the process default before
|
|
151
|
+
# the page's own charset declaration is read, and destroy every byte that
|
|
152
|
+
# was not UTF-8. extract() and to_markdown() both decode bytes properly.
|
|
153
|
+
data = path.read_bytes()
|
|
154
|
+
|
|
155
|
+
# An empty file is a user mistake and gets a message about the file;
|
|
156
|
+
# load() would quietly read it as a page that declares nothing.
|
|
157
|
+
if not data.strip():
|
|
158
|
+
_fail("This file contains no HTML.")
|
|
159
|
+
|
|
160
|
+
# A path is not an address: handed to the readers as the page's URL it
|
|
161
|
+
# would resolve every relative link against the file name.
|
|
162
|
+
return data, base_url, None
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
@main.command()
|
|
166
|
+
@click.argument("source")
|
|
167
|
+
@click.option(
|
|
168
|
+
"--induce",
|
|
169
|
+
is_flag=True,
|
|
170
|
+
help="Also read the rows a page repeats when it declares nothing about them.",
|
|
171
|
+
)
|
|
172
|
+
@click.option(
|
|
173
|
+
"--microformats",
|
|
174
|
+
is_flag=True,
|
|
175
|
+
help="Also read microformats2 (needs sluicer[microformats]).",
|
|
176
|
+
)
|
|
177
|
+
@_with_fetch_options
|
|
178
|
+
def extract(
|
|
179
|
+
source: str,
|
|
180
|
+
induce: bool,
|
|
181
|
+
microformats: bool,
|
|
182
|
+
stealth: bool,
|
|
183
|
+
no_robots: bool,
|
|
184
|
+
base_url: str | None,
|
|
185
|
+
) -> None:
|
|
186
|
+
"""Read the structured data a URL, a file or stdin declares."""
|
|
187
|
+
html, url, fetched = _read_source(source, stealth, no_robots, base_url)
|
|
188
|
+
try:
|
|
189
|
+
result = extract_html(html, url=url, induce=induce, microformats=microformats)
|
|
190
|
+
except MicroformatsExtraMissing as missing:
|
|
191
|
+
_fail(str(missing), missing)
|
|
192
|
+
if not result.records and not result.summary:
|
|
193
|
+
click.echo("This page gives nothing: no record and no summary.", err=True)
|
|
194
|
+
if fetched is not None:
|
|
195
|
+
# What the page cost is reported even when it declared nothing.
|
|
196
|
+
click.echo(f"Fetch reached the '{fetched.rung}' rung.", err=True)
|
|
197
|
+
for climb in fetched.climbs:
|
|
198
|
+
click.echo(
|
|
199
|
+
f" {climb.from_rung} -> {climb.to_rung}: {climb.reason}",
|
|
200
|
+
err=True,
|
|
201
|
+
)
|
|
202
|
+
raise SystemExit(NOTHING_FOUND)
|
|
203
|
+
payload = asdict(result)
|
|
204
|
+
if fetched is not None:
|
|
205
|
+
payload["fetch"] = {
|
|
206
|
+
"rung": fetched.rung,
|
|
207
|
+
"status": fetched.status,
|
|
208
|
+
"climbs": [asdict(climb) for climb in fetched.climbs],
|
|
209
|
+
}
|
|
210
|
+
click.echo(json.dumps(payload, indent=2, ensure_ascii=False))
|
|
211
|
+
|
|
212
|
+
|
|
213
|
+
@main.command()
|
|
214
|
+
@click.argument("source")
|
|
215
|
+
@_with_fetch_options
|
|
216
|
+
def markdown(source: str, stealth: bool, no_robots: bool, base_url: str | None) -> None:
|
|
217
|
+
"""Print the main content of a URL, a file or stdin as markdown."""
|
|
218
|
+
html, url, _fetched = _read_source(source, stealth, no_robots, base_url)
|
|
219
|
+
# MarkdownExtraMissing: trafilatura is not installed; the message says how.
|
|
220
|
+
try:
|
|
221
|
+
content = to_markdown(html, url=url)
|
|
222
|
+
except MarkdownExtraMissing as missing:
|
|
223
|
+
_fail(str(missing), missing)
|
|
224
|
+
if not content:
|
|
225
|
+
click.echo("This page has no main content.", err=True)
|
|
226
|
+
raise SystemExit(NOTHING_FOUND)
|
|
227
|
+
click.echo(content)
|
|
228
|
+
|
|
229
|
+
|
|
230
|
+
def _read_pages(
|
|
231
|
+
sources: tuple[str, ...], stealth: bool, no_robots: bool
|
|
232
|
+
) -> list[tuple[str | bytes, str | None]]:
|
|
233
|
+
return [_read_source(source, stealth, no_robots, None)[:2] for source in sources]
|
|
234
|
+
|
|
235
|
+
|
|
236
|
+
def _load_extractor(path: str) -> Extractor:
|
|
237
|
+
try:
|
|
238
|
+
return Extractor.from_json(Path(path).read_text(encoding="utf-8"))
|
|
239
|
+
except (OSError, ValueError) as failure:
|
|
240
|
+
_fail(f"{path} is not an extractor: {failure}", failure)
|
|
241
|
+
|
|
242
|
+
|
|
243
|
+
@main.command("compile")
|
|
244
|
+
@click.argument("sources", nargs=-1, required=True)
|
|
245
|
+
@click.option("-o", "--output", required=True, help="Where to write the extractor.")
|
|
246
|
+
@click.option(
|
|
247
|
+
"--listing/--no-listing",
|
|
248
|
+
default=None,
|
|
249
|
+
help="Learn the rows the pages repeat (default: only if they declare no thing).",
|
|
250
|
+
)
|
|
251
|
+
@click.option("--stealth", is_flag=True, help="Allow the stealth rung.")
|
|
252
|
+
@click.option("--no-robots", is_flag=True, help="Fetch even where robots.txt says no.")
|
|
253
|
+
def compile_command(
|
|
254
|
+
sources: tuple[str, ...],
|
|
255
|
+
output: str,
|
|
256
|
+
listing: bool | None,
|
|
257
|
+
stealth: bool,
|
|
258
|
+
no_robots: bool,
|
|
259
|
+
) -> None:
|
|
260
|
+
"""Learn an extractor from pages of one template, and write it to a file."""
|
|
261
|
+
pages = _read_pages(sources, stealth, no_robots)
|
|
262
|
+
try:
|
|
263
|
+
extractor = compile_extractor(pages, listing=listing, names=list(sources))
|
|
264
|
+
except NothingToLearn as nothing:
|
|
265
|
+
click.echo(f"Learnt nothing: {nothing}.", err=True)
|
|
266
|
+
raise SystemExit(NOTHING_FOUND) from nothing
|
|
267
|
+
_write(output, extractor.to_json())
|
|
268
|
+
learnt = []
|
|
269
|
+
if extractor.listing is not None:
|
|
270
|
+
rows = extractor.listing.rows
|
|
271
|
+
learnt.append(
|
|
272
|
+
f"a listing at {extractor.listing.container}, {rows[0]}-{rows[1]} rows "
|
|
273
|
+
f"of {len(extractor.listing.fields)} fields"
|
|
274
|
+
)
|
|
275
|
+
if extractor.summary:
|
|
276
|
+
learnt.append(f"{len(extractor.summary)} summary answers")
|
|
277
|
+
if extractor.types:
|
|
278
|
+
learnt.append("declared " + ", ".join(extractor.types))
|
|
279
|
+
click.echo(f"Learnt {'; '.join(learnt)}. Wrote {output}.", err=True)
|
|
280
|
+
for note in extractor.notes:
|
|
281
|
+
click.echo(f"Note: {note}.", err=True)
|
|
282
|
+
|
|
283
|
+
|
|
284
|
+
def _write(path: str, text: str) -> None:
|
|
285
|
+
try:
|
|
286
|
+
Path(path).write_text(text, encoding="utf-8")
|
|
287
|
+
except OSError as failure:
|
|
288
|
+
_fail(f"Could not write {path}: {failure.strerror or failure}", failure)
|
|
289
|
+
|
|
290
|
+
|
|
291
|
+
@main.command("run")
|
|
292
|
+
@click.argument("extractor_file")
|
|
293
|
+
@click.argument("sources", nargs=-1, required=True)
|
|
294
|
+
@click.option("--stealth", is_flag=True, help="Allow the stealth rung.")
|
|
295
|
+
@click.option("--no-robots", is_flag=True, help="Fetch even where robots.txt says no.")
|
|
296
|
+
def run_command(
|
|
297
|
+
extractor_file: str, sources: tuple[str, ...], stealth: bool, no_robots: bool
|
|
298
|
+
) -> None:
|
|
299
|
+
"""Replay an extractor on pages, and exit 3 if any page broke its contract."""
|
|
300
|
+
extractor = _load_extractor(extractor_file)
|
|
301
|
+
pages = []
|
|
302
|
+
broken = False
|
|
303
|
+
for source, (html, url) in zip(
|
|
304
|
+
sources, _read_pages(sources, stealth, no_robots), strict=True
|
|
305
|
+
):
|
|
306
|
+
run = run_extractor(extractor, html, url=url)
|
|
307
|
+
failed = [asdict(check) for check in run.checks if not check.ok]
|
|
308
|
+
pages.append(
|
|
309
|
+
{
|
|
310
|
+
"source": source,
|
|
311
|
+
"url": run.url,
|
|
312
|
+
"ok": run.ok,
|
|
313
|
+
"rows": run.rows,
|
|
314
|
+
"summary": run.summary,
|
|
315
|
+
"failed": failed,
|
|
316
|
+
}
|
|
317
|
+
)
|
|
318
|
+
for check in run.checks:
|
|
319
|
+
if not check.ok:
|
|
320
|
+
broken = True
|
|
321
|
+
click.echo(
|
|
322
|
+
f"FAILED {source}: expected {check.expected}, got {check.got}",
|
|
323
|
+
err=True,
|
|
324
|
+
)
|
|
325
|
+
click.echo(
|
|
326
|
+
json.dumps(
|
|
327
|
+
{"extractor": extractor_file, "pages": pages}, indent=2, ensure_ascii=False
|
|
328
|
+
)
|
|
329
|
+
)
|
|
330
|
+
if broken:
|
|
331
|
+
raise SystemExit(CONTRACT_BROKEN)
|
|
332
|
+
|
|
333
|
+
|
|
334
|
+
@main.command("heal")
|
|
335
|
+
@click.argument("extractor_file")
|
|
336
|
+
@click.argument("sources", nargs=-1, required=True)
|
|
337
|
+
@click.option("-o", "--output", help="Where to write the healed extractor.")
|
|
338
|
+
@click.option(
|
|
339
|
+
"--force",
|
|
340
|
+
is_flag=True,
|
|
341
|
+
help="Write the healed extractor even when healing lost something.",
|
|
342
|
+
)
|
|
343
|
+
@click.option("--stealth", is_flag=True, help="Allow the stealth rung.")
|
|
344
|
+
@click.option("--no-robots", is_flag=True, help="Fetch even where robots.txt says no.")
|
|
345
|
+
def heal_command(
|
|
346
|
+
extractor_file: str,
|
|
347
|
+
sources: tuple[str, ...],
|
|
348
|
+
output: str | None,
|
|
349
|
+
force: bool,
|
|
350
|
+
stealth: bool,
|
|
351
|
+
no_robots: bool,
|
|
352
|
+
) -> None:
|
|
353
|
+
"""Learn pages again and say what moved; write the result only with -o.
|
|
354
|
+
|
|
355
|
+
Exits 3 when a field, a summary answer, a type or the listing was lost for
|
|
356
|
+
good: healing moved what it could, and what it could not needs a person.
|
|
357
|
+
Nothing is written then without --force, so a lossy extractor never
|
|
358
|
+
quietly replaces the one that would have kept failing.
|
|
359
|
+
"""
|
|
360
|
+
extractor = _load_extractor(extractor_file)
|
|
361
|
+
pages = _read_pages(sources, stealth, no_robots)
|
|
362
|
+
try:
|
|
363
|
+
healed, changes = heal_extractor(extractor, pages, names=list(sources))
|
|
364
|
+
except NothingToLearn as nothing:
|
|
365
|
+
click.echo(f"The pages hold nothing to heal from: {nothing}.", err=True)
|
|
366
|
+
raise SystemExit(CONTRACT_BROKEN) from nothing
|
|
367
|
+
for change in changes:
|
|
368
|
+
if change.kind == "kept":
|
|
369
|
+
continue
|
|
370
|
+
if change.before and change.after:
|
|
371
|
+
click.echo(f"{change.kind}: {change.before} -> {change.after}", err=True)
|
|
372
|
+
else:
|
|
373
|
+
click.echo(f"{change.kind}: {change.before or change.after}", err=True)
|
|
374
|
+
lost = any(c.kind in LOSSES for c in changes)
|
|
375
|
+
if output and (force or not lost):
|
|
376
|
+
_write(output, healed.to_json())
|
|
377
|
+
click.echo(f"Wrote {output}.", err=True)
|
|
378
|
+
elif output:
|
|
379
|
+
click.echo(
|
|
380
|
+
f"Did not write {output}: healing lost data. Pass --force to write it.",
|
|
381
|
+
err=True,
|
|
382
|
+
)
|
|
383
|
+
click.echo(
|
|
384
|
+
json.dumps(
|
|
385
|
+
{"changes": [asdict(c) for c in changes]}, indent=2, ensure_ascii=False
|
|
386
|
+
)
|
|
387
|
+
)
|
|
388
|
+
if lost:
|
|
389
|
+
raise SystemExit(CONTRACT_BROKEN)
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""Readers for the structured data a page already declares."""
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
"""Read Dublin Core meta tags.
|
|
2
|
+
|
|
3
|
+
Older than schema.org, and still the only structured data on many library,
|
|
4
|
+
university, repository and government pages: flat ``<meta name="DC.title">``
|
|
5
|
+
tags, sometimes with a ``scheme`` attribute naming the value's encoding rather
|
|
6
|
+
than adding a field. ``DC.`` (the fifteen original elements) and ``DCTERMS.``
|
|
7
|
+
(the refined set) turn up on the same page and answer the same questions, so
|
|
8
|
+
both are read into one flat mapping.
|
|
9
|
+
|
|
10
|
+
The whole name is matched case-insensitively and the key lowercased: the 1997
|
|
11
|
+
examples wrote ``DC.Title``, the HTML that copied them ``dc.title``, and
|
|
12
|
+
first-wins only means something across one spelling.
|
|
13
|
+
|
|
14
|
+
Dublin Core describes the document, so it is not in
|
|
15
|
+
``sluicer.declared.merge.ABOUT_A_THING``.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
from sluicer.document import Document
|
|
21
|
+
|
|
22
|
+
_PREFIXES = ("dc.", "dcterms.")
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def read_dublincore(doc: Document) -> dict[str, str]:
|
|
26
|
+
"""Return the Dublin Core meta tags, prefixes stripped, keys lowercased."""
|
|
27
|
+
found: dict[str, str] = {}
|
|
28
|
+
for meta in doc.tree.xpath("//meta[@name]"):
|
|
29
|
+
name = (meta.get("name") or "").strip().lower()
|
|
30
|
+
content = (meta.get("content") or "").strip()
|
|
31
|
+
if not content:
|
|
32
|
+
continue
|
|
33
|
+
for prefix in _PREFIXES:
|
|
34
|
+
if name.startswith(prefix):
|
|
35
|
+
found.setdefault(name[len(prefix) :], content)
|
|
36
|
+
break
|
|
37
|
+
return found
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
"""Read the metadata names HTML defines for itself.
|
|
2
|
+
|
|
3
|
+
The oldest declaration on the web and still the most common: pages with no
|
|
4
|
+
JSON-LD, microdata or OpenGraph very often carry ``<meta name="description">``
|
|
5
|
+
and ``<meta name="author">``. Measured on 2026-09-22 across the 359 WCXB pages
|
|
6
|
+
Sluicer targets, ``description`` is on 87% of them and ``author`` on 29%.
|
|
7
|
+
|
|
8
|
+
It is a closed list: ``author``, ``description``, ``keywords``, ``generator``,
|
|
9
|
+
``application-name`` and ``theme-color``. Every ``name=`` attribute would
|
|
10
|
+
report a page's ``csrf-token`` and build id as statements about it; browser
|
|
11
|
+
directives (``robots``, ``viewport``) are instructions to a client, not
|
|
12
|
+
statements about the page's subject.
|
|
13
|
+
|
|
14
|
+
These values arrive as ``source="html"``: no vocabulary, no schema, no type.
|
|
15
|
+
Extruct files ``<meta name="description">`` under Dublin Core, which never
|
|
16
|
+
claimed it. The reader runs last, so any real vocabulary wins and a bare tag
|
|
17
|
+
only fills a gap. It describes the document, so it is not in
|
|
18
|
+
``sluicer.declared.merge.ABOUT_A_THING``.
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
from __future__ import annotations
|
|
22
|
+
|
|
23
|
+
from sluicer.declared.meta import read_named_meta
|
|
24
|
+
from sluicer.document import Document
|
|
25
|
+
|
|
26
|
+
# The metadata names the HTML standard defines, and nothing else. Written as
|
|
27
|
+
# a frozenset because it is a membership test and a closed list, and the
|
|
28
|
+
# closedness is the design: every name added here is a claim that the name
|
|
29
|
+
# says something about the page's subject rather than about its plumbing.
|
|
30
|
+
NAMES = frozenset(
|
|
31
|
+
{
|
|
32
|
+
"author",
|
|
33
|
+
"description",
|
|
34
|
+
"keywords",
|
|
35
|
+
"generator",
|
|
36
|
+
"application-name",
|
|
37
|
+
"theme-color",
|
|
38
|
+
}
|
|
39
|
+
)
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def read_htmlmeta(doc: Document) -> dict[str, str]:
|
|
43
|
+
"""Return the standard ``<meta name=...>`` metadata, keys lowercased."""
|
|
44
|
+
return read_named_meta(doc, NAMES)
|