webpage-parser 0.2.0__tar.gz → 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- webpage_parser-0.3.0/PKG-INFO +193 -0
- webpage_parser-0.3.0/README_PYPI.md +183 -0
- webpage_parser-0.3.0/__init__.py +11 -0
- webpage_parser-0.3.0/extract.py +3206 -0
- {webpage_parser-0.2.0 → webpage_parser-0.3.0}/pyproject.toml +1 -1
- webpage_parser-0.3.0/requirements.txt +3 -0
- webpage_parser-0.3.0/tests/test_extract.py +108 -0
- webpage_parser-0.3.0/webpage_parser.egg-info/PKG-INFO +193 -0
- webpage_parser-0.3.0/webpage_parser.egg-info/SOURCES.txt +15 -0
- webpage_parser-0.3.0/webpage_parser.egg-info/requires.txt +3 -0
- webpage_parser-0.2.0/PKG-INFO +0 -98
- webpage_parser-0.2.0/README_PYPI.md +0 -88
- webpage_parser-0.2.0/__init__.py +0 -11
- webpage_parser-0.2.0/author.py +0 -503
- webpage_parser-0.2.0/content.py +0 -915
- webpage_parser-0.2.0/extract.py +0 -42
- webpage_parser-0.2.0/htmlprep.py +0 -767
- webpage_parser-0.2.0/pagegate.py +0 -177
- webpage_parser-0.2.0/pipeline.py +0 -325
- webpage_parser-0.2.0/pubtime.py +0 -562
- webpage_parser-0.2.0/requirements.txt +0 -3
- webpage_parser-0.2.0/siteconfig.py +0 -110
- webpage_parser-0.2.0/sitespecific.py +0 -170
- webpage_parser-0.2.0/timetext.py +0 -365
- webpage_parser-0.2.0/title.py +0 -222
- webpage_parser-0.2.0/util.py +0 -250
- webpage_parser-0.2.0/webpage_parser.egg-info/PKG-INFO +0 -98
- webpage_parser-0.2.0/webpage_parser.egg-info/SOURCES.txt +0 -36
- webpage_parser-0.2.0/webpage_parser.egg-info/requires.txt +0 -3
- {webpage_parser-0.2.0 → webpage_parser-0.3.0}/MANIFEST.in +0 -0
- {webpage_parser-0.2.0 → webpage_parser-0.3.0}/setup.cfg +0 -0
- {webpage_parser-0.2.0 → webpage_parser-0.3.0}/webpage_parser.egg-info/dependency_links.txt +0 -0
- {webpage_parser-0.2.0 → webpage_parser-0.3.0}/webpage_parser.egg-info/entry_points.txt +0 -0
- {webpage_parser-0.2.0 → webpage_parser-0.3.0}/webpage_parser.egg-info/top_level.txt +0 -0
|
@@ -0,0 +1,193 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: webpage-parser
|
|
3
|
+
Version: 0.3.0
|
|
4
|
+
Summary: Extract structured fields (title, body, publication time, authors, language and more) from raw HTML pages
|
|
5
|
+
Requires-Python: >=3.10
|
|
6
|
+
Description-Content-Type: text/markdown
|
|
7
|
+
Requires-Dist: beautifulsoup4==4.15.0
|
|
8
|
+
Requires-Dist: lxml==6.1.2
|
|
9
|
+
Requires-Dist: tldextract==5.3.2
|
|
10
|
+
|
|
11
|
+
# webpage-parser
|
|
12
|
+
|
|
13
|
+
Extract structured fields from a raw HTML page. Given one crawled web record,
|
|
14
|
+
`process_row` returns a fixed set of **35 fields** covering document identity,
|
|
15
|
+
URLs, the main article (title, body, headings, description, authors),
|
|
16
|
+
timestamps, language and content type.
|
|
17
|
+
|
|
18
|
+
## Install
|
|
19
|
+
|
|
20
|
+
```bash
|
|
21
|
+
pip install webpage-parser
|
|
22
|
+
```
|
|
23
|
+
|
|
24
|
+
Python 3.10+ is required. (The dependencies themselves only need 3.9 at most;
|
|
25
|
+
the floor is aligned with the cp311 runtime used to build the MaxCompute UDF.)
|
|
26
|
+
|
|
27
|
+
## Python API
|
|
28
|
+
|
|
29
|
+
`process_row` is the only entry point: hand it one dict, get back a dict with
|
|
30
|
+
exactly 35 keys.
|
|
31
|
+
|
|
32
|
+
```python
|
|
33
|
+
import json
|
|
34
|
+
|
|
35
|
+
from webpage_parser import empty_result, process_row
|
|
36
|
+
|
|
37
|
+
HTML = """<!DOCTYPE html>
|
|
38
|
+
<html lang="en">
|
|
39
|
+
<head>
|
|
40
|
+
<meta charset="utf-8">
|
|
41
|
+
<title>Quantum computing reaches 100 qubits - Example Times</title>
|
|
42
|
+
<meta name="description" content="Researchers entangled 100 superconducting qubits.">
|
|
43
|
+
<meta property="og:site_name" content="Example Times">
|
|
44
|
+
<meta name="author" content="Jane Doe">
|
|
45
|
+
<meta property="article:published_time" content="2026-09-05T10:30:00+00:00">
|
|
46
|
+
<link rel="canonical" href="https://tech.example.com/news/quantum-2026">
|
|
47
|
+
</head>
|
|
48
|
+
<body>
|
|
49
|
+
<article>
|
|
50
|
+
<h1>Quantum computing reaches 100 qubits</h1>
|
|
51
|
+
<p>The team said a superconducting chip held a stable entangled state across
|
|
52
|
+
100 qubits, tripling coherence time compared with the previous generation.</p>
|
|
53
|
+
<h2>How it works</h2>
|
|
54
|
+
<p>Tunable couplers suppress crosstalk, and dynamical decoupling extends the
|
|
55
|
+
coherence time.</p>
|
|
56
|
+
</article>
|
|
57
|
+
</body>
|
|
58
|
+
</html>
|
|
59
|
+
"""
|
|
60
|
+
|
|
61
|
+
# 1) Minimal call: html is the only key that really matters
|
|
62
|
+
result = process_row({"html": HTML})
|
|
63
|
+
print(result["title"], result["language"], len(result))
|
|
64
|
+
# Quantum computing reaches 100 qubits en 35
|
|
65
|
+
|
|
66
|
+
# 2) Full call: every key the parser looks at
|
|
67
|
+
row = {
|
|
68
|
+
"raw_id": "doc-0001",
|
|
69
|
+
"raw_id_version": "v1",
|
|
70
|
+
"url": "https://tech.example.com/news/quantum?utm_source=x",
|
|
71
|
+
"durl": None,
|
|
72
|
+
"redirect_url": None,
|
|
73
|
+
"purl": "https://tech.example.com/",
|
|
74
|
+
"domain": "example.com",
|
|
75
|
+
"host": "tech.example.com",
|
|
76
|
+
"html": HTML,
|
|
77
|
+
"headers": '{"Content-Type": "text/html; charset=utf-8"}',
|
|
78
|
+
"crawled_time": 1789000000,
|
|
79
|
+
"first_found_time": 1788000000,
|
|
80
|
+
"last_found_time": 1789000000,
|
|
81
|
+
"anchor": "quantum milestone",
|
|
82
|
+
"in_domain_links": 12,
|
|
83
|
+
"out_domain_links": 3,
|
|
84
|
+
}
|
|
85
|
+
out = process_row(row)
|
|
86
|
+
|
|
87
|
+
# 3) Batch: one JSON object per line in, one 35-field object per line out
|
|
88
|
+
with open("parsed.jsonl", "w") as f:
|
|
89
|
+
for one in [row]:
|
|
90
|
+
f.write(json.dumps(process_row(one), ensure_ascii=False) + "\n")
|
|
91
|
+
|
|
92
|
+
# 4) Fallback: the empty result carries the very same key set
|
|
93
|
+
blank = empty_result(row)
|
|
94
|
+
assert set(blank) == set(out)
|
|
95
|
+
```
|
|
96
|
+
|
|
97
|
+
What `out` actually holds:
|
|
98
|
+
|
|
99
|
+
```text
|
|
100
|
+
document_id = 'doc-0001'
|
|
101
|
+
document_version = 'v1'
|
|
102
|
+
url = 'https://tech.example.com/news/quantum?utm_source=x'
|
|
103
|
+
canonical_url = 'https://tech.example.com/news/quantum-2026'
|
|
104
|
+
domain = 'tech.example.com'
|
|
105
|
+
registrable_domain = 'example.com'
|
|
106
|
+
site_name = 'Example Times'
|
|
107
|
+
title = 'Quantum computing reaches 100 qubits'
|
|
108
|
+
headings = ['How it works']
|
|
109
|
+
description = 'Researchers entangled 100 superconducting qubits.'
|
|
110
|
+
authors = ['Jane Doe']
|
|
111
|
+
published_at = '2026-09-05T10:30:00+00:00'
|
|
112
|
+
published_at_source = 'meta.article_published_time'
|
|
113
|
+
published_at_confidence = 0.95
|
|
114
|
+
language = 'en'
|
|
115
|
+
language_confidence = 0.7
|
|
116
|
+
content_type = 'html'
|
|
117
|
+
mime_type = 'text/html'
|
|
118
|
+
crawled_at = '2026-09-10T00:26:40Z'
|
|
119
|
+
first_seen_at = '2026-08-29T10:40:00Z'
|
|
120
|
+
last_seen_at = '2026-09-10T00:26:40Z'
|
|
121
|
+
anchor_text = 'quantum milestone'
|
|
122
|
+
in_domain_links = 12
|
|
123
|
+
out_domain_links = 3
|
|
124
|
+
```
|
|
125
|
+
|
|
126
|
+
Four things that tend to surprise people:
|
|
127
|
+
|
|
128
|
+
- `title` drops the trailing site suffix (`- Example Times`); the site name is
|
|
129
|
+
reported separately as `site_name`.
|
|
130
|
+
- The `h1` was taken as the title, so it is not repeated in `headings`.
|
|
131
|
+
- `published_at` keeps the page's own UTC offset. The page declares
|
|
132
|
+
`10:30:00+00:00`, so the output stays `2026-09-05T10:30:00+00:00`; see the
|
|
133
|
+
Time zone section below.
|
|
134
|
+
- Every call returns all 35 keys, parsed or not.
|
|
135
|
+
|
|
136
|
+
`process_row(row)` reads the following keys from the input dict. Only `html`
|
|
137
|
+
really matters; without usable HTML most output fields stay empty.
|
|
138
|
+
|
|
139
|
+
| key | meaning |
|
|
140
|
+
| --- | --- |
|
|
141
|
+
| `html` | raw HTML, the main input |
|
|
142
|
+
| `url`, `durl`, `redirect_url` | candidate page URLs; first non-empty of `redirect_url` > `durl` > `url` wins |
|
|
143
|
+
| `purl` | passed through unchanged |
|
|
144
|
+
| `domain` | registrable-domain hint |
|
|
145
|
+
| `host` | full host-name hint |
|
|
146
|
+
| `headers` | HTTP response headers, a dict or a JSON string |
|
|
147
|
+
| `crawled_time`, `first_found_time`, `last_found_time` | epoch seconds (or milliseconds) |
|
|
148
|
+
| `anchor` | inbound anchor text |
|
|
149
|
+
| `in_domain_links`, `out_domain_links` | link counts; `(0, 0)` is treated as unknown |
|
|
150
|
+
| `raw_id`, `raw_id_version` | opaque document id / version |
|
|
151
|
+
|
|
152
|
+
## Output
|
|
153
|
+
|
|
154
|
+
`process_row` returns a dict with exactly these 35 keys:
|
|
155
|
+
|
|
156
|
+
`document_id`, `document_version`, `url`, `url_hash`, `canonical_url`,
|
|
157
|
+
`canonical_url_hash`, `domain`, `registrable_domain`, `purl`, `site_name`,
|
|
158
|
+
`favicon_url`, `host_logo_url`, `image_url`, `title`, `headings`, `body`,
|
|
159
|
+
`description`, `authors`, `published_at`, `published_at_confidence`,
|
|
160
|
+
`published_at_source`, `updated_at`, `language`, `language_confidence`,
|
|
161
|
+
`country_codes`, `content_type`, `mime_type`, `content_return_policy`,
|
|
162
|
+
`parser_version`, `crawled_at`, `first_seen_at`, `last_seen_at`,
|
|
163
|
+
`anchor_text`, `in_domain_links`, `out_domain_links`.
|
|
164
|
+
|
|
165
|
+
Notes on selected fields:
|
|
166
|
+
|
|
167
|
+
- `title`, `body`, `description` are strings; `headings` and `authors` are lists.
|
|
168
|
+
- `domain` is the full host name; `registrable_domain` is the eTLD+1.
|
|
169
|
+
- `favicon_url`, `host_logo_url`, `image_url`, `country_codes` and
|
|
170
|
+
`content_return_policy` are placeholders that are currently always `None`.
|
|
171
|
+
- `parser_version` is a constant string (currently `'1.1.0'`).
|
|
172
|
+
|
|
173
|
+
`empty_result(row)` is also exported and returns the same 35-key dict with the
|
|
174
|
+
HTML-derived fields left empty. `process_row` degrades to it whenever a page
|
|
175
|
+
cannot be parsed, so callers always receive the complete key set.
|
|
176
|
+
|
|
177
|
+
## Time zone
|
|
178
|
+
|
|
179
|
+
- `published_at` and `updated_at` are ISO 8601 strings that **preserve the
|
|
180
|
+
page's own UTC offset** — `Z` stays `Z`, `+00:00` stays `+00:00`, `+08:00`
|
|
181
|
+
stays `+08:00`. When the page carries no explicit offset, **UTC+8** is
|
|
182
|
+
assumed (a date-only value becomes `...T00:00:00+08:00`).
|
|
183
|
+
- `crawled_at`, `first_seen_at` and `last_seen_at` are RFC 3339 timestamps in
|
|
184
|
+
**UTC** with a trailing `Z`, derived from the epoch inputs.
|
|
185
|
+
|
|
186
|
+
## Command line
|
|
187
|
+
|
|
188
|
+
```bash
|
|
189
|
+
webpage-parser --input rows.jsonl --output parsed.jsonl
|
|
190
|
+
```
|
|
191
|
+
|
|
192
|
+
Each input line is a JSON object shaped like the `process_row` argument; each
|
|
193
|
+
output line is the 35-field JSON result.
|
|
@@ -0,0 +1,183 @@
|
|
|
1
|
+
# webpage-parser
|
|
2
|
+
|
|
3
|
+
Extract structured fields from a raw HTML page. Given one crawled web record,
|
|
4
|
+
`process_row` returns a fixed set of **35 fields** covering document identity,
|
|
5
|
+
URLs, the main article (title, body, headings, description, authors),
|
|
6
|
+
timestamps, language and content type.
|
|
7
|
+
|
|
8
|
+
## Install
|
|
9
|
+
|
|
10
|
+
```bash
|
|
11
|
+
pip install webpage-parser
|
|
12
|
+
```
|
|
13
|
+
|
|
14
|
+
Python 3.10+ is required. (The dependencies themselves only need 3.9 at most;
|
|
15
|
+
the floor is aligned with the cp311 runtime used to build the MaxCompute UDF.)
|
|
16
|
+
|
|
17
|
+
## Python API
|
|
18
|
+
|
|
19
|
+
`process_row` is the only entry point: hand it one dict, get back a dict with
|
|
20
|
+
exactly 35 keys.
|
|
21
|
+
|
|
22
|
+
```python
|
|
23
|
+
import json
|
|
24
|
+
|
|
25
|
+
from webpage_parser import empty_result, process_row
|
|
26
|
+
|
|
27
|
+
HTML = """<!DOCTYPE html>
|
|
28
|
+
<html lang="en">
|
|
29
|
+
<head>
|
|
30
|
+
<meta charset="utf-8">
|
|
31
|
+
<title>Quantum computing reaches 100 qubits - Example Times</title>
|
|
32
|
+
<meta name="description" content="Researchers entangled 100 superconducting qubits.">
|
|
33
|
+
<meta property="og:site_name" content="Example Times">
|
|
34
|
+
<meta name="author" content="Jane Doe">
|
|
35
|
+
<meta property="article:published_time" content="2026-09-05T10:30:00+00:00">
|
|
36
|
+
<link rel="canonical" href="https://tech.example.com/news/quantum-2026">
|
|
37
|
+
</head>
|
|
38
|
+
<body>
|
|
39
|
+
<article>
|
|
40
|
+
<h1>Quantum computing reaches 100 qubits</h1>
|
|
41
|
+
<p>The team said a superconducting chip held a stable entangled state across
|
|
42
|
+
100 qubits, tripling coherence time compared with the previous generation.</p>
|
|
43
|
+
<h2>How it works</h2>
|
|
44
|
+
<p>Tunable couplers suppress crosstalk, and dynamical decoupling extends the
|
|
45
|
+
coherence time.</p>
|
|
46
|
+
</article>
|
|
47
|
+
</body>
|
|
48
|
+
</html>
|
|
49
|
+
"""
|
|
50
|
+
|
|
51
|
+
# 1) Minimal call: html is the only key that really matters
|
|
52
|
+
result = process_row({"html": HTML})
|
|
53
|
+
print(result["title"], result["language"], len(result))
|
|
54
|
+
# Quantum computing reaches 100 qubits en 35
|
|
55
|
+
|
|
56
|
+
# 2) Full call: every key the parser looks at
|
|
57
|
+
row = {
|
|
58
|
+
"raw_id": "doc-0001",
|
|
59
|
+
"raw_id_version": "v1",
|
|
60
|
+
"url": "https://tech.example.com/news/quantum?utm_source=x",
|
|
61
|
+
"durl": None,
|
|
62
|
+
"redirect_url": None,
|
|
63
|
+
"purl": "https://tech.example.com/",
|
|
64
|
+
"domain": "example.com",
|
|
65
|
+
"host": "tech.example.com",
|
|
66
|
+
"html": HTML,
|
|
67
|
+
"headers": '{"Content-Type": "text/html; charset=utf-8"}',
|
|
68
|
+
"crawled_time": 1789000000,
|
|
69
|
+
"first_found_time": 1788000000,
|
|
70
|
+
"last_found_time": 1789000000,
|
|
71
|
+
"anchor": "quantum milestone",
|
|
72
|
+
"in_domain_links": 12,
|
|
73
|
+
"out_domain_links": 3,
|
|
74
|
+
}
|
|
75
|
+
out = process_row(row)
|
|
76
|
+
|
|
77
|
+
# 3) Batch: one JSON object per line in, one 35-field object per line out
|
|
78
|
+
with open("parsed.jsonl", "w") as f:
|
|
79
|
+
for one in [row]:
|
|
80
|
+
f.write(json.dumps(process_row(one), ensure_ascii=False) + "\n")
|
|
81
|
+
|
|
82
|
+
# 4) Fallback: the empty result carries the very same key set
|
|
83
|
+
blank = empty_result(row)
|
|
84
|
+
assert set(blank) == set(out)
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
What `out` actually holds:
|
|
88
|
+
|
|
89
|
+
```text
|
|
90
|
+
document_id = 'doc-0001'
|
|
91
|
+
document_version = 'v1'
|
|
92
|
+
url = 'https://tech.example.com/news/quantum?utm_source=x'
|
|
93
|
+
canonical_url = 'https://tech.example.com/news/quantum-2026'
|
|
94
|
+
domain = 'tech.example.com'
|
|
95
|
+
registrable_domain = 'example.com'
|
|
96
|
+
site_name = 'Example Times'
|
|
97
|
+
title = 'Quantum computing reaches 100 qubits'
|
|
98
|
+
headings = ['How it works']
|
|
99
|
+
description = 'Researchers entangled 100 superconducting qubits.'
|
|
100
|
+
authors = ['Jane Doe']
|
|
101
|
+
published_at = '2026-09-05T10:30:00+00:00'
|
|
102
|
+
published_at_source = 'meta.article_published_time'
|
|
103
|
+
published_at_confidence = 0.95
|
|
104
|
+
language = 'en'
|
|
105
|
+
language_confidence = 0.7
|
|
106
|
+
content_type = 'html'
|
|
107
|
+
mime_type = 'text/html'
|
|
108
|
+
crawled_at = '2026-09-10T00:26:40Z'
|
|
109
|
+
first_seen_at = '2026-08-29T10:40:00Z'
|
|
110
|
+
last_seen_at = '2026-09-10T00:26:40Z'
|
|
111
|
+
anchor_text = 'quantum milestone'
|
|
112
|
+
in_domain_links = 12
|
|
113
|
+
out_domain_links = 3
|
|
114
|
+
```
|
|
115
|
+
|
|
116
|
+
Four things that tend to surprise people:
|
|
117
|
+
|
|
118
|
+
- `title` drops the trailing site suffix (`- Example Times`); the site name is
|
|
119
|
+
reported separately as `site_name`.
|
|
120
|
+
- The `h1` was taken as the title, so it is not repeated in `headings`.
|
|
121
|
+
- `published_at` keeps the page's own UTC offset. The page declares
|
|
122
|
+
`10:30:00+00:00`, so the output stays `2026-09-05T10:30:00+00:00`; see the
|
|
123
|
+
Time zone section below.
|
|
124
|
+
- Every call returns all 35 keys, parsed or not.
|
|
125
|
+
|
|
126
|
+
`process_row(row)` reads the following keys from the input dict. Only `html`
|
|
127
|
+
really matters; without usable HTML most output fields stay empty.
|
|
128
|
+
|
|
129
|
+
| key | meaning |
|
|
130
|
+
| --- | --- |
|
|
131
|
+
| `html` | raw HTML, the main input |
|
|
132
|
+
| `url`, `durl`, `redirect_url` | candidate page URLs; first non-empty of `redirect_url` > `durl` > `url` wins |
|
|
133
|
+
| `purl` | passed through unchanged |
|
|
134
|
+
| `domain` | registrable-domain hint |
|
|
135
|
+
| `host` | full host-name hint |
|
|
136
|
+
| `headers` | HTTP response headers, a dict or a JSON string |
|
|
137
|
+
| `crawled_time`, `first_found_time`, `last_found_time` | epoch seconds (or milliseconds) |
|
|
138
|
+
| `anchor` | inbound anchor text |
|
|
139
|
+
| `in_domain_links`, `out_domain_links` | link counts; `(0, 0)` is treated as unknown |
|
|
140
|
+
| `raw_id`, `raw_id_version` | opaque document id / version |
|
|
141
|
+
|
|
142
|
+
## Output
|
|
143
|
+
|
|
144
|
+
`process_row` returns a dict with exactly these 35 keys:
|
|
145
|
+
|
|
146
|
+
`document_id`, `document_version`, `url`, `url_hash`, `canonical_url`,
|
|
147
|
+
`canonical_url_hash`, `domain`, `registrable_domain`, `purl`, `site_name`,
|
|
148
|
+
`favicon_url`, `host_logo_url`, `image_url`, `title`, `headings`, `body`,
|
|
149
|
+
`description`, `authors`, `published_at`, `published_at_confidence`,
|
|
150
|
+
`published_at_source`, `updated_at`, `language`, `language_confidence`,
|
|
151
|
+
`country_codes`, `content_type`, `mime_type`, `content_return_policy`,
|
|
152
|
+
`parser_version`, `crawled_at`, `first_seen_at`, `last_seen_at`,
|
|
153
|
+
`anchor_text`, `in_domain_links`, `out_domain_links`.
|
|
154
|
+
|
|
155
|
+
Notes on selected fields:
|
|
156
|
+
|
|
157
|
+
- `title`, `body`, `description` are strings; `headings` and `authors` are lists.
|
|
158
|
+
- `domain` is the full host name; `registrable_domain` is the eTLD+1.
|
|
159
|
+
- `favicon_url`, `host_logo_url`, `image_url`, `country_codes` and
|
|
160
|
+
`content_return_policy` are placeholders that are currently always `None`.
|
|
161
|
+
- `parser_version` is a constant string (currently `'1.1.0'`).
|
|
162
|
+
|
|
163
|
+
`empty_result(row)` is also exported and returns the same 35-key dict with the
|
|
164
|
+
HTML-derived fields left empty. `process_row` degrades to it whenever a page
|
|
165
|
+
cannot be parsed, so callers always receive the complete key set.
|
|
166
|
+
|
|
167
|
+
## Time zone
|
|
168
|
+
|
|
169
|
+
- `published_at` and `updated_at` are ISO 8601 strings that **preserve the
|
|
170
|
+
page's own UTC offset** — `Z` stays `Z`, `+00:00` stays `+00:00`, `+08:00`
|
|
171
|
+
stays `+08:00`. When the page carries no explicit offset, **UTC+8** is
|
|
172
|
+
assumed (a date-only value becomes `...T00:00:00+08:00`).
|
|
173
|
+
- `crawled_at`, `first_seen_at` and `last_seen_at` are RFC 3339 timestamps in
|
|
174
|
+
**UTC** with a trailing `Z`, derived from the epoch inputs.
|
|
175
|
+
|
|
176
|
+
## Command line
|
|
177
|
+
|
|
178
|
+
```bash
|
|
179
|
+
webpage-parser --input rows.jsonl --output parsed.jsonl
|
|
180
|
+
```
|
|
181
|
+
|
|
182
|
+
Each input line is a JSON object shaped like the `process_row` argument; each
|
|
183
|
+
output line is the 35-field JSON result.
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
"""网页解析算子:从爬取到的 HTML 原始行抽取结构化字段。
|
|
2
|
+
|
|
3
|
+
解析逻辑已全部合并进 extract.py,既可以按包导入(`from webpage_parser import
|
|
4
|
+
process_row`),也可以把 extract.py 当脚本跑(`python3 extract.py`)。推荐按包导入。
|
|
5
|
+
"""
|
|
6
|
+
__version__ = "0.3.0"
|
|
7
|
+
|
|
8
|
+
# 历史 empty_result(row) 现由 extract.build_base 提供,保留别名以稳定对外 API
|
|
9
|
+
from .extract import build_base as empty_result, process_row
|
|
10
|
+
|
|
11
|
+
__all__ = ["empty_result", "process_row", "__version__"]
|