webpage-parser 0.2.0__tar.gz → 0.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (34) hide show
  1. webpage_parser-0.3.0/PKG-INFO +193 -0
  2. webpage_parser-0.3.0/README_PYPI.md +183 -0
  3. webpage_parser-0.3.0/__init__.py +11 -0
  4. webpage_parser-0.3.0/extract.py +3206 -0
  5. {webpage_parser-0.2.0 → webpage_parser-0.3.0}/pyproject.toml +1 -1
  6. webpage_parser-0.3.0/requirements.txt +3 -0
  7. webpage_parser-0.3.0/tests/test_extract.py +108 -0
  8. webpage_parser-0.3.0/webpage_parser.egg-info/PKG-INFO +193 -0
  9. webpage_parser-0.3.0/webpage_parser.egg-info/SOURCES.txt +15 -0
  10. webpage_parser-0.3.0/webpage_parser.egg-info/requires.txt +3 -0
  11. webpage_parser-0.2.0/PKG-INFO +0 -98
  12. webpage_parser-0.2.0/README_PYPI.md +0 -88
  13. webpage_parser-0.2.0/__init__.py +0 -11
  14. webpage_parser-0.2.0/author.py +0 -503
  15. webpage_parser-0.2.0/content.py +0 -915
  16. webpage_parser-0.2.0/extract.py +0 -42
  17. webpage_parser-0.2.0/htmlprep.py +0 -767
  18. webpage_parser-0.2.0/pagegate.py +0 -177
  19. webpage_parser-0.2.0/pipeline.py +0 -325
  20. webpage_parser-0.2.0/pubtime.py +0 -562
  21. webpage_parser-0.2.0/requirements.txt +0 -3
  22. webpage_parser-0.2.0/siteconfig.py +0 -110
  23. webpage_parser-0.2.0/sitespecific.py +0 -170
  24. webpage_parser-0.2.0/timetext.py +0 -365
  25. webpage_parser-0.2.0/title.py +0 -222
  26. webpage_parser-0.2.0/util.py +0 -250
  27. webpage_parser-0.2.0/webpage_parser.egg-info/PKG-INFO +0 -98
  28. webpage_parser-0.2.0/webpage_parser.egg-info/SOURCES.txt +0 -36
  29. webpage_parser-0.2.0/webpage_parser.egg-info/requires.txt +0 -3
  30. {webpage_parser-0.2.0 → webpage_parser-0.3.0}/MANIFEST.in +0 -0
  31. {webpage_parser-0.2.0 → webpage_parser-0.3.0}/setup.cfg +0 -0
  32. {webpage_parser-0.2.0 → webpage_parser-0.3.0}/webpage_parser.egg-info/dependency_links.txt +0 -0
  33. {webpage_parser-0.2.0 → webpage_parser-0.3.0}/webpage_parser.egg-info/entry_points.txt +0 -0
  34. {webpage_parser-0.2.0 → webpage_parser-0.3.0}/webpage_parser.egg-info/top_level.txt +0 -0
@@ -0,0 +1,193 @@
1
+ Metadata-Version: 2.4
2
+ Name: webpage-parser
3
+ Version: 0.3.0
4
+ Summary: Extract structured fields (title, body, publication time, authors, language and more) from raw HTML pages
5
+ Requires-Python: >=3.10
6
+ Description-Content-Type: text/markdown
7
+ Requires-Dist: beautifulsoup4==4.15.0
8
+ Requires-Dist: lxml==6.1.2
9
+ Requires-Dist: tldextract==5.3.2
10
+
11
+ # webpage-parser
12
+
13
+ Extract structured fields from a raw HTML page. Given one crawled web record,
14
+ `process_row` returns a fixed set of **35 fields** covering document identity,
15
+ URLs, the main article (title, body, headings, description, authors),
16
+ timestamps, language and content type.
17
+
18
+ ## Install
19
+
20
+ ```bash
21
+ pip install webpage-parser
22
+ ```
23
+
24
+ Python 3.10+ is required. (The dependencies themselves only need 3.9 at most;
25
+ the floor is aligned with the cp311 runtime used to build the MaxCompute UDF.)
26
+
27
+ ## Python API
28
+
29
+ `process_row` is the only entry point: hand it one dict, get back a dict with
30
+ exactly 35 keys.
31
+
32
+ ```python
33
+ import json
34
+
35
+ from webpage_parser import empty_result, process_row
36
+
37
+ HTML = """<!DOCTYPE html>
38
+ <html lang="en">
39
+ <head>
40
+ <meta charset="utf-8">
41
+ <title>Quantum computing reaches 100 qubits - Example Times</title>
42
+ <meta name="description" content="Researchers entangled 100 superconducting qubits.">
43
+ <meta property="og:site_name" content="Example Times">
44
+ <meta name="author" content="Jane Doe">
45
+ <meta property="article:published_time" content="2026-09-05T10:30:00+00:00">
46
+ <link rel="canonical" href="https://tech.example.com/news/quantum-2026">
47
+ </head>
48
+ <body>
49
+ <article>
50
+ <h1>Quantum computing reaches 100 qubits</h1>
51
+ <p>The team said a superconducting chip held a stable entangled state across
52
+ 100 qubits, tripling coherence time compared with the previous generation.</p>
53
+ <h2>How it works</h2>
54
+ <p>Tunable couplers suppress crosstalk, and dynamical decoupling extends the
55
+ coherence time.</p>
56
+ </article>
57
+ </body>
58
+ </html>
59
+ """
60
+
61
+ # 1) Minimal call: html is the only key that really matters
62
+ result = process_row({"html": HTML})
63
+ print(result["title"], result["language"], len(result))
64
+ # Quantum computing reaches 100 qubits en 35
65
+
66
+ # 2) Full call: every key the parser looks at
67
+ row = {
68
+ "raw_id": "doc-0001",
69
+ "raw_id_version": "v1",
70
+ "url": "https://tech.example.com/news/quantum?utm_source=x",
71
+ "durl": None,
72
+ "redirect_url": None,
73
+ "purl": "https://tech.example.com/",
74
+ "domain": "example.com",
75
+ "host": "tech.example.com",
76
+ "html": HTML,
77
+ "headers": '{"Content-Type": "text/html; charset=utf-8"}',
78
+ "crawled_time": 1789000000,
79
+ "first_found_time": 1788000000,
80
+ "last_found_time": 1789000000,
81
+ "anchor": "quantum milestone",
82
+ "in_domain_links": 12,
83
+ "out_domain_links": 3,
84
+ }
85
+ out = process_row(row)
86
+
87
+ # 3) Batch: one JSON object per line in, one 35-field object per line out
88
+ with open("parsed.jsonl", "w") as f:
89
+ for one in [row]:
90
+ f.write(json.dumps(process_row(one), ensure_ascii=False) + "\n")
91
+
92
+ # 4) Fallback: the empty result carries the very same key set
93
+ blank = empty_result(row)
94
+ assert set(blank) == set(out)
95
+ ```
96
+
97
+ What `out` actually holds:
98
+
99
+ ```text
100
+ document_id = 'doc-0001'
101
+ document_version = 'v1'
102
+ url = 'https://tech.example.com/news/quantum?utm_source=x'
103
+ canonical_url = 'https://tech.example.com/news/quantum-2026'
104
+ domain = 'tech.example.com'
105
+ registrable_domain = 'example.com'
106
+ site_name = 'Example Times'
107
+ title = 'Quantum computing reaches 100 qubits'
108
+ headings = ['How it works']
109
+ description = 'Researchers entangled 100 superconducting qubits.'
110
+ authors = ['Jane Doe']
111
+ published_at = '2026-09-05T10:30:00+00:00'
112
+ published_at_source = 'meta.article_published_time'
113
+ published_at_confidence = 0.95
114
+ language = 'en'
115
+ language_confidence = 0.7
116
+ content_type = 'html'
117
+ mime_type = 'text/html'
118
+ crawled_at = '2026-09-10T00:26:40Z'
119
+ first_seen_at = '2026-08-29T10:40:00Z'
120
+ last_seen_at = '2026-09-10T00:26:40Z'
121
+ anchor_text = 'quantum milestone'
122
+ in_domain_links = 12
123
+ out_domain_links = 3
124
+ ```
125
+
126
+ Four things that tend to surprise people:
127
+
128
+ - `title` drops the trailing site suffix (`- Example Times`); the site name is
129
+ reported separately as `site_name`.
130
+ - The `h1` was taken as the title, so it is not repeated in `headings`.
131
+ - `published_at` keeps the page's own UTC offset. The page declares
132
+ `10:30:00+00:00`, so the output stays `2026-09-05T10:30:00+00:00`; see the
133
+ Time zone section below.
134
+ - Every call returns all 35 keys, parsed or not.
135
+
136
+ `process_row(row)` reads the following keys from the input dict. Only `html`
137
+ really matters; without usable HTML most output fields stay empty.
138
+
139
+ | key | meaning |
140
+ | --- | --- |
141
+ | `html` | raw HTML, the main input |
142
+ | `url`, `durl`, `redirect_url` | candidate page URLs; first non-empty of `redirect_url` > `durl` > `url` wins |
143
+ | `purl` | passed through unchanged |
144
+ | `domain` | registrable-domain hint |
145
+ | `host` | full host-name hint |
146
+ | `headers` | HTTP response headers, a dict or a JSON string |
147
+ | `crawled_time`, `first_found_time`, `last_found_time` | epoch seconds (or milliseconds) |
148
+ | `anchor` | inbound anchor text |
149
+ | `in_domain_links`, `out_domain_links` | link counts; `(0, 0)` is treated as unknown |
150
+ | `raw_id`, `raw_id_version` | opaque document id / version |
151
+
152
+ ## Output
153
+
154
+ `process_row` returns a dict with exactly these 35 keys:
155
+
156
+ `document_id`, `document_version`, `url`, `url_hash`, `canonical_url`,
157
+ `canonical_url_hash`, `domain`, `registrable_domain`, `purl`, `site_name`,
158
+ `favicon_url`, `host_logo_url`, `image_url`, `title`, `headings`, `body`,
159
+ `description`, `authors`, `published_at`, `published_at_confidence`,
160
+ `published_at_source`, `updated_at`, `language`, `language_confidence`,
161
+ `country_codes`, `content_type`, `mime_type`, `content_return_policy`,
162
+ `parser_version`, `crawled_at`, `first_seen_at`, `last_seen_at`,
163
+ `anchor_text`, `in_domain_links`, `out_domain_links`.
164
+
165
+ Notes on selected fields:
166
+
167
+ - `title`, `body`, `description` are strings; `headings` and `authors` are lists.
168
+ - `domain` is the full host name; `registrable_domain` is the eTLD+1.
169
+ - `favicon_url`, `host_logo_url`, `image_url`, `country_codes` and
170
+ `content_return_policy` are placeholders that are currently always `None`.
171
+ - `parser_version` is a constant string (currently `'1.1.0'`).
172
+
173
+ `empty_result(row)` is also exported and returns the same 35-key dict with the
174
+ HTML-derived fields left empty. `process_row` degrades to it whenever a page
175
+ cannot be parsed, so callers always receive the complete key set.
176
+
177
+ ## Time zone
178
+
179
+ - `published_at` and `updated_at` are ISO 8601 strings that **preserve the
180
+ page's own UTC offset** — `Z` stays `Z`, `+00:00` stays `+00:00`, `+08:00`
181
+ stays `+08:00`. When the page carries no explicit offset, **UTC+8** is
182
+ assumed (a date-only value becomes `...T00:00:00+08:00`).
183
+ - `crawled_at`, `first_seen_at` and `last_seen_at` are RFC 3339 timestamps in
184
+ **UTC** with a trailing `Z`, derived from the epoch inputs.
185
+
186
+ ## Command line
187
+
188
+ ```bash
189
+ webpage-parser --input rows.jsonl --output parsed.jsonl
190
+ ```
191
+
192
+ Each input line is a JSON object shaped like the `process_row` argument; each
193
+ output line is the 35-field JSON result.
@@ -0,0 +1,183 @@
1
+ # webpage-parser
2
+
3
+ Extract structured fields from a raw HTML page. Given one crawled web record,
4
+ `process_row` returns a fixed set of **35 fields** covering document identity,
5
+ URLs, the main article (title, body, headings, description, authors),
6
+ timestamps, language and content type.
7
+
8
+ ## Install
9
+
10
+ ```bash
11
+ pip install webpage-parser
12
+ ```
13
+
14
+ Python 3.10+ is required. (The dependencies themselves only need 3.9 at most;
15
+ the floor is aligned with the cp311 runtime used to build the MaxCompute UDF.)
16
+
17
+ ## Python API
18
+
19
+ `process_row` is the only entry point: hand it one dict, get back a dict with
20
+ exactly 35 keys.
21
+
22
+ ```python
23
+ import json
24
+
25
+ from webpage_parser import empty_result, process_row
26
+
27
+ HTML = """<!DOCTYPE html>
28
+ <html lang="en">
29
+ <head>
30
+ <meta charset="utf-8">
31
+ <title>Quantum computing reaches 100 qubits - Example Times</title>
32
+ <meta name="description" content="Researchers entangled 100 superconducting qubits.">
33
+ <meta property="og:site_name" content="Example Times">
34
+ <meta name="author" content="Jane Doe">
35
+ <meta property="article:published_time" content="2026-09-05T10:30:00+00:00">
36
+ <link rel="canonical" href="https://tech.example.com/news/quantum-2026">
37
+ </head>
38
+ <body>
39
+ <article>
40
+ <h1>Quantum computing reaches 100 qubits</h1>
41
+ <p>The team said a superconducting chip held a stable entangled state across
42
+ 100 qubits, tripling coherence time compared with the previous generation.</p>
43
+ <h2>How it works</h2>
44
+ <p>Tunable couplers suppress crosstalk, and dynamical decoupling extends the
45
+ coherence time.</p>
46
+ </article>
47
+ </body>
48
+ </html>
49
+ """
50
+
51
+ # 1) Minimal call: html is the only key that really matters
52
+ result = process_row({"html": HTML})
53
+ print(result["title"], result["language"], len(result))
54
+ # Quantum computing reaches 100 qubits en 35
55
+
56
+ # 2) Full call: every key the parser looks at
57
+ row = {
58
+ "raw_id": "doc-0001",
59
+ "raw_id_version": "v1",
60
+ "url": "https://tech.example.com/news/quantum?utm_source=x",
61
+ "durl": None,
62
+ "redirect_url": None,
63
+ "purl": "https://tech.example.com/",
64
+ "domain": "example.com",
65
+ "host": "tech.example.com",
66
+ "html": HTML,
67
+ "headers": '{"Content-Type": "text/html; charset=utf-8"}',
68
+ "crawled_time": 1789000000,
69
+ "first_found_time": 1788000000,
70
+ "last_found_time": 1789000000,
71
+ "anchor": "quantum milestone",
72
+ "in_domain_links": 12,
73
+ "out_domain_links": 3,
74
+ }
75
+ out = process_row(row)
76
+
77
+ # 3) Batch: one JSON object per line in, one 35-field object per line out
78
+ with open("parsed.jsonl", "w") as f:
79
+ for one in [row]:
80
+ f.write(json.dumps(process_row(one), ensure_ascii=False) + "\n")
81
+
82
+ # 4) Fallback: the empty result carries the very same key set
83
+ blank = empty_result(row)
84
+ assert set(blank) == set(out)
85
+ ```
86
+
87
+ What `out` actually holds:
88
+
89
+ ```text
90
+ document_id = 'doc-0001'
91
+ document_version = 'v1'
92
+ url = 'https://tech.example.com/news/quantum?utm_source=x'
93
+ canonical_url = 'https://tech.example.com/news/quantum-2026'
94
+ domain = 'tech.example.com'
95
+ registrable_domain = 'example.com'
96
+ site_name = 'Example Times'
97
+ title = 'Quantum computing reaches 100 qubits'
98
+ headings = ['How it works']
99
+ description = 'Researchers entangled 100 superconducting qubits.'
100
+ authors = ['Jane Doe']
101
+ published_at = '2026-09-05T10:30:00+00:00'
102
+ published_at_source = 'meta.article_published_time'
103
+ published_at_confidence = 0.95
104
+ language = 'en'
105
+ language_confidence = 0.7
106
+ content_type = 'html'
107
+ mime_type = 'text/html'
108
+ crawled_at = '2026-09-10T00:26:40Z'
109
+ first_seen_at = '2026-08-29T10:40:00Z'
110
+ last_seen_at = '2026-09-10T00:26:40Z'
111
+ anchor_text = 'quantum milestone'
112
+ in_domain_links = 12
113
+ out_domain_links = 3
114
+ ```
115
+
116
+ Four things that tend to surprise people:
117
+
118
+ - `title` drops the trailing site suffix (`- Example Times`); the site name is
119
+ reported separately as `site_name`.
120
+ - The `h1` was taken as the title, so it is not repeated in `headings`.
121
+ - `published_at` keeps the page's own UTC offset. The page declares
122
+ `10:30:00+00:00`, so the output stays `2026-09-05T10:30:00+00:00`; see the
123
+ Time zone section below.
124
+ - Every call returns all 35 keys, parsed or not.
125
+
126
+ `process_row(row)` reads the following keys from the input dict. Only `html`
127
+ really matters; without usable HTML most output fields stay empty.
128
+
129
+ | key | meaning |
130
+ | --- | --- |
131
+ | `html` | raw HTML, the main input |
132
+ | `url`, `durl`, `redirect_url` | candidate page URLs; first non-empty of `redirect_url` > `durl` > `url` wins |
133
+ | `purl` | passed through unchanged |
134
+ | `domain` | registrable-domain hint |
135
+ | `host` | full host-name hint |
136
+ | `headers` | HTTP response headers, a dict or a JSON string |
137
+ | `crawled_time`, `first_found_time`, `last_found_time` | epoch seconds (or milliseconds) |
138
+ | `anchor` | inbound anchor text |
139
+ | `in_domain_links`, `out_domain_links` | link counts; `(0, 0)` is treated as unknown |
140
+ | `raw_id`, `raw_id_version` | opaque document id / version |
141
+
142
+ ## Output
143
+
144
+ `process_row` returns a dict with exactly these 35 keys:
145
+
146
+ `document_id`, `document_version`, `url`, `url_hash`, `canonical_url`,
147
+ `canonical_url_hash`, `domain`, `registrable_domain`, `purl`, `site_name`,
148
+ `favicon_url`, `host_logo_url`, `image_url`, `title`, `headings`, `body`,
149
+ `description`, `authors`, `published_at`, `published_at_confidence`,
150
+ `published_at_source`, `updated_at`, `language`, `language_confidence`,
151
+ `country_codes`, `content_type`, `mime_type`, `content_return_policy`,
152
+ `parser_version`, `crawled_at`, `first_seen_at`, `last_seen_at`,
153
+ `anchor_text`, `in_domain_links`, `out_domain_links`.
154
+
155
+ Notes on selected fields:
156
+
157
+ - `title`, `body`, `description` are strings; `headings` and `authors` are lists.
158
+ - `domain` is the full host name; `registrable_domain` is the eTLD+1.
159
+ - `favicon_url`, `host_logo_url`, `image_url`, `country_codes` and
160
+ `content_return_policy` are placeholders that are currently always `None`.
161
+ - `parser_version` is a constant string (currently `'1.1.0'`).
162
+
163
+ `empty_result(row)` is also exported and returns the same 35-key dict with the
164
+ HTML-derived fields left empty. `process_row` degrades to it whenever a page
165
+ cannot be parsed, so callers always receive the complete key set.
166
+
167
+ ## Time zone
168
+
169
+ - `published_at` and `updated_at` are ISO 8601 strings that **preserve the
170
+ page's own UTC offset** — `Z` stays `Z`, `+00:00` stays `+00:00`, `+08:00`
171
+ stays `+08:00`. When the page carries no explicit offset, **UTC+8** is
172
+ assumed (a date-only value becomes `...T00:00:00+08:00`).
173
+ - `crawled_at`, `first_seen_at` and `last_seen_at` are RFC 3339 timestamps in
174
+ **UTC** with a trailing `Z`, derived from the epoch inputs.
175
+
176
+ ## Command line
177
+
178
+ ```bash
179
+ webpage-parser --input rows.jsonl --output parsed.jsonl
180
+ ```
181
+
182
+ Each input line is a JSON object shaped like the `process_row` argument; each
183
+ output line is the 35-field JSON result.
@@ -0,0 +1,11 @@
1
+ """网页解析算子:从爬取到的 HTML 原始行抽取结构化字段。
2
+
3
+ 解析逻辑已全部合并进 extract.py,既可以按包导入(`from webpage_parser import
4
+ process_row`),也可以把 extract.py 当脚本跑(`python3 extract.py`)。推荐按包导入。
5
+ """
6
+ __version__ = "0.3.0"
7
+
8
+ # 历史 empty_result(row) 现由 extract.build_base 提供,保留别名以稳定对外 API
9
+ from .extract import build_base as empty_result, process_row
10
+
11
+ __all__ = ["empty_result", "process_row", "__version__"]