scrapy-ingest 1.0.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- scrapy_ingest-1.0.0/LICENSE +21 -0
- scrapy_ingest-1.0.0/PKG-INFO +163 -0
- scrapy_ingest-1.0.0/README.md +98 -0
- scrapy_ingest-1.0.0/scrapy_ingest/__init__.py +36 -0
- scrapy_ingest-1.0.0/scrapy_ingest/bootstrap/__init__.py +5 -0
- scrapy_ingest-1.0.0/scrapy_ingest/bootstrap/bootstrap.py +85 -0
- scrapy_ingest-1.0.0/scrapy_ingest/collector/__init__.py +5 -0
- scrapy_ingest-1.0.0/scrapy_ingest/collector/collector.py +78 -0
- scrapy_ingest-1.0.0/scrapy_ingest/config/__init__.py +2 -0
- scrapy_ingest-1.0.0/scrapy_ingest/config/settings.py +133 -0
- scrapy_ingest-1.0.0/scrapy_ingest/database/__init__.py +2 -0
- scrapy_ingest-1.0.0/scrapy_ingest/database/connection.py +144 -0
- scrapy_ingest-1.0.0/scrapy_ingest/database/flusher.py +135 -0
- scrapy_ingest-1.0.0/scrapy_ingest/database/schema.py +266 -0
- scrapy_ingest-1.0.0/scrapy_ingest/database/writer.py +232 -0
- scrapy_ingest-1.0.0/scrapy_ingest/extensions/__init__.py +7 -0
- scrapy_ingest-1.0.0/scrapy_ingest/extensions/base.py +81 -0
- scrapy_ingest-1.0.0/scrapy_ingest/extensions/log_handler.py +185 -0
- scrapy_ingest-1.0.0/scrapy_ingest/extensions/logging.py +64 -0
- scrapy_ingest-1.0.0/scrapy_ingest/extensions/request_logger.py +45 -0
- scrapy_ingest-1.0.0/scrapy_ingest/extensions/stats.py +31 -0
- scrapy_ingest-1.0.0/scrapy_ingest/middleware/__init__.py +5 -0
- scrapy_ingest-1.0.0/scrapy_ingest/middleware/middleware.py +184 -0
- scrapy_ingest-1.0.0/scrapy_ingest/pipelines/__init__.py +11 -0
- scrapy_ingest-1.0.0/scrapy_ingest/pipelines/base.py +20 -0
- scrapy_ingest-1.0.0/scrapy_ingest/pipelines/items.py +39 -0
- scrapy_ingest-1.0.0/scrapy_ingest/pipelines/main.py +20 -0
- scrapy_ingest-1.0.0/scrapy_ingest/pipelines/requests.py +36 -0
- scrapy_ingest-1.0.0/scrapy_ingest/utils/__init__.py +2 -0
- scrapy_ingest-1.0.0/scrapy_ingest/utils/fingerprint.py +18 -0
- scrapy_ingest-1.0.0/scrapy_ingest/utils/job_id.py +27 -0
- scrapy_ingest-1.0.0/scrapy_ingest/utils/parent.py +31 -0
- scrapy_ingest-1.0.0/scrapy_ingest/utils/serialization.py +20 -0
- scrapy_ingest-1.0.0/scrapy_ingest/utils/time.py +19 -0
- scrapy_ingest-1.0.0/scrapy_ingest.egg-info/PKG-INFO +163 -0
- scrapy_ingest-1.0.0/scrapy_ingest.egg-info/SOURCES.txt +41 -0
- scrapy_ingest-1.0.0/scrapy_ingest.egg-info/dependency_links.txt +1 -0
- scrapy_ingest-1.0.0/scrapy_ingest.egg-info/entry_points.txt +6 -0
- scrapy_ingest-1.0.0/scrapy_ingest.egg-info/not-zip-safe +1 -0
- scrapy_ingest-1.0.0/scrapy_ingest.egg-info/requires.txt +26 -0
- scrapy_ingest-1.0.0/scrapy_ingest.egg-info/top_level.txt +1 -0
- scrapy_ingest-1.0.0/setup.cfg +4 -0
- scrapy_ingest-1.0.0/setup.py +115 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2025 Fawad Ali
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,163 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: scrapy-ingest
|
|
3
|
+
Version: 1.0.0
|
|
4
|
+
Summary: Scrapy extension for database ingestion with job/spider tracking
|
|
5
|
+
Home-page: https://github.com/fawadss1/scrapy_item_ingest
|
|
6
|
+
Author: Fawad Ali
|
|
7
|
+
Author-email: fawadstar6@gmail.com
|
|
8
|
+
Project-URL: Documentation, https://scrapy-ingest.readthedocs.io/
|
|
9
|
+
Project-URL: Source, https://github.com/fawadss1/scrapy_item_ingest
|
|
10
|
+
Project-URL: Tracker, https://github.com/fawadss1/scrapy_item_ingest/issues
|
|
11
|
+
Keywords: scrapy,database,postgresql,web-scraping,data-pipeline
|
|
12
|
+
Classifier: Development Status :: 4 - Beta
|
|
13
|
+
Classifier: Intended Audience :: Developers
|
|
14
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
15
|
+
Classifier: Operating System :: OS Independent
|
|
16
|
+
Classifier: Programming Language :: Python :: 3
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.7
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.8
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
22
|
+
Classifier: Framework :: Scrapy
|
|
23
|
+
Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
24
|
+
Classifier: Topic :: Internet :: WWW/HTTP
|
|
25
|
+
Classifier: Topic :: Database
|
|
26
|
+
Requires-Python: >=3.7
|
|
27
|
+
Description-Content-Type: text/markdown
|
|
28
|
+
License-File: LICENSE
|
|
29
|
+
Requires-Dist: scrapy>=2.13.3
|
|
30
|
+
Requires-Dist: psycopg2-binary>=2.9.10
|
|
31
|
+
Requires-Dist: itemadapter>=0.11.0
|
|
32
|
+
Requires-Dist: SQLAlchemy>=2.0.41
|
|
33
|
+
Requires-Dist: pytz>=2025.2
|
|
34
|
+
Requires-Dist: w3lib>=1.22.0
|
|
35
|
+
Provides-Extra: docs
|
|
36
|
+
Requires-Dist: sphinx>=5.0.0; extra == "docs"
|
|
37
|
+
Requires-Dist: sphinx_rtd_theme>=1.2.0; extra == "docs"
|
|
38
|
+
Requires-Dist: myst-parser>=0.18.0; extra == "docs"
|
|
39
|
+
Requires-Dist: sphinx-autodoc-typehints>=1.19.0; extra == "docs"
|
|
40
|
+
Requires-Dist: sphinx-copybutton>=0.5.0; extra == "docs"
|
|
41
|
+
Provides-Extra: dev
|
|
42
|
+
Requires-Dist: pytest>=7.0.0; extra == "dev"
|
|
43
|
+
Requires-Dist: pytest-cov>=4.0.0; extra == "dev"
|
|
44
|
+
Requires-Dist: black>=22.0.0; extra == "dev"
|
|
45
|
+
Requires-Dist: flake8>=5.0.0; extra == "dev"
|
|
46
|
+
Requires-Dist: mypy>=0.991; extra == "dev"
|
|
47
|
+
Requires-Dist: pre-commit>=2.20.0; extra == "dev"
|
|
48
|
+
Provides-Extra: test
|
|
49
|
+
Requires-Dist: pytest>=7.0.0; extra == "test"
|
|
50
|
+
Requires-Dist: pytest-cov>=4.0.0; extra == "test"
|
|
51
|
+
Requires-Dist: pytest-mock>=3.8.0; extra == "test"
|
|
52
|
+
Dynamic: author
|
|
53
|
+
Dynamic: author-email
|
|
54
|
+
Dynamic: classifier
|
|
55
|
+
Dynamic: description
|
|
56
|
+
Dynamic: description-content-type
|
|
57
|
+
Dynamic: home-page
|
|
58
|
+
Dynamic: keywords
|
|
59
|
+
Dynamic: license-file
|
|
60
|
+
Dynamic: project-url
|
|
61
|
+
Dynamic: provides-extra
|
|
62
|
+
Dynamic: requires-dist
|
|
63
|
+
Dynamic: requires-python
|
|
64
|
+
Dynamic: summary
|
|
65
|
+
|
|
66
|
+
# Scrapy Ingest
|
|
67
|
+
|
|
68
|
+
A Scrapy addon that saves **items, requests, logs, and stats** to PostgreSQL — with parent_url tracking, failed-request errors, and full job log capture (including `print()`).
|
|
69
|
+
|
|
70
|
+
## Install
|
|
71
|
+
|
|
72
|
+
```bash
|
|
73
|
+
pip install scrapy-ingest
|
|
74
|
+
```
|
|
75
|
+
|
|
76
|
+
## Minimal setup (settings.py)
|
|
77
|
+
|
|
78
|
+
Only the item pipeline is required — requests, logs, stats, parent_url, and error logging are enabled automatically:
|
|
79
|
+
|
|
80
|
+
```python
|
|
81
|
+
ITEM_PIPELINES = {
|
|
82
|
+
"scrapy_ingest.pipelines.DbInsertPipeline": 300,
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
# Pick ONE of the two database config styles:
|
|
86
|
+
DB_URL = "postgresql://user:password@localhost:5432/database"
|
|
87
|
+
# Or use discrete fields (avoids URL encoding):
|
|
88
|
+
# DB_TYPE = "postgres"
|
|
89
|
+
# DB_HOST = "localhost"
|
|
90
|
+
# DB_PORT = 5432
|
|
91
|
+
# DB_USER = "user"
|
|
92
|
+
# DB_PASSWORD = "password"
|
|
93
|
+
# DB_NAME = "database"
|
|
94
|
+
|
|
95
|
+
# Optional
|
|
96
|
+
CREATE_TABLES = True # auto-create tables on first run (default True)
|
|
97
|
+
JOB_ID = 1 # or omit; a unique id is generated per crawl
|
|
98
|
+
INGEST_BATCH_SIZE = 50 # flush when this many rows are buffered
|
|
99
|
+
```
|
|
100
|
+
|
|
101
|
+
Run your spider:
|
|
102
|
+
|
|
103
|
+
```bash
|
|
104
|
+
scrapy crawl your_spider
|
|
105
|
+
```
|
|
106
|
+
|
|
107
|
+
Log level follows Scrapy `LOG_LEVEL`.
|
|
108
|
+
|
|
109
|
+
## What is stored
|
|
110
|
+
|
|
111
|
+
| Table | Contents |
|
|
112
|
+
|----------------|---------------------------------------------------------------------------------------------------------|
|
|
113
|
+
| `jobs` | One row per crawl: `id`, unique `job_id` string, spider, status, start/finish, counts, items/min, stats |
|
|
114
|
+
| `job_items` | JSON items (`crawled_at` added). `job_id` = `jobs.id` (CASCADE) |
|
|
115
|
+
| `job_requests` | url, parent_url, parent_id, status, response_time, error, success. `job_id` = `jobs.id` (CASCADE) |
|
|
116
|
+
| `job_logs` | time, logger, level, message, exception. `job_id` = `jobs.id` (CASCADE) |
|
|
117
|
+
|
|
118
|
+
Request `parent_url` is the page that scheduled the request (e.g. sitemap → product). Start URLs are `null`.
|
|
119
|
+
|
|
120
|
+
Data flushes on batch size, every 10s, and on engine/process stop.
|
|
121
|
+
|
|
122
|
+
## Troubleshooting
|
|
123
|
+
|
|
124
|
+
- Password has special characters like `@` or `$`?
|
|
125
|
+
- In a URL, encode them: `@` -> `%40`, `$` -> `%24`.
|
|
126
|
+
- Example: `postgresql://user:PAK%40swat1%24@localhost:5432/db`
|
|
127
|
+
- Or use the discrete fields (no encoding needed).
|
|
128
|
+
- **Yield items** from callbacks (not only `return` inside a generator).
|
|
129
|
+
|
|
130
|
+
## Useful settings (optional)
|
|
131
|
+
|
|
132
|
+
- `DB_TYPE` (default: `postgres`) — used when building a URL from `DB_HOST` / `DB_*` fields
|
|
133
|
+
- `INGEST_BATCH_SIZE` (default: `50`) — flush when this many items+requests+logs are buffered
|
|
134
|
+
- `INGEST_FLUSH_INTERVAL` (default: `10`) — periodic flush in seconds
|
|
135
|
+
- `CREATE_TABLES` (default: `True`) — create tables on startup
|
|
136
|
+
- `ITEMS_TABLE`, `REQUESTS_TABLE`, `LOGS_TABLE`, `JOBS_TABLE` — override table names
|
|
137
|
+
- `TIMEZONE` (default: `Asia/Karachi`) — timezone for `created_at`
|
|
138
|
+
- `JOB_ID` — omit to auto-generate a unique id (`spider-YYYYMMDDHHMMSS-xxxxxxxx`)
|
|
139
|
+
|
|
140
|
+
## Standalone components
|
|
141
|
+
|
|
142
|
+
If you only want part of the collection:
|
|
143
|
+
|
|
144
|
+
```python
|
|
145
|
+
# Items only
|
|
146
|
+
ITEM_PIPELINES = {"scrapy_ingest.pipelines.ItemsPipeline": 300}
|
|
147
|
+
|
|
148
|
+
# Requests only (parent_url + errors)
|
|
149
|
+
ITEM_PIPELINES = {"scrapy_ingest.pipelines.RequestsPipeline": 300}
|
|
150
|
+
|
|
151
|
+
# Logs only
|
|
152
|
+
EXTENSIONS = {"scrapy_ingest.extensions.LoggingExtension": 500}
|
|
153
|
+
```
|
|
154
|
+
|
|
155
|
+
## Links
|
|
156
|
+
|
|
157
|
+
- Docs: https://scrapy-ingest.readthedocs.io/
|
|
158
|
+
- Changelog: docs/development/changelog.rst
|
|
159
|
+
- Issues: https://github.com/fawadss1/scrapy_item_ingest/issues
|
|
160
|
+
|
|
161
|
+
## License
|
|
162
|
+
|
|
163
|
+
MIT License. See [LICENSE](LICENSE).
|
|
@@ -0,0 +1,98 @@
|
|
|
1
|
+
# Scrapy Ingest
|
|
2
|
+
|
|
3
|
+
A Scrapy addon that saves **items, requests, logs, and stats** to PostgreSQL — with parent_url tracking, failed-request errors, and full job log capture (including `print()`).
|
|
4
|
+
|
|
5
|
+
## Install
|
|
6
|
+
|
|
7
|
+
```bash
|
|
8
|
+
pip install scrapy-ingest
|
|
9
|
+
```
|
|
10
|
+
|
|
11
|
+
## Minimal setup (settings.py)
|
|
12
|
+
|
|
13
|
+
Only the item pipeline is required — requests, logs, stats, parent_url, and error logging are enabled automatically:
|
|
14
|
+
|
|
15
|
+
```python
|
|
16
|
+
ITEM_PIPELINES = {
|
|
17
|
+
"scrapy_ingest.pipelines.DbInsertPipeline": 300,
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
# Pick ONE of the two database config styles:
|
|
21
|
+
DB_URL = "postgresql://user:password@localhost:5432/database"
|
|
22
|
+
# Or use discrete fields (avoids URL encoding):
|
|
23
|
+
# DB_TYPE = "postgres"
|
|
24
|
+
# DB_HOST = "localhost"
|
|
25
|
+
# DB_PORT = 5432
|
|
26
|
+
# DB_USER = "user"
|
|
27
|
+
# DB_PASSWORD = "password"
|
|
28
|
+
# DB_NAME = "database"
|
|
29
|
+
|
|
30
|
+
# Optional
|
|
31
|
+
CREATE_TABLES = True # auto-create tables on first run (default True)
|
|
32
|
+
JOB_ID = 1 # or omit; a unique id is generated per crawl
|
|
33
|
+
INGEST_BATCH_SIZE = 50 # flush when this many rows are buffered
|
|
34
|
+
```
|
|
35
|
+
|
|
36
|
+
Run your spider:
|
|
37
|
+
|
|
38
|
+
```bash
|
|
39
|
+
scrapy crawl your_spider
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
Log level follows Scrapy `LOG_LEVEL`.
|
|
43
|
+
|
|
44
|
+
## What is stored
|
|
45
|
+
|
|
46
|
+
| Table | Contents |
|
|
47
|
+
|----------------|---------------------------------------------------------------------------------------------------------|
|
|
48
|
+
| `jobs` | One row per crawl: `id`, unique `job_id` string, spider, status, start/finish, counts, items/min, stats |
|
|
49
|
+
| `job_items` | JSON items (`crawled_at` added). `job_id` = `jobs.id` (CASCADE) |
|
|
50
|
+
| `job_requests` | url, parent_url, parent_id, status, response_time, error, success. `job_id` = `jobs.id` (CASCADE) |
|
|
51
|
+
| `job_logs` | time, logger, level, message, exception. `job_id` = `jobs.id` (CASCADE) |
|
|
52
|
+
|
|
53
|
+
Request `parent_url` is the page that scheduled the request (e.g. sitemap → product). Start URLs are `null`.
|
|
54
|
+
|
|
55
|
+
Data flushes on batch size, every 10s, and on engine/process stop.
|
|
56
|
+
|
|
57
|
+
## Troubleshooting
|
|
58
|
+
|
|
59
|
+
- Password has special characters like `@` or `$`?
|
|
60
|
+
- In a URL, encode them: `@` -> `%40`, `$` -> `%24`.
|
|
61
|
+
- Example: `postgresql://user:PAK%40swat1%24@localhost:5432/db`
|
|
62
|
+
- Or use the discrete fields (no encoding needed).
|
|
63
|
+
- **Yield items** from callbacks (not only `return` inside a generator).
|
|
64
|
+
|
|
65
|
+
## Useful settings (optional)
|
|
66
|
+
|
|
67
|
+
- `DB_TYPE` (default: `postgres`) — used when building a URL from `DB_HOST` / `DB_*` fields
|
|
68
|
+
- `INGEST_BATCH_SIZE` (default: `50`) — flush when this many items+requests+logs are buffered
|
|
69
|
+
- `INGEST_FLUSH_INTERVAL` (default: `10`) — periodic flush in seconds
|
|
70
|
+
- `CREATE_TABLES` (default: `True`) — create tables on startup
|
|
71
|
+
- `ITEMS_TABLE`, `REQUESTS_TABLE`, `LOGS_TABLE`, `JOBS_TABLE` — override table names
|
|
72
|
+
- `TIMEZONE` (default: `Asia/Karachi`) — timezone for `created_at`
|
|
73
|
+
- `JOB_ID` — omit to auto-generate a unique id (`spider-YYYYMMDDHHMMSS-xxxxxxxx`)
|
|
74
|
+
|
|
75
|
+
## Standalone components
|
|
76
|
+
|
|
77
|
+
If you only want part of the collection:
|
|
78
|
+
|
|
79
|
+
```python
|
|
80
|
+
# Items only
|
|
81
|
+
ITEM_PIPELINES = {"scrapy_ingest.pipelines.ItemsPipeline": 300}
|
|
82
|
+
|
|
83
|
+
# Requests only (parent_url + errors)
|
|
84
|
+
ITEM_PIPELINES = {"scrapy_ingest.pipelines.RequestsPipeline": 300}
|
|
85
|
+
|
|
86
|
+
# Logs only
|
|
87
|
+
EXTENSIONS = {"scrapy_ingest.extensions.LoggingExtension": 500}
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
## Links
|
|
91
|
+
|
|
92
|
+
- Docs: https://scrapy-ingest.readthedocs.io/
|
|
93
|
+
- Changelog: docs/development/changelog.rst
|
|
94
|
+
- Issues: https://github.com/fawadss1/scrapy_item_ingest/issues
|
|
95
|
+
|
|
96
|
+
## License
|
|
97
|
+
|
|
98
|
+
MIT License. See [LICENSE](LICENSE).
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
"""
|
|
2
|
+
scrapy_ingest - A Scrapy extension for ingesting items, requests, logs, and stats into PostgreSQL.
|
|
3
|
+
|
|
4
|
+
Enabling DbInsertPipeline auto-enables request logging (with parent_url),
|
|
5
|
+
error logging, full job logs (including print()), and crawl stats.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from .extensions.log_handler import install_early
|
|
9
|
+
|
|
10
|
+
install_early()
|
|
11
|
+
|
|
12
|
+
__version__ = "1.0.0"
|
|
13
|
+
__author__ = "Fawad Ali"
|
|
14
|
+
__description__ = "Scrapy extension for database ingestion with job/spider tracking"
|
|
15
|
+
|
|
16
|
+
from .pipelines.main import DbInsertPipeline
|
|
17
|
+
from .extensions.logging import LoggingExtension
|
|
18
|
+
from .extensions.stats import StatsExtension
|
|
19
|
+
from .pipelines.items import ItemsPipeline
|
|
20
|
+
from .pipelines.requests import RequestsPipeline
|
|
21
|
+
from .extensions.request_logger import RequestLogger
|
|
22
|
+
from .config.settings import Settings, validate_settings
|
|
23
|
+
|
|
24
|
+
__all__ = [
|
|
25
|
+
"DbInsertPipeline",
|
|
26
|
+
"LoggingExtension",
|
|
27
|
+
"StatsExtension",
|
|
28
|
+
"ItemsPipeline",
|
|
29
|
+
"RequestsPipeline",
|
|
30
|
+
"RequestLogger",
|
|
31
|
+
"Settings",
|
|
32
|
+
"validate_settings",
|
|
33
|
+
"__version__",
|
|
34
|
+
"__author__",
|
|
35
|
+
"__description__",
|
|
36
|
+
]
|
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
"""Enable all ingest components from a single entry point (the item pipeline)."""
|
|
2
|
+
|
|
3
|
+
import inspect
|
|
4
|
+
|
|
5
|
+
from scrapy.core.engine import ExecutionEngine
|
|
6
|
+
from scrapy.core.scraper import Scraper
|
|
7
|
+
|
|
8
|
+
from ..collector import ensure_collector
|
|
9
|
+
from ..extensions.logging import LoggingExtension
|
|
10
|
+
from ..extensions.stats import StatsExtension
|
|
11
|
+
from ..extensions.request_logger import RequestLogger
|
|
12
|
+
from ..middleware import ErrorMiddleware, _inject_into_scraper
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def _inject_error_middleware(crawler, engine=None):
|
|
16
|
+
if getattr(crawler, "_ingest_error_mw_hooked", False):
|
|
17
|
+
return
|
|
18
|
+
try:
|
|
19
|
+
engine = engine or crawler.engine
|
|
20
|
+
mwman = engine.downloader.middleware
|
|
21
|
+
except (AttributeError, RuntimeError):
|
|
22
|
+
return
|
|
23
|
+
|
|
24
|
+
mw = ErrorMiddleware()
|
|
25
|
+
mw.crawler = crawler
|
|
26
|
+
mw.collector = crawler.ingest_collector
|
|
27
|
+
mwman._add_middleware(mw)
|
|
28
|
+
crawler._ingest_error_mw_hooked = True
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def _find_building(cls):
|
|
32
|
+
"""Find an in-progress constructor instance of *cls* on the call stack."""
|
|
33
|
+
for frame_info in inspect.stack()[1:]:
|
|
34
|
+
obj = frame_info.frame.f_locals.get("self")
|
|
35
|
+
if isinstance(obj, cls):
|
|
36
|
+
return obj
|
|
37
|
+
return None
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def attach_runtime_hooks(crawler):
|
|
41
|
+
"""
|
|
42
|
+
Attach parent_url + error hooks.
|
|
43
|
+
|
|
44
|
+
ITEM_PIPELINES load inside Scraper.__init__ (before crawler.engine is set),
|
|
45
|
+
so we locate the in-progress Scraper/Engine on the stack and patch those.
|
|
46
|
+
"""
|
|
47
|
+
if not hasattr(crawler, "ingest_parent_by_fp"):
|
|
48
|
+
crawler.ingest_parent_by_fp = {}
|
|
49
|
+
|
|
50
|
+
scraper = _find_building(Scraper)
|
|
51
|
+
if scraper is not None:
|
|
52
|
+
_inject_into_scraper(crawler, scraper)
|
|
53
|
+
else:
|
|
54
|
+
try:
|
|
55
|
+
_inject_into_scraper(crawler, crawler.engine.scraper)
|
|
56
|
+
except (AttributeError, RuntimeError):
|
|
57
|
+
pass
|
|
58
|
+
|
|
59
|
+
engine = _find_building(ExecutionEngine)
|
|
60
|
+
if engine is not None and getattr(engine, "downloader", None) is not None:
|
|
61
|
+
_inject_error_middleware(crawler, engine)
|
|
62
|
+
else:
|
|
63
|
+
try:
|
|
64
|
+
_inject_error_middleware(crawler, crawler.engine)
|
|
65
|
+
except (AttributeError, RuntimeError):
|
|
66
|
+
pass
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def enable_ingest(crawler):
|
|
70
|
+
"""
|
|
71
|
+
Idempotently enable request logging, error logging, parent_url tracking,
|
|
72
|
+
job logs, and stats. Called from DbInsertPipeline so projects only
|
|
73
|
+
need ITEM_PIPELINES.
|
|
74
|
+
"""
|
|
75
|
+
if getattr(crawler, "_ingest_enabled", False):
|
|
76
|
+
return
|
|
77
|
+
crawler._ingest_enabled = True
|
|
78
|
+
|
|
79
|
+
ensure_collector(crawler)
|
|
80
|
+
|
|
81
|
+
LoggingExtension.from_crawler(crawler)
|
|
82
|
+
StatsExtension.from_crawler(crawler)
|
|
83
|
+
RequestLogger.from_crawler(crawler)
|
|
84
|
+
|
|
85
|
+
attach_runtime_hooks(crawler)
|
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
"""Thread-safe batch buffer for requests, items, logs, and stats."""
|
|
2
|
+
from threading import Lock
|
|
3
|
+
|
|
4
|
+
|
|
5
|
+
def ensure_collector(crawler):
|
|
6
|
+
"""Return the crawler's shared DataCollector, creating it if needed."""
|
|
7
|
+
if not hasattr(crawler, "ingest_collector"):
|
|
8
|
+
crawler.ingest_collector = DataCollector()
|
|
9
|
+
return crawler.ingest_collector
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class DataCollector:
|
|
13
|
+
"""
|
|
14
|
+
Collects requests, items, logs, and stats in batches.
|
|
15
|
+
|
|
16
|
+
Thread-safe. One instance is stored on the crawler and shared by the
|
|
17
|
+
pipeline, request logger, error middleware, and logging extension.
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
def __init__(self):
|
|
21
|
+
self.requests = []
|
|
22
|
+
self.items = []
|
|
23
|
+
self.logs = []
|
|
24
|
+
self.stats = None
|
|
25
|
+
self.lock = Lock()
|
|
26
|
+
|
|
27
|
+
def add_request(self, request_log: dict):
|
|
28
|
+
with self.lock:
|
|
29
|
+
self.requests.append(request_log)
|
|
30
|
+
|
|
31
|
+
def add_item(self, item: dict):
|
|
32
|
+
with self.lock:
|
|
33
|
+
self.items.append(item)
|
|
34
|
+
|
|
35
|
+
def add_log(self, log_entry: dict):
|
|
36
|
+
with self.lock:
|
|
37
|
+
self.logs.append(log_entry)
|
|
38
|
+
|
|
39
|
+
def set_stats(self, stats: dict):
|
|
40
|
+
with self.lock:
|
|
41
|
+
self.stats = stats
|
|
42
|
+
|
|
43
|
+
def get_and_clear(self):
|
|
44
|
+
"""Return collected data and clear buffers. None if empty."""
|
|
45
|
+
with self.lock:
|
|
46
|
+
if not (self.requests or self.items or self.logs or self.stats):
|
|
47
|
+
return None
|
|
48
|
+
|
|
49
|
+
data = {
|
|
50
|
+
"requests": self.requests[:],
|
|
51
|
+
"items": self.items[:],
|
|
52
|
+
"logs": self.logs[:],
|
|
53
|
+
}
|
|
54
|
+
if self.stats:
|
|
55
|
+
data["stats"] = self.stats
|
|
56
|
+
|
|
57
|
+
self.requests.clear()
|
|
58
|
+
self.items.clear()
|
|
59
|
+
self.logs.clear()
|
|
60
|
+
self.stats = None
|
|
61
|
+
return data
|
|
62
|
+
|
|
63
|
+
def requeue(self, data: dict):
|
|
64
|
+
"""Put data back at the front of the buffers after a failed flush."""
|
|
65
|
+
with self.lock:
|
|
66
|
+
self.requests = data.get("requests", []) + self.requests
|
|
67
|
+
self.items = data.get("items", []) + self.items
|
|
68
|
+
self.logs = data.get("logs", []) + self.logs
|
|
69
|
+
if data.get("stats") is not None and self.stats is None:
|
|
70
|
+
self.stats = data["stats"]
|
|
71
|
+
|
|
72
|
+
def has_data(self) -> bool:
|
|
73
|
+
with self.lock:
|
|
74
|
+
return bool(self.requests or self.items or self.logs or self.stats)
|
|
75
|
+
|
|
76
|
+
def size(self) -> int:
|
|
77
|
+
with self.lock:
|
|
78
|
+
return len(self.requests) + len(self.items) + len(self.logs)
|
|
@@ -0,0 +1,133 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Module for managing and validating crawler settings.
|
|
3
|
+
"""
|
|
4
|
+
from urllib.parse import quote_plus
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
class Settings:
|
|
8
|
+
"""
|
|
9
|
+
Handles settings configuration for crawlers, providing access to default values,
|
|
10
|
+
database table names, and other operational parameters defined in crawler settings.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
DEFAULT_ITEMS_TABLE = "job_items"
|
|
14
|
+
DEFAULT_REQUESTS_TABLE = "job_requests"
|
|
15
|
+
DEFAULT_LOGS_TABLE = "job_logs"
|
|
16
|
+
DEFAULT_JOBS_TABLE = "jobs"
|
|
17
|
+
DEFAULT_DB_TYPE = "postgres"
|
|
18
|
+
DEFAULT_TIMEZONE = "Asia/Karachi"
|
|
19
|
+
DEFAULT_BATCH_SIZE = 50
|
|
20
|
+
DEFAULT_FLUSH_INTERVAL = 10.0
|
|
21
|
+
_DB_SCHEMES = {
|
|
22
|
+
"postgres": "postgresql",
|
|
23
|
+
"postgresql": "postgresql",
|
|
24
|
+
}
|
|
25
|
+
_DB_PORTS = {
|
|
26
|
+
"postgres": 5432,
|
|
27
|
+
"postgresql": 5432,
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
def __init__(self, crawler_settings):
|
|
31
|
+
self.crawler_settings = crawler_settings
|
|
32
|
+
|
|
33
|
+
@property
|
|
34
|
+
def db_url(self):
|
|
35
|
+
"""Database URL from DB_URL, or built from discrete DB_* fields."""
|
|
36
|
+
url = self.crawler_settings.get("DB_URL")
|
|
37
|
+
if url:
|
|
38
|
+
return url
|
|
39
|
+
|
|
40
|
+
host = self.crawler_settings.get("DB_HOST")
|
|
41
|
+
if not host:
|
|
42
|
+
return None
|
|
43
|
+
|
|
44
|
+
user = self.crawler_settings.get("DB_USER") or ""
|
|
45
|
+
password = self.crawler_settings.get("DB_PASSWORD") or ""
|
|
46
|
+
port = self.crawler_settings.get("DB_PORT", self._default_port())
|
|
47
|
+
name = self.crawler_settings.get("DB_NAME") or ""
|
|
48
|
+
scheme = self._DB_SCHEMES.get(self.db_type, self.db_type)
|
|
49
|
+
return (
|
|
50
|
+
f"{scheme}://{quote_plus(str(user))}:{quote_plus(str(password))}"
|
|
51
|
+
f"@{host}:{port}/{name}"
|
|
52
|
+
)
|
|
53
|
+
|
|
54
|
+
@property
|
|
55
|
+
def db_type(self):
|
|
56
|
+
"""Database engine from ``DB_TYPE`` (default: postgres)."""
|
|
57
|
+
raw = self.crawler_settings.get("DB_TYPE", self.DEFAULT_DB_TYPE)
|
|
58
|
+
if raw in (None, ""):
|
|
59
|
+
return self.DEFAULT_DB_TYPE
|
|
60
|
+
return str(raw).strip().lower()
|
|
61
|
+
|
|
62
|
+
def _default_port(self):
|
|
63
|
+
return self._DB_PORTS.get(self.db_type, 5432)
|
|
64
|
+
|
|
65
|
+
@property
|
|
66
|
+
def db_items_table(self):
|
|
67
|
+
return self.crawler_settings.get("ITEMS_TABLE", self.DEFAULT_ITEMS_TABLE)
|
|
68
|
+
|
|
69
|
+
@property
|
|
70
|
+
def db_requests_table(self):
|
|
71
|
+
return self.crawler_settings.get("REQUESTS_TABLE", self.DEFAULT_REQUESTS_TABLE)
|
|
72
|
+
|
|
73
|
+
@property
|
|
74
|
+
def db_logs_table(self):
|
|
75
|
+
return self.crawler_settings.get("LOGS_TABLE", self.DEFAULT_LOGS_TABLE)
|
|
76
|
+
|
|
77
|
+
@property
|
|
78
|
+
def db_jobs_table(self):
|
|
79
|
+
return (
|
|
80
|
+
self.crawler_settings.get("JOBS_TABLE")
|
|
81
|
+
or self.crawler_settings.get("DETAILS_TABLE")
|
|
82
|
+
or self.DEFAULT_JOBS_TABLE
|
|
83
|
+
)
|
|
84
|
+
|
|
85
|
+
@property
|
|
86
|
+
def create_tables(self):
|
|
87
|
+
return self.crawler_settings.getbool("CREATE_TABLES", True)
|
|
88
|
+
|
|
89
|
+
@property
|
|
90
|
+
def ingest_batch_size(self):
|
|
91
|
+
return self.crawler_settings.getint("INGEST_BATCH_SIZE", self.DEFAULT_BATCH_SIZE)
|
|
92
|
+
|
|
93
|
+
@property
|
|
94
|
+
def ingest_flush_interval(self):
|
|
95
|
+
return self.crawler_settings.getfloat(
|
|
96
|
+
"INGEST_FLUSH_INTERVAL", self.DEFAULT_FLUSH_INTERVAL
|
|
97
|
+
)
|
|
98
|
+
|
|
99
|
+
def get_tz(self):
|
|
100
|
+
return self.crawler_settings.get("TIMEZONE", self.DEFAULT_TIMEZONE)
|
|
101
|
+
|
|
102
|
+
@staticmethod
|
|
103
|
+
def get_identifier_column():
|
|
104
|
+
return "job_id"
|
|
105
|
+
|
|
106
|
+
def get_identifier_value(self, spider):
|
|
107
|
+
"""Use JOB_ID if set; otherwise a unique generated id (cached per crawl)."""
|
|
108
|
+
from ..utils.job_id import cache_job_id, cached_job_id, generate_job_id
|
|
109
|
+
|
|
110
|
+
configured = self.crawler_settings.get("JOB_ID", None)
|
|
111
|
+
if configured not in (None, ""):
|
|
112
|
+
return cache_job_id(spider, str(configured))
|
|
113
|
+
|
|
114
|
+
existing = cached_job_id(spider)
|
|
115
|
+
if existing:
|
|
116
|
+
return existing
|
|
117
|
+
|
|
118
|
+
return cache_job_id(spider, generate_job_id(getattr(spider, "name", "spider")))
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def validate_settings(settings):
|
|
122
|
+
"""Validate configuration settings. Accepts DB_URL or discrete DB_* fields."""
|
|
123
|
+
if settings.db_type not in settings._DB_SCHEMES:
|
|
124
|
+
supported = ", ".join(sorted(settings._DB_SCHEMES))
|
|
125
|
+
raise ValueError(
|
|
126
|
+
f"Unsupported DB_TYPE={settings.db_type!r}. Supported: {supported}"
|
|
127
|
+
)
|
|
128
|
+
if not settings.db_url:
|
|
129
|
+
raise ValueError(
|
|
130
|
+
"Database connection is required: set DB_URL or "
|
|
131
|
+
"DB_HOST / DB_USER / DB_PASSWORD / DB_NAME"
|
|
132
|
+
)
|
|
133
|
+
return True
|