scrapy-ingest 1.0.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (43) hide show
  1. scrapy_ingest-1.0.0/LICENSE +21 -0
  2. scrapy_ingest-1.0.0/PKG-INFO +163 -0
  3. scrapy_ingest-1.0.0/README.md +98 -0
  4. scrapy_ingest-1.0.0/scrapy_ingest/__init__.py +36 -0
  5. scrapy_ingest-1.0.0/scrapy_ingest/bootstrap/__init__.py +5 -0
  6. scrapy_ingest-1.0.0/scrapy_ingest/bootstrap/bootstrap.py +85 -0
  7. scrapy_ingest-1.0.0/scrapy_ingest/collector/__init__.py +5 -0
  8. scrapy_ingest-1.0.0/scrapy_ingest/collector/collector.py +78 -0
  9. scrapy_ingest-1.0.0/scrapy_ingest/config/__init__.py +2 -0
  10. scrapy_ingest-1.0.0/scrapy_ingest/config/settings.py +133 -0
  11. scrapy_ingest-1.0.0/scrapy_ingest/database/__init__.py +2 -0
  12. scrapy_ingest-1.0.0/scrapy_ingest/database/connection.py +144 -0
  13. scrapy_ingest-1.0.0/scrapy_ingest/database/flusher.py +135 -0
  14. scrapy_ingest-1.0.0/scrapy_ingest/database/schema.py +266 -0
  15. scrapy_ingest-1.0.0/scrapy_ingest/database/writer.py +232 -0
  16. scrapy_ingest-1.0.0/scrapy_ingest/extensions/__init__.py +7 -0
  17. scrapy_ingest-1.0.0/scrapy_ingest/extensions/base.py +81 -0
  18. scrapy_ingest-1.0.0/scrapy_ingest/extensions/log_handler.py +185 -0
  19. scrapy_ingest-1.0.0/scrapy_ingest/extensions/logging.py +64 -0
  20. scrapy_ingest-1.0.0/scrapy_ingest/extensions/request_logger.py +45 -0
  21. scrapy_ingest-1.0.0/scrapy_ingest/extensions/stats.py +31 -0
  22. scrapy_ingest-1.0.0/scrapy_ingest/middleware/__init__.py +5 -0
  23. scrapy_ingest-1.0.0/scrapy_ingest/middleware/middleware.py +184 -0
  24. scrapy_ingest-1.0.0/scrapy_ingest/pipelines/__init__.py +11 -0
  25. scrapy_ingest-1.0.0/scrapy_ingest/pipelines/base.py +20 -0
  26. scrapy_ingest-1.0.0/scrapy_ingest/pipelines/items.py +39 -0
  27. scrapy_ingest-1.0.0/scrapy_ingest/pipelines/main.py +20 -0
  28. scrapy_ingest-1.0.0/scrapy_ingest/pipelines/requests.py +36 -0
  29. scrapy_ingest-1.0.0/scrapy_ingest/utils/__init__.py +2 -0
  30. scrapy_ingest-1.0.0/scrapy_ingest/utils/fingerprint.py +18 -0
  31. scrapy_ingest-1.0.0/scrapy_ingest/utils/job_id.py +27 -0
  32. scrapy_ingest-1.0.0/scrapy_ingest/utils/parent.py +31 -0
  33. scrapy_ingest-1.0.0/scrapy_ingest/utils/serialization.py +20 -0
  34. scrapy_ingest-1.0.0/scrapy_ingest/utils/time.py +19 -0
  35. scrapy_ingest-1.0.0/scrapy_ingest.egg-info/PKG-INFO +163 -0
  36. scrapy_ingest-1.0.0/scrapy_ingest.egg-info/SOURCES.txt +41 -0
  37. scrapy_ingest-1.0.0/scrapy_ingest.egg-info/dependency_links.txt +1 -0
  38. scrapy_ingest-1.0.0/scrapy_ingest.egg-info/entry_points.txt +6 -0
  39. scrapy_ingest-1.0.0/scrapy_ingest.egg-info/not-zip-safe +1 -0
  40. scrapy_ingest-1.0.0/scrapy_ingest.egg-info/requires.txt +26 -0
  41. scrapy_ingest-1.0.0/scrapy_ingest.egg-info/top_level.txt +1 -0
  42. scrapy_ingest-1.0.0/setup.cfg +4 -0
  43. scrapy_ingest-1.0.0/setup.py +115 -0
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2025 Fawad Ali
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,163 @@
1
+ Metadata-Version: 2.4
2
+ Name: scrapy-ingest
3
+ Version: 1.0.0
4
+ Summary: Scrapy extension for database ingestion with job/spider tracking
5
+ Home-page: https://github.com/fawadss1/scrapy_item_ingest
6
+ Author: Fawad Ali
7
+ Author-email: fawadstar6@gmail.com
8
+ Project-URL: Documentation, https://scrapy-ingest.readthedocs.io/
9
+ Project-URL: Source, https://github.com/fawadss1/scrapy_item_ingest
10
+ Project-URL: Tracker, https://github.com/fawadss1/scrapy_item_ingest/issues
11
+ Keywords: scrapy,database,postgresql,web-scraping,data-pipeline
12
+ Classifier: Development Status :: 4 - Beta
13
+ Classifier: Intended Audience :: Developers
14
+ Classifier: License :: OSI Approved :: MIT License
15
+ Classifier: Operating System :: OS Independent
16
+ Classifier: Programming Language :: Python :: 3
17
+ Classifier: Programming Language :: Python :: 3.7
18
+ Classifier: Programming Language :: Python :: 3.8
19
+ Classifier: Programming Language :: Python :: 3.9
20
+ Classifier: Programming Language :: Python :: 3.10
21
+ Classifier: Programming Language :: Python :: 3.11
22
+ Classifier: Framework :: Scrapy
23
+ Classifier: Topic :: Software Development :: Libraries :: Python Modules
24
+ Classifier: Topic :: Internet :: WWW/HTTP
25
+ Classifier: Topic :: Database
26
+ Requires-Python: >=3.7
27
+ Description-Content-Type: text/markdown
28
+ License-File: LICENSE
29
+ Requires-Dist: scrapy>=2.13.3
30
+ Requires-Dist: psycopg2-binary>=2.9.10
31
+ Requires-Dist: itemadapter>=0.11.0
32
+ Requires-Dist: SQLAlchemy>=2.0.41
33
+ Requires-Dist: pytz>=2025.2
34
+ Requires-Dist: w3lib>=1.22.0
35
+ Provides-Extra: docs
36
+ Requires-Dist: sphinx>=5.0.0; extra == "docs"
37
+ Requires-Dist: sphinx_rtd_theme>=1.2.0; extra == "docs"
38
+ Requires-Dist: myst-parser>=0.18.0; extra == "docs"
39
+ Requires-Dist: sphinx-autodoc-typehints>=1.19.0; extra == "docs"
40
+ Requires-Dist: sphinx-copybutton>=0.5.0; extra == "docs"
41
+ Provides-Extra: dev
42
+ Requires-Dist: pytest>=7.0.0; extra == "dev"
43
+ Requires-Dist: pytest-cov>=4.0.0; extra == "dev"
44
+ Requires-Dist: black>=22.0.0; extra == "dev"
45
+ Requires-Dist: flake8>=5.0.0; extra == "dev"
46
+ Requires-Dist: mypy>=0.991; extra == "dev"
47
+ Requires-Dist: pre-commit>=2.20.0; extra == "dev"
48
+ Provides-Extra: test
49
+ Requires-Dist: pytest>=7.0.0; extra == "test"
50
+ Requires-Dist: pytest-cov>=4.0.0; extra == "test"
51
+ Requires-Dist: pytest-mock>=3.8.0; extra == "test"
52
+ Dynamic: author
53
+ Dynamic: author-email
54
+ Dynamic: classifier
55
+ Dynamic: description
56
+ Dynamic: description-content-type
57
+ Dynamic: home-page
58
+ Dynamic: keywords
59
+ Dynamic: license-file
60
+ Dynamic: project-url
61
+ Dynamic: provides-extra
62
+ Dynamic: requires-dist
63
+ Dynamic: requires-python
64
+ Dynamic: summary
65
+
66
+ # Scrapy Ingest
67
+
68
+ A Scrapy addon that saves **items, requests, logs, and stats** to PostgreSQL — with parent_url tracking, failed-request errors, and full job log capture (including `print()`).
69
+
70
+ ## Install
71
+
72
+ ```bash
73
+ pip install scrapy-ingest
74
+ ```
75
+
76
+ ## Minimal setup (settings.py)
77
+
78
+ Only the item pipeline is required — requests, logs, stats, parent_url, and error logging are enabled automatically:
79
+
80
+ ```python
81
+ ITEM_PIPELINES = {
82
+ "scrapy_ingest.pipelines.DbInsertPipeline": 300,
83
+ }
84
+
85
+ # Pick ONE of the two database config styles:
86
+ DB_URL = "postgresql://user:password@localhost:5432/database"
87
+ # Or use discrete fields (avoids URL encoding):
88
+ # DB_TYPE = "postgres"
89
+ # DB_HOST = "localhost"
90
+ # DB_PORT = 5432
91
+ # DB_USER = "user"
92
+ # DB_PASSWORD = "password"
93
+ # DB_NAME = "database"
94
+
95
+ # Optional
96
+ CREATE_TABLES = True # auto-create tables on first run (default True)
97
+ JOB_ID = 1 # or omit; a unique id is generated per crawl
98
+ INGEST_BATCH_SIZE = 50 # flush when this many rows are buffered
99
+ ```
100
+
101
+ Run your spider:
102
+
103
+ ```bash
104
+ scrapy crawl your_spider
105
+ ```
106
+
107
+ Log level follows Scrapy `LOG_LEVEL`.
108
+
109
+ ## What is stored
110
+
111
+ | Table | Contents |
112
+ |----------------|---------------------------------------------------------------------------------------------------------|
113
+ | `jobs` | One row per crawl: `id`, unique `job_id` string, spider, status, start/finish, counts, items/min, stats |
114
+ | `job_items` | JSON items (`crawled_at` added). `job_id` = `jobs.id` (CASCADE) |
115
+ | `job_requests` | url, parent_url, parent_id, status, response_time, error, success. `job_id` = `jobs.id` (CASCADE) |
116
+ | `job_logs` | time, logger, level, message, exception. `job_id` = `jobs.id` (CASCADE) |
117
+
118
+ Request `parent_url` is the page that scheduled the request (e.g. sitemap → product). Start URLs are `null`.
119
+
120
+ Data flushes on batch size, every 10s, and on engine/process stop.
121
+
122
+ ## Troubleshooting
123
+
124
+ - Password has special characters like `@` or `$`?
125
+ - In a URL, encode them: `@` -> `%40`, `$` -> `%24`.
126
+ - Example: `postgresql://user:PAK%40swat1%24@localhost:5432/db`
127
+ - Or use the discrete fields (no encoding needed).
128
+ - **Yield items** from callbacks (not only `return` inside a generator).
129
+
130
+ ## Useful settings (optional)
131
+
132
+ - `DB_TYPE` (default: `postgres`) — used when building a URL from `DB_HOST` / `DB_*` fields
133
+ - `INGEST_BATCH_SIZE` (default: `50`) — flush when this many items+requests+logs are buffered
134
+ - `INGEST_FLUSH_INTERVAL` (default: `10`) — periodic flush in seconds
135
+ - `CREATE_TABLES` (default: `True`) — create tables on startup
136
+ - `ITEMS_TABLE`, `REQUESTS_TABLE`, `LOGS_TABLE`, `JOBS_TABLE` — override table names
137
+ - `TIMEZONE` (default: `Asia/Karachi`) — timezone for `created_at`
138
+ - `JOB_ID` — omit to auto-generate a unique id (`spider-YYYYMMDDHHMMSS-xxxxxxxx`)
139
+
140
+ ## Standalone components
141
+
142
+ If you only want part of the collection:
143
+
144
+ ```python
145
+ # Items only
146
+ ITEM_PIPELINES = {"scrapy_ingest.pipelines.ItemsPipeline": 300}
147
+
148
+ # Requests only (parent_url + errors)
149
+ ITEM_PIPELINES = {"scrapy_ingest.pipelines.RequestsPipeline": 300}
150
+
151
+ # Logs only
152
+ EXTENSIONS = {"scrapy_ingest.extensions.LoggingExtension": 500}
153
+ ```
154
+
155
+ ## Links
156
+
157
+ - Docs: https://scrapy-ingest.readthedocs.io/
158
+ - Changelog: docs/development/changelog.rst
159
+ - Issues: https://github.com/fawadss1/scrapy_item_ingest/issues
160
+
161
+ ## License
162
+
163
+ MIT License. See [LICENSE](LICENSE).
@@ -0,0 +1,98 @@
1
+ # Scrapy Ingest
2
+
3
+ A Scrapy addon that saves **items, requests, logs, and stats** to PostgreSQL — with parent_url tracking, failed-request errors, and full job log capture (including `print()`).
4
+
5
+ ## Install
6
+
7
+ ```bash
8
+ pip install scrapy-ingest
9
+ ```
10
+
11
+ ## Minimal setup (settings.py)
12
+
13
+ Only the item pipeline is required — requests, logs, stats, parent_url, and error logging are enabled automatically:
14
+
15
+ ```python
16
+ ITEM_PIPELINES = {
17
+ "scrapy_ingest.pipelines.DbInsertPipeline": 300,
18
+ }
19
+
20
+ # Pick ONE of the two database config styles:
21
+ DB_URL = "postgresql://user:password@localhost:5432/database"
22
+ # Or use discrete fields (avoids URL encoding):
23
+ # DB_TYPE = "postgres"
24
+ # DB_HOST = "localhost"
25
+ # DB_PORT = 5432
26
+ # DB_USER = "user"
27
+ # DB_PASSWORD = "password"
28
+ # DB_NAME = "database"
29
+
30
+ # Optional
31
+ CREATE_TABLES = True # auto-create tables on first run (default True)
32
+ JOB_ID = 1 # or omit; a unique id is generated per crawl
33
+ INGEST_BATCH_SIZE = 50 # flush when this many rows are buffered
34
+ ```
35
+
36
+ Run your spider:
37
+
38
+ ```bash
39
+ scrapy crawl your_spider
40
+ ```
41
+
42
+ Log level follows Scrapy `LOG_LEVEL`.
43
+
44
+ ## What is stored
45
+
46
+ | Table | Contents |
47
+ |----------------|---------------------------------------------------------------------------------------------------------|
48
+ | `jobs` | One row per crawl: `id`, unique `job_id` string, spider, status, start/finish, counts, items/min, stats |
49
+ | `job_items` | JSON items (`crawled_at` added). `job_id` = `jobs.id` (CASCADE) |
50
+ | `job_requests` | url, parent_url, parent_id, status, response_time, error, success. `job_id` = `jobs.id` (CASCADE) |
51
+ | `job_logs` | time, logger, level, message, exception. `job_id` = `jobs.id` (CASCADE) |
52
+
53
+ Request `parent_url` is the page that scheduled the request (e.g. sitemap → product). Start URLs are `null`.
54
+
55
+ Data flushes on batch size, every 10s, and on engine/process stop.
56
+
57
+ ## Troubleshooting
58
+
59
+ - Password has special characters like `@` or `$`?
60
+ - In a URL, encode them: `@` -> `%40`, `$` -> `%24`.
61
+ - Example: `postgresql://user:PAK%40swat1%24@localhost:5432/db`
62
+ - Or use the discrete fields (no encoding needed).
63
+ - **Yield items** from callbacks (not only `return` inside a generator).
64
+
65
+ ## Useful settings (optional)
66
+
67
+ - `DB_TYPE` (default: `postgres`) — used when building a URL from `DB_HOST` / `DB_*` fields
68
+ - `INGEST_BATCH_SIZE` (default: `50`) — flush when this many items+requests+logs are buffered
69
+ - `INGEST_FLUSH_INTERVAL` (default: `10`) — periodic flush in seconds
70
+ - `CREATE_TABLES` (default: `True`) — create tables on startup
71
+ - `ITEMS_TABLE`, `REQUESTS_TABLE`, `LOGS_TABLE`, `JOBS_TABLE` — override table names
72
+ - `TIMEZONE` (default: `Asia/Karachi`) — timezone for `created_at`
73
+ - `JOB_ID` — omit to auto-generate a unique id (`spider-YYYYMMDDHHMMSS-xxxxxxxx`)
74
+
75
+ ## Standalone components
76
+
77
+ If you only want part of the collection:
78
+
79
+ ```python
80
+ # Items only
81
+ ITEM_PIPELINES = {"scrapy_ingest.pipelines.ItemsPipeline": 300}
82
+
83
+ # Requests only (parent_url + errors)
84
+ ITEM_PIPELINES = {"scrapy_ingest.pipelines.RequestsPipeline": 300}
85
+
86
+ # Logs only
87
+ EXTENSIONS = {"scrapy_ingest.extensions.LoggingExtension": 500}
88
+ ```
89
+
90
+ ## Links
91
+
92
+ - Docs: https://scrapy-ingest.readthedocs.io/
93
+ - Changelog: docs/development/changelog.rst
94
+ - Issues: https://github.com/fawadss1/scrapy_item_ingest/issues
95
+
96
+ ## License
97
+
98
+ MIT License. See [LICENSE](LICENSE).
@@ -0,0 +1,36 @@
1
+ """
2
+ scrapy_ingest - A Scrapy extension for ingesting items, requests, logs, and stats into PostgreSQL.
3
+
4
+ Enabling DbInsertPipeline auto-enables request logging (with parent_url),
5
+ error logging, full job logs (including print()), and crawl stats.
6
+ """
7
+
8
+ from .extensions.log_handler import install_early
9
+
10
+ install_early()
11
+
12
+ __version__ = "1.0.0"
13
+ __author__ = "Fawad Ali"
14
+ __description__ = "Scrapy extension for database ingestion with job/spider tracking"
15
+
16
+ from .pipelines.main import DbInsertPipeline
17
+ from .extensions.logging import LoggingExtension
18
+ from .extensions.stats import StatsExtension
19
+ from .pipelines.items import ItemsPipeline
20
+ from .pipelines.requests import RequestsPipeline
21
+ from .extensions.request_logger import RequestLogger
22
+ from .config.settings import Settings, validate_settings
23
+
24
+ __all__ = [
25
+ "DbInsertPipeline",
26
+ "LoggingExtension",
27
+ "StatsExtension",
28
+ "ItemsPipeline",
29
+ "RequestsPipeline",
30
+ "RequestLogger",
31
+ "Settings",
32
+ "validate_settings",
33
+ "__version__",
34
+ "__author__",
35
+ "__description__",
36
+ ]
@@ -0,0 +1,5 @@
1
+ """Bootstrap helpers that auto-enable ingest components."""
2
+
3
+ from .bootstrap import attach_runtime_hooks, enable_ingest
4
+
5
+ __all__ = ["attach_runtime_hooks", "enable_ingest"]
@@ -0,0 +1,85 @@
1
+ """Enable all ingest components from a single entry point (the item pipeline)."""
2
+
3
+ import inspect
4
+
5
+ from scrapy.core.engine import ExecutionEngine
6
+ from scrapy.core.scraper import Scraper
7
+
8
+ from ..collector import ensure_collector
9
+ from ..extensions.logging import LoggingExtension
10
+ from ..extensions.stats import StatsExtension
11
+ from ..extensions.request_logger import RequestLogger
12
+ from ..middleware import ErrorMiddleware, _inject_into_scraper
13
+
14
+
15
+ def _inject_error_middleware(crawler, engine=None):
16
+ if getattr(crawler, "_ingest_error_mw_hooked", False):
17
+ return
18
+ try:
19
+ engine = engine or crawler.engine
20
+ mwman = engine.downloader.middleware
21
+ except (AttributeError, RuntimeError):
22
+ return
23
+
24
+ mw = ErrorMiddleware()
25
+ mw.crawler = crawler
26
+ mw.collector = crawler.ingest_collector
27
+ mwman._add_middleware(mw)
28
+ crawler._ingest_error_mw_hooked = True
29
+
30
+
31
+ def _find_building(cls):
32
+ """Find an in-progress constructor instance of *cls* on the call stack."""
33
+ for frame_info in inspect.stack()[1:]:
34
+ obj = frame_info.frame.f_locals.get("self")
35
+ if isinstance(obj, cls):
36
+ return obj
37
+ return None
38
+
39
+
40
+ def attach_runtime_hooks(crawler):
41
+ """
42
+ Attach parent_url + error hooks.
43
+
44
+ ITEM_PIPELINES load inside Scraper.__init__ (before crawler.engine is set),
45
+ so we locate the in-progress Scraper/Engine on the stack and patch those.
46
+ """
47
+ if not hasattr(crawler, "ingest_parent_by_fp"):
48
+ crawler.ingest_parent_by_fp = {}
49
+
50
+ scraper = _find_building(Scraper)
51
+ if scraper is not None:
52
+ _inject_into_scraper(crawler, scraper)
53
+ else:
54
+ try:
55
+ _inject_into_scraper(crawler, crawler.engine.scraper)
56
+ except (AttributeError, RuntimeError):
57
+ pass
58
+
59
+ engine = _find_building(ExecutionEngine)
60
+ if engine is not None and getattr(engine, "downloader", None) is not None:
61
+ _inject_error_middleware(crawler, engine)
62
+ else:
63
+ try:
64
+ _inject_error_middleware(crawler, crawler.engine)
65
+ except (AttributeError, RuntimeError):
66
+ pass
67
+
68
+
69
+ def enable_ingest(crawler):
70
+ """
71
+ Idempotently enable request logging, error logging, parent_url tracking,
72
+ job logs, and stats. Called from DbInsertPipeline so projects only
73
+ need ITEM_PIPELINES.
74
+ """
75
+ if getattr(crawler, "_ingest_enabled", False):
76
+ return
77
+ crawler._ingest_enabled = True
78
+
79
+ ensure_collector(crawler)
80
+
81
+ LoggingExtension.from_crawler(crawler)
82
+ StatsExtension.from_crawler(crawler)
83
+ RequestLogger.from_crawler(crawler)
84
+
85
+ attach_runtime_hooks(crawler)
@@ -0,0 +1,5 @@
1
+ """Shared batch collector for items, requests, logs, and stats."""
2
+
3
+ from .collector import DataCollector, ensure_collector
4
+
5
+ __all__ = ["DataCollector", "ensure_collector"]
@@ -0,0 +1,78 @@
1
+ """Thread-safe batch buffer for requests, items, logs, and stats."""
2
+ from threading import Lock
3
+
4
+
5
+ def ensure_collector(crawler):
6
+ """Return the crawler's shared DataCollector, creating it if needed."""
7
+ if not hasattr(crawler, "ingest_collector"):
8
+ crawler.ingest_collector = DataCollector()
9
+ return crawler.ingest_collector
10
+
11
+
12
+ class DataCollector:
13
+ """
14
+ Collects requests, items, logs, and stats in batches.
15
+
16
+ Thread-safe. One instance is stored on the crawler and shared by the
17
+ pipeline, request logger, error middleware, and logging extension.
18
+ """
19
+
20
+ def __init__(self):
21
+ self.requests = []
22
+ self.items = []
23
+ self.logs = []
24
+ self.stats = None
25
+ self.lock = Lock()
26
+
27
+ def add_request(self, request_log: dict):
28
+ with self.lock:
29
+ self.requests.append(request_log)
30
+
31
+ def add_item(self, item: dict):
32
+ with self.lock:
33
+ self.items.append(item)
34
+
35
+ def add_log(self, log_entry: dict):
36
+ with self.lock:
37
+ self.logs.append(log_entry)
38
+
39
+ def set_stats(self, stats: dict):
40
+ with self.lock:
41
+ self.stats = stats
42
+
43
+ def get_and_clear(self):
44
+ """Return collected data and clear buffers. None if empty."""
45
+ with self.lock:
46
+ if not (self.requests or self.items or self.logs or self.stats):
47
+ return None
48
+
49
+ data = {
50
+ "requests": self.requests[:],
51
+ "items": self.items[:],
52
+ "logs": self.logs[:],
53
+ }
54
+ if self.stats:
55
+ data["stats"] = self.stats
56
+
57
+ self.requests.clear()
58
+ self.items.clear()
59
+ self.logs.clear()
60
+ self.stats = None
61
+ return data
62
+
63
+ def requeue(self, data: dict):
64
+ """Put data back at the front of the buffers after a failed flush."""
65
+ with self.lock:
66
+ self.requests = data.get("requests", []) + self.requests
67
+ self.items = data.get("items", []) + self.items
68
+ self.logs = data.get("logs", []) + self.logs
69
+ if data.get("stats") is not None and self.stats is None:
70
+ self.stats = data["stats"]
71
+
72
+ def has_data(self) -> bool:
73
+ with self.lock:
74
+ return bool(self.requests or self.items or self.logs or self.stats)
75
+
76
+ def size(self) -> int:
77
+ with self.lock:
78
+ return len(self.requests) + len(self.items) + len(self.logs)
@@ -0,0 +1,2 @@
1
+ """Configuration modules for scrapy_ingest."""
2
+
@@ -0,0 +1,133 @@
1
+ """
2
+ Module for managing and validating crawler settings.
3
+ """
4
+ from urllib.parse import quote_plus
5
+
6
+
7
+ class Settings:
8
+ """
9
+ Handles settings configuration for crawlers, providing access to default values,
10
+ database table names, and other operational parameters defined in crawler settings.
11
+ """
12
+
13
+ DEFAULT_ITEMS_TABLE = "job_items"
14
+ DEFAULT_REQUESTS_TABLE = "job_requests"
15
+ DEFAULT_LOGS_TABLE = "job_logs"
16
+ DEFAULT_JOBS_TABLE = "jobs"
17
+ DEFAULT_DB_TYPE = "postgres"
18
+ DEFAULT_TIMEZONE = "Asia/Karachi"
19
+ DEFAULT_BATCH_SIZE = 50
20
+ DEFAULT_FLUSH_INTERVAL = 10.0
21
+ _DB_SCHEMES = {
22
+ "postgres": "postgresql",
23
+ "postgresql": "postgresql",
24
+ }
25
+ _DB_PORTS = {
26
+ "postgres": 5432,
27
+ "postgresql": 5432,
28
+ }
29
+
30
+ def __init__(self, crawler_settings):
31
+ self.crawler_settings = crawler_settings
32
+
33
+ @property
34
+ def db_url(self):
35
+ """Database URL from DB_URL, or built from discrete DB_* fields."""
36
+ url = self.crawler_settings.get("DB_URL")
37
+ if url:
38
+ return url
39
+
40
+ host = self.crawler_settings.get("DB_HOST")
41
+ if not host:
42
+ return None
43
+
44
+ user = self.crawler_settings.get("DB_USER") or ""
45
+ password = self.crawler_settings.get("DB_PASSWORD") or ""
46
+ port = self.crawler_settings.get("DB_PORT", self._default_port())
47
+ name = self.crawler_settings.get("DB_NAME") or ""
48
+ scheme = self._DB_SCHEMES.get(self.db_type, self.db_type)
49
+ return (
50
+ f"{scheme}://{quote_plus(str(user))}:{quote_plus(str(password))}"
51
+ f"@{host}:{port}/{name}"
52
+ )
53
+
54
+ @property
55
+ def db_type(self):
56
+ """Database engine from ``DB_TYPE`` (default: postgres)."""
57
+ raw = self.crawler_settings.get("DB_TYPE", self.DEFAULT_DB_TYPE)
58
+ if raw in (None, ""):
59
+ return self.DEFAULT_DB_TYPE
60
+ return str(raw).strip().lower()
61
+
62
+ def _default_port(self):
63
+ return self._DB_PORTS.get(self.db_type, 5432)
64
+
65
+ @property
66
+ def db_items_table(self):
67
+ return self.crawler_settings.get("ITEMS_TABLE", self.DEFAULT_ITEMS_TABLE)
68
+
69
+ @property
70
+ def db_requests_table(self):
71
+ return self.crawler_settings.get("REQUESTS_TABLE", self.DEFAULT_REQUESTS_TABLE)
72
+
73
+ @property
74
+ def db_logs_table(self):
75
+ return self.crawler_settings.get("LOGS_TABLE", self.DEFAULT_LOGS_TABLE)
76
+
77
+ @property
78
+ def db_jobs_table(self):
79
+ return (
80
+ self.crawler_settings.get("JOBS_TABLE")
81
+ or self.crawler_settings.get("DETAILS_TABLE")
82
+ or self.DEFAULT_JOBS_TABLE
83
+ )
84
+
85
+ @property
86
+ def create_tables(self):
87
+ return self.crawler_settings.getbool("CREATE_TABLES", True)
88
+
89
+ @property
90
+ def ingest_batch_size(self):
91
+ return self.crawler_settings.getint("INGEST_BATCH_SIZE", self.DEFAULT_BATCH_SIZE)
92
+
93
+ @property
94
+ def ingest_flush_interval(self):
95
+ return self.crawler_settings.getfloat(
96
+ "INGEST_FLUSH_INTERVAL", self.DEFAULT_FLUSH_INTERVAL
97
+ )
98
+
99
+ def get_tz(self):
100
+ return self.crawler_settings.get("TIMEZONE", self.DEFAULT_TIMEZONE)
101
+
102
+ @staticmethod
103
+ def get_identifier_column():
104
+ return "job_id"
105
+
106
+ def get_identifier_value(self, spider):
107
+ """Use JOB_ID if set; otherwise a unique generated id (cached per crawl)."""
108
+ from ..utils.job_id import cache_job_id, cached_job_id, generate_job_id
109
+
110
+ configured = self.crawler_settings.get("JOB_ID", None)
111
+ if configured not in (None, ""):
112
+ return cache_job_id(spider, str(configured))
113
+
114
+ existing = cached_job_id(spider)
115
+ if existing:
116
+ return existing
117
+
118
+ return cache_job_id(spider, generate_job_id(getattr(spider, "name", "spider")))
119
+
120
+
121
+ def validate_settings(settings):
122
+ """Validate configuration settings. Accepts DB_URL or discrete DB_* fields."""
123
+ if settings.db_type not in settings._DB_SCHEMES:
124
+ supported = ", ".join(sorted(settings._DB_SCHEMES))
125
+ raise ValueError(
126
+ f"Unsupported DB_TYPE={settings.db_type!r}. Supported: {supported}"
127
+ )
128
+ if not settings.db_url:
129
+ raise ValueError(
130
+ "Database connection is required: set DB_URL or "
131
+ "DB_HOST / DB_USER / DB_PASSWORD / DB_NAME"
132
+ )
133
+ return True
@@ -0,0 +1,2 @@
1
+ """Database modules for scrapy_ingest."""
2
+