folioflex 1.2.0__tar.gz → 1.2.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {folioflex-1.2.0 → folioflex-1.2.2}/MANIFEST.in +2 -1
- {folioflex-1.2.0/folioflex.egg-info → folioflex-1.2.2}/PKG-INFO +34 -32
- {folioflex-1.2.0 → folioflex-1.2.2}/README.md +22 -23
- folioflex-1.2.2/folioflex/chatbot/scraper.py +363 -0
- {folioflex-1.2.0 → folioflex-1.2.2}/folioflex/configs/config.ini +1 -0
- folioflex-1.2.2/folioflex/dashboard/app.py +157 -0
- folioflex-1.2.2/folioflex/dashboard/assets/folioflex_logo.ico +0 -0
- folioflex-1.2.2/folioflex/dashboard/assets/folioflex_logo.png +0 -0
- folioflex-1.2.2/folioflex/dashboard/assets/github-mark-white.png +0 -0
- folioflex-1.2.2/folioflex/dashboard/components/__init__.py +1 -0
- folioflex-1.2.2/folioflex/dashboard/components/auth.py +78 -0
- folioflex-1.2.2/folioflex/dashboard/pages/budget.py +194 -0
- folioflex-1.2.2/folioflex/dashboard/pages/ideas.py +102 -0
- folioflex-1.2.2/folioflex/dashboard/pages/login.py +33 -0
- folioflex-1.2.2/folioflex/dashboard/pages/logout.py +18 -0
- folioflex-1.2.2/folioflex/dashboard/pages/macro.py +157 -0
- folioflex-1.2.2/folioflex/dashboard/pages/personal.py +447 -0
- folioflex-1.2.2/folioflex/dashboard/pages/sectors.py +206 -0
- folioflex-1.2.2/folioflex/dashboard/pages/stocks.py +267 -0
- folioflex-1.2.2/folioflex/dashboard/utils/__init__.py +1 -0
- {folioflex-1.2.0/folioflex/dashboard → folioflex-1.2.2/folioflex/dashboard/utils}/dashboard_helper.py +0 -38
- {folioflex-1.2.0 → folioflex-1.2.2}/folioflex/portfolio/broker.py +1 -1
- {folioflex-1.2.0 → folioflex-1.2.2}/folioflex/portfolio/heatmap.py +1 -0
- {folioflex-1.2.0 → folioflex-1.2.2}/folioflex/portfolio/helper.py +48 -7
- {folioflex-1.2.0 → folioflex-1.2.2}/folioflex/portfolio/portfolio.py +8 -1
- {folioflex-1.2.0 → folioflex-1.2.2}/folioflex/portfolio/wrappers.py +6 -1
- {folioflex-1.2.0 → folioflex-1.2.2}/folioflex/utils/cli.py +26 -0
- {folioflex-1.2.0 → folioflex-1.2.2}/folioflex/utils/config_helper.py +4 -0
- {folioflex-1.2.0 → folioflex-1.2.2}/folioflex/utils/cq.py +1 -1
- {folioflex-1.2.0 → folioflex-1.2.2}/folioflex/utils/custom_logger.py +33 -2
- {folioflex-1.2.0 → folioflex-1.2.2}/folioflex/utils/mailer.py +9 -13
- {folioflex-1.2.0 → folioflex-1.2.2}/folioflex/version.py +1 -1
- {folioflex-1.2.0 → folioflex-1.2.2/folioflex.egg-info}/PKG-INFO +34 -32
- {folioflex-1.2.0 → folioflex-1.2.2}/folioflex.egg-info/SOURCES.txt +10 -2
- {folioflex-1.2.0 → folioflex-1.2.2}/folioflex.egg-info/requires.txt +6 -4
- {folioflex-1.2.0 → folioflex-1.2.2}/folioflex.egg-info/top_level.txt +0 -1
- {folioflex-1.2.0 → folioflex-1.2.2}/pyproject.toml +10 -7
- {folioflex-1.2.0 → folioflex-1.2.2}/tests/test_chatbot.py +2 -2
- {folioflex-1.2.0 → folioflex-1.2.2}/tests/test_utils.py +1 -1
- folioflex-1.2.0/folioflex/chatbot/scraper.py +0 -136
- folioflex-1.2.0/folioflex/dashboard/pages/budget.py +0 -60
- folioflex-1.2.0/folioflex/dashboard/pages/ideas.py +0 -59
- folioflex-1.2.0/folioflex/dashboard/pages/login.py +0 -21
- folioflex-1.2.0/folioflex/dashboard/pages/macro.py +0 -159
- folioflex-1.2.0/folioflex/dashboard/pages/personal.py +0 -115
- folioflex-1.2.0/folioflex/dashboard/pages/sectors.py +0 -61
- folioflex-1.2.0/folioflex/dashboard/pages/stocks.py +0 -119
- {folioflex-1.2.0 → folioflex-1.2.2}/LICENSE.md +0 -0
- {folioflex-1.2.0 → folioflex-1.2.2}/docs/source/conf.py +0 -0
- {folioflex-1.2.0 → folioflex-1.2.2}/folioflex/__init__.py +0 -0
- {folioflex-1.2.0 → folioflex-1.2.2}/folioflex/budget/__init__.py +0 -0
- {folioflex-1.2.0 → folioflex-1.2.2}/folioflex/budget/budget.py +0 -0
- {folioflex-1.2.0 → folioflex-1.2.2}/folioflex/budget/models.py +0 -0
- {folioflex-1.2.0 → folioflex-1.2.2}/folioflex/chatbot/__init__.py +0 -0
- {folioflex-1.2.0 → folioflex-1.2.2}/folioflex/chatbot/providers.py +0 -0
- {folioflex-1.2.0 → folioflex-1.2.2}/folioflex/configs/__init__.py +0 -0
- {folioflex-1.2.0 → folioflex-1.2.2}/folioflex/configs/budget_personal.ini +0 -0
- {folioflex-1.2.0 → folioflex-1.2.2}/folioflex/configs/portfolio_dash.ini +0 -0
- {folioflex-1.2.0 → folioflex-1.2.2}/folioflex/configs/portfolio_demo.ini +0 -0
- {folioflex-1.2.0 → folioflex-1.2.2}/folioflex/configs/portfolio_personal.ini +0 -0
- {folioflex-1.2.0 → folioflex-1.2.2}/folioflex/configs/transactions_dash.csv +0 -0
- {folioflex-1.2.0 → folioflex-1.2.2}/folioflex/configs/transactions_demo.csv +0 -0
- {folioflex-1.2.0 → folioflex-1.2.2}/folioflex/dashboard/__init__.py +0 -0
- {folioflex-1.2.0/folioflex/dashboard → folioflex-1.2.2/folioflex/dashboard/components}/layouts.py +0 -0
- {folioflex-1.2.0 → folioflex-1.2.2}/folioflex/dashboard/pages/__init__.py +0 -0
- {folioflex-1.2.0 → folioflex-1.2.2}/folioflex/market/__init__.py +0 -0
- {folioflex-1.2.0 → folioflex-1.2.2}/folioflex/market/screener.py +0 -0
- {folioflex-1.2.0 → folioflex-1.2.2}/folioflex/portfolio/__init__.py +0 -0
- {folioflex-1.2.0 → folioflex-1.2.2}/folioflex/utils/__init__.py +0 -0
- {folioflex-1.2.0 → folioflex-1.2.2}/folioflex.egg-info/dependency_links.txt +0 -0
- {folioflex-1.2.0 → folioflex-1.2.2}/folioflex.egg-info/entry_points.txt +0 -0
- {folioflex-1.2.0 → folioflex-1.2.2}/setup.cfg +0 -0
- {folioflex-1.2.0 → folioflex-1.2.2}/setup.py +0 -0
- {folioflex-1.2.0 → folioflex-1.2.2}/tests/test_budget.py +0 -0
- {folioflex-1.2.0 → folioflex-1.2.2}/tests/test_portfolio.py +0 -0
- {folioflex-1.2.0 → folioflex-1.2.2}/tests/test_wrappers.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.1
|
|
2
2
|
Name: folioflex
|
|
3
|
-
Version: 1.2.
|
|
3
|
+
Version: 1.2.2
|
|
4
4
|
Summary: A collection of portfolio tracking capabilities
|
|
5
5
|
Author-email: John Koestner <johnkoestner@outlook.com>
|
|
6
6
|
License: The MIT License (MIT)
|
|
@@ -33,8 +33,7 @@ License-File: LICENSE.md
|
|
|
33
33
|
Requires-Dist: ipywidgets>=8.1.0
|
|
34
34
|
Requires-Dist: fredapi>=0.4.3
|
|
35
35
|
Requires-Dist: jupyter-dash>=0.4.2
|
|
36
|
-
Requires-Dist: jupyterlab>=4.0.
|
|
37
|
-
Requires-Dist: jupyterlab-code-formatter>=2.2.1
|
|
36
|
+
Requires-Dist: jupyterlab>=4.0.10
|
|
38
37
|
Requires-Dist: kaleido<0.2.0,>=0.1.0
|
|
39
38
|
Requires-Dist: numpy>=1.22.3
|
|
40
39
|
Requires-Dist: openpyxl>=3.0.7
|
|
@@ -43,13 +42,16 @@ Requires-Dist: pandas-market-calendars>=4.1.4
|
|
|
43
42
|
Requires-Dist: pyxirr>=0.7.2
|
|
44
43
|
Requires-Dist: sqlalchemy>=2.0.23
|
|
45
44
|
Requires-Dist: yfinance>=0.2.32
|
|
45
|
+
Requires-Dist: tzlocal>=5.2
|
|
46
46
|
Requires-Dist: dash>=1.0.2
|
|
47
|
+
Requires-Dist: dash_bootstrap_components>=1.6.0
|
|
47
48
|
Requires-Dist: dash-core-components>=1.0.0
|
|
48
49
|
Requires-Dist: dash-html-components>=1.0.0
|
|
49
50
|
Requires-Dist: dash-renderer>=1.0.0
|
|
50
51
|
Requires-Dist: dash-table>=4.0.2
|
|
51
52
|
Requires-Dist: Flask>=1.1.1
|
|
52
53
|
Requires-Dist: Flask-Compress>=1.4.0
|
|
54
|
+
Requires-Dist: Flask-Login>=0.6.3
|
|
53
55
|
Requires-Dist: gunicorn>=19.9.0
|
|
54
56
|
Requires-Dist: celery>=5.3.1
|
|
55
57
|
Requires-Dist: flower>=2.0.0
|
|
@@ -57,16 +59,17 @@ Requires-Dist: redis>=3.3.8
|
|
|
57
59
|
Requires-Dist: g4f==0.2.4.1
|
|
58
60
|
Requires-Dist: hugchat>=0.3.8
|
|
59
61
|
Requires-Dist: openai>=1.3.7
|
|
62
|
+
Requires-Dist: pyautogui>=0.9.53
|
|
60
63
|
Requires-Dist: seleniumbase>=4.22.0
|
|
61
|
-
|
|
62
|
-
Requires-Dist:
|
|
63
|
-
Requires-Dist:
|
|
64
|
-
Requires-Dist:
|
|
65
|
-
Requires-Dist:
|
|
66
|
-
Requires-Dist: psycopg2-binary; extra == "budget"
|
|
64
|
+
Requires-Dist: emoji>=2.9.0
|
|
65
|
+
Requires-Dist: gensim>=4.3.2
|
|
66
|
+
Requires-Dist: scikit-learn>=1.3.2
|
|
67
|
+
Requires-Dist: scipy==1.10.1
|
|
68
|
+
Requires-Dist: psycopg2-binary
|
|
67
69
|
Provides-Extra: dev
|
|
68
70
|
Requires-Dist: black>=23.7.0; extra == "dev"
|
|
69
71
|
Requires-Dist: isort>=5.13.2; extra == "dev"
|
|
72
|
+
Requires-Dist: jupyterlab-code-formatter>=2.2.1; extra == "dev"
|
|
70
73
|
Requires-Dist: pytest>=7.4.0; extra == "dev"
|
|
71
74
|
Requires-Dist: pytest-cov>=4.1.0; extra == "dev"
|
|
72
75
|
Requires-Dist: ruff>=0.1.5; extra == "dev"
|
|
@@ -88,19 +91,21 @@ Simple investment portfolio tool that will track stock and provide returns and o
|
|
|
88
91
|
|
|
89
92
|
|
|
90
93
|
## Table of Contents
|
|
91
|
-
- [
|
|
92
|
-
- [
|
|
93
|
-
- [
|
|
94
|
-
- [
|
|
95
|
-
- [
|
|
96
|
-
|
|
97
|
-
- [
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
- [
|
|
101
|
-
|
|
102
|
-
- [
|
|
103
|
-
|
|
94
|
+
- [Portfolio](#portfolio)
|
|
95
|
+
- [Table of Contents](#table-of-contents)
|
|
96
|
+
- [Overview](#overview)
|
|
97
|
+
- [Installation](#installation)
|
|
98
|
+
- [Local Install](#local-install)
|
|
99
|
+
- [Docker Install](#docker-install)
|
|
100
|
+
- [Usage](#usage)
|
|
101
|
+
- [CLI](#cli)
|
|
102
|
+
- [Python](#python)
|
|
103
|
+
- [Web Dashboard - Invest](#web-dashboard---invest)
|
|
104
|
+
- [Plaid Dashboard](#plaid-dashboard)
|
|
105
|
+
- [Other Tools](#other-tools)
|
|
106
|
+
- [Jupyter Lab Usage](#jupyter-lab-usage)
|
|
107
|
+
- [Logging](#logging)
|
|
108
|
+
- [Coverage](#coverage)
|
|
104
109
|
|
|
105
110
|
## Overview
|
|
106
111
|
|
|
@@ -113,13 +118,13 @@ Simple investment portfolio tool that will track stock and provide returns and o
|
|
|
113
118
|
**🔧 Features:**
|
|
114
119
|
|
|
115
120
|
- **Market Screener**: Filter and find trending stocks. 🔍
|
|
116
|
-

|
|
122
|
+

|
|
118
123
|
- **Portfolio Management**: Organize and track, your investments. 💼
|
|
119
|
-

|
|
120
125
|
- **Budget Tool**: Create and monitor a budget. 💰
|
|
121
|
-

|
|
127
|
+

|
|
123
128
|
|
|
124
129
|
**📚 Documentation:**
|
|
125
130
|
|
|
@@ -157,11 +162,8 @@ pip install folioflex
|
|
|
157
162
|
Other options can be installed if using more functionality
|
|
158
163
|
|
|
159
164
|
```
|
|
160
|
-
pip install folioflex
|
|
165
|
+
pip install folioflex
|
|
161
166
|
pip install folioflex[dev] # if needing to develop or lint
|
|
162
|
-
pip install folioflex[gpt] # if using the mailer or gpt code
|
|
163
|
-
pip install folioflex[web] # if using the web dashboard
|
|
164
|
-
pip install folioflex[worker] # if using the web dashboard
|
|
165
167
|
``````
|
|
166
168
|
|
|
167
169
|
Or could be done using GitHub.
|
|
@@ -194,7 +196,7 @@ services:
|
|
|
194
196
|
folioflex-web:
|
|
195
197
|
image: dmbymdt/folioflex:latest
|
|
196
198
|
container_name: folioflex-web
|
|
197
|
-
command: gunicorn -b 0.0.0.0:8001 app:server
|
|
199
|
+
command: gunicorn -b 0.0.0.0:8001 folioflex.dashboard.app:server
|
|
198
200
|
restart: unless-stopped
|
|
199
201
|
environment:
|
|
200
202
|
FFX_CONFIG_PATH: /code/folioflex/configs
|
|
@@ -13,19 +13,21 @@ Simple investment portfolio tool that will track stock and provide returns and o
|
|
|
13
13
|
|
|
14
14
|
|
|
15
15
|
## Table of Contents
|
|
16
|
-
- [
|
|
17
|
-
- [
|
|
18
|
-
- [
|
|
19
|
-
- [
|
|
20
|
-
- [
|
|
21
|
-
|
|
22
|
-
- [
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
- [
|
|
26
|
-
|
|
27
|
-
- [
|
|
28
|
-
|
|
16
|
+
- [Portfolio](#portfolio)
|
|
17
|
+
- [Table of Contents](#table-of-contents)
|
|
18
|
+
- [Overview](#overview)
|
|
19
|
+
- [Installation](#installation)
|
|
20
|
+
- [Local Install](#local-install)
|
|
21
|
+
- [Docker Install](#docker-install)
|
|
22
|
+
- [Usage](#usage)
|
|
23
|
+
- [CLI](#cli)
|
|
24
|
+
- [Python](#python)
|
|
25
|
+
- [Web Dashboard - Invest](#web-dashboard---invest)
|
|
26
|
+
- [Plaid Dashboard](#plaid-dashboard)
|
|
27
|
+
- [Other Tools](#other-tools)
|
|
28
|
+
- [Jupyter Lab Usage](#jupyter-lab-usage)
|
|
29
|
+
- [Logging](#logging)
|
|
30
|
+
- [Coverage](#coverage)
|
|
29
31
|
|
|
30
32
|
## Overview
|
|
31
33
|
|
|
@@ -38,13 +40,13 @@ Simple investment portfolio tool that will track stock and provide returns and o
|
|
|
38
40
|
**🔧 Features:**
|
|
39
41
|
|
|
40
42
|
- **Market Screener**: Filter and find trending stocks. 🔍
|
|
41
|
-

|
|
44
|
+

|
|
43
45
|
- **Portfolio Management**: Organize and track, your investments. 💼
|
|
44
|
-

|
|
45
47
|
- **Budget Tool**: Create and monitor a budget. 💰
|
|
46
|
-

|
|
49
|
+

|
|
48
50
|
|
|
49
51
|
**📚 Documentation:**
|
|
50
52
|
|
|
@@ -82,11 +84,8 @@ pip install folioflex
|
|
|
82
84
|
Other options can be installed if using more functionality
|
|
83
85
|
|
|
84
86
|
```
|
|
85
|
-
pip install folioflex
|
|
87
|
+
pip install folioflex
|
|
86
88
|
pip install folioflex[dev] # if needing to develop or lint
|
|
87
|
-
pip install folioflex[gpt] # if using the mailer or gpt code
|
|
88
|
-
pip install folioflex[web] # if using the web dashboard
|
|
89
|
-
pip install folioflex[worker] # if using the web dashboard
|
|
90
89
|
``````
|
|
91
90
|
|
|
92
91
|
Or could be done using GitHub.
|
|
@@ -119,7 +118,7 @@ services:
|
|
|
119
118
|
folioflex-web:
|
|
120
119
|
image: dmbymdt/folioflex:latest
|
|
121
120
|
container_name: folioflex-web
|
|
122
|
-
command: gunicorn -b 0.0.0.0:8001 app:server
|
|
121
|
+
command: gunicorn -b 0.0.0.0:8001 folioflex.dashboard.app:server
|
|
123
122
|
restart: unless-stopped
|
|
124
123
|
environment:
|
|
125
124
|
FFX_CONFIG_PATH: /code/folioflex/configs
|
|
@@ -0,0 +1,363 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Scraper module.
|
|
3
|
+
|
|
4
|
+
This module contains functions to scrape the html of a website to be able
|
|
5
|
+
to share to a gpt.
|
|
6
|
+
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
import http.client
|
|
10
|
+
import os
|
|
11
|
+
import re
|
|
12
|
+
|
|
13
|
+
import requests
|
|
14
|
+
from bs4 import BeautifulSoup
|
|
15
|
+
from selenium.webdriver.chrome.options import Options
|
|
16
|
+
from selenium.webdriver.remote.webdriver import WebDriver
|
|
17
|
+
from seleniumbase import SB, Driver
|
|
18
|
+
|
|
19
|
+
from folioflex.utils import config_helper, custom_logger
|
|
20
|
+
|
|
21
|
+
logger = custom_logger.setup_logging(__name__)
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def scrape_html(
|
|
25
|
+
url,
|
|
26
|
+
scraper="selenium",
|
|
27
|
+
**kwargs,
|
|
28
|
+
):
|
|
29
|
+
"""
|
|
30
|
+
Scrape the html of a website.
|
|
31
|
+
|
|
32
|
+
Parameters
|
|
33
|
+
----------
|
|
34
|
+
url : str
|
|
35
|
+
url of the website to scrape
|
|
36
|
+
scraper : str (optional)
|
|
37
|
+
scraper to use, seleniumbase by default
|
|
38
|
+
**kwargs : dict (optional)
|
|
39
|
+
keyword arguments for the options of the driver
|
|
40
|
+
|
|
41
|
+
Returns
|
|
42
|
+
-------
|
|
43
|
+
scrape_results : dict
|
|
44
|
+
dictionary with the url and the text of the website
|
|
45
|
+
|
|
46
|
+
"""
|
|
47
|
+
scrape_results = {"url": url, "text": None}
|
|
48
|
+
logger.info(f"scraping '{url}' with '{scraper}'")
|
|
49
|
+
if scraper == "selenium":
|
|
50
|
+
soup, url = scrape_selenium(url, **kwargs)
|
|
51
|
+
elif scraper == "bee":
|
|
52
|
+
soup = scrape_bee(url, **kwargs)
|
|
53
|
+
|
|
54
|
+
# removing the html tags
|
|
55
|
+
scrape_text = soup.get_text(separator=" ", strip=True)
|
|
56
|
+
scrape_text = scrape_text.replace("\xa0", " ").replace("\\", "")
|
|
57
|
+
|
|
58
|
+
if url.startswith("https://www.wsj.com/livecoverage"):
|
|
59
|
+
# Use regex to find everything between "LIVE UPDATES"
|
|
60
|
+
# and "What to Read Next"
|
|
61
|
+
logger.info("cleaning the text")
|
|
62
|
+
pattern = r"LIVE(.*?)By "
|
|
63
|
+
match = re.search(pattern, scrape_text, re.DOTALL)
|
|
64
|
+
try:
|
|
65
|
+
scrape_text = match.group(1)
|
|
66
|
+
except AttributeError:
|
|
67
|
+
logger.info("no match found")
|
|
68
|
+
|
|
69
|
+
scrape_results = {"url": url, "text": scrape_text}
|
|
70
|
+
|
|
71
|
+
return scrape_results
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def close_windows(sb, url):
|
|
75
|
+
"""
|
|
76
|
+
Close windows except url.
|
|
77
|
+
|
|
78
|
+
Parameters
|
|
79
|
+
----------
|
|
80
|
+
sb : seleniumbase.SB
|
|
81
|
+
seleniumbase instance
|
|
82
|
+
url : str
|
|
83
|
+
url of the website to scrape
|
|
84
|
+
|
|
85
|
+
"""
|
|
86
|
+
open_windows = sb.driver.window_handles
|
|
87
|
+
nbr_windows = len(open_windows)
|
|
88
|
+
logger.info(f"close {nbr_windows-1} open windows")
|
|
89
|
+
for window in open_windows:
|
|
90
|
+
sb.driver.switch_to.window(window)
|
|
91
|
+
nbr_windows = len(sb.driver.window_handles)
|
|
92
|
+
if url not in sb.get_current_url() and nbr_windows > 1:
|
|
93
|
+
sb.driver.close()
|
|
94
|
+
open_windows = sb.driver.window_handles
|
|
95
|
+
sb.driver.switch_to.window(open_windows[0])
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def scrape_selenium(
|
|
99
|
+
url,
|
|
100
|
+
xvfb=None,
|
|
101
|
+
screenshot=False,
|
|
102
|
+
port=None,
|
|
103
|
+
**kwargs,
|
|
104
|
+
):
|
|
105
|
+
"""
|
|
106
|
+
Scrape the html of a website using seleniumbase.
|
|
107
|
+
|
|
108
|
+
seleniumbase has a lot of methods that can be used in kwargs shown here:
|
|
109
|
+
https://github.com/seleniumbase/SeleniumBase/blob/af3d9545473e55b2a25cdbab8be0b1ed5e1f6afa/seleniumbase/plugins/sb_manager.py
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
Parameters
|
|
114
|
+
----------
|
|
115
|
+
url : str
|
|
116
|
+
url of the website to scrape
|
|
117
|
+
xvfb : bool (optional)
|
|
118
|
+
It's recommended if using uc=True to not run the headless2=True option and to
|
|
119
|
+
have xvfb=True if running in linux.
|
|
120
|
+
screenshot : bool (optional)
|
|
121
|
+
take a screenshot of the website
|
|
122
|
+
port : int (optional)
|
|
123
|
+
port to use for debugging
|
|
124
|
+
**kwargs : dict (optional)
|
|
125
|
+
keyword arguments for the options of the driver
|
|
126
|
+
|
|
127
|
+
Returns
|
|
128
|
+
-------
|
|
129
|
+
soup : bs4.BeautifulSoup
|
|
130
|
+
beautiful soup object with the html of the website
|
|
131
|
+
url : str
|
|
132
|
+
url of the website
|
|
133
|
+
|
|
134
|
+
"""
|
|
135
|
+
# kwargs defaults
|
|
136
|
+
binary_location = kwargs.pop("binary_location", None)
|
|
137
|
+
extension_dir = kwargs.pop("extension_dir", None)
|
|
138
|
+
headless2 = kwargs.pop("headless2", False)
|
|
139
|
+
wait_time = kwargs.pop("wait_time", 10)
|
|
140
|
+
proxy = kwargs.pop("proxy", None)
|
|
141
|
+
chromium_arg = None
|
|
142
|
+
|
|
143
|
+
# use xvfb if running a linux os and xvfb is not specified
|
|
144
|
+
if xvfb is None and os.name == "posix" and not headless2:
|
|
145
|
+
logger.info("using xvfb for browser")
|
|
146
|
+
xvfb = True
|
|
147
|
+
if port:
|
|
148
|
+
logger.info(f"using port {port} for browser debugging")
|
|
149
|
+
chromium_arg = f"--remote-debugging-port={port}"
|
|
150
|
+
|
|
151
|
+
if binary_location or config_helper.BROWSER_LOCATION:
|
|
152
|
+
logger.info("using binary location for browser")
|
|
153
|
+
binary_location = binary_location or config_helper.BROWSER_LOCATION
|
|
154
|
+
if extension_dir or config_helper.BROWSER_EXTENSION:
|
|
155
|
+
logger.info("using extension location for browser")
|
|
156
|
+
extension_dir = extension_dir or config_helper.BROWSER_EXTENSION
|
|
157
|
+
if proxy:
|
|
158
|
+
logger.info("using proxy for browser")
|
|
159
|
+
|
|
160
|
+
# specific landing page
|
|
161
|
+
#
|
|
162
|
+
# this breaks frequently. added the following due to breaks
|
|
163
|
+
# good issue describing the detection:
|
|
164
|
+
# https://github.com/seleniumbase/SeleniumBase/issues/2842
|
|
165
|
+
#
|
|
166
|
+
# incognito=True, to avoid detection
|
|
167
|
+
# xvfb=True, uc works better when display is shown and linux usually needs xvfb
|
|
168
|
+
# headless2=False, uc works better when display is shown
|
|
169
|
+
if re.match(r"https://www\.w.j\.com/finance", url, re.IGNORECASE):
|
|
170
|
+
with SB(
|
|
171
|
+
uc=True,
|
|
172
|
+
incognito=True,
|
|
173
|
+
xvfb=xvfb,
|
|
174
|
+
headless2=headless2,
|
|
175
|
+
binary_location=binary_location,
|
|
176
|
+
extension_dir=extension_dir,
|
|
177
|
+
proxy=proxy,
|
|
178
|
+
chromium_arg=chromium_arg,
|
|
179
|
+
**kwargs,
|
|
180
|
+
) as sb:
|
|
181
|
+
sb.sleep(2)
|
|
182
|
+
close_windows(sb, url)
|
|
183
|
+
logger.info("obtaining the landing page")
|
|
184
|
+
sb.driver.uc_open_with_reconnect(url, reconnect_time=wait_time + 5)
|
|
185
|
+
try:
|
|
186
|
+
sb.driver.uc_click(
|
|
187
|
+
"(//p[contains(text(), 'View All')])[1]", reconnect_time=wait_time
|
|
188
|
+
)
|
|
189
|
+
url = sb.get_current_url()
|
|
190
|
+
logger.info(f"scraping {url}")
|
|
191
|
+
soup = sb.get_beautiful_soup()
|
|
192
|
+
if screenshot:
|
|
193
|
+
logger.info("screenshot saved to 'screenshot.png'")
|
|
194
|
+
sb.driver.save_screenshot("screenshot.png")
|
|
195
|
+
except Exception:
|
|
196
|
+
logger.error("probably flagged bot: returning None")
|
|
197
|
+
html_content = "<html><body><p>could not scrape</p></body></html>"
|
|
198
|
+
soup = BeautifulSoup(html_content, "html.parser")
|
|
199
|
+
if screenshot:
|
|
200
|
+
logger.info("screenshot saved to 'screenshot.png'")
|
|
201
|
+
sb.driver.save_screenshot("screenshot.png")
|
|
202
|
+
return soup, url
|
|
203
|
+
|
|
204
|
+
else:
|
|
205
|
+
with SB(
|
|
206
|
+
uc=True,
|
|
207
|
+
incognito=True,
|
|
208
|
+
xvfb=xvfb,
|
|
209
|
+
headless2=headless2,
|
|
210
|
+
binary_location=binary_location,
|
|
211
|
+
extension_dir=extension_dir,
|
|
212
|
+
proxy=proxy,
|
|
213
|
+
chromium_arg=chromium_arg,
|
|
214
|
+
**kwargs,
|
|
215
|
+
) as sb:
|
|
216
|
+
sb.sleep(2)
|
|
217
|
+
close_windows(sb, url)
|
|
218
|
+
logger.info("initializing the driver")
|
|
219
|
+
sb.driver.uc_open_with_reconnect(url, reconnect_time=wait_time)
|
|
220
|
+
logger.info(f"scraping {url}")
|
|
221
|
+
soup = sb.get_beautiful_soup()
|
|
222
|
+
if screenshot:
|
|
223
|
+
logger.info("screenshot saved to 'screenshot.png'")
|
|
224
|
+
sb.driver.save_screenshot("screenshot.png")
|
|
225
|
+
|
|
226
|
+
logger.info("scraped")
|
|
227
|
+
|
|
228
|
+
return soup, url
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
def scrape_bee(url, **kwargs):
|
|
232
|
+
"""
|
|
233
|
+
Scrape the html of a website using scrapingbee.
|
|
234
|
+
|
|
235
|
+
Parameters
|
|
236
|
+
----------
|
|
237
|
+
url : str
|
|
238
|
+
url of the website to scrape
|
|
239
|
+
**kwargs : dict (optional)
|
|
240
|
+
keyword arguments for the options of the driver
|
|
241
|
+
|
|
242
|
+
Returns
|
|
243
|
+
-------
|
|
244
|
+
soup : bs4.BeautifulSoup
|
|
245
|
+
beautiful soup object with the html of the website
|
|
246
|
+
|
|
247
|
+
"""
|
|
248
|
+
# increase the max headers to avoid error
|
|
249
|
+
http.client._MAXHEADERS = 1000
|
|
250
|
+
api_key = config_helper.SCRAPINGBEE_API
|
|
251
|
+
stealth_proxy = kwargs.get("stealth_proxy", "true")
|
|
252
|
+
response = requests.get(
|
|
253
|
+
url="https://app.scrapingbee.com/api/v1/",
|
|
254
|
+
params={
|
|
255
|
+
"api_key": api_key,
|
|
256
|
+
"url": url,
|
|
257
|
+
"stealth_proxy": stealth_proxy,
|
|
258
|
+
},
|
|
259
|
+
)
|
|
260
|
+
soup = BeautifulSoup(response.content, "html.parser")
|
|
261
|
+
|
|
262
|
+
return soup
|
|
263
|
+
|
|
264
|
+
|
|
265
|
+
def scrape_test(url, **kwargs):
|
|
266
|
+
"""
|
|
267
|
+
Test basic functionality of seleniumbase scraper.
|
|
268
|
+
|
|
269
|
+
When debugging a website it is useful if able to able to find the cause of
|
|
270
|
+
the error from the function or from another source. This function creates
|
|
271
|
+
a pause when connecting to the website that will temporarily stop the selenium
|
|
272
|
+
driver.
|
|
273
|
+
|
|
274
|
+
Here are some common test sites:
|
|
275
|
+
- pixelscan.net
|
|
276
|
+
- fingerprint.com/products/bot-detection/
|
|
277
|
+
- nowsecure.nl
|
|
278
|
+
|
|
279
|
+
Parameters
|
|
280
|
+
----------
|
|
281
|
+
url : str
|
|
282
|
+
url of the website to scrape
|
|
283
|
+
**kwargs : dict (optional)
|
|
284
|
+
keyword arguments for the options of the driver
|
|
285
|
+
|
|
286
|
+
Returns
|
|
287
|
+
-------
|
|
288
|
+
soup : bs4.BeautifulSoup
|
|
289
|
+
beautiful soup object with the html of the website
|
|
290
|
+
|
|
291
|
+
"""
|
|
292
|
+
driver = Driver(
|
|
293
|
+
uc=True,
|
|
294
|
+
headless2=False,
|
|
295
|
+
)
|
|
296
|
+
driver.sleep(2)
|
|
297
|
+
close_windows(driver, url)
|
|
298
|
+
logger.info("connecting to website")
|
|
299
|
+
driver.uc_open_with_reconnect(url, reconnect_time="breakpoint")
|
|
300
|
+
logger.info("exit site")
|
|
301
|
+
driver.quit()
|
|
302
|
+
|
|
303
|
+
|
|
304
|
+
def attach_to_session(executor_url, session_id, options=None):
|
|
305
|
+
"""
|
|
306
|
+
Attach to an existing browser session.
|
|
307
|
+
|
|
308
|
+
Parameters
|
|
309
|
+
----------
|
|
310
|
+
executor_url : str
|
|
311
|
+
The URL of the WebDriver server to connect to.
|
|
312
|
+
session_id : str
|
|
313
|
+
The ID of the session to attach to.
|
|
314
|
+
options : Options, optional
|
|
315
|
+
The options to use when attaching to the session.
|
|
316
|
+
|
|
317
|
+
Returns
|
|
318
|
+
-------
|
|
319
|
+
WebDriver
|
|
320
|
+
A WebDriver instance that is attached to the existing session.
|
|
321
|
+
|
|
322
|
+
"""
|
|
323
|
+
# save the original execute method of driver
|
|
324
|
+
original_execute = WebDriver.execute
|
|
325
|
+
|
|
326
|
+
def override_execute_method(self, command, params=None):
|
|
327
|
+
"""
|
|
328
|
+
Override for the newSession command.
|
|
329
|
+
|
|
330
|
+
Parameters
|
|
331
|
+
----------
|
|
332
|
+
self : WebDriver
|
|
333
|
+
The WebDriver instance to execute the command on.
|
|
334
|
+
command : str
|
|
335
|
+
The name of the command to execute.
|
|
336
|
+
params : dict, optional
|
|
337
|
+
The parameters for the command.
|
|
338
|
+
|
|
339
|
+
Returns
|
|
340
|
+
-------
|
|
341
|
+
dict
|
|
342
|
+
The result of the command execution.
|
|
343
|
+
|
|
344
|
+
"""
|
|
345
|
+
# point to the existing session
|
|
346
|
+
if command == "newSession":
|
|
347
|
+
return {"value": {"sessionId": session_id, "capabilities": {}}}
|
|
348
|
+
else:
|
|
349
|
+
return original_execute(self, command, params)
|
|
350
|
+
|
|
351
|
+
# override the execute method with the original session
|
|
352
|
+
WebDriver.execute = override_execute_method
|
|
353
|
+
|
|
354
|
+
# create the driver
|
|
355
|
+
if options is None:
|
|
356
|
+
options = Options()
|
|
357
|
+
driver = WebDriver(command_executor=executor_url, options=options)
|
|
358
|
+
driver.session_id = session_id
|
|
359
|
+
|
|
360
|
+
# restore the method
|
|
361
|
+
WebDriver.execute = original_execute
|
|
362
|
+
|
|
363
|
+
return driver
|