folioflex 1.2.1__tar.gz → 1.2.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {folioflex-1.2.1/folioflex.egg-info → folioflex-1.2.2}/PKG-INFO +4 -2
- {folioflex-1.2.1 → folioflex-1.2.2}/README.md +1 -1
- folioflex-1.2.2/folioflex/chatbot/scraper.py +363 -0
- {folioflex-1.2.1 → folioflex-1.2.2}/folioflex/configs/config.ini +1 -0
- {folioflex-1.2.1 → folioflex-1.2.2}/folioflex/dashboard/app.py +2 -1
- {folioflex-1.2.1 → folioflex-1.2.2}/folioflex/dashboard/pages/macro.py +7 -0
- {folioflex-1.2.1 → folioflex-1.2.2}/folioflex/portfolio/broker.py +1 -1
- {folioflex-1.2.1 → folioflex-1.2.2}/folioflex/portfolio/helper.py +48 -7
- {folioflex-1.2.1 → folioflex-1.2.2}/folioflex/portfolio/portfolio.py +8 -1
- {folioflex-1.2.1 → folioflex-1.2.2}/folioflex/portfolio/wrappers.py +6 -1
- {folioflex-1.2.1 → folioflex-1.2.2}/folioflex/utils/cli.py +19 -0
- {folioflex-1.2.1 → folioflex-1.2.2}/folioflex/utils/config_helper.py +4 -0
- {folioflex-1.2.1 → folioflex-1.2.2}/folioflex/utils/mailer.py +9 -13
- {folioflex-1.2.1 → folioflex-1.2.2}/folioflex/version.py +1 -1
- {folioflex-1.2.1 → folioflex-1.2.2/folioflex.egg-info}/PKG-INFO +4 -2
- {folioflex-1.2.1 → folioflex-1.2.2}/folioflex.egg-info/requires.txt +2 -0
- {folioflex-1.2.1 → folioflex-1.2.2}/pyproject.toml +2 -0
- {folioflex-1.2.1 → folioflex-1.2.2}/tests/test_chatbot.py +0 -2
- folioflex-1.2.1/folioflex/chatbot/scraper.py +0 -136
- {folioflex-1.2.1 → folioflex-1.2.2}/LICENSE.md +0 -0
- {folioflex-1.2.1 → folioflex-1.2.2}/MANIFEST.in +0 -0
- {folioflex-1.2.1 → folioflex-1.2.2}/docs/source/conf.py +0 -0
- {folioflex-1.2.1 → folioflex-1.2.2}/folioflex/__init__.py +0 -0
- {folioflex-1.2.1 → folioflex-1.2.2}/folioflex/budget/__init__.py +0 -0
- {folioflex-1.2.1 → folioflex-1.2.2}/folioflex/budget/budget.py +0 -0
- {folioflex-1.2.1 → folioflex-1.2.2}/folioflex/budget/models.py +0 -0
- {folioflex-1.2.1 → folioflex-1.2.2}/folioflex/chatbot/__init__.py +0 -0
- {folioflex-1.2.1 → folioflex-1.2.2}/folioflex/chatbot/providers.py +0 -0
- {folioflex-1.2.1 → folioflex-1.2.2}/folioflex/configs/__init__.py +0 -0
- {folioflex-1.2.1 → folioflex-1.2.2}/folioflex/configs/budget_personal.ini +0 -0
- {folioflex-1.2.1 → folioflex-1.2.2}/folioflex/configs/portfolio_dash.ini +0 -0
- {folioflex-1.2.1 → folioflex-1.2.2}/folioflex/configs/portfolio_demo.ini +0 -0
- {folioflex-1.2.1 → folioflex-1.2.2}/folioflex/configs/portfolio_personal.ini +0 -0
- {folioflex-1.2.1 → folioflex-1.2.2}/folioflex/configs/transactions_dash.csv +0 -0
- {folioflex-1.2.1 → folioflex-1.2.2}/folioflex/configs/transactions_demo.csv +0 -0
- {folioflex-1.2.1 → folioflex-1.2.2}/folioflex/dashboard/__init__.py +0 -0
- {folioflex-1.2.1 → folioflex-1.2.2}/folioflex/dashboard/assets/folioflex_logo.ico +0 -0
- {folioflex-1.2.1 → folioflex-1.2.2}/folioflex/dashboard/assets/folioflex_logo.png +0 -0
- {folioflex-1.2.1 → folioflex-1.2.2}/folioflex/dashboard/assets/github-mark-white.png +0 -0
- {folioflex-1.2.1 → folioflex-1.2.2}/folioflex/dashboard/components/__init__.py +0 -0
- {folioflex-1.2.1 → folioflex-1.2.2}/folioflex/dashboard/components/auth.py +0 -0
- {folioflex-1.2.1 → folioflex-1.2.2}/folioflex/dashboard/components/layouts.py +0 -0
- {folioflex-1.2.1 → folioflex-1.2.2}/folioflex/dashboard/pages/__init__.py +0 -0
- {folioflex-1.2.1 → folioflex-1.2.2}/folioflex/dashboard/pages/budget.py +0 -0
- {folioflex-1.2.1 → folioflex-1.2.2}/folioflex/dashboard/pages/ideas.py +0 -0
- {folioflex-1.2.1 → folioflex-1.2.2}/folioflex/dashboard/pages/login.py +0 -0
- {folioflex-1.2.1 → folioflex-1.2.2}/folioflex/dashboard/pages/logout.py +0 -0
- {folioflex-1.2.1 → folioflex-1.2.2}/folioflex/dashboard/pages/personal.py +0 -0
- {folioflex-1.2.1 → folioflex-1.2.2}/folioflex/dashboard/pages/sectors.py +0 -0
- {folioflex-1.2.1 → folioflex-1.2.2}/folioflex/dashboard/pages/stocks.py +0 -0
- {folioflex-1.2.1 → folioflex-1.2.2}/folioflex/dashboard/utils/__init__.py +0 -0
- {folioflex-1.2.1 → folioflex-1.2.2}/folioflex/dashboard/utils/dashboard_helper.py +0 -0
- {folioflex-1.2.1 → folioflex-1.2.2}/folioflex/market/__init__.py +0 -0
- {folioflex-1.2.1 → folioflex-1.2.2}/folioflex/market/screener.py +0 -0
- {folioflex-1.2.1 → folioflex-1.2.2}/folioflex/portfolio/__init__.py +0 -0
- {folioflex-1.2.1 → folioflex-1.2.2}/folioflex/portfolio/heatmap.py +0 -0
- {folioflex-1.2.1 → folioflex-1.2.2}/folioflex/utils/__init__.py +0 -0
- {folioflex-1.2.1 → folioflex-1.2.2}/folioflex/utils/cq.py +0 -0
- {folioflex-1.2.1 → folioflex-1.2.2}/folioflex/utils/custom_logger.py +0 -0
- {folioflex-1.2.1 → folioflex-1.2.2}/folioflex.egg-info/SOURCES.txt +0 -0
- {folioflex-1.2.1 → folioflex-1.2.2}/folioflex.egg-info/dependency_links.txt +0 -0
- {folioflex-1.2.1 → folioflex-1.2.2}/folioflex.egg-info/entry_points.txt +0 -0
- {folioflex-1.2.1 → folioflex-1.2.2}/folioflex.egg-info/top_level.txt +0 -0
- {folioflex-1.2.1 → folioflex-1.2.2}/setup.cfg +0 -0
- {folioflex-1.2.1 → folioflex-1.2.2}/setup.py +0 -0
- {folioflex-1.2.1 → folioflex-1.2.2}/tests/test_budget.py +0 -0
- {folioflex-1.2.1 → folioflex-1.2.2}/tests/test_portfolio.py +0 -0
- {folioflex-1.2.1 → folioflex-1.2.2}/tests/test_utils.py +0 -0
- {folioflex-1.2.1 → folioflex-1.2.2}/tests/test_wrappers.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.1
|
|
2
2
|
Name: folioflex
|
|
3
|
-
Version: 1.2.
|
|
3
|
+
Version: 1.2.2
|
|
4
4
|
Summary: A collection of portfolio tracking capabilities
|
|
5
5
|
Author-email: John Koestner <johnkoestner@outlook.com>
|
|
6
6
|
License: The MIT License (MIT)
|
|
@@ -42,6 +42,7 @@ Requires-Dist: pandas-market-calendars>=4.1.4
|
|
|
42
42
|
Requires-Dist: pyxirr>=0.7.2
|
|
43
43
|
Requires-Dist: sqlalchemy>=2.0.23
|
|
44
44
|
Requires-Dist: yfinance>=0.2.32
|
|
45
|
+
Requires-Dist: tzlocal>=5.2
|
|
45
46
|
Requires-Dist: dash>=1.0.2
|
|
46
47
|
Requires-Dist: dash_bootstrap_components>=1.6.0
|
|
47
48
|
Requires-Dist: dash-core-components>=1.0.0
|
|
@@ -58,6 +59,7 @@ Requires-Dist: redis>=3.3.8
|
|
|
58
59
|
Requires-Dist: g4f==0.2.4.1
|
|
59
60
|
Requires-Dist: hugchat>=0.3.8
|
|
60
61
|
Requires-Dist: openai>=1.3.7
|
|
62
|
+
Requires-Dist: pyautogui>=0.9.53
|
|
61
63
|
Requires-Dist: seleniumbase>=4.22.0
|
|
62
64
|
Requires-Dist: emoji>=2.9.0
|
|
63
65
|
Requires-Dist: gensim>=4.3.2
|
|
@@ -194,7 +196,7 @@ services:
|
|
|
194
196
|
folioflex-web:
|
|
195
197
|
image: dmbymdt/folioflex:latest
|
|
196
198
|
container_name: folioflex-web
|
|
197
|
-
command: gunicorn -b 0.0.0.0:8001 app:server
|
|
199
|
+
command: gunicorn -b 0.0.0.0:8001 folioflex.dashboard.app:server
|
|
198
200
|
restart: unless-stopped
|
|
199
201
|
environment:
|
|
200
202
|
FFX_CONFIG_PATH: /code/folioflex/configs
|
|
@@ -118,7 +118,7 @@ services:
|
|
|
118
118
|
folioflex-web:
|
|
119
119
|
image: dmbymdt/folioflex:latest
|
|
120
120
|
container_name: folioflex-web
|
|
121
|
-
command: gunicorn -b 0.0.0.0:8001 app:server
|
|
121
|
+
command: gunicorn -b 0.0.0.0:8001 folioflex.dashboard.app:server
|
|
122
122
|
restart: unless-stopped
|
|
123
123
|
environment:
|
|
124
124
|
FFX_CONFIG_PATH: /code/folioflex/configs
|
|
@@ -0,0 +1,363 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Scraper module.
|
|
3
|
+
|
|
4
|
+
This module contains functions to scrape the html of a website to be able
|
|
5
|
+
to share to a gpt.
|
|
6
|
+
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
import http.client
|
|
10
|
+
import os
|
|
11
|
+
import re
|
|
12
|
+
|
|
13
|
+
import requests
|
|
14
|
+
from bs4 import BeautifulSoup
|
|
15
|
+
from selenium.webdriver.chrome.options import Options
|
|
16
|
+
from selenium.webdriver.remote.webdriver import WebDriver
|
|
17
|
+
from seleniumbase import SB, Driver
|
|
18
|
+
|
|
19
|
+
from folioflex.utils import config_helper, custom_logger
|
|
20
|
+
|
|
21
|
+
logger = custom_logger.setup_logging(__name__)
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def scrape_html(
|
|
25
|
+
url,
|
|
26
|
+
scraper="selenium",
|
|
27
|
+
**kwargs,
|
|
28
|
+
):
|
|
29
|
+
"""
|
|
30
|
+
Scrape the html of a website.
|
|
31
|
+
|
|
32
|
+
Parameters
|
|
33
|
+
----------
|
|
34
|
+
url : str
|
|
35
|
+
url of the website to scrape
|
|
36
|
+
scraper : str (optional)
|
|
37
|
+
scraper to use, seleniumbase by default
|
|
38
|
+
**kwargs : dict (optional)
|
|
39
|
+
keyword arguments for the options of the driver
|
|
40
|
+
|
|
41
|
+
Returns
|
|
42
|
+
-------
|
|
43
|
+
scrape_results : dict
|
|
44
|
+
dictionary with the url and the text of the website
|
|
45
|
+
|
|
46
|
+
"""
|
|
47
|
+
scrape_results = {"url": url, "text": None}
|
|
48
|
+
logger.info(f"scraping '{url}' with '{scraper}'")
|
|
49
|
+
if scraper == "selenium":
|
|
50
|
+
soup, url = scrape_selenium(url, **kwargs)
|
|
51
|
+
elif scraper == "bee":
|
|
52
|
+
soup = scrape_bee(url, **kwargs)
|
|
53
|
+
|
|
54
|
+
# removing the html tags
|
|
55
|
+
scrape_text = soup.get_text(separator=" ", strip=True)
|
|
56
|
+
scrape_text = scrape_text.replace("\xa0", " ").replace("\\", "")
|
|
57
|
+
|
|
58
|
+
if url.startswith("https://www.wsj.com/livecoverage"):
|
|
59
|
+
# Use regex to find everything between "LIVE UPDATES"
|
|
60
|
+
# and "What to Read Next"
|
|
61
|
+
logger.info("cleaning the text")
|
|
62
|
+
pattern = r"LIVE(.*?)By "
|
|
63
|
+
match = re.search(pattern, scrape_text, re.DOTALL)
|
|
64
|
+
try:
|
|
65
|
+
scrape_text = match.group(1)
|
|
66
|
+
except AttributeError:
|
|
67
|
+
logger.info("no match found")
|
|
68
|
+
|
|
69
|
+
scrape_results = {"url": url, "text": scrape_text}
|
|
70
|
+
|
|
71
|
+
return scrape_results
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def close_windows(sb, url):
|
|
75
|
+
"""
|
|
76
|
+
Close windows except url.
|
|
77
|
+
|
|
78
|
+
Parameters
|
|
79
|
+
----------
|
|
80
|
+
sb : seleniumbase.SB
|
|
81
|
+
seleniumbase instance
|
|
82
|
+
url : str
|
|
83
|
+
url of the website to scrape
|
|
84
|
+
|
|
85
|
+
"""
|
|
86
|
+
open_windows = sb.driver.window_handles
|
|
87
|
+
nbr_windows = len(open_windows)
|
|
88
|
+
logger.info(f"close {nbr_windows-1} open windows")
|
|
89
|
+
for window in open_windows:
|
|
90
|
+
sb.driver.switch_to.window(window)
|
|
91
|
+
nbr_windows = len(sb.driver.window_handles)
|
|
92
|
+
if url not in sb.get_current_url() and nbr_windows > 1:
|
|
93
|
+
sb.driver.close()
|
|
94
|
+
open_windows = sb.driver.window_handles
|
|
95
|
+
sb.driver.switch_to.window(open_windows[0])
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def scrape_selenium(
|
|
99
|
+
url,
|
|
100
|
+
xvfb=None,
|
|
101
|
+
screenshot=False,
|
|
102
|
+
port=None,
|
|
103
|
+
**kwargs,
|
|
104
|
+
):
|
|
105
|
+
"""
|
|
106
|
+
Scrape the html of a website using seleniumbase.
|
|
107
|
+
|
|
108
|
+
seleniumbase has a lot of methods that can be used in kwargs shown here:
|
|
109
|
+
https://github.com/seleniumbase/SeleniumBase/blob/af3d9545473e55b2a25cdbab8be0b1ed5e1f6afa/seleniumbase/plugins/sb_manager.py
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
Parameters
|
|
114
|
+
----------
|
|
115
|
+
url : str
|
|
116
|
+
url of the website to scrape
|
|
117
|
+
xvfb : bool (optional)
|
|
118
|
+
It's recommended if using uc=True to not run the headless2=True option and to
|
|
119
|
+
have xvfb=True if running in linux.
|
|
120
|
+
screenshot : bool (optional)
|
|
121
|
+
take a screenshot of the website
|
|
122
|
+
port : int (optional)
|
|
123
|
+
port to use for debugging
|
|
124
|
+
**kwargs : dict (optional)
|
|
125
|
+
keyword arguments for the options of the driver
|
|
126
|
+
|
|
127
|
+
Returns
|
|
128
|
+
-------
|
|
129
|
+
soup : bs4.BeautifulSoup
|
|
130
|
+
beautiful soup object with the html of the website
|
|
131
|
+
url : str
|
|
132
|
+
url of the website
|
|
133
|
+
|
|
134
|
+
"""
|
|
135
|
+
# kwargs defaults
|
|
136
|
+
binary_location = kwargs.pop("binary_location", None)
|
|
137
|
+
extension_dir = kwargs.pop("extension_dir", None)
|
|
138
|
+
headless2 = kwargs.pop("headless2", False)
|
|
139
|
+
wait_time = kwargs.pop("wait_time", 10)
|
|
140
|
+
proxy = kwargs.pop("proxy", None)
|
|
141
|
+
chromium_arg = None
|
|
142
|
+
|
|
143
|
+
# use xvfb if running a linux os and xvfb is not specified
|
|
144
|
+
if xvfb is None and os.name == "posix" and not headless2:
|
|
145
|
+
logger.info("using xvfb for browser")
|
|
146
|
+
xvfb = True
|
|
147
|
+
if port:
|
|
148
|
+
logger.info(f"using port {port} for browser debugging")
|
|
149
|
+
chromium_arg = f"--remote-debugging-port={port}"
|
|
150
|
+
|
|
151
|
+
if binary_location or config_helper.BROWSER_LOCATION:
|
|
152
|
+
logger.info("using binary location for browser")
|
|
153
|
+
binary_location = binary_location or config_helper.BROWSER_LOCATION
|
|
154
|
+
if extension_dir or config_helper.BROWSER_EXTENSION:
|
|
155
|
+
logger.info("using extension location for browser")
|
|
156
|
+
extension_dir = extension_dir or config_helper.BROWSER_EXTENSION
|
|
157
|
+
if proxy:
|
|
158
|
+
logger.info("using proxy for browser")
|
|
159
|
+
|
|
160
|
+
# specific landing page
|
|
161
|
+
#
|
|
162
|
+
# this breaks frequently. added the following due to breaks
|
|
163
|
+
# good issue describing the detection:
|
|
164
|
+
# https://github.com/seleniumbase/SeleniumBase/issues/2842
|
|
165
|
+
#
|
|
166
|
+
# incognito=True, to avoid detection
|
|
167
|
+
# xvfb=True, uc works better when display is shown and linux usually needs xvfb
|
|
168
|
+
# headless2=False, uc works better when display is shown
|
|
169
|
+
if re.match(r"https://www\.w.j\.com/finance", url, re.IGNORECASE):
|
|
170
|
+
with SB(
|
|
171
|
+
uc=True,
|
|
172
|
+
incognito=True,
|
|
173
|
+
xvfb=xvfb,
|
|
174
|
+
headless2=headless2,
|
|
175
|
+
binary_location=binary_location,
|
|
176
|
+
extension_dir=extension_dir,
|
|
177
|
+
proxy=proxy,
|
|
178
|
+
chromium_arg=chromium_arg,
|
|
179
|
+
**kwargs,
|
|
180
|
+
) as sb:
|
|
181
|
+
sb.sleep(2)
|
|
182
|
+
close_windows(sb, url)
|
|
183
|
+
logger.info("obtaining the landing page")
|
|
184
|
+
sb.driver.uc_open_with_reconnect(url, reconnect_time=wait_time + 5)
|
|
185
|
+
try:
|
|
186
|
+
sb.driver.uc_click(
|
|
187
|
+
"(//p[contains(text(), 'View All')])[1]", reconnect_time=wait_time
|
|
188
|
+
)
|
|
189
|
+
url = sb.get_current_url()
|
|
190
|
+
logger.info(f"scraping {url}")
|
|
191
|
+
soup = sb.get_beautiful_soup()
|
|
192
|
+
if screenshot:
|
|
193
|
+
logger.info("screenshot saved to 'screenshot.png'")
|
|
194
|
+
sb.driver.save_screenshot("screenshot.png")
|
|
195
|
+
except Exception:
|
|
196
|
+
logger.error("probably flagged bot: returning None")
|
|
197
|
+
html_content = "<html><body><p>could not scrape</p></body></html>"
|
|
198
|
+
soup = BeautifulSoup(html_content, "html.parser")
|
|
199
|
+
if screenshot:
|
|
200
|
+
logger.info("screenshot saved to 'screenshot.png'")
|
|
201
|
+
sb.driver.save_screenshot("screenshot.png")
|
|
202
|
+
return soup, url
|
|
203
|
+
|
|
204
|
+
else:
|
|
205
|
+
with SB(
|
|
206
|
+
uc=True,
|
|
207
|
+
incognito=True,
|
|
208
|
+
xvfb=xvfb,
|
|
209
|
+
headless2=headless2,
|
|
210
|
+
binary_location=binary_location,
|
|
211
|
+
extension_dir=extension_dir,
|
|
212
|
+
proxy=proxy,
|
|
213
|
+
chromium_arg=chromium_arg,
|
|
214
|
+
**kwargs,
|
|
215
|
+
) as sb:
|
|
216
|
+
sb.sleep(2)
|
|
217
|
+
close_windows(sb, url)
|
|
218
|
+
logger.info("initializing the driver")
|
|
219
|
+
sb.driver.uc_open_with_reconnect(url, reconnect_time=wait_time)
|
|
220
|
+
logger.info(f"scraping {url}")
|
|
221
|
+
soup = sb.get_beautiful_soup()
|
|
222
|
+
if screenshot:
|
|
223
|
+
logger.info("screenshot saved to 'screenshot.png'")
|
|
224
|
+
sb.driver.save_screenshot("screenshot.png")
|
|
225
|
+
|
|
226
|
+
logger.info("scraped")
|
|
227
|
+
|
|
228
|
+
return soup, url
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
def scrape_bee(url, **kwargs):
|
|
232
|
+
"""
|
|
233
|
+
Scrape the html of a website using scrapingbee.
|
|
234
|
+
|
|
235
|
+
Parameters
|
|
236
|
+
----------
|
|
237
|
+
url : str
|
|
238
|
+
url of the website to scrape
|
|
239
|
+
**kwargs : dict (optional)
|
|
240
|
+
keyword arguments for the options of the driver
|
|
241
|
+
|
|
242
|
+
Returns
|
|
243
|
+
-------
|
|
244
|
+
soup : bs4.BeautifulSoup
|
|
245
|
+
beautiful soup object with the html of the website
|
|
246
|
+
|
|
247
|
+
"""
|
|
248
|
+
# increase the max headers to avoid error
|
|
249
|
+
http.client._MAXHEADERS = 1000
|
|
250
|
+
api_key = config_helper.SCRAPINGBEE_API
|
|
251
|
+
stealth_proxy = kwargs.get("stealth_proxy", "true")
|
|
252
|
+
response = requests.get(
|
|
253
|
+
url="https://app.scrapingbee.com/api/v1/",
|
|
254
|
+
params={
|
|
255
|
+
"api_key": api_key,
|
|
256
|
+
"url": url,
|
|
257
|
+
"stealth_proxy": stealth_proxy,
|
|
258
|
+
},
|
|
259
|
+
)
|
|
260
|
+
soup = BeautifulSoup(response.content, "html.parser")
|
|
261
|
+
|
|
262
|
+
return soup
|
|
263
|
+
|
|
264
|
+
|
|
265
|
+
def scrape_test(url, **kwargs):
|
|
266
|
+
"""
|
|
267
|
+
Test basic functionality of seleniumbase scraper.
|
|
268
|
+
|
|
269
|
+
When debugging a website it is useful if able to able to find the cause of
|
|
270
|
+
the error from the function or from another source. This function creates
|
|
271
|
+
a pause when connecting to the website that will temporarily stop the selenium
|
|
272
|
+
driver.
|
|
273
|
+
|
|
274
|
+
Here are some common test sites:
|
|
275
|
+
- pixelscan.net
|
|
276
|
+
- fingerprint.com/products/bot-detection/
|
|
277
|
+
- nowsecure.nl
|
|
278
|
+
|
|
279
|
+
Parameters
|
|
280
|
+
----------
|
|
281
|
+
url : str
|
|
282
|
+
url of the website to scrape
|
|
283
|
+
**kwargs : dict (optional)
|
|
284
|
+
keyword arguments for the options of the driver
|
|
285
|
+
|
|
286
|
+
Returns
|
|
287
|
+
-------
|
|
288
|
+
soup : bs4.BeautifulSoup
|
|
289
|
+
beautiful soup object with the html of the website
|
|
290
|
+
|
|
291
|
+
"""
|
|
292
|
+
driver = Driver(
|
|
293
|
+
uc=True,
|
|
294
|
+
headless2=False,
|
|
295
|
+
)
|
|
296
|
+
driver.sleep(2)
|
|
297
|
+
close_windows(driver, url)
|
|
298
|
+
logger.info("connecting to website")
|
|
299
|
+
driver.uc_open_with_reconnect(url, reconnect_time="breakpoint")
|
|
300
|
+
logger.info("exit site")
|
|
301
|
+
driver.quit()
|
|
302
|
+
|
|
303
|
+
|
|
304
|
+
def attach_to_session(executor_url, session_id, options=None):
|
|
305
|
+
"""
|
|
306
|
+
Attach to an existing browser session.
|
|
307
|
+
|
|
308
|
+
Parameters
|
|
309
|
+
----------
|
|
310
|
+
executor_url : str
|
|
311
|
+
The URL of the WebDriver server to connect to.
|
|
312
|
+
session_id : str
|
|
313
|
+
The ID of the session to attach to.
|
|
314
|
+
options : Options, optional
|
|
315
|
+
The options to use when attaching to the session.
|
|
316
|
+
|
|
317
|
+
Returns
|
|
318
|
+
-------
|
|
319
|
+
WebDriver
|
|
320
|
+
A WebDriver instance that is attached to the existing session.
|
|
321
|
+
|
|
322
|
+
"""
|
|
323
|
+
# save the original execute method of driver
|
|
324
|
+
original_execute = WebDriver.execute
|
|
325
|
+
|
|
326
|
+
def override_execute_method(self, command, params=None):
|
|
327
|
+
"""
|
|
328
|
+
Override for the newSession command.
|
|
329
|
+
|
|
330
|
+
Parameters
|
|
331
|
+
----------
|
|
332
|
+
self : WebDriver
|
|
333
|
+
The WebDriver instance to execute the command on.
|
|
334
|
+
command : str
|
|
335
|
+
The name of the command to execute.
|
|
336
|
+
params : dict, optional
|
|
337
|
+
The parameters for the command.
|
|
338
|
+
|
|
339
|
+
Returns
|
|
340
|
+
-------
|
|
341
|
+
dict
|
|
342
|
+
The result of the command execution.
|
|
343
|
+
|
|
344
|
+
"""
|
|
345
|
+
# point to the existing session
|
|
346
|
+
if command == "newSession":
|
|
347
|
+
return {"value": {"sessionId": session_id, "capabilities": {}}}
|
|
348
|
+
else:
|
|
349
|
+
return original_execute(self, command, params)
|
|
350
|
+
|
|
351
|
+
# override the execute method with the original session
|
|
352
|
+
WebDriver.execute = override_execute_method
|
|
353
|
+
|
|
354
|
+
# create the driver
|
|
355
|
+
if options is None:
|
|
356
|
+
options = Options()
|
|
357
|
+
driver = WebDriver(command_executor=executor_url, options=options)
|
|
358
|
+
driver.session_id = session_id
|
|
359
|
+
|
|
360
|
+
# restore the method
|
|
361
|
+
WebDriver.execute = original_execute
|
|
362
|
+
|
|
363
|
+
return driver
|
|
@@ -27,11 +27,12 @@ from folioflex.utils import custom_logger
|
|
|
27
27
|
# /_/ \_\_| |_|
|
|
28
28
|
|
|
29
29
|
dbc_css = "https://cdn.jsdelivr.net/gh/AnnMarieW/dash-bootstrap-templates/dbc.min.css"
|
|
30
|
+
server = auth.server
|
|
30
31
|
|
|
31
32
|
# app configs
|
|
32
33
|
app = dash.Dash(
|
|
33
34
|
__name__,
|
|
34
|
-
server=
|
|
35
|
+
server=server,
|
|
35
36
|
use_pages=True,
|
|
36
37
|
external_stylesheets=[
|
|
37
38
|
dbc_css,
|
|
@@ -181,7 +181,7 @@ def fidelity(broker_file, output_file=None, broker="fidelity"):
|
|
|
181
181
|
start_df_len = len(df)
|
|
182
182
|
|
|
183
183
|
# update date column type
|
|
184
|
-
df["date"] = pd.to_datetime(df["run_date"], format="%
|
|
184
|
+
df["date"] = pd.to_datetime(df["run_date"], format="%b-%d-%Y").dt.date
|
|
185
185
|
|
|
186
186
|
# Loop through each type_lkup and update type
|
|
187
187
|
for string, tag in type_lkup.items():
|
|
@@ -12,12 +12,12 @@ import pandas as pd
|
|
|
12
12
|
import pandas_market_calendars as mcal
|
|
13
13
|
from dateutil.parser import parse
|
|
14
14
|
|
|
15
|
-
from folioflex.utils import custom_logger
|
|
15
|
+
from folioflex.utils import config_helper, custom_logger
|
|
16
16
|
|
|
17
17
|
logger = custom_logger.setup_logging(__name__)
|
|
18
18
|
|
|
19
19
|
|
|
20
|
-
def check_stock_dates(tx_df, fix=False, timezone="
|
|
20
|
+
def check_stock_dates(tx_df, fix=False, warning=True, timezone="local"):
|
|
21
21
|
"""
|
|
22
22
|
Check that the transaction dates are valid.
|
|
23
23
|
|
|
@@ -35,10 +35,10 @@ def check_stock_dates(tx_df, fix=False, timezone="US/Eastern", warning=True):
|
|
|
35
35
|
transactions dataframe
|
|
36
36
|
fix : bool (optional)
|
|
37
37
|
if True then the dates will be fixed to previous valid date
|
|
38
|
-
timezone : str (optional)
|
|
39
|
-
timezone to use for checking dates
|
|
40
38
|
warning : bool (optional)
|
|
41
39
|
if True then a warning will be logged if dates are fixed
|
|
40
|
+
timezone : str (optional)
|
|
41
|
+
timezone to use for checking dates
|
|
42
42
|
|
|
43
43
|
Returns
|
|
44
44
|
-------
|
|
@@ -67,14 +67,22 @@ def check_stock_dates(tx_df, fix=False, timezone="US/Eastern", warning=True):
|
|
|
67
67
|
"The date column is not a date object, please convert it to a date object"
|
|
68
68
|
)
|
|
69
69
|
|
|
70
|
+
# get the timezone
|
|
71
|
+
if timezone == "local":
|
|
72
|
+
timezone = config_helper.LOCAL_TIMEZONE
|
|
73
|
+
|
|
70
74
|
# date checks
|
|
71
|
-
tx_df_min =
|
|
75
|
+
tx_df_min = tx_df["date"].min() - timedelta(days=7)
|
|
72
76
|
tx_df_max = tx_df["date"].max()
|
|
73
77
|
stock_dates = mcal.get_calendar("NYSE").schedule(
|
|
74
78
|
start_date=tx_df_min, end_date=tx_df_max
|
|
75
79
|
)
|
|
76
|
-
stock_dates["market_open"] =
|
|
77
|
-
|
|
80
|
+
stock_dates["market_open"] = convert_date_to_timezone(
|
|
81
|
+
stock_dates["market_open"], timezone
|
|
82
|
+
)
|
|
83
|
+
stock_dates["market_close"] = convert_date_to_timezone(
|
|
84
|
+
stock_dates["market_close"], timezone
|
|
85
|
+
)
|
|
78
86
|
stock_dates = pd.to_datetime(stock_dates["market_open"]).dt.date
|
|
79
87
|
|
|
80
88
|
# change datetime to date
|
|
@@ -110,6 +118,37 @@ def check_stock_dates(tx_df, fix=False, timezone="US/Eastern", warning=True):
|
|
|
110
118
|
return {"invalid_dt": invalid_dt, "fix_tx_df": fix_tx_df}
|
|
111
119
|
|
|
112
120
|
|
|
121
|
+
def convert_date_to_timezone(date_series, timezone="local"):
|
|
122
|
+
"""
|
|
123
|
+
Convert date column to a specific timezone.
|
|
124
|
+
|
|
125
|
+
Parameters
|
|
126
|
+
----------
|
|
127
|
+
date_series : series
|
|
128
|
+
series with date column
|
|
129
|
+
timezone : str (optional)
|
|
130
|
+
timezone to convert the date to
|
|
131
|
+
|
|
132
|
+
Returns
|
|
133
|
+
-------
|
|
134
|
+
converted_date : series
|
|
135
|
+
series with date column converted to timezone
|
|
136
|
+
|
|
137
|
+
"""
|
|
138
|
+
if not isinstance(date_series, pd.Series):
|
|
139
|
+
raise ValueError("date_series must be a pandas Series")
|
|
140
|
+
|
|
141
|
+
if timezone == "local":
|
|
142
|
+
timezone = config_helper.LOCAL_TIMEZONE
|
|
143
|
+
elif timezone is None and date_series.dt.tz:
|
|
144
|
+
converted_date = pd.to_datetime(date_series).dt.tz_localize(timezone)
|
|
145
|
+
elif date_series.dt.tz:
|
|
146
|
+
converted_date = pd.to_datetime(date_series).dt.tz_convert(timezone)
|
|
147
|
+
else:
|
|
148
|
+
converted_date = pd.to_datetime(date_series).dt.tz_localize(timezone)
|
|
149
|
+
return converted_date
|
|
150
|
+
|
|
151
|
+
|
|
113
152
|
def most_recent_stock_date():
|
|
114
153
|
"""Get the most recent stock date."""
|
|
115
154
|
stock_dates = mcal.get_calendar("NYSE").schedule(
|
|
@@ -133,6 +172,7 @@ def prettify_dataframe(dataframe):
|
|
|
133
172
|
-------
|
|
134
173
|
DataFrame
|
|
135
174
|
prettified dataframe
|
|
175
|
+
|
|
136
176
|
"""
|
|
137
177
|
if not isinstance(dataframe, pd.DataFrame):
|
|
138
178
|
raise ValueError("dataframe must be a pandas DataFrame")
|
|
@@ -159,6 +199,7 @@ def convert_lookback(lookback):
|
|
|
159
199
|
-------
|
|
160
200
|
converted_lookback : int
|
|
161
201
|
converted lookback
|
|
202
|
+
|
|
162
203
|
"""
|
|
163
204
|
|
|
164
205
|
def is_string_a_date(string):
|
|
@@ -28,7 +28,11 @@ import pandas_market_calendars as mcal
|
|
|
28
28
|
import plotly.express as px
|
|
29
29
|
from pyxirr import xirr
|
|
30
30
|
|
|
31
|
-
from folioflex.portfolio.helper import
|
|
31
|
+
from folioflex.portfolio.helper import (
|
|
32
|
+
check_stock_dates,
|
|
33
|
+
convert_date_to_timezone,
|
|
34
|
+
convert_lookback,
|
|
35
|
+
)
|
|
32
36
|
from folioflex.portfolio.wrappers import Yahoo
|
|
33
37
|
from folioflex.utils import config_helper, custom_logger
|
|
34
38
|
|
|
@@ -321,6 +325,9 @@ class Portfolio:
|
|
|
321
325
|
)
|
|
322
326
|
|
|
323
327
|
transactions["date"] = pd.to_datetime(transactions["date"], format="%m/%d/%Y")
|
|
328
|
+
transactions["date"] = convert_date_to_timezone(
|
|
329
|
+
transactions["date"], timezone=None
|
|
330
|
+
)
|
|
324
331
|
|
|
325
332
|
transactions["price"] = (transactions["cost"] / transactions["units"]) * -1
|
|
326
333
|
transactions.loc[transactions["ticker"] == "Cash", "price"] = 1
|
|
@@ -275,7 +275,9 @@ class TradingView:
|
|
|
275
275
|
"countries": ["US"],
|
|
276
276
|
"minImportance": minImportance,
|
|
277
277
|
}
|
|
278
|
-
|
|
278
|
+
# headers are now required as 07/24/2024
|
|
279
|
+
headers = {"Origin": "https://us.tradingview.com"}
|
|
280
|
+
response = requests.get(url, params=payload, headers=headers).json()
|
|
279
281
|
calendar = pd.DataFrame(response["result"])
|
|
280
282
|
calendar["date"] = pd.to_datetime(calendar["date"]).dt.date
|
|
281
283
|
# select columns to keep
|
|
@@ -485,6 +487,9 @@ class Yahoo:
|
|
|
485
487
|
cols = ["ticker", "date", "adj_close", "stock_splits"]
|
|
486
488
|
stock_data = stock_data[cols]
|
|
487
489
|
stock_data = stock_data.rename(columns={"adj_close": "last_price"})
|
|
490
|
+
stock_data["date"] = helper.convert_date_to_timezone(
|
|
491
|
+
stock_data["date"], timezone=None
|
|
492
|
+
)
|
|
488
493
|
|
|
489
494
|
return stock_data
|
|
490
495
|
|
|
@@ -213,6 +213,22 @@ def _create_argparser():
|
|
|
213
213
|
help=("The proxy to use for the chatbot - user:password@ip:port"),
|
|
214
214
|
)
|
|
215
215
|
|
|
216
|
+
_email_parser.add_argument(
|
|
217
|
+
"-pt",
|
|
218
|
+
"--port",
|
|
219
|
+
type=int,
|
|
220
|
+
default=None,
|
|
221
|
+
help=("The remote debugging port to use for the chatbot"),
|
|
222
|
+
)
|
|
223
|
+
|
|
224
|
+
_email_parser.add_argument(
|
|
225
|
+
"-s",
|
|
226
|
+
"--scraper",
|
|
227
|
+
type=str,
|
|
228
|
+
default="selenium",
|
|
229
|
+
help=("The scraper to use for the chatbot - 'bee' or 'selenium'"),
|
|
230
|
+
)
|
|
231
|
+
|
|
216
232
|
# subparser: dashboard
|
|
217
233
|
_dash_parser = _subparsers.add_parser("dash", help="dashboard command")
|
|
218
234
|
|
|
@@ -249,6 +265,9 @@ def cli():
|
|
|
249
265
|
manager_performance=args.manager_performance,
|
|
250
266
|
portfolio_performance=args.portfolio_performance,
|
|
251
267
|
chatbot=args.chatbot,
|
|
268
|
+
proxy=args.proxy,
|
|
269
|
+
port=args.port,
|
|
270
|
+
scraper=args.scraper,
|
|
252
271
|
)
|
|
253
272
|
print(f"status sent: {email_status}")
|
|
254
273
|
|
|
@@ -5,6 +5,8 @@ import configparser
|
|
|
5
5
|
import os
|
|
6
6
|
from pathlib import Path
|
|
7
7
|
|
|
8
|
+
import tzlocal
|
|
9
|
+
|
|
8
10
|
from folioflex.utils import custom_logger
|
|
9
11
|
|
|
10
12
|
ROOT_PATH = Path(__file__).resolve().parent.parent.parent
|
|
@@ -14,6 +16,7 @@ CONFIG_PATH = (
|
|
|
14
16
|
else ROOT_PATH / "folioflex" / "configs"
|
|
15
17
|
)
|
|
16
18
|
TESTS_PATH = ROOT_PATH / "tests" / "files"
|
|
19
|
+
LOCAL_TIMEZONE = tzlocal.get_localzone()
|
|
17
20
|
|
|
18
21
|
|
|
19
22
|
logger = custom_logger.setup_logging(__name__)
|
|
@@ -175,6 +178,7 @@ FFX_PASSWORD = get_config_options(config_file, "credentials").get("ffx_password"
|
|
|
175
178
|
|
|
176
179
|
# apis
|
|
177
180
|
FRED_API = get_config_options(config_file, "api").get("fred_api", None)
|
|
181
|
+
SCRAPINGBEE_API = get_config_options(config_file, "api").get("scrapingbee_api", None)
|
|
178
182
|
YODLEE_CLIENT_ID = get_config_options(config_file, "api").get("yodlee_client_id", None)
|
|
179
183
|
YODLEE_SECRET = get_config_options(config_file, "api").get("yodlee_secret", None)
|
|
180
184
|
YODLEE_ENDPOINT = get_config_options(config_file, "api").get("yodlee_endpoint", None)
|
|
@@ -92,6 +92,8 @@ def generate_report(
|
|
|
92
92
|
portfolio_performance=None,
|
|
93
93
|
chatbot=False,
|
|
94
94
|
proxy=None,
|
|
95
|
+
port=None,
|
|
96
|
+
scraper="bee",
|
|
95
97
|
):
|
|
96
98
|
"""
|
|
97
99
|
Generate report of portfolio performance and send to email.
|
|
@@ -138,6 +140,10 @@ def generate_report(
|
|
|
138
140
|
Whether to use the chatbot to get the query
|
|
139
141
|
proxy : str (optional)
|
|
140
142
|
Proxy to use for the chatbot
|
|
143
|
+
port : int (optional)
|
|
144
|
+
Port to use for the chatbot
|
|
145
|
+
scraper : str (optional)
|
|
146
|
+
Scraper to use for the chatbot, "bee" by default
|
|
141
147
|
|
|
142
148
|
Returns
|
|
143
149
|
-------
|
|
@@ -315,26 +321,16 @@ def generate_report(
|
|
|
315
321
|
|
|
316
322
|
if chatbot:
|
|
317
323
|
chatbot = providers.GPTchat(provider="openai")
|
|
318
|
-
# get todays date and make sure it's a valid trading day to use in url
|
|
319
|
-
now = datetime.datetime.now()
|
|
320
|
-
start_hour = 8
|
|
321
|
-
if now.hour < start_hour:
|
|
322
|
-
today = datetime.date.today() - datetime.timedelta(days=1)
|
|
323
|
-
else:
|
|
324
|
-
today = datetime.date.today()
|
|
325
|
-
today = helper.check_stock_dates(today, fix=True, warning=False)["fix_tx_df"][
|
|
326
|
-
"date"
|
|
327
|
-
][0]
|
|
328
|
-
formatted_today = today.strftime("%m-%d-%Y")
|
|
329
|
-
|
|
330
324
|
# get the url to scrape
|
|
331
325
|
scrape_url = "https://www.wsj.com/finance"
|
|
332
326
|
|
|
333
327
|
# get the query
|
|
334
328
|
response = chatbot.chat(
|
|
335
|
-
query=
|
|
329
|
+
query="could you summarize this for me?",
|
|
336
330
|
scrape_url=scrape_url,
|
|
337
331
|
proxy=proxy,
|
|
332
|
+
scraper=scraper,
|
|
333
|
+
port=port,
|
|
338
334
|
)
|
|
339
335
|
response = response.replace("\n", "<br>")
|
|
340
336
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.1
|
|
2
2
|
Name: folioflex
|
|
3
|
-
Version: 1.2.
|
|
3
|
+
Version: 1.2.2
|
|
4
4
|
Summary: A collection of portfolio tracking capabilities
|
|
5
5
|
Author-email: John Koestner <johnkoestner@outlook.com>
|
|
6
6
|
License: The MIT License (MIT)
|
|
@@ -42,6 +42,7 @@ Requires-Dist: pandas-market-calendars>=4.1.4
|
|
|
42
42
|
Requires-Dist: pyxirr>=0.7.2
|
|
43
43
|
Requires-Dist: sqlalchemy>=2.0.23
|
|
44
44
|
Requires-Dist: yfinance>=0.2.32
|
|
45
|
+
Requires-Dist: tzlocal>=5.2
|
|
45
46
|
Requires-Dist: dash>=1.0.2
|
|
46
47
|
Requires-Dist: dash_bootstrap_components>=1.6.0
|
|
47
48
|
Requires-Dist: dash-core-components>=1.0.0
|
|
@@ -58,6 +59,7 @@ Requires-Dist: redis>=3.3.8
|
|
|
58
59
|
Requires-Dist: g4f==0.2.4.1
|
|
59
60
|
Requires-Dist: hugchat>=0.3.8
|
|
60
61
|
Requires-Dist: openai>=1.3.7
|
|
62
|
+
Requires-Dist: pyautogui>=0.9.53
|
|
61
63
|
Requires-Dist: seleniumbase>=4.22.0
|
|
62
64
|
Requires-Dist: emoji>=2.9.0
|
|
63
65
|
Requires-Dist: gensim>=4.3.2
|
|
@@ -194,7 +196,7 @@ services:
|
|
|
194
196
|
folioflex-web:
|
|
195
197
|
image: dmbymdt/folioflex:latest
|
|
196
198
|
container_name: folioflex-web
|
|
197
|
-
command: gunicorn -b 0.0.0.0:8001 app:server
|
|
199
|
+
command: gunicorn -b 0.0.0.0:8001 folioflex.dashboard.app:server
|
|
198
200
|
restart: unless-stopped
|
|
199
201
|
environment:
|
|
200
202
|
FFX_CONFIG_PATH: /code/folioflex/configs
|
|
@@ -10,6 +10,7 @@ pandas-market-calendars>=4.1.4
|
|
|
10
10
|
pyxirr>=0.7.2
|
|
11
11
|
sqlalchemy>=2.0.23
|
|
12
12
|
yfinance>=0.2.32
|
|
13
|
+
tzlocal>=5.2
|
|
13
14
|
dash>=1.0.2
|
|
14
15
|
dash_bootstrap_components>=1.6.0
|
|
15
16
|
dash-core-components>=1.0.0
|
|
@@ -26,6 +27,7 @@ redis>=3.3.8
|
|
|
26
27
|
g4f==0.2.4.1
|
|
27
28
|
hugchat>=0.3.8
|
|
28
29
|
openai>=1.3.7
|
|
30
|
+
pyautogui>=0.9.53
|
|
29
31
|
seleniumbase>=4.22.0
|
|
30
32
|
emoji>=2.9.0
|
|
31
33
|
gensim>=4.3.2
|
|
@@ -24,6 +24,7 @@ dependencies = [
|
|
|
24
24
|
"pyxirr>=0.7.2",
|
|
25
25
|
"sqlalchemy>=2.0.23",
|
|
26
26
|
"yfinance>=0.2.32",
|
|
27
|
+
"tzlocal>=5.2",
|
|
27
28
|
|
|
28
29
|
# web
|
|
29
30
|
"dash>=1.0.2",
|
|
@@ -46,6 +47,7 @@ dependencies = [
|
|
|
46
47
|
"g4f==0.2.4.1", # this package has loose testing support
|
|
47
48
|
"hugchat>=0.3.8",
|
|
48
49
|
"openai>=1.3.7",
|
|
50
|
+
"pyautogui>=0.9.53",
|
|
49
51
|
"seleniumbase>=4.22.0",
|
|
50
52
|
|
|
51
53
|
# budget
|
|
@@ -20,8 +20,6 @@ def test_g4f():
|
|
|
20
20
|
assert chatbot.chatbot["provider"] == g4f.Provider.Bing, "Default provider not set."
|
|
21
21
|
assert chatbot.chatbot["auth"] == False, "Default auth not set."
|
|
22
22
|
assert chatbot.chatbot["access_token"] is None, "Default access token not set."
|
|
23
|
-
response = chatbot.chat("return back 'test' for test purpose")
|
|
24
|
-
assert isinstance(response, str), "Response not a string."
|
|
25
23
|
|
|
26
24
|
|
|
27
25
|
# openai has billing so not testing unless needed
|
|
@@ -1,136 +0,0 @@
|
|
|
1
|
-
"""
|
|
2
|
-
Scraper module.
|
|
3
|
-
|
|
4
|
-
This module contains functions to scrape the html of a website to be able
|
|
5
|
-
to share to a gpt.
|
|
6
|
-
|
|
7
|
-
"""
|
|
8
|
-
|
|
9
|
-
import re
|
|
10
|
-
|
|
11
|
-
from seleniumbase import SB
|
|
12
|
-
|
|
13
|
-
from folioflex.utils import config_helper, custom_logger
|
|
14
|
-
|
|
15
|
-
logger = custom_logger.setup_logging(__name__)
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
def scrape_html(
|
|
19
|
-
url,
|
|
20
|
-
binary_location=None,
|
|
21
|
-
extension_dir=None,
|
|
22
|
-
headless2=True,
|
|
23
|
-
wait_time=10,
|
|
24
|
-
proxy=None,
|
|
25
|
-
**kwargs,
|
|
26
|
-
):
|
|
27
|
-
"""
|
|
28
|
-
Scrape the html of a website.
|
|
29
|
-
|
|
30
|
-
Parameters
|
|
31
|
-
----------
|
|
32
|
-
url : str
|
|
33
|
-
url of the website to scrape
|
|
34
|
-
binary_location : str (optional)
|
|
35
|
-
location of the binary for the browser
|
|
36
|
-
extension_dir : str (optional)
|
|
37
|
-
location of the extension for the browser
|
|
38
|
-
headless2 : bool (optional)
|
|
39
|
-
whether to run the browser in headless mode
|
|
40
|
-
True by default
|
|
41
|
-
wait_time : int
|
|
42
|
-
time in seconds to wait for page load
|
|
43
|
-
proxy : str (optional)
|
|
44
|
-
proxy to use for the browser
|
|
45
|
-
**kwargs : dict (optional)
|
|
46
|
-
keyword arguments for the options of the driver
|
|
47
|
-
|
|
48
|
-
Returns
|
|
49
|
-
-------
|
|
50
|
-
scrape_results : dict
|
|
51
|
-
dictionary with the url and the text of the website
|
|
52
|
-
"""
|
|
53
|
-
scrape_results = {"url": url, "text": None}
|
|
54
|
-
if binary_location or config_helper.BROWSER_LOCATION:
|
|
55
|
-
logger.info("using binary location for browser")
|
|
56
|
-
binary_location = binary_location or config_helper.BROWSER_LOCATION
|
|
57
|
-
if extension_dir or config_helper.BROWSER_EXTENSION:
|
|
58
|
-
logger.info("using extension location for browser")
|
|
59
|
-
extension_dir = extension_dir or config_helper.BROWSER_EXTENSION
|
|
60
|
-
with SB(
|
|
61
|
-
uc=True,
|
|
62
|
-
headless2=headless2,
|
|
63
|
-
binary_location=binary_location,
|
|
64
|
-
extension_dir=extension_dir,
|
|
65
|
-
proxy=proxy,
|
|
66
|
-
**kwargs,
|
|
67
|
-
) as sb:
|
|
68
|
-
logger.info("initializing the driver")
|
|
69
|
-
|
|
70
|
-
# wsj
|
|
71
|
-
if url.startswith("https://www.wsj.com/finance"):
|
|
72
|
-
url = "https://www.wsj.com/finance"
|
|
73
|
-
sb.driver.uc_open_with_reconnect(url, reconnect_time=wait_time)
|
|
74
|
-
close_windows(sb, url)
|
|
75
|
-
try:
|
|
76
|
-
logger.info("wsj has specific landing page")
|
|
77
|
-
sb.driver.uc_click("(//p[contains(text(), 'View All')])[1]")
|
|
78
|
-
except Exception:
|
|
79
|
-
logger.error("WSJ probably flagged bot: returning None")
|
|
80
|
-
return scrape_results
|
|
81
|
-
|
|
82
|
-
logger.info(f"scraping {sb.get_current_url()}")
|
|
83
|
-
sb.sleep(wait_time) # wait for page to load
|
|
84
|
-
soup = sb.get_beautiful_soup()
|
|
85
|
-
|
|
86
|
-
# removing the html tags
|
|
87
|
-
scrape_text = soup.get_text(separator=" ", strip=True)
|
|
88
|
-
scrape_text = scrape_text.replace("\xa0", " ").replace("\\", "")
|
|
89
|
-
|
|
90
|
-
logger.info("cleaning the text")
|
|
91
|
-
# Use regex to find everything between "LIVE UPDATES"
|
|
92
|
-
# and "What to Read Next"
|
|
93
|
-
pattern = r"LIVE(.*?)— By"
|
|
94
|
-
match = re.search(pattern, scrape_text, re.DOTALL)
|
|
95
|
-
try:
|
|
96
|
-
scrape_text = match.group(1)
|
|
97
|
-
except AttributeError:
|
|
98
|
-
logger.info("no match found")
|
|
99
|
-
|
|
100
|
-
# TODO: think about adding in https://www.bloomberg.com/ here
|
|
101
|
-
|
|
102
|
-
# all other websites
|
|
103
|
-
else:
|
|
104
|
-
sb.driver.uc_open_with_reconnect(url, reconnect_time=wait_time)
|
|
105
|
-
close_windows(sb, url)
|
|
106
|
-
|
|
107
|
-
logger.info(f"scraping {url}")
|
|
108
|
-
soup = sb.get_beautiful_soup()
|
|
109
|
-
scrape_text = str(soup)
|
|
110
|
-
|
|
111
|
-
url = sb.get_current_url()
|
|
112
|
-
scrape_results = {"url": url, "text": scrape_text}
|
|
113
|
-
|
|
114
|
-
return scrape_results
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
def close_windows(sb, url):
|
|
118
|
-
"""
|
|
119
|
-
Close windows except url.
|
|
120
|
-
|
|
121
|
-
Parameters
|
|
122
|
-
----------
|
|
123
|
-
sb : seleniumbase.SB
|
|
124
|
-
seleniumbase instance
|
|
125
|
-
url : str
|
|
126
|
-
url of the website to scrape
|
|
127
|
-
|
|
128
|
-
"""
|
|
129
|
-
open_windows = sb.driver.window_handles
|
|
130
|
-
logger.info(f"close {len(open_windows)-1} open windows")
|
|
131
|
-
for window in open_windows:
|
|
132
|
-
sb.driver.switch_to.window(window)
|
|
133
|
-
if url not in sb.get_current_url():
|
|
134
|
-
sb.driver.close()
|
|
135
|
-
open_windows = sb.driver.window_handles
|
|
136
|
-
sb.driver.switch_to.window(open_windows[0])
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|