reddit-stock-analyzer 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- reddit_stock_analyzer-0.1.0/LICENSE +21 -0
- reddit_stock_analyzer-0.1.0/PKG-INFO +307 -0
- reddit_stock_analyzer-0.1.0/README.md +282 -0
- reddit_stock_analyzer-0.1.0/pyproject.toml +70 -0
- reddit_stock_analyzer-0.1.0/reddit_stock_analyzer/__init__.py +55 -0
- reddit_stock_analyzer-0.1.0/reddit_stock_analyzer/__main__.py +171 -0
- reddit_stock_analyzer-0.1.0/reddit_stock_analyzer/_version.py +13 -0
- reddit_stock_analyzer-0.1.0/reddit_stock_analyzer/aggregator.py +243 -0
- reddit_stock_analyzer-0.1.0/reddit_stock_analyzer/analyzer.py +250 -0
- reddit_stock_analyzer-0.1.0/reddit_stock_analyzer/client.py +425 -0
- reddit_stock_analyzer-0.1.0/reddit_stock_analyzer/config.py +173 -0
- reddit_stock_analyzer-0.1.0/reddit_stock_analyzer/models.py +248 -0
- reddit_stock_analyzer-0.1.0/reddit_stock_analyzer/py.typed +0 -0
- reddit_stock_analyzer-0.1.0/reddit_stock_analyzer/recognizer.py +162 -0
- reddit_stock_analyzer-0.1.0/reddit_stock_analyzer/sentiment.py +241 -0
- reddit_stock_analyzer-0.1.0/reddit_stock_analyzer/trends.py +439 -0
- reddit_stock_analyzer-0.1.0/reddit_stock_analyzer.egg-info/PKG-INFO +307 -0
- reddit_stock_analyzer-0.1.0/reddit_stock_analyzer.egg-info/SOURCES.txt +31 -0
- reddit_stock_analyzer-0.1.0/reddit_stock_analyzer.egg-info/dependency_links.txt +1 -0
- reddit_stock_analyzer-0.1.0/reddit_stock_analyzer.egg-info/entry_points.txt +2 -0
- reddit_stock_analyzer-0.1.0/reddit_stock_analyzer.egg-info/requires.txt +15 -0
- reddit_stock_analyzer-0.1.0/reddit_stock_analyzer.egg-info/top_level.txt +1 -0
- reddit_stock_analyzer-0.1.0/setup.cfg +4 -0
- reddit_stock_analyzer-0.1.0/tests/test_aggregator.py +132 -0
- reddit_stock_analyzer-0.1.0/tests/test_analyzer.py +179 -0
- reddit_stock_analyzer-0.1.0/tests/test_cli.py +86 -0
- reddit_stock_analyzer-0.1.0/tests/test_client.py +262 -0
- reddit_stock_analyzer-0.1.0/tests/test_config.py +88 -0
- reddit_stock_analyzer-0.1.0/tests/test_integration.py +140 -0
- reddit_stock_analyzer-0.1.0/tests/test_recognizer.py +144 -0
- reddit_stock_analyzer-0.1.0/tests/test_sentiment.py +144 -0
- reddit_stock_analyzer-0.1.0/tests/test_trends.py +464 -0
- reddit_stock_analyzer-0.1.0/tests/test_version.py +77 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2024 Stephan Akkerman
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,307 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: reddit-stock-analyzer
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Scrape finance subreddits, recognise the stocks discussed, and rank what is trending.
|
|
5
|
+
Author-email: Stephan Akkerman <stephan@akkerman.ai>
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/StephanAkkerman/reddit-stock-analyzer
|
|
8
|
+
Keywords: reddit,stocks,wallstreetbets,sentiment,trends
|
|
9
|
+
Requires-Python: >=3.10
|
|
10
|
+
Description-Content-Type: text/markdown
|
|
11
|
+
License-File: LICENSE
|
|
12
|
+
Requires-Dist: httpx
|
|
13
|
+
Requires-Dist: stock-recognizer>=0.1.1
|
|
14
|
+
Provides-Extra: praw
|
|
15
|
+
Requires-Dist: asyncpraw; extra == "praw"
|
|
16
|
+
Requires-Dist: python-dotenv; extra == "praw"
|
|
17
|
+
Provides-Extra: sentiment
|
|
18
|
+
Requires-Dist: transformers; extra == "sentiment"
|
|
19
|
+
Requires-Dist: torch; extra == "sentiment"
|
|
20
|
+
Provides-Extra: test
|
|
21
|
+
Requires-Dist: pytest; extra == "test"
|
|
22
|
+
Requires-Dist: pytest-asyncio; extra == "test"
|
|
23
|
+
Requires-Dist: httpx; extra == "test"
|
|
24
|
+
Dynamic: license-file
|
|
25
|
+
|
|
26
|
+
# Reddit Stock Analyzer 📊
|
|
27
|
+
|
|
28
|
+
<!-- Add a banner here like: https://github.com/StephanAkkerman/fintwit-bot/blob/main/img/logo/fintwit-banner.png -->
|
|
29
|
+
|
|
30
|
+
---
|
|
31
|
+
<p align="center">
|
|
32
|
+
<img alt="GitHub Actions Workflow Status" src="https://img.shields.io/github/actions/workflow/status/StephanAkkerman/reddit-stock-analyzer/pyversions.yml?label=python%203.10%20%7C%203.11%20%7C%203.12%20%7C%203.13&logo=python&style=flat-square">
|
|
33
|
+
<img src="https://img.shields.io/github/license/StephanAkkerman/reddit-stock-analyzer.svg?color=brightgreen" alt="License">
|
|
34
|
+
<a href="https://github.com/psf/black"><img src="https://img.shields.io/badge/code%20style-black-000000.svg" alt="Code style: black"></a>
|
|
35
|
+
</p>
|
|
36
|
+
|
|
37
|
+
## Introduction
|
|
38
|
+
|
|
39
|
+
Scrapes recent posts from finance subreddits, works out **which stocks each
|
|
40
|
+
post is about** using
|
|
41
|
+
[`stock-recognizer`](https://github.com/StephanAkkerman/stock-recognizer),
|
|
42
|
+
scores their sentiment with
|
|
43
|
+
[FinTwitBERT](https://huggingface.co/StephanAkkerman/FinTwitBERT-wsb-sentiment),
|
|
44
|
+
and ranks what is **trending** — not just what is loud, but what is getting
|
|
45
|
+
louder.
|
|
46
|
+
|
|
47
|
+
It is built as a library first: every entry point returns plain dataclasses
|
|
48
|
+
with `to_dict()`, so [`fintwit-web`](https://github.com/StephanAkkerman/fintwit-web)
|
|
49
|
+
can serve it straight out of a FastAPI route. See
|
|
50
|
+
[`docs/fintwit-web-integration.md`](docs/fintwit-web-integration.md).
|
|
51
|
+
|
|
52
|
+
## Table of Contents 🗂
|
|
53
|
+
|
|
54
|
+
- [Key Features](#key-features-)
|
|
55
|
+
- [Installation](#installation-)
|
|
56
|
+
- [Usage](#usage-)
|
|
57
|
+
- [Sentiment](#sentiment-)
|
|
58
|
+
- [Trend Metrics](#trend-metrics-)
|
|
59
|
+
- [Configuration](#configuration-)
|
|
60
|
+
- [Citation](#citation-)
|
|
61
|
+
- [Contributing](#contributing-)
|
|
62
|
+
- [License](#license-)
|
|
63
|
+
|
|
64
|
+
## Key Features 🔑
|
|
65
|
+
|
|
66
|
+
- **Curated subreddit catalogue** — `retail`, `stocks`, `research`, `trading`,
|
|
67
|
+
`macro` and `crypto` groups, so you can ask for "everything retail traders
|
|
68
|
+
shout about" instead of hard-coding names.
|
|
69
|
+
- **Credentials optional** — uses authenticated `asyncpraw` when Reddit API
|
|
70
|
+
keys are configured, and the public JSON endpoints when they are not.
|
|
71
|
+
- **Real ticker recognition** — `stock-recognizer` handles cashtags, market-data
|
|
72
|
+
validation, "DD" the company vs. "DD" the due diligence, and company-name
|
|
73
|
+
mapping (`TSMC` → `TSM`).
|
|
74
|
+
- **WSB-native sentiment** — [FinTwitBERT-wsb](https://huggingface.co/StephanAkkerman/FinTwitBERT-wsb-sentiment)
|
|
75
|
+
reads "puts printing" the way a trader does, per post *and per ticker*.
|
|
76
|
+
- **Trend analytics** — two-window momentum, spike z-scores, emerging/fading
|
|
77
|
+
detection, per-subreddit breakdowns and an hourly mention timeline.
|
|
78
|
+
- **Async and shareable** — concurrent scraping, models loaded once, CPU-bound
|
|
79
|
+
work pushed off the event loop.
|
|
80
|
+
|
|
81
|
+
## Installation ⚙️
|
|
82
|
+
|
|
83
|
+
```bash
|
|
84
|
+
pip install git+https://github.com/StephanAkkerman/reddit-stock-analyzer.git
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
Optional extras:
|
|
88
|
+
|
|
89
|
+
```bash
|
|
90
|
+
pip install "reddit-stock-analyzer[praw]" # authenticated scraping
|
|
91
|
+
pip install "reddit-stock-analyzer[sentiment]" # FinTwitBERT post sentiment
|
|
92
|
+
```
|
|
93
|
+
|
|
94
|
+
`[sentiment]` is usually redundant: `transformers` and `torch` already arrive
|
|
95
|
+
with `stock-recognizer`, so sentiment works out of the box. If they are
|
|
96
|
+
missing, every post is labelled `neutral` and everything else still works.
|
|
97
|
+
Credentials are read from the environment or a `.env` file — see
|
|
98
|
+
[`.env.example`](.env.example).
|
|
99
|
+
|
|
100
|
+
## Usage ⌨️
|
|
101
|
+
|
|
102
|
+
### Library
|
|
103
|
+
|
|
104
|
+
```python
|
|
105
|
+
import asyncio
|
|
106
|
+
from reddit_stock_analyzer import RedditTrendService, subreddits_for
|
|
107
|
+
|
|
108
|
+
async def main():
|
|
109
|
+
async with RedditTrendService() as service:
|
|
110
|
+
report = await service.trend_report(
|
|
111
|
+
subreddits_for("retail", "trading"),
|
|
112
|
+
window_hours=24, # the period under analysis
|
|
113
|
+
baseline_hours=24, # the period it is compared against
|
|
114
|
+
)
|
|
115
|
+
|
|
116
|
+
for trend in report.tickers[:5]:
|
|
117
|
+
print(
|
|
118
|
+
f"{trend.symbol:<6} {trend.mentions:>3} mentions "
|
|
119
|
+
f"(was {trend.previous_mentions}) "
|
|
120
|
+
f"momentum {trend.momentum:+.2f} "
|
|
121
|
+
f"{trend.sentiment} ({trend.sentiment_score:+.2f})"
|
|
122
|
+
)
|
|
123
|
+
|
|
124
|
+
print("emerging:", report.emerging)
|
|
125
|
+
print("fading:", report.fading)
|
|
126
|
+
|
|
127
|
+
asyncio.run(main())
|
|
128
|
+
```
|
|
129
|
+
|
|
130
|
+
A single subreddit snapshot, without the trend maths:
|
|
131
|
+
|
|
132
|
+
```python
|
|
133
|
+
overview = await service.subreddit_overview("wallstreetbets", limit=50)
|
|
134
|
+
# {'subreddit': ..., 'overall_mood': 'bullish', 'top_tickers': {'NVDA': 7, ...},
|
|
135
|
+
# 'sentiment_breakdown': {...}, 'posts': [...]}
|
|
136
|
+
```
|
|
137
|
+
|
|
138
|
+
Scraping and analysis are separable — cache a scrape and re-rank it with
|
|
139
|
+
different parameters for free:
|
|
140
|
+
|
|
141
|
+
```python
|
|
142
|
+
from reddit_stock_analyzer import compute_trends
|
|
143
|
+
|
|
144
|
+
posts = await service.scrape(["wallstreetbets"], max_age_hours=48)
|
|
145
|
+
day = compute_trends(posts, window_hours=24)
|
|
146
|
+
morning = compute_trends(posts, window_hours=4, baseline_hours=44)
|
|
147
|
+
```
|
|
148
|
+
|
|
149
|
+
### Command line
|
|
150
|
+
|
|
151
|
+
```bash
|
|
152
|
+
python -m reddit_stock_analyzer --category retail --window 12 --top 10
|
|
153
|
+
python -m reddit_stock_analyzer -s wallstreetbets -s options --json
|
|
154
|
+
python -m reddit_stock_analyzer --no-ai # skip the GLiNER2 model
|
|
155
|
+
```
|
|
156
|
+
|
|
157
|
+
```text
|
|
158
|
+
r/wallstreetbets + r/options
|
|
159
|
+
412 posts in the last 12h (of 780 scraped) — mood: bullish (+0.21)
|
|
160
|
+
|
|
161
|
+
TICKER MENTIONS PREV MOM SPIKE HEAT SENTIMENT
|
|
162
|
+
NVDA 41 18 +1.21 +2.84 0.912 bullish (+0.44)
|
|
163
|
+
SPY 33 35 -0.06 +2.01 0.604 neutral (+0.03)
|
|
164
|
+
...
|
|
165
|
+
|
|
166
|
+
rising: NVDA, SMCI
|
|
167
|
+
emerging: RKLB
|
|
168
|
+
fading: AMC
|
|
169
|
+
```
|
|
170
|
+
|
|
171
|
+
## Sentiment 🐂🐻
|
|
172
|
+
|
|
173
|
+
Posts are classified by
|
|
174
|
+
[`StephanAkkerman/FinTwitBERT-wsb-sentiment`](https://huggingface.co/StephanAkkerman/FinTwitBERT-wsb-sentiment)
|
|
175
|
+
— BERT pre-trained on financial tweets and fine-tuned on WallStreetBets-style
|
|
176
|
+
text, so it reads "puts printing" and "she's gonna rip" as a trader would.
|
|
177
|
+
Labels are normalised to `bullish` / `bearish` / `neutral`, and every score is
|
|
178
|
+
**signed by direction**: `+0.9` is confidently bullish, `-0.9` confidently
|
|
179
|
+
bearish, `0.0` neutral. That makes them averageable, which is what every
|
|
180
|
+
aggregate in a report is built on.
|
|
181
|
+
|
|
182
|
+
Sentiment surfaces at four levels:
|
|
183
|
+
|
|
184
|
+
| Level | Field |
|
|
185
|
+
| --- | --- |
|
|
186
|
+
| Post | `AnalyzedPost.sentiment`, `.sentiment_score`, `.sentiment_confidence` |
|
|
187
|
+
| Ticker within a post | `AnalyzedPost.sentiment_for("INTC")` |
|
|
188
|
+
| Ticker across the window | `TickerTrend.sentiment`, `.sentiment_score`, `.sentiment_breakdown` |
|
|
189
|
+
| Subreddit / overall | `SubredditSummary.mood`, `TrendReport.mood` |
|
|
190
|
+
|
|
191
|
+
### Per-ticker attribution
|
|
192
|
+
|
|
193
|
+
A post reading *"Long NVDA, short INTC"* is bullish and bearish at once. One
|
|
194
|
+
post-level label would give one of the two the wrong sign, so posts that
|
|
195
|
+
mention **more than one** ticker are split into segments, and each ticker is
|
|
196
|
+
scored from the segments that name **only it** — a sentence mentioning both
|
|
197
|
+
says nothing about either in particular, and is used only when a ticker has
|
|
198
|
+
no sentence of its own:
|
|
199
|
+
|
|
200
|
+
```python
|
|
201
|
+
analyzed.sentiment # 'bullish' — the post as a whole
|
|
202
|
+
analyzed.sentiment_for("NVDA") # +0.9
|
|
203
|
+
analyzed.sentiment_for("INTC") # -0.8
|
|
204
|
+
```
|
|
205
|
+
|
|
206
|
+
Single-ticker posts — the majority — cost nothing extra: the post label is
|
|
207
|
+
already about that ticker. A ticker the recognizer resolved from a company
|
|
208
|
+
name ("TSMC" → `TSM`) has no literal segment to match, and falls back to the
|
|
209
|
+
post's score. Switch the whole thing off with
|
|
210
|
+
`PostAnalyzer(per_ticker_sentiment=False)`.
|
|
211
|
+
|
|
212
|
+
`TickerTrend.sentiment_score` is the mean of these attributed scores, so a
|
|
213
|
+
ticker that everyone is shorting reads bearish even in a bullish subreddit.
|
|
214
|
+
|
|
215
|
+
### Swapping the model
|
|
216
|
+
|
|
217
|
+
```python
|
|
218
|
+
from reddit_stock_analyzer import PostAnalyzer, SentimentAnalyzer
|
|
219
|
+
|
|
220
|
+
analyzer = PostAnalyzer(sentiment=SentimentAnalyzer("your-org/your-model"))
|
|
221
|
+
```
|
|
222
|
+
|
|
223
|
+
Labels named `BULLISH`/`BEARISH`/`NEUTRAL` (or `POSITIVE`/`NEGATIVE`) are
|
|
224
|
+
understood as-is. A checkpoint published without an `id2label` mapping reports
|
|
225
|
+
`LABEL_0`, `LABEL_1`... — that order is **not** guessed, since guessing wrong
|
|
226
|
+
silently inverts every score. Those read neutral and log a warning until you
|
|
227
|
+
say which is which:
|
|
228
|
+
|
|
229
|
+
```python
|
|
230
|
+
SentimentAnalyzer(
|
|
231
|
+
"your-org/your-model",
|
|
232
|
+
label_map={"LABEL_0": "bearish", "LABEL_1": "neutral", "LABEL_2": "bullish"},
|
|
233
|
+
)
|
|
234
|
+
```
|
|
235
|
+
|
|
236
|
+
## Trend Metrics 📈
|
|
237
|
+
|
|
238
|
+
Posts are split into a **window** (the recent period, default 24h) and an
|
|
239
|
+
equally long **baseline** immediately before it. Raw counts say how loud a
|
|
240
|
+
ticker is; the comparison says whether it is getting louder.
|
|
241
|
+
|
|
242
|
+
| Metric | Meaning |
|
|
243
|
+
| --- | --- |
|
|
244
|
+
| `mentions` | Posts in the window mentioning the ticker — counted **once per post**, so one ranting post cannot manufacture a trend |
|
|
245
|
+
| `previous_mentions` | The same count over the baseline window |
|
|
246
|
+
| `momentum` | `(mentions - previous) / (previous + 1)` — smoothed so 0 → 5 is finite (4.0) and ranks above 50 → 60 (0.2) |
|
|
247
|
+
| `change_ratio` | `mentions / previous`, or `null` when there is no baseline |
|
|
248
|
+
| `spike_score` | Standard deviations above this scrape's mean mention count — "unusual *for today*" |
|
|
249
|
+
| `heat_score` | 0–1 blend of mentions (40%), engagement (25%), unique authors (15%) and momentum (20%); the default ranking |
|
|
250
|
+
| `is_emerging` | No baseline mentions at all, and at least `emerging_min_mentions` now |
|
|
251
|
+
|
|
252
|
+
Alongside the ranked tickers a report carries `rising` / `emerging` / `fading`
|
|
253
|
+
shortlists, a `by_subreddit` breakdown (is this hype confined to r/wallstreetbets,
|
|
254
|
+
or has it reached r/investing?) and an hourly `timeline` per top ticker.
|
|
255
|
+
|
|
256
|
+
Every weight and threshold is a keyword argument:
|
|
257
|
+
|
|
258
|
+
```python
|
|
259
|
+
report = compute_trends(
|
|
260
|
+
posts,
|
|
261
|
+
window_hours=6,
|
|
262
|
+
min_mentions=3, # ignore one-off noise
|
|
263
|
+
emerging_min_mentions=5,
|
|
264
|
+
weights={"momentum": 0.5, "mentions": 0.2}, # rank by acceleration
|
|
265
|
+
)
|
|
266
|
+
```
|
|
267
|
+
|
|
268
|
+
## Configuration 🔧
|
|
269
|
+
|
|
270
|
+
| Subreddit category | Contents |
|
|
271
|
+
| --- | --- |
|
|
272
|
+
| `retail` | wallstreetbets, smallstreetbets, WallStreetbetsELITE, Shortsqueeze, pennystocks |
|
|
273
|
+
| `stocks` | stocks, StockMarket, investing, dividends |
|
|
274
|
+
| `research` | SecurityAnalysis, ValueInvesting, stocks |
|
|
275
|
+
| `trading` | Daytrading, swingtrading, options, thetagang, algotrading |
|
|
276
|
+
| `macro` | economics, finance, economy |
|
|
277
|
+
| `crypto` | CryptoCurrency, CryptoMarkets, satoshistreetbets, ethtrader, Bitcoin |
|
|
278
|
+
|
|
279
|
+
```python
|
|
280
|
+
from reddit_stock_analyzer import DEFAULT_SUBREDDITS, all_subreddits, subreddits_for
|
|
281
|
+
|
|
282
|
+
subreddits_for("retail", "trading") # merged, de-duplicated, declaration order
|
|
283
|
+
```
|
|
284
|
+
|
|
285
|
+
## Citation ✍️
|
|
286
|
+
|
|
287
|
+
If you use this project in your research, please cite as follows:
|
|
288
|
+
|
|
289
|
+
```bibtex
|
|
290
|
+
@misc{reddit_stock_analyzer,
|
|
291
|
+
author = {Stephan Akkerman},
|
|
292
|
+
title = {Reddit Stock Analyzer},
|
|
293
|
+
year = {2026},
|
|
294
|
+
publisher = {GitHub},
|
|
295
|
+
journal = {GitHub repository},
|
|
296
|
+
howpublished = {\url{https://github.com/StephanAkkerman/reddit-stock-analyzer}}
|
|
297
|
+
}
|
|
298
|
+
```
|
|
299
|
+
|
|
300
|
+
## Contributing 🛠
|
|
301
|
+
|
|
302
|
+
Contributions are welcome! If you have a feature request, bug report, or proposal for code refactoring, please feel free to open an issue on GitHub. We appreciate your help in improving this project.\
|
|
303
|
+

|
|
304
|
+
|
|
305
|
+
## License 📜
|
|
306
|
+
|
|
307
|
+
This project is licensed under the MIT License. See the [LICENSE](LICENSE) file for details.
|
|
@@ -0,0 +1,282 @@
|
|
|
1
|
+
# Reddit Stock Analyzer 📊
|
|
2
|
+
|
|
3
|
+
<!-- Add a banner here like: https://github.com/StephanAkkerman/fintwit-bot/blob/main/img/logo/fintwit-banner.png -->
|
|
4
|
+
|
|
5
|
+
---
|
|
6
|
+
<p align="center">
|
|
7
|
+
<img alt="GitHub Actions Workflow Status" src="https://img.shields.io/github/actions/workflow/status/StephanAkkerman/reddit-stock-analyzer/pyversions.yml?label=python%203.10%20%7C%203.11%20%7C%203.12%20%7C%203.13&logo=python&style=flat-square">
|
|
8
|
+
<img src="https://img.shields.io/github/license/StephanAkkerman/reddit-stock-analyzer.svg?color=brightgreen" alt="License">
|
|
9
|
+
<a href="https://github.com/psf/black"><img src="https://img.shields.io/badge/code%20style-black-000000.svg" alt="Code style: black"></a>
|
|
10
|
+
</p>
|
|
11
|
+
|
|
12
|
+
## Introduction
|
|
13
|
+
|
|
14
|
+
Scrapes recent posts from finance subreddits, works out **which stocks each
|
|
15
|
+
post is about** using
|
|
16
|
+
[`stock-recognizer`](https://github.com/StephanAkkerman/stock-recognizer),
|
|
17
|
+
scores their sentiment with
|
|
18
|
+
[FinTwitBERT](https://huggingface.co/StephanAkkerman/FinTwitBERT-wsb-sentiment),
|
|
19
|
+
and ranks what is **trending** — not just what is loud, but what is getting
|
|
20
|
+
louder.
|
|
21
|
+
|
|
22
|
+
It is built as a library first: every entry point returns plain dataclasses
|
|
23
|
+
with `to_dict()`, so [`fintwit-web`](https://github.com/StephanAkkerman/fintwit-web)
|
|
24
|
+
can serve it straight out of a FastAPI route. See
|
|
25
|
+
[`docs/fintwit-web-integration.md`](docs/fintwit-web-integration.md).
|
|
26
|
+
|
|
27
|
+
## Table of Contents 🗂
|
|
28
|
+
|
|
29
|
+
- [Key Features](#key-features-)
|
|
30
|
+
- [Installation](#installation-)
|
|
31
|
+
- [Usage](#usage-)
|
|
32
|
+
- [Sentiment](#sentiment-)
|
|
33
|
+
- [Trend Metrics](#trend-metrics-)
|
|
34
|
+
- [Configuration](#configuration-)
|
|
35
|
+
- [Citation](#citation-)
|
|
36
|
+
- [Contributing](#contributing-)
|
|
37
|
+
- [License](#license-)
|
|
38
|
+
|
|
39
|
+
## Key Features 🔑
|
|
40
|
+
|
|
41
|
+
- **Curated subreddit catalogue** — `retail`, `stocks`, `research`, `trading`,
|
|
42
|
+
`macro` and `crypto` groups, so you can ask for "everything retail traders
|
|
43
|
+
shout about" instead of hard-coding names.
|
|
44
|
+
- **Credentials optional** — uses authenticated `asyncpraw` when Reddit API
|
|
45
|
+
keys are configured, and the public JSON endpoints when they are not.
|
|
46
|
+
- **Real ticker recognition** — `stock-recognizer` handles cashtags, market-data
|
|
47
|
+
validation, "DD" the company vs. "DD" the due diligence, and company-name
|
|
48
|
+
mapping (`TSMC` → `TSM`).
|
|
49
|
+
- **WSB-native sentiment** — [FinTwitBERT-wsb](https://huggingface.co/StephanAkkerman/FinTwitBERT-wsb-sentiment)
|
|
50
|
+
reads "puts printing" the way a trader does, per post *and per ticker*.
|
|
51
|
+
- **Trend analytics** — two-window momentum, spike z-scores, emerging/fading
|
|
52
|
+
detection, per-subreddit breakdowns and an hourly mention timeline.
|
|
53
|
+
- **Async and shareable** — concurrent scraping, models loaded once, CPU-bound
|
|
54
|
+
work pushed off the event loop.
|
|
55
|
+
|
|
56
|
+
## Installation ⚙️
|
|
57
|
+
|
|
58
|
+
```bash
|
|
59
|
+
pip install git+https://github.com/StephanAkkerman/reddit-stock-analyzer.git
|
|
60
|
+
```
|
|
61
|
+
|
|
62
|
+
Optional extras:
|
|
63
|
+
|
|
64
|
+
```bash
|
|
65
|
+
pip install "reddit-stock-analyzer[praw]" # authenticated scraping
|
|
66
|
+
pip install "reddit-stock-analyzer[sentiment]" # FinTwitBERT post sentiment
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
`[sentiment]` is usually redundant: `transformers` and `torch` already arrive
|
|
70
|
+
with `stock-recognizer`, so sentiment works out of the box. If they are
|
|
71
|
+
missing, every post is labelled `neutral` and everything else still works.
|
|
72
|
+
Credentials are read from the environment or a `.env` file — see
|
|
73
|
+
[`.env.example`](.env.example).
|
|
74
|
+
|
|
75
|
+
## Usage ⌨️
|
|
76
|
+
|
|
77
|
+
### Library
|
|
78
|
+
|
|
79
|
+
```python
|
|
80
|
+
import asyncio
|
|
81
|
+
from reddit_stock_analyzer import RedditTrendService, subreddits_for
|
|
82
|
+
|
|
83
|
+
async def main():
|
|
84
|
+
async with RedditTrendService() as service:
|
|
85
|
+
report = await service.trend_report(
|
|
86
|
+
subreddits_for("retail", "trading"),
|
|
87
|
+
window_hours=24, # the period under analysis
|
|
88
|
+
baseline_hours=24, # the period it is compared against
|
|
89
|
+
)
|
|
90
|
+
|
|
91
|
+
for trend in report.tickers[:5]:
|
|
92
|
+
print(
|
|
93
|
+
f"{trend.symbol:<6} {trend.mentions:>3} mentions "
|
|
94
|
+
f"(was {trend.previous_mentions}) "
|
|
95
|
+
f"momentum {trend.momentum:+.2f} "
|
|
96
|
+
f"{trend.sentiment} ({trend.sentiment_score:+.2f})"
|
|
97
|
+
)
|
|
98
|
+
|
|
99
|
+
print("emerging:", report.emerging)
|
|
100
|
+
print("fading:", report.fading)
|
|
101
|
+
|
|
102
|
+
asyncio.run(main())
|
|
103
|
+
```
|
|
104
|
+
|
|
105
|
+
A single subreddit snapshot, without the trend maths:
|
|
106
|
+
|
|
107
|
+
```python
|
|
108
|
+
overview = await service.subreddit_overview("wallstreetbets", limit=50)
|
|
109
|
+
# {'subreddit': ..., 'overall_mood': 'bullish', 'top_tickers': {'NVDA': 7, ...},
|
|
110
|
+
# 'sentiment_breakdown': {...}, 'posts': [...]}
|
|
111
|
+
```
|
|
112
|
+
|
|
113
|
+
Scraping and analysis are separable — cache a scrape and re-rank it with
|
|
114
|
+
different parameters for free:
|
|
115
|
+
|
|
116
|
+
```python
|
|
117
|
+
from reddit_stock_analyzer import compute_trends
|
|
118
|
+
|
|
119
|
+
posts = await service.scrape(["wallstreetbets"], max_age_hours=48)
|
|
120
|
+
day = compute_trends(posts, window_hours=24)
|
|
121
|
+
morning = compute_trends(posts, window_hours=4, baseline_hours=44)
|
|
122
|
+
```
|
|
123
|
+
|
|
124
|
+
### Command line
|
|
125
|
+
|
|
126
|
+
```bash
|
|
127
|
+
python -m reddit_stock_analyzer --category retail --window 12 --top 10
|
|
128
|
+
python -m reddit_stock_analyzer -s wallstreetbets -s options --json
|
|
129
|
+
python -m reddit_stock_analyzer --no-ai # skip the GLiNER2 model
|
|
130
|
+
```
|
|
131
|
+
|
|
132
|
+
```text
|
|
133
|
+
r/wallstreetbets + r/options
|
|
134
|
+
412 posts in the last 12h (of 780 scraped) — mood: bullish (+0.21)
|
|
135
|
+
|
|
136
|
+
TICKER MENTIONS PREV MOM SPIKE HEAT SENTIMENT
|
|
137
|
+
NVDA 41 18 +1.21 +2.84 0.912 bullish (+0.44)
|
|
138
|
+
SPY 33 35 -0.06 +2.01 0.604 neutral (+0.03)
|
|
139
|
+
...
|
|
140
|
+
|
|
141
|
+
rising: NVDA, SMCI
|
|
142
|
+
emerging: RKLB
|
|
143
|
+
fading: AMC
|
|
144
|
+
```
|
|
145
|
+
|
|
146
|
+
## Sentiment 🐂🐻
|
|
147
|
+
|
|
148
|
+
Posts are classified by
|
|
149
|
+
[`StephanAkkerman/FinTwitBERT-wsb-sentiment`](https://huggingface.co/StephanAkkerman/FinTwitBERT-wsb-sentiment)
|
|
150
|
+
— BERT pre-trained on financial tweets and fine-tuned on WallStreetBets-style
|
|
151
|
+
text, so it reads "puts printing" and "she's gonna rip" as a trader would.
|
|
152
|
+
Labels are normalised to `bullish` / `bearish` / `neutral`, and every score is
|
|
153
|
+
**signed by direction**: `+0.9` is confidently bullish, `-0.9` confidently
|
|
154
|
+
bearish, `0.0` neutral. That makes them averageable, which is what every
|
|
155
|
+
aggregate in a report is built on.
|
|
156
|
+
|
|
157
|
+
Sentiment surfaces at four levels:
|
|
158
|
+
|
|
159
|
+
| Level | Field |
|
|
160
|
+
| --- | --- |
|
|
161
|
+
| Post | `AnalyzedPost.sentiment`, `.sentiment_score`, `.sentiment_confidence` |
|
|
162
|
+
| Ticker within a post | `AnalyzedPost.sentiment_for("INTC")` |
|
|
163
|
+
| Ticker across the window | `TickerTrend.sentiment`, `.sentiment_score`, `.sentiment_breakdown` |
|
|
164
|
+
| Subreddit / overall | `SubredditSummary.mood`, `TrendReport.mood` |
|
|
165
|
+
|
|
166
|
+
### Per-ticker attribution
|
|
167
|
+
|
|
168
|
+
A post reading *"Long NVDA, short INTC"* is bullish and bearish at once. One
|
|
169
|
+
post-level label would give one of the two the wrong sign, so posts that
|
|
170
|
+
mention **more than one** ticker are split into segments, and each ticker is
|
|
171
|
+
scored from the segments that name **only it** — a sentence mentioning both
|
|
172
|
+
says nothing about either in particular, and is used only when a ticker has
|
|
173
|
+
no sentence of its own:
|
|
174
|
+
|
|
175
|
+
```python
|
|
176
|
+
analyzed.sentiment # 'bullish' — the post as a whole
|
|
177
|
+
analyzed.sentiment_for("NVDA") # +0.9
|
|
178
|
+
analyzed.sentiment_for("INTC") # -0.8
|
|
179
|
+
```
|
|
180
|
+
|
|
181
|
+
Single-ticker posts — the majority — cost nothing extra: the post label is
|
|
182
|
+
already about that ticker. A ticker the recognizer resolved from a company
|
|
183
|
+
name ("TSMC" → `TSM`) has no literal segment to match, and falls back to the
|
|
184
|
+
post's score. Switch the whole thing off with
|
|
185
|
+
`PostAnalyzer(per_ticker_sentiment=False)`.
|
|
186
|
+
|
|
187
|
+
`TickerTrend.sentiment_score` is the mean of these attributed scores, so a
|
|
188
|
+
ticker that everyone is shorting reads bearish even in a bullish subreddit.
|
|
189
|
+
|
|
190
|
+
### Swapping the model
|
|
191
|
+
|
|
192
|
+
```python
|
|
193
|
+
from reddit_stock_analyzer import PostAnalyzer, SentimentAnalyzer
|
|
194
|
+
|
|
195
|
+
analyzer = PostAnalyzer(sentiment=SentimentAnalyzer("your-org/your-model"))
|
|
196
|
+
```
|
|
197
|
+
|
|
198
|
+
Labels named `BULLISH`/`BEARISH`/`NEUTRAL` (or `POSITIVE`/`NEGATIVE`) are
|
|
199
|
+
understood as-is. A checkpoint published without an `id2label` mapping reports
|
|
200
|
+
`LABEL_0`, `LABEL_1`... — that order is **not** guessed, since guessing wrong
|
|
201
|
+
silently inverts every score. Those read neutral and log a warning until you
|
|
202
|
+
say which is which:
|
|
203
|
+
|
|
204
|
+
```python
|
|
205
|
+
SentimentAnalyzer(
|
|
206
|
+
"your-org/your-model",
|
|
207
|
+
label_map={"LABEL_0": "bearish", "LABEL_1": "neutral", "LABEL_2": "bullish"},
|
|
208
|
+
)
|
|
209
|
+
```
|
|
210
|
+
|
|
211
|
+
## Trend Metrics 📈
|
|
212
|
+
|
|
213
|
+
Posts are split into a **window** (the recent period, default 24h) and an
|
|
214
|
+
equally long **baseline** immediately before it. Raw counts say how loud a
|
|
215
|
+
ticker is; the comparison says whether it is getting louder.
|
|
216
|
+
|
|
217
|
+
| Metric | Meaning |
|
|
218
|
+
| --- | --- |
|
|
219
|
+
| `mentions` | Posts in the window mentioning the ticker — counted **once per post**, so one ranting post cannot manufacture a trend |
|
|
220
|
+
| `previous_mentions` | The same count over the baseline window |
|
|
221
|
+
| `momentum` | `(mentions - previous) / (previous + 1)` — smoothed so 0 → 5 is finite (4.0) and ranks above 50 → 60 (0.2) |
|
|
222
|
+
| `change_ratio` | `mentions / previous`, or `null` when there is no baseline |
|
|
223
|
+
| `spike_score` | Standard deviations above this scrape's mean mention count — "unusual *for today*" |
|
|
224
|
+
| `heat_score` | 0–1 blend of mentions (40%), engagement (25%), unique authors (15%) and momentum (20%); the default ranking |
|
|
225
|
+
| `is_emerging` | No baseline mentions at all, and at least `emerging_min_mentions` now |
|
|
226
|
+
|
|
227
|
+
Alongside the ranked tickers a report carries `rising` / `emerging` / `fading`
|
|
228
|
+
shortlists, a `by_subreddit` breakdown (is this hype confined to r/wallstreetbets,
|
|
229
|
+
or has it reached r/investing?) and an hourly `timeline` per top ticker.
|
|
230
|
+
|
|
231
|
+
Every weight and threshold is a keyword argument:
|
|
232
|
+
|
|
233
|
+
```python
|
|
234
|
+
report = compute_trends(
|
|
235
|
+
posts,
|
|
236
|
+
window_hours=6,
|
|
237
|
+
min_mentions=3, # ignore one-off noise
|
|
238
|
+
emerging_min_mentions=5,
|
|
239
|
+
weights={"momentum": 0.5, "mentions": 0.2}, # rank by acceleration
|
|
240
|
+
)
|
|
241
|
+
```
|
|
242
|
+
|
|
243
|
+
## Configuration 🔧
|
|
244
|
+
|
|
245
|
+
| Subreddit category | Contents |
|
|
246
|
+
| --- | --- |
|
|
247
|
+
| `retail` | wallstreetbets, smallstreetbets, WallStreetbetsELITE, Shortsqueeze, pennystocks |
|
|
248
|
+
| `stocks` | stocks, StockMarket, investing, dividends |
|
|
249
|
+
| `research` | SecurityAnalysis, ValueInvesting, stocks |
|
|
250
|
+
| `trading` | Daytrading, swingtrading, options, thetagang, algotrading |
|
|
251
|
+
| `macro` | economics, finance, economy |
|
|
252
|
+
| `crypto` | CryptoCurrency, CryptoMarkets, satoshistreetbets, ethtrader, Bitcoin |
|
|
253
|
+
|
|
254
|
+
```python
|
|
255
|
+
from reddit_stock_analyzer import DEFAULT_SUBREDDITS, all_subreddits, subreddits_for
|
|
256
|
+
|
|
257
|
+
subreddits_for("retail", "trading") # merged, de-duplicated, declaration order
|
|
258
|
+
```
|
|
259
|
+
|
|
260
|
+
## Citation ✍️
|
|
261
|
+
|
|
262
|
+
If you use this project in your research, please cite as follows:
|
|
263
|
+
|
|
264
|
+
```bibtex
|
|
265
|
+
@misc{reddit_stock_analyzer,
|
|
266
|
+
author = {Stephan Akkerman},
|
|
267
|
+
title = {Reddit Stock Analyzer},
|
|
268
|
+
year = {2026},
|
|
269
|
+
publisher = {GitHub},
|
|
270
|
+
journal = {GitHub repository},
|
|
271
|
+
howpublished = {\url{https://github.com/StephanAkkerman/reddit-stock-analyzer}}
|
|
272
|
+
}
|
|
273
|
+
```
|
|
274
|
+
|
|
275
|
+
## Contributing 🛠
|
|
276
|
+
|
|
277
|
+
Contributions are welcome! If you have a feature request, bug report, or proposal for code refactoring, please feel free to open an issue on GitHub. We appreciate your help in improving this project.\
|
|
278
|
+

|
|
279
|
+
|
|
280
|
+
## License 📜
|
|
281
|
+
|
|
282
|
+
This project is licensed under the MIT License. See the [LICENSE](LICENSE) file for details.
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "reddit-stock-analyzer"
|
|
3
|
+
dynamic = ["version"]
|
|
4
|
+
authors = [{ name = "Stephan Akkerman", email = "stephan@akkerman.ai" }]
|
|
5
|
+
description = "Scrape finance subreddits, recognise the stocks discussed, and rank what is trending."
|
|
6
|
+
readme = "README.md"
|
|
7
|
+
license = { text = "MIT" }
|
|
8
|
+
requires-python = ">=3.10"
|
|
9
|
+
keywords = ["reddit", "stocks", "wallstreetbets", "sentiment", "trends"]
|
|
10
|
+
dependencies = [
|
|
11
|
+
"httpx",
|
|
12
|
+
# Ticker + company recognition. Pulls torch/transformers transitively for
|
|
13
|
+
# its GLiNER2 path; `TickerExtractor(use_ai=False)` skips loading it.
|
|
14
|
+
"stock-recognizer>=0.1.1",
|
|
15
|
+
]
|
|
16
|
+
|
|
17
|
+
[project.optional-dependencies]
|
|
18
|
+
# Authenticated scraping. Without it the anonymous JSON endpoints are used,
|
|
19
|
+
# which work but are rate limited harder.
|
|
20
|
+
praw = ["asyncpraw", "python-dotenv"]
|
|
21
|
+
# FinTwitBERT-wsb post sentiment. Normally redundant — stock-recognizer
|
|
22
|
+
# already pulls transformers and torch — but pinned here so the sentiment
|
|
23
|
+
# path keeps working if that ever stops being true. Without them every post
|
|
24
|
+
# is labelled neutral.
|
|
25
|
+
sentiment = ["transformers", "torch"]
|
|
26
|
+
test = ["pytest", "pytest-asyncio", "httpx"]
|
|
27
|
+
|
|
28
|
+
[project.scripts]
|
|
29
|
+
reddit-stock-analyzer = "reddit_stock_analyzer.__main__:main"
|
|
30
|
+
|
|
31
|
+
[project.urls]
|
|
32
|
+
"Homepage" = "https://github.com/StephanAkkerman/reddit-stock-analyzer"
|
|
33
|
+
|
|
34
|
+
[build-system]
|
|
35
|
+
requires = ["setuptools>=61.0"]
|
|
36
|
+
build-backend = "setuptools.build_meta"
|
|
37
|
+
|
|
38
|
+
# Single source of truth for the version: `reddit_stock_analyzer/_version.py`.
|
|
39
|
+
# A release bumps that one literal; this reads it, so the two can never drift.
|
|
40
|
+
[tool.setuptools.dynamic]
|
|
41
|
+
version = { attr = "reddit_stock_analyzer._version.__version__" }
|
|
42
|
+
|
|
43
|
+
[tool.setuptools.packages.find]
|
|
44
|
+
where = ["."]
|
|
45
|
+
include = ["reddit_stock_analyzer*"]
|
|
46
|
+
|
|
47
|
+
[tool.setuptools.package-data]
|
|
48
|
+
reddit_stock_analyzer = ["py.typed"]
|
|
49
|
+
|
|
50
|
+
[tool.pytest.ini_options]
|
|
51
|
+
asyncio_mode = "auto"
|
|
52
|
+
testpaths = ["tests"]
|
|
53
|
+
# Import the package from the working tree, so the suite runs without an
|
|
54
|
+
# editable install too.
|
|
55
|
+
pythonpath = ["."]
|
|
56
|
+
|
|
57
|
+
[tool.isort]
|
|
58
|
+
multi_line_output = 3
|
|
59
|
+
include_trailing_comma = true
|
|
60
|
+
force_grid_wrap = 0
|
|
61
|
+
line_length = 88
|
|
62
|
+
profile = "black"
|
|
63
|
+
|
|
64
|
+
[tool.ruff]
|
|
65
|
+
line-length = 88
|
|
66
|
+
#select = ["I001"]
|
|
67
|
+
|
|
68
|
+
[tool.ruff.lint.pydocstyle]
|
|
69
|
+
# Use Google-style docstrings.
|
|
70
|
+
convention = "numpy"
|