TradeAssist 0.6.7__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
app/__init__.py ADDED
@@ -0,0 +1,4 @@
1
+ # TradeAssist Gemini-SEC Analyzer Package
2
+ from app.config import VERSION
3
+
4
+ __version__ = VERSION
@@ -0,0 +1,2 @@
1
+ # Backend logic for TradeAssist Gemini-SEC Analyzer
2
+ # Keeping this package 100% pure Python/Pydantic/Gemini-SDK, free of PyQt6 GUI dependencies.
@@ -0,0 +1,324 @@
1
+ import yfinance as yf
2
+ from typing import Type, Optional, Union
3
+ from pydantic import BaseModel
4
+ from app.backend.sec_client import SECClient
5
+ from app.backend.gemini_client import GeminiClient
6
+ from app.backend.data_models import StockAnalysis, CustomAnalysisResult
7
+ from app.config import get_logger
8
+ from app.settings_manager import SettingsManager
9
+ from datetime import datetime, timedelta
10
+ import time
11
+
12
+ logger = get_logger("analyzer")
13
+
14
+ class GeminiSecAnalyzer:
15
+ """
16
+ Coordinating Engine that integrates SECClient and GeminiClient.
17
+ Provides decoupled analytical routines suitable for both the PyQt GUI
18
+ and integration into an automated trading bot.
19
+ """
20
+ def __init__(self, gemini_api_key: Optional[str] = None):
21
+ self.sec_client = SECClient()
22
+ self.gemini_client = GeminiClient(api_key=gemini_api_key)
23
+
24
+ def gather_recent_news(
25
+ self,
26
+ ticker_symbol: str,
27
+ days_back: int = 7,
28
+ limit: int = 10
29
+ ) -> str:
30
+ """
31
+ Gathers recent news for a ticker symbol using Finnhub (if API key is available)
32
+ or falls back to yfinance news feed.
33
+ Applies date filtering and article limits.
34
+ """
35
+ from app.config import FINNHUB_API_KEY
36
+
37
+ news_headlines = []
38
+ used_source = "None"
39
+
40
+ # 1. Try Finnhub News
41
+ if FINNHUB_API_KEY:
42
+ try:
43
+ import finnhub
44
+ logger.info(f"Attempting to fetch news from Finnhub for {ticker_symbol}")
45
+ finnhub_client = finnhub.Client(api_key=FINNHUB_API_KEY)
46
+
47
+ to_date = datetime.now()
48
+ from_date = to_date - timedelta(days=days_back)
49
+
50
+ from_str = from_date.strftime("%Y-%m-%d")
51
+ to_str = to_date.strftime("%Y-%m-%d")
52
+
53
+ raw_news = finnhub_client.company_news(ticker_symbol.upper(), _from=from_str, to=to_str)
54
+ if raw_news:
55
+ # Sort by datetime descending
56
+ raw_news = sorted(raw_news, key=lambda x: x.get('datetime', 0), reverse=True)
57
+ raw_news = raw_news[:limit]
58
+ for article in raw_news:
59
+ headline = article.get('headline', '').strip()
60
+ source = article.get('source', 'Finnhub')
61
+ summary = article.get('summary', '').strip()
62
+ if headline:
63
+ news_item = f"[{source}] {headline}"
64
+ if summary:
65
+ if len(summary) > 200:
66
+ summary = summary[:200] + "..."
67
+ news_item += f"\n Summary: {summary}"
68
+ news_headlines.append(news_item)
69
+ used_source = "Finnhub"
70
+ except Exception as e:
71
+ logger.error(f"Error fetching Finnhub news for {ticker_symbol}, falling back to yfinance: {e}")
72
+
73
+ # 2. Fallback to yfinance if Finnhub failed or wasn't configured
74
+ if not news_headlines:
75
+ try:
76
+ logger.info(f"Fetching news from yfinance for {ticker_symbol}")
77
+ ticker = yf.Ticker(ticker_symbol)
78
+ yf_news = ticker.news
79
+ if yf_news:
80
+ now_ts = time.time()
81
+ from_ts = now_ts - (days_back * 86400)
82
+
83
+ filtered_yf_news = []
84
+ for article in yf_news:
85
+ content = article.get('content', {})
86
+ pub_time_val = article.get('providerPublishTime', content.get('pubDate', 0))
87
+
88
+ ts = 0
89
+ if isinstance(pub_time_val, (int, float)):
90
+ ts = pub_time_val
91
+ elif isinstance(pub_time_val, str):
92
+ try:
93
+ ts = datetime.fromisoformat(pub_time_val.replace('Z', '+00:00')).timestamp()
94
+ except Exception:
95
+ pass
96
+
97
+ if ts == 0 or ts >= from_ts:
98
+ filtered_yf_news.append(article)
99
+
100
+ # Limit count
101
+ filtered_yf_news = filtered_yf_news[:limit]
102
+
103
+ for article in filtered_yf_news:
104
+ content = article.get('content', {})
105
+ headline = content.get('title', article.get('title', '')).strip()
106
+ publisher = content.get('publisher', article.get('publisher', 'YF'))
107
+ summary = content.get('summary', article.get('summary', '')).strip()
108
+ if not headline:
109
+ headline = article.get('title', '').strip()
110
+ if headline:
111
+ news_item = f"[{publisher}] {headline}"
112
+ if summary:
113
+ if len(summary) > 200:
114
+ summary = summary[:200] + "..."
115
+ news_item += f"\n Summary: {summary}"
116
+ news_headlines.append(news_item)
117
+ used_source = "yfinance"
118
+ except Exception as e:
119
+ logger.error(f"Error fetching yfinance news fallback for {ticker_symbol}: {e}")
120
+
121
+ formatted_news = "\n\n".join(news_headlines) if news_headlines else "No recent news found."
122
+ logger.info(f"Gathered {len(news_headlines)} news items using {used_source}")
123
+ return formatted_news
124
+
125
+ def analyze_ticker_comprehensive(
126
+ self,
127
+ ticker_symbol: str,
128
+ model: str = "gemini-3.8-flash",
129
+ days_back: Optional[int] = None,
130
+ limit: Optional[int] = None
131
+ ) -> StockAnalysis:
132
+ """
133
+ Gathers general news, recent filings RSS summary, historical share count,
134
+ and financial statements for a ticker, and feeds them into the standard
135
+ comprehensive StockAnalysis model. Highly suitable for periodic bot scanning.
136
+ """
137
+ logger.info(f"Starting comprehensive analyzer for {ticker_symbol}")
138
+ ticker = yf.Ticker(ticker_symbol)
139
+
140
+ # Load settings for news limits if not passed explicitly
141
+ settings = SettingsManager.load_settings()
142
+ if days_back is None:
143
+ days_back = settings.get("news_scan_days", 7)
144
+ if limit is None:
145
+ limit = settings.get("news_limit", 10)
146
+
147
+ # 1. Gather recent news (Using customizable Finnhub/yfinance news feed)
148
+ formatted_news = self.gather_recent_news(ticker_symbol, days_back=days_back, limit=limit)
149
+
150
+ # 2. Gather SEC filings list (using RSS / metadata endpoint)
151
+ recent_filings_list = self.sec_client.get_recent_filings(ticker_symbol, count=10)
152
+ formatted_filings = []
153
+ for f in recent_filings_list:
154
+ formatted_filings.append(
155
+ f"- [Filed: {f['filing_date']}] Form {f['form']} | {f['description']} | Link: {f['url']}"
156
+ )
157
+ sec_filings_summary = "\n".join(formatted_filings) if formatted_filings else "No recent filings found."
158
+
159
+ # 3. Gather Fundamentals (Dilution & Cash Burn)
160
+ try:
161
+ shares_df = ticker.get_shares_full(start="2024-01-01")
162
+ shares_data = shares_df.tail(10).to_string() if shares_df is not None else "No share history available."
163
+ except Exception:
164
+ shares_data = "No share history available."
165
+
166
+ try:
167
+ balance_sheet = ticker.balance_sheet.to_string()
168
+ financials = ticker.financials.to_string()
169
+ except Exception:
170
+ balance_sheet = "Financial statement data unavailable."
171
+ financials = "Income statement data unavailable."
172
+
173
+ # 4. Build prompt
174
+ prompt = f"""
175
+ Analyze the following financial and news data for the ticker '{ticker_symbol.upper()}'.
176
+
177
+ 1. RECENT NEWS HEADLINES:
178
+ {formatted_news}
179
+
180
+ 1b. RECENT SEC FILINGS (LAST SEVERAL WEEKS):
181
+ {sec_filings_summary}
182
+
183
+ 2. HISTORICAL SHARE COUNT DATA (Look for increasing share count indicating dilution):
184
+ {shares_data}
185
+
186
+ 3. BALANCE SHEET & FINANCIAL DATA (Look for Cash & Cash Equivalents vs Net Income/Operating Expenses to evaluate cash burn):
187
+ {balance_sheet[:4000]}
188
+ {financials[:4000]}
189
+ """
190
+
191
+ # 5. Call structured generation
192
+ result = self.gemini_client.generate_structured_analysis(
193
+ prompt=prompt,
194
+ response_schema=StockAnalysis,
195
+ model=model,
196
+ temperature=0.0
197
+ )
198
+ return result
199
+
200
+ def analyze_specific_filings(
201
+ self,
202
+ ticker_symbol: str,
203
+ filings: list,
204
+ custom_prompt: str,
205
+ response_schema: Optional[Type[BaseModel]] = None,
206
+ model: str = "gemini-3.8-flash",
207
+ temperature: float = 0.1,
208
+ max_chars: int = 150000,
209
+ include_news: Optional[bool] = None,
210
+ days_back: Optional[int] = None,
211
+ limit: Optional[int] = None
212
+ ) -> Union[BaseModel, str]:
213
+ """
214
+ Downloads multiple specific filings, extracts their text content,
215
+ concatenates them with distinct structural separators and metadata,
216
+ and sends the combined text to Gemini along with a custom analytical prompt.
217
+ Supports structured JSON output (if response_schema is provided) or plaintext markdown.
218
+ Optionally includes recent news feed context for the ticker.
219
+ """
220
+ logger.info(f"Analyzing {len(filings)} filings for {ticker_symbol}")
221
+
222
+ # Load default settings if not passed explicitly
223
+ settings = SettingsManager.load_settings()
224
+ if include_news is None:
225
+ include_news = settings.get("include_news_custom", False)
226
+ if days_back is None:
227
+ days_back = settings.get("news_scan_days", 7)
228
+ if limit is None:
229
+ limit = settings.get("news_limit", 10)
230
+
231
+ news_context = ""
232
+ if include_news:
233
+ logger.info(f"Fetching news context to couple with custom prompt for {ticker_symbol}")
234
+ news_data = self.gather_recent_news(ticker_symbol, days_back=days_back, limit=limit)
235
+ news_context = f"""
236
+ === RECENT NEWS ({days_back} days, limit {limit}) ===
237
+ {news_data}
238
+ ==================================================
239
+ """
240
+
241
+ combined_texts = []
242
+ for idx, filing in enumerate(filings, start=1):
243
+ url = filing["url"]
244
+ form = filing["form"]
245
+ date = filing["filing_date"]
246
+ desc = filing.get("description", "[No Description]")
247
+
248
+ # Fetch text of the specific filing using specified max_chars limit
249
+ filing_text = self.sec_client.fetch_filing_text(url, max_chars=max_chars)
250
+ if not filing_text:
251
+ logger.warning(f"Could not retrieve text for filing Form {form} from {url}")
252
+ continue
253
+
254
+ combined_texts.append(f"""
255
+ === FILING DOCUMENT {idx} OF {len(filings)} ===
256
+ Form Type: {form}
257
+ Filing Date: {date}
258
+ Description: {desc}
259
+ SEC Document URL: {url}
260
+
261
+ --- START OF FILING TEXT ---
262
+ {filing_text}
263
+ --- END OF FILING TEXT ---
264
+ """)
265
+
266
+ if not combined_texts:
267
+ raise ValueError("Could not retrieve or parse filing text from any of the selected filings.")
268
+
269
+ # Build the combined prompt
270
+ full_prompt = f"""
271
+ You are an expert financial analyst. Please execute the custom instruction below on the provided SEC filing document(s) and recent news for ticker '{ticker_symbol.upper()}'.
272
+
273
+ {news_context}
274
+
275
+ CUSTOM ANALYSIS INSTRUCTION:
276
+ {custom_prompt}
277
+
278
+ SEC FILING DOCUMENTS PROVIDED:
279
+ {"".join(combined_texts)}
280
+ """
281
+
282
+ # Request Gemini analysis (either structured or unstructured)
283
+ if response_schema:
284
+ return self.gemini_client.generate_structured_analysis(
285
+ prompt=full_prompt,
286
+ response_schema=response_schema,
287
+ model=model,
288
+ temperature=temperature
289
+ )
290
+ else:
291
+ return self.gemini_client.generate_text_analysis(
292
+ prompt=full_prompt,
293
+ model=model,
294
+ temperature=temperature
295
+ )
296
+
297
+ def analyze_specific_filing(
298
+ self,
299
+ ticker_symbol: str,
300
+ filing_url: str,
301
+ custom_prompt: str,
302
+ response_schema: Optional[Type[BaseModel]] = None,
303
+ model: str = "gemini-3.8-flash",
304
+ temperature: float = 0.1
305
+ ) -> Union[BaseModel, str]:
306
+ """
307
+ Downloads a specific filing (10-K, 10-Q, 8-K), extracts its text content,
308
+ and sends it to Gemini along with a custom analytical prompt.
309
+ Supports structured JSON output (if response_schema is provided) or plaintext markdown.
310
+ """
311
+ # Backward compatibility wrapper
312
+ filings = [{
313
+ "url": filing_url,
314
+ "form": "Unknown",
315
+ "filing_date": "Unknown"
316
+ }]
317
+ return self.analyze_specific_filings(
318
+ ticker_symbol=ticker_symbol,
319
+ filings=filings,
320
+ custom_prompt=custom_prompt,
321
+ response_schema=response_schema,
322
+ model=model,
323
+ temperature=temperature
324
+ )
@@ -0,0 +1,24 @@
1
+ from pydantic import BaseModel, Field
2
+ from typing import List, Dict, Any, Optional
3
+
4
+ class StockAnalysis(BaseModel):
5
+ news_score: int = Field(description="Score from 0 to 10")
6
+ news_explanation: str = Field(description="Explanation")
7
+ headline_count_fed: int = Field(description="Count")
8
+ SEC_score: int = Field(description="Score from 0 to 10")
9
+ SEC_explanation: str = Field(description="Explanation")
10
+ dilution_score: int = Field(description="Score")
11
+ dilution_explanation: str = Field(description="Explanation")
12
+ cash_burn_score: int = Field(description="Score")
13
+ cash_burn_explanation: str = Field(description="Explanation")
14
+
15
+ class KeyValueMetric(BaseModel):
16
+ key: str = Field(description="The name of the metric")
17
+ value: str = Field(description="The value")
18
+ context: str = Field(description="Context")
19
+
20
+ class CustomAnalysisResult(BaseModel):
21
+ executive_summary: str = Field(description="Overview")
22
+ sentiment_score: int = Field(description="Sentiment 0-10")
23
+ key_insights: List[str] = Field(description="Bullet points")
24
+ metrics_table: List[KeyValueMetric] = Field(description="Structured key-values")
@@ -0,0 +1,176 @@
1
+ from google import genai
2
+ from google.genai import types
3
+ from google.genai.errors import APIError
4
+ import time
5
+ from typing import Type, Optional, Union
6
+ from pydantic import BaseModel
7
+ from app.config import GEMINI_API_KEY, get_logger
8
+
9
+ logger = get_logger("gemini_client")
10
+
11
+ class TextAnalysisResult(str):
12
+ """
13
+ Subclass of str that holds metadata about token usage and model.
14
+ """
15
+ def __new__(cls, content, *args, **kwargs):
16
+ return super().__new__(cls, content)
17
+
18
+ def __init__(self, content, model="", prompt_tokens=0, candidates_tokens=0, total_tokens=0):
19
+ super().__init__()
20
+ self.model = model
21
+ self.prompt_tokens = prompt_tokens
22
+ self.candidates_tokens = candidates_tokens
23
+ self.total_tokens = total_tokens
24
+
25
+ class GeminiClient:
26
+ """
27
+ Pure Python Gemini API Client wrapper.
28
+ Responsible for initializing the Google GenAI SDK, formatting contents,
29
+ and handling rate limits/service unavailability with exponential backoff.
30
+ """
31
+ def __init__(self, api_key: Optional[str] = None):
32
+ key = api_key or GEMINI_API_KEY
33
+ self.client = None
34
+ if not key:
35
+ logger.warning("No Gemini API key supplied. Ensure GEMINI_API_KEY is in your environment.")
36
+ else:
37
+ # Initialize client
38
+ self.client = genai.Client(api_key=key)
39
+
40
+ def generate_structured_analysis(
41
+ self,
42
+ prompt: str,
43
+ response_schema: Type[BaseModel],
44
+ model: str = "gemini-3.8-flash",
45
+ temperature: float = 0.0,
46
+ max_retries: int = 5,
47
+ initial_delay: float = 3.0
48
+ ) -> BaseModel:
49
+ """
50
+ Sends a prompt to Gemini requesting a structured output conforming to response_schema.
51
+ Implements exponential backoff to handle 429 (Rate Limits) and 503 (Overloaded) errors.
52
+ """
53
+ if not self.client:
54
+ raise ValueError(
55
+ "Gemini API Key is not set! Please go to File -> Settings & API Keys... "
56
+ "to configure your Gemini API Key before running analyses."
57
+ )
58
+
59
+ for attempt in range(max_retries):
60
+ try:
61
+ logger.info(f"Sending prompt to Gemini using model {model} (Attempt {attempt+1}/{max_retries})")
62
+
63
+ response = self.client.models.generate_content(
64
+ model=model,
65
+ contents=prompt,
66
+ config=types.GenerateContentConfig(
67
+ response_mime_type="application/json",
68
+ response_schema=response_schema,
69
+ temperature=temperature
70
+ )
71
+ )
72
+
73
+ # Use Pydantic to validate and parse the JSON string returned by Gemini
74
+ parsed_response = response_schema.model_validate_json(response.text)
75
+
76
+ # Get usage metadata or estimate
77
+ usage = getattr(response, 'usage_metadata', None)
78
+ prompt_tokens = getattr(usage, 'prompt_token_count', 0) if usage else 0
79
+ candidates_tokens = getattr(usage, 'candidates_token_count', 0) if usage else 0
80
+ total_tokens = getattr(usage, 'total_token_count', 0) if usage else 0
81
+
82
+ if not prompt_tokens:
83
+ prompt_tokens = int(len(prompt) / 4)
84
+ if not candidates_tokens:
85
+ candidates_tokens = int(len(response.text) / 4)
86
+ if not total_tokens:
87
+ total_tokens = prompt_tokens + candidates_tokens
88
+
89
+ # Attach metadata safely
90
+ object.__setattr__(parsed_response, 'model', model)
91
+ object.__setattr__(parsed_response, 'prompt_tokens', prompt_tokens)
92
+ object.__setattr__(parsed_response, 'candidates_tokens', candidates_tokens)
93
+ object.__setattr__(parsed_response, 'total_tokens', total_tokens)
94
+
95
+ return parsed_response
96
+
97
+ except APIError as e:
98
+ if e.code in [429, 503] and attempt < max_retries - 1:
99
+ delay = initial_delay * (2 ** attempt)
100
+ logger.warning(f"Gemini API returned error {e.code} ({e.message}). Retrying in {delay}s...")
101
+ time.sleep(delay)
102
+ else:
103
+ logger.error(f"Gemini API non-retryable error: {e}")
104
+ raise e
105
+ except Exception as e:
106
+ logger.error(f"Unexpected error in Gemini generation: {e}")
107
+ raise e
108
+
109
+ raise Exception(f"Gemini API generation failed after {max_retries} attempts.")
110
+
111
+ def generate_text_analysis(
112
+ self,
113
+ prompt: str,
114
+ model: str = "gemini-3.8-flash",
115
+ temperature: float = 0.2,
116
+ max_retries: int = 5,
117
+ initial_delay: float = 3.0
118
+ ) -> str:
119
+ """
120
+ Sends a prompt to Gemini requesting a standard unstructured Markdown or text response.
121
+ Useful for general, open-ended research prompts.
122
+ """
123
+ if not self.client:
124
+ raise ValueError(
125
+ "Gemini API Key is not set! Please go to File -> Settings & API Keys... "
126
+ "to configure your Gemini API Key before running analyses."
127
+ )
128
+
129
+ for attempt in range(max_retries):
130
+ try:
131
+ logger.info(f"Sending text prompt to Gemini using model {model} (Attempt {attempt+1}/{max_retries})")
132
+
133
+ response = self.client.models.generate_content(
134
+ model=model,
135
+ contents=prompt,
136
+ config=types.GenerateContentConfig(
137
+ temperature=temperature
138
+ )
139
+ )
140
+
141
+ response_text = response.text
142
+
143
+ # Get usage metadata or estimate
144
+ usage = getattr(response, 'usage_metadata', None)
145
+ prompt_tokens = getattr(usage, 'prompt_token_count', 0) if usage else 0
146
+ candidates_tokens = getattr(usage, 'candidates_token_count', 0) if usage else 0
147
+ total_tokens = getattr(usage, 'total_token_count', 0) if usage else 0
148
+
149
+ if not prompt_tokens:
150
+ prompt_tokens = int(len(prompt) / 4)
151
+ if not candidates_tokens:
152
+ candidates_tokens = int(len(response_text) / 4)
153
+ if not total_tokens:
154
+ total_tokens = prompt_tokens + candidates_tokens
155
+
156
+ return TextAnalysisResult(
157
+ response_text,
158
+ model=model,
159
+ prompt_tokens=prompt_tokens,
160
+ candidates_tokens=candidates_tokens,
161
+ total_tokens=total_tokens
162
+ )
163
+
164
+ except APIError as e:
165
+ if e.code in [429, 503] and attempt < max_retries - 1:
166
+ delay = initial_delay * (2 ** attempt)
167
+ logger.warning(f"Gemini API returned error {e.code} ({e.message}). Retrying in {delay}s...")
168
+ time.sleep(delay)
169
+ else:
170
+ logger.error(f"Gemini API non-retryable error: {e}")
171
+ raise e
172
+ except Exception as e:
173
+ logger.error(f"Unexpected error in Gemini generation: {e}")
174
+ raise e
175
+
176
+ raise Exception(f"Gemini API generation failed after {max_retries} attempts.")
@@ -0,0 +1,126 @@
1
+ import warnings
2
+ import requests
3
+ from bs4 import BeautifulSoup, XMLParsedAsHTMLWarning
4
+ from app.config import SEC_HEADERS, get_logger
5
+
6
+ # Suppress BS4 XML-parsed-as-HTML warning for SEC XML-format filings
7
+ warnings.filterwarnings("ignore", category=XMLParsedAsHTMLWarning)
8
+
9
+ logger = get_logger("sec_client")
10
+
11
+ class SECClient:
12
+ """
13
+ Pure Python SEC EDGAR Client.
14
+ Handles mapping tickers to CIKs, listing recent company filings,
15
+ and fetching/parsing the text content of specific filing documents.
16
+ """
17
+ def __init__(self):
18
+ self.headers = SEC_HEADERS
19
+
20
+ def get_cik(self, ticker: str) -> str:
21
+ """
22
+ Maps a stock ticker (e.g. 'AAPL') to its zero-padded 10-digit CIK.
23
+ """
24
+ ticker_clean = ticker.strip().upper()
25
+ url = "https://www.sec.gov/files/company_tickers.json"
26
+ try:
27
+ response = requests.get(url, headers=self.headers, timeout=10)
28
+ if response.status_code != 200:
29
+ logger.error(f"Failed to fetch SEC mapping list. Status: {response.status_code}")
30
+ return None
31
+
32
+ data = response.json()
33
+ for entry in data.values():
34
+ if entry.get('ticker') == ticker_clean:
35
+ return str(entry.get('cik_str')).zfill(10)
36
+
37
+ logger.warning(f"Ticker '{ticker_clean}' not found in SEC company_tickers.json")
38
+ return None
39
+ except Exception as e:
40
+ logger.error(f"Error mapping CIK for {ticker_clean}: {e}")
41
+ return None
42
+
43
+ def get_recent_filings(self, ticker: str, count: int = 20) -> list:
44
+ """
45
+ Fetches the recent filings list for a given ticker from the SEC Submissions API.
46
+ Returns a list of structured dictionaries containing filing metadata.
47
+ """
48
+ cik = self.get_cik(ticker)
49
+ if not cik:
50
+ return []
51
+
52
+ url = f"https://data.sec.gov/submissions/CIK{cik}.json"
53
+ try:
54
+ response = requests.get(url, headers=self.headers, timeout=10)
55
+ if response.status_code != 200:
56
+ logger.error(f"SEC Submissions API returned status {response.status_code} for CIK {cik}")
57
+ return []
58
+
59
+ data = response.json()
60
+ filings_raw = data.get('filings', {}).get('recent', {})
61
+ if not filings_raw:
62
+ return []
63
+
64
+ # Structure the lists into individual records
65
+ filings = []
66
+ num_filings = len(filings_raw.get('accessionNumber', []))
67
+
68
+ for i in range(min(num_filings, count)):
69
+ acc_num = filings_raw['accessionNumber'][i]
70
+ acc_num_no_dashes = acc_num.replace("-", "")
71
+ primary_doc = filings_raw['primaryDocument'][i]
72
+
73
+ # Construct standard EDGAR URL for the filing document
74
+ doc_url = f"https://www.sec.gov/Archives/edgar/data/{int(cik)}/{acc_num_no_dashes}/{primary_doc}"
75
+
76
+ filings.append({
77
+ "ticker": ticker.upper(),
78
+ "cik": cik,
79
+ "accession_number": acc_num,
80
+ "filing_date": filings_raw['filingDate'][i],
81
+ "form": filings_raw['form'][i],
82
+ "size": filings_raw['size'][i],
83
+ "primary_doc": primary_doc,
84
+ "description": filings_raw['primaryDocDescription'][i],
85
+ "url": doc_url
86
+ })
87
+
88
+ return filings
89
+ except Exception as e:
90
+ logger.error(f"Error fetching filings for {ticker}: {e}")
91
+ return []
92
+
93
+ def fetch_filing_text(self, url: str, max_chars: int = 150000) -> str:
94
+ """
95
+ Downloads an SEC filing HTML/TXT document and extracts cleaned plaintext.
96
+ Strips HTML tags, inline styling, and limits length to avoid excessive tokens.
97
+ """
98
+ try:
99
+ logger.info(f"Downloading filing from {url}")
100
+ response = requests.get(url, headers=self.headers, timeout=15)
101
+ if response.status_code != 200:
102
+ logger.error(f"Failed to fetch filing document. Status: {response.status_code}")
103
+ return ""
104
+
105
+ # Use BeautifulSoup to parse HTML and extract text
106
+ soup = BeautifulSoup(response.text, "lxml")
107
+
108
+ # Remove script and style elements
109
+ for element in soup(["script", "style"]):
110
+ element.decompose()
111
+
112
+ # Get text and clean up whitespace
113
+ text = soup.get_text(separator="\n")
114
+ lines = [line.strip() for line in text.splitlines()]
115
+ chunks = [phrase for phrase in lines if phrase]
116
+ clean_text = "\n".join(chunks)
117
+
118
+ # Limit length to prevent extreme context window consumption
119
+ if len(clean_text) > max_chars:
120
+ logger.info(f"Filing text truncated from {len(clean_text)} to {max_chars} chars.")
121
+ return clean_text[:max_chars] + "\n\n...[TRUNCATED FOR LENGTH]..."
122
+
123
+ return clean_text
124
+ except Exception as e:
125
+ logger.error(f"Error fetching/parsing filing text from {url}: {e}")
126
+ return ""