fable-engine 1.3.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (104) hide show
  1. fable_compressor.py +356 -0
  2. fable_engine/__init__.py +1 -0
  3. fable_engine/actions/__init__.py +291 -0
  4. fable_engine/actions/cas.py +182 -0
  5. fable_engine/actions/deliberation.py +523 -0
  6. fable_engine/actions/fleet.py +807 -0
  7. fable_engine/actions/lifecycle.py +298 -0
  8. fable_engine/actions/scrapers.py +116 -0
  9. fable_engine/actions/system3.py +815 -0
  10. fable_engine/browser.py +824 -0
  11. fable_engine/cas.py +974 -0
  12. fable_engine/fable_session.json +510 -0
  13. fable_engine/guards.py +283 -0
  14. fable_engine/schema.py +714 -0
  15. fable_engine/scrapers/__init__.py +32 -0
  16. fable_engine/scrapers/arxiv.py +115 -0
  17. fable_engine/scrapers/base.py +386 -0
  18. fable_engine/scrapers/github.py +129 -0
  19. fable_engine/scrapers/reddit.py +154 -0
  20. fable_engine/scrapers/web.py +120 -0
  21. fable_engine/scrapers/x.py +125 -0
  22. fable_engine/scrapers/youtube.py +132 -0
  23. fable_engine/server.py +414 -0
  24. fable_engine/session.py +1819 -0
  25. fable_engine/test_server.py +1362 -0
  26. fable_engine/updater.py +541 -0
  27. fable_engine-1.3.1.dist-info/LICENSE +22 -0
  28. fable_engine-1.3.1.dist-info/METADATA +173 -0
  29. fable_engine-1.3.1.dist-info/RECORD +104 -0
  30. fable_engine-1.3.1.dist-info/WHEEL +5 -0
  31. fable_engine-1.3.1.dist-info/entry_points.txt +5 -0
  32. fable_engine-1.3.1.dist-info/top_level.txt +6 -0
  33. fable_mode/__init__.py +3 -0
  34. fable_mode/__main__.py +4 -0
  35. fable_mode/adapters.py +1014 -0
  36. fable_mode/installer.py +553 -0
  37. fable_mode/launcher.py +437 -0
  38. fable_mode/manifest.py +142 -0
  39. fable_mode/resources.json +114 -0
  40. fable_mode/safety.py +103 -0
  41. fable_mode_entry.py +10 -0
  42. fable_v2/__init__.py +146 -0
  43. fable_v2/adapters.py +151 -0
  44. fable_v2/coder_fleet/__init__.py +100 -0
  45. fable_v2/coder_fleet/ast_tools.py +158 -0
  46. fable_v2/coder_fleet/compute.py +199 -0
  47. fable_v2/coder_fleet/design_engine.py +1316 -0
  48. fable_v2/coder_fleet/diagnostics.py +293 -0
  49. fable_v2/coder_fleet/fleet_dispatcher.py +214 -0
  50. fable_v2/coder_fleet/mock_auditor.py +306 -0
  51. fable_v2/coder_fleet/mutation.py +216 -0
  52. fable_v2/coder_fleet/property_oracle.py +260 -0
  53. fable_v2/coder_fleet/receipt_attestor.py +122 -0
  54. fable_v2/coder_fleet/red_team_swarm.py +908 -0
  55. fable_v2/coder_fleet/test_harness.py +198 -0
  56. fable_v2/coder_fleet/vector_engine.py +1287 -0
  57. fable_v2/coder_fleet/visual.py +357 -0
  58. fable_v2/coder_fleet/workspace.py +153 -0
  59. fable_v2/cortical/__init__.py +20 -0
  60. fable_v2/cortical/plasticity_engine.py +992 -0
  61. fable_v2/execution_broker.py +811 -0
  62. fable_v2/proof_engine.py +1141 -0
  63. fable_v2/protocol.py +485 -0
  64. fable_v2/runtime.py +1010 -0
  65. fable_v2/system3/__init__.py +204 -0
  66. fable_v2/system3/causal.py +558 -0
  67. fable_v2/system3/dialectical.py +577 -0
  68. fable_v2/system3/evolution.py +503 -0
  69. fable_v2/system3/executive.py +338 -0
  70. fable_v2/system3/free_energy.py +479 -0
  71. fable_v2/system3/hyperbolic.py +555 -0
  72. fable_v2/system3/induction.py +336 -0
  73. fable_v2/system3/kripke.py +548 -0
  74. fable_v2/system3/oracle.py +745 -0
  75. fable_v2/verifiers.py +72 -0
  76. tests/__init__.py +1 -0
  77. tests/test_anti_loop_circuit_breaker.py +64 -0
  78. tests/test_auto_updater.py +407 -0
  79. tests/test_coder_fleet.py +535 -0
  80. tests/test_delegation_compiler.py +54 -0
  81. tests/test_descriptor_boundaries.py +126 -0
  82. tests/test_design_engine.py +603 -0
  83. tests/test_epistemic_evidence_validator.py +66 -0
  84. tests/test_execution_broker.py +233 -0
  85. tests/test_fable_v2.py +406 -0
  86. tests/test_fleet_transitions.py +116 -0
  87. tests/test_fsm_redteam_evolution.py +406 -0
  88. tests/test_goal_rubric_and_pipeline.py +367 -0
  89. tests/test_hebbian_plasticity.py +585 -0
  90. tests/test_packaging_runtime.py +194 -0
  91. tests/test_proof_engine.py +259 -0
  92. tests/test_red_team_swarm.py +645 -0
  93. tests/test_redteam_remediation.py +169 -0
  94. tests/test_registration_transaction.py +375 -0
  95. tests/test_requested_regressions.py +467 -0
  96. tests/test_scrapers.py +370 -0
  97. tests/test_server_actions.py +93 -0
  98. tests/test_server_frontier_actions.py +269 -0
  99. tests/test_server_protocol.py +88 -0
  100. tests/test_stealth_browser.py +970 -0
  101. tests/test_system3.py +381 -0
  102. tests/test_system3_deep_integration.py +385 -0
  103. tests/test_system3_frontier.py +436 -0
  104. tests/test_vector_engine.py +608 -0
@@ -0,0 +1,32 @@
1
+ """
2
+ Fable Engine Research Scrapers Package.
3
+ Exporting zero-cost, standard-library based scrapers for Web, YouTube, Reddit, X, GitHub, and arXiv.
4
+ """
5
+
6
+ from __future__ import annotations
7
+
8
+ from fable_engine.scrapers.arxiv import ArxivScraper, scrape_arxiv
9
+ from fable_engine.scrapers.base import ResearchResult, ResearchSource, fetch_url
10
+ from fable_engine.scrapers.github import GitHubScraper, scrape_github
11
+ from fable_engine.scrapers.reddit import RedditScraper, scrape_reddit
12
+ from fable_engine.scrapers.web import WebScraper, scrape_web
13
+ from fable_engine.scrapers.x import XScraper, scrape_x
14
+ from fable_engine.scrapers.youtube import YouTubeScraper, scrape_youtube
15
+
16
+ __all__ = [
17
+ "ResearchResult",
18
+ "ResearchSource",
19
+ "fetch_url",
20
+ "WebScraper",
21
+ "scrape_web",
22
+ "YouTubeScraper",
23
+ "scrape_youtube",
24
+ "RedditScraper",
25
+ "scrape_reddit",
26
+ "XScraper",
27
+ "scrape_x",
28
+ "GitHubScraper",
29
+ "scrape_github",
30
+ "ArxivScraper",
31
+ "scrape_arxiv",
32
+ ]
@@ -0,0 +1,115 @@
1
+ """
2
+ arXiv research paper and search query scraper implementation using arXiv export API.
3
+ """
4
+
5
+ from __future__ import annotations
6
+
7
+ import re
8
+ import urllib.parse
9
+ import xml.etree.ElementTree as ET
10
+ from typing import Optional
11
+
12
+ from fable_engine.scrapers.base import ResearchResult, ResearchSource, fetch_url
13
+
14
+
15
+ class ArxivScraper(ResearchSource):
16
+ source_type = "arxiv"
17
+
18
+ def fetch(self, target: str, timeout: int = 15, max_content_length: int = 10000) -> ResearchResult:
19
+ target = target.strip()
20
+ if not target:
21
+ return ResearchResult(
22
+ ok=False,
23
+ source_type=self.source_type,
24
+ canonical_url="",
25
+ error="Target arXiv paper ID or search query cannot be empty."
26
+ )
27
+
28
+ paper_id = None
29
+ if "arxiv.org/" in target:
30
+ id_match = re.search(r"(?:abs|pdf)/([0-9]+\.[0-9]+(?:v[0-9]+)?)", target)
31
+ if id_match:
32
+ paper_id = id_match.group(1)
33
+ elif re.match(r"^[0-9]{4}\.[0-9]{4,5}(?:v[0-9]+)?$", target):
34
+ paper_id = target
35
+
36
+ if paper_id:
37
+ api_url = f"https://export.arxiv.org/api/query?id_list={paper_id}"
38
+ canonical_url = f"https://arxiv.org/abs/{paper_id}"
39
+ else:
40
+ query_enc = urllib.parse.quote(target)
41
+ api_url = f"https://export.arxiv.org/api/query?search_query=all:{query_enc}&max_results=5"
42
+ canonical_url = f"https://arxiv.org/search/?query={query_enc}&searchtype=all"
43
+
44
+ try:
45
+ xml_str = fetch_url(api_url, timeout=timeout)
46
+ root = ET.fromstring(xml_str)
47
+ ns = {"atom": "http://www.w3.org/2005/Atom", "arxiv": "http://arxiv.org/schemas/atom"}
48
+
49
+ entries = root.findall("atom:entry", ns)
50
+ if not entries:
51
+ return ResearchResult(
52
+ ok=True,
53
+ source_type=self.source_type,
54
+ canonical_url=canonical_url,
55
+ title=f"arXiv Search: '{target}'",
56
+ content="*(No papers found for this query)*"
57
+ )
58
+
59
+ blocks = []
60
+ first_title = ""
61
+ first_author = ""
62
+ for entry in entries:
63
+ title_elem = entry.find("atom:title", ns)
64
+ title = re.sub(r"\s+", " ", title_elem.text).strip() if title_elem is not None and title_elem.text else "Untitled"
65
+ if not first_title:
66
+ first_title = title
67
+
68
+ id_elem = entry.find("atom:id", ns)
69
+ entry_id = id_elem.text.strip() if id_elem is not None and id_elem.text else ""
70
+
71
+ published_elem = entry.find("atom:published", ns)
72
+ published = published_elem.text[:10] if published_elem is not None and published_elem.text else ""
73
+
74
+ summary_elem = entry.find("atom:summary", ns)
75
+ summary = re.sub(r"\s+", " ", summary_elem.text).strip() if summary_elem is not None and summary_elem.text else ""
76
+
77
+ authors = []
78
+ for author_elem in entry.findall("atom:author", ns):
79
+ name_elem = author_elem.find("atom:name", ns)
80
+ if name_elem is not None and name_elem.text:
81
+ authors.append(name_elem.text.strip())
82
+ if not first_author and authors:
83
+ first_author = ", ".join(authors)
84
+
85
+ pdf_link = entry_id.replace("/abs/", "/pdf/") + ".pdf" if "/abs/" in entry_id else entry_id
86
+
87
+ blocks.append(f"## {title}")
88
+ blocks.append(f"**Authors**: {', '.join(authors)}")
89
+ blocks.append(f"**Published**: {published} | **arXiv**: [{entry_id}]({entry_id}) | **PDF**: [{pdf_link}]({pdf_link})\n")
90
+ blocks.append(f"### Abstract\n{summary}\n")
91
+
92
+ full_content = "\n".join(blocks)
93
+ if len(full_content) > max_content_length:
94
+ full_content = full_content[:max_content_length] + "\n\n*(Abstracts truncated)*"
95
+
96
+ return ResearchResult(
97
+ ok=True,
98
+ source_type=self.source_type,
99
+ canonical_url=canonical_url,
100
+ title=first_title if paper_id else f"arXiv Papers for: '{target}'",
101
+ author=first_author,
102
+ content=full_content,
103
+ metadata={"paper_count": len(entries)}
104
+ )
105
+ except Exception as e:
106
+ return ResearchResult(
107
+ ok=False,
108
+ source_type=self.source_type,
109
+ canonical_url=canonical_url,
110
+ error=f"arXiv fetch failed: {str(e)}"
111
+ )
112
+
113
+
114
+ def scrape_arxiv(target: str, timeout: int = 15) -> ResearchResult:
115
+ return ArxivScraper().fetch(target, timeout=timeout)
@@ -0,0 +1,386 @@
1
+ """
2
+ Base research scraper interfaces, structured result objects, and HTTP transport utilities.
3
+ Enforces standard TLS certificate verification, strict SSRF protection with alternate IP
4
+ canonicalization & IP pinning (preventing DNS rebinding TOCTOU window), host rate limiting,
5
+ clamped retry/backoff, response bounds, and explicit untrusted external content boundaries.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import datetime
11
+ import http.client
12
+ import ipaddress
13
+ import json
14
+ import logging
15
+ import re
16
+ import socket
17
+ import ssl
18
+ import threading
19
+ import time
20
+ import urllib.error
21
+ import urllib.parse
22
+ from dataclasses import dataclass, field
23
+ from html.parser import HTMLParser
24
+ from typing import Any, Dict, List, Optional, Tuple
25
+
26
+ logger = logging.getLogger("fable-engine.scrapers.base")
27
+
28
+ USER_AGENT = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/122.0.0.0 Safari/537.36 FableResearch/1.3.0"
29
+
30
+ MAX_RETRY_DELAY_SECONDS = 10.0
31
+
32
+ # Standard TLS context enforcing certificate verification
33
+ _SSL_CONTEXT = ssl.create_default_context()
34
+
35
+
36
+ def is_safe_ip(ip_str: str) -> bool:
37
+ """Verifies that an IP address is public and safe (not loopback, private, link-local, reserved)."""
38
+ try:
39
+ ip = ipaddress.ip_address(ip_str)
40
+ if (
41
+ ip.is_loopback
42
+ or ip.is_private
43
+ or ip.is_link_local
44
+ or ip.is_reserved
45
+ or ip.is_multicast
46
+ or ip.is_unspecified
47
+ ):
48
+ return False
49
+ # Cloud metadata service check (169.254.169.254)
50
+ if str(ip) == "169.254.169.254":
51
+ return False
52
+ return True
53
+ except ValueError:
54
+ return False
55
+
56
+
57
+ def parse_canonical_ip(hostname_str: str) -> Optional[str]:
58
+ """Attempts to parse hostname as standard or alternate IPv4/IPv6 address (hex, octal, short IPv4)."""
59
+ try:
60
+ return str(ipaddress.ip_address(hostname_str))
61
+ except ValueError:
62
+ pass
63
+ try:
64
+ raw_bytes = socket.inet_aton(hostname_str)
65
+ return str(ipaddress.ip_address(raw_bytes))
66
+ except (socket.error, ValueError, OverflowError):
67
+ return None
68
+
69
+
70
+ def validate_safe_url(url: str) -> Tuple[bool, str, Optional[str]]:
71
+ """
72
+ Validates that a URL uses http/https scheme and its hostname resolves
73
+ exclusively to safe, public IP addresses (SSRF prevention).
74
+ Returns (is_safe, reason, canonical_resolved_ip).
75
+ """
76
+ try:
77
+ parsed = urllib.parse.urlparse(url)
78
+ scheme = parsed.scheme.lower()
79
+ if scheme not in ("http", "https"):
80
+ return False, f"Invalid URL scheme '{parsed.scheme}'. Strictly http and https are permitted.", None
81
+
82
+ hostname = parsed.hostname
83
+ if not hostname:
84
+ return False, "URL missing valid hostname.", None
85
+
86
+ hostname_clean = hostname.strip().lower()
87
+
88
+ # Reject explicit local hostnames
89
+ if hostname_clean in ("localhost", "localhost.localdomain") or hostname_clean.endswith(".local"):
90
+ return False, f"SSRF blocked: local hostname '{hostname_clean}' is not permitted.", None
91
+
92
+ # Check if hostname is a standard or alternate numeric IP string (octal, hex, decimal)
93
+ canonical_ip = parse_canonical_ip(hostname_clean)
94
+ if canonical_ip is not None:
95
+ if not is_safe_ip(canonical_ip):
96
+ return False, f"SSRF blocked: target IP '{hostname_clean}' ({canonical_ip}) is private, loopback, or reserved.", None
97
+ return True, "ok", canonical_ip
98
+
99
+ # Resolve hostname DNS records
100
+ port = parsed.port or (443 if scheme == "https" else 80)
101
+ try:
102
+ addr_info = socket.getaddrinfo(hostname_clean, port, socket.AF_UNSPEC, socket.SOCK_STREAM)
103
+ except socket.gaierror as e:
104
+ return False, f"DNS resolution failed for host '{hostname_clean}': {e}", None
105
+
106
+ if not addr_info:
107
+ return False, f"Could not resolve IP addresses for host '{hostname_clean}'.", None
108
+
109
+ safe_ips = []
110
+ for family, _, _, _, sockaddr in addr_info:
111
+ ip_str = sockaddr[0]
112
+ if not is_safe_ip(ip_str):
113
+ return False, f"SSRF blocked: host '{hostname_clean}' resolved to restricted IP '{ip_str}'.", None
114
+ safe_ips.append(ip_str)
115
+
116
+ return True, "ok", safe_ips[0]
117
+ except Exception as e:
118
+ return False, f"URL validation error: {e}", None
119
+
120
+
121
+ class PinnedHTTPConnection(http.client.HTTPConnection):
122
+ """HTTP connection connecting directly to a pre-validated safe IP address."""
123
+
124
+ def __init__(self, host: str, resolved_ip: str, port: int = 80, **kwargs):
125
+ super().__init__(host, port=port, **kwargs)
126
+ self.resolved_ip = resolved_ip
127
+
128
+ def connect(self):
129
+ self.sock = socket.create_connection((self.resolved_ip, self.port), self.timeout)
130
+
131
+
132
+ class PinnedHTTPSConnection(http.client.HTTPSConnection):
133
+ """HTTPS connection connecting directly to a pre-validated safe IP address with TLS SNI."""
134
+
135
+ def __init__(self, host: str, resolved_ip: str, port: int = 443, **kwargs):
136
+ super().__init__(host, port=port, **kwargs)
137
+ self.resolved_ip = resolved_ip
138
+
139
+ def connect(self):
140
+ sock = socket.create_connection((self.resolved_ip, self.port), self.timeout)
141
+ self.sock = self._context.wrap_socket(sock, server_hostname=self.host)
142
+
143
+
144
+ class DomainRateLimiter:
145
+ """Thread-safe outbound domain rate limiter ensuring minimum spacing between requests."""
146
+
147
+ def __init__(self, min_interval_seconds: float = 0.3):
148
+ self.min_interval = min_interval_seconds
149
+ self._last_request: Dict[str, float] = {}
150
+ self._lock = threading.Lock()
151
+
152
+ def wait_if_needed(self, url: str):
153
+ try:
154
+ hostname = urllib.parse.urlparse(url).hostname or "default"
155
+ domain = hostname.lower()
156
+ except Exception:
157
+ domain = "default"
158
+
159
+ with self._lock:
160
+ now = time.monotonic()
161
+ last = self._last_request.get(domain, 0.0)
162
+ elapsed = now - last
163
+ if elapsed < self.min_interval:
164
+ sleep_time = self.min_interval - elapsed
165
+ else:
166
+ sleep_time = 0.0
167
+ self._last_request[domain] = now + sleep_time
168
+
169
+ if sleep_time > 0:
170
+ time.sleep(sleep_time)
171
+
172
+
173
+ GLOBAL_RATE_LIMITER = DomainRateLimiter(min_interval_seconds=0.3)
174
+
175
+
176
+ @dataclass
177
+ class ResearchResult:
178
+ """Structured research retrieval result."""
179
+ ok: bool
180
+ source_type: str
181
+ canonical_url: str
182
+ title: str = ""
183
+ author: str = ""
184
+ content: str = ""
185
+ error: Optional[str] = None
186
+ retrieved_at: str = field(default_factory=lambda: datetime.datetime.now(datetime.timezone.utc).isoformat())
187
+ metadata: Dict[str, Any] = field(default_factory=dict)
188
+
189
+ def __post_init__(self):
190
+ if "trust_level" not in self.metadata:
191
+ self.metadata["trust_level"] = "untrusted_external_content"
192
+
193
+ def to_markdown(self) -> str:
194
+ """Formats the structured research result as readable Markdown with explicit untrusted content boundaries."""
195
+ if not self.ok:
196
+ return (
197
+ f"# Research Retrieval Failed: {self.source_type.title()}\n"
198
+ f"**Canonical URL**: {self.canonical_url or 'N/A'}\n"
199
+ f"**Retrieved At**: {self.retrieved_at}\n"
200
+ f"**Error**: {self.error or 'Unknown error'}\n"
201
+ )
202
+ md = [
203
+ f"# {self.source_type.title()}: {self.title or 'Untitled'}",
204
+ f"**Canonical URL**: {self.canonical_url}",
205
+ ]
206
+ if self.author:
207
+ md.append(f"**Author**: {self.author}")
208
+ md.append(f"**Retrieved At**: {self.retrieved_at}\n")
209
+ if self.content:
210
+ md.append("[BEGIN UNTRUSTED EXTERNAL RESEARCH CONTENT]")
211
+ md.append(self.content)
212
+ md.append("[END UNTRUSTED EXTERNAL RESEARCH CONTENT]")
213
+ return "\n".join(md)
214
+
215
+
216
+ def fetch_url(
217
+ url: str,
218
+ headers: Optional[Dict[str, str]] = None,
219
+ timeout: int = 15,
220
+ max_bytes: int = 5 * 1024 * 1024,
221
+ max_retries: int = 2,
222
+ backoff_factor: float = 0.5,
223
+ ) -> str:
224
+ """
225
+ Fetches raw string content from URL using pinned IP transport:
226
+ 1. Validates host & DNS records (including alternate numeric IPs) for SSRF safety
227
+ 2. Pins connection directly to validated IP (immune to DNS rebinding)
228
+ 3. Handles HTTP 301/302/303/307/308 redirects with re-validation on every redirect
229
+ 4. Applies domain rate limiting
230
+ 5. Clamps server Retry-After delays to a hard maximum ceiling (10s)
231
+ 6. Standard TLS certificate verification & retries on 429/5xx status
232
+ """
233
+ current_url = url
234
+ max_redirects = 5
235
+
236
+ for redirect_count in range(max_redirects + 1):
237
+ safe, reason, resolved_ip = validate_safe_url(current_url)
238
+ if not safe or not resolved_ip:
239
+ raise ValueError(f"SSRF validation failed for '{current_url}': {reason}")
240
+
241
+ GLOBAL_RATE_LIMITER.wait_if_needed(current_url)
242
+
243
+ parsed = urllib.parse.urlparse(current_url)
244
+ scheme = parsed.scheme.lower()
245
+ host = parsed.hostname
246
+ port = parsed.port or (443 if scheme == "https" else 80)
247
+ path = parsed.path or "/"
248
+ if parsed.query:
249
+ path += f"?{parsed.query}"
250
+
251
+ req_headers = {"User-Agent": USER_AGENT, "Host": host}
252
+ if headers:
253
+ req_headers.update(headers)
254
+
255
+ last_exc: Optional[Exception] = None
256
+
257
+ for attempt in range(max_retries + 1):
258
+ try:
259
+ if scheme == "https":
260
+ conn = PinnedHTTPSConnection(host, resolved_ip, port=port, context=_SSL_CONTEXT, timeout=timeout)
261
+ else:
262
+ conn = PinnedHTTPConnection(host, resolved_ip, port=port, timeout=timeout)
263
+
264
+ conn.request("GET", path, headers=req_headers)
265
+ resp = conn.getresponse()
266
+
267
+ # Handle HTTP redirects safely
268
+ if resp.status in (301, 302, 303, 307, 308):
269
+ location = resp.getheader("Location")
270
+ conn.close()
271
+ if not location:
272
+ raise urllib.error.HTTPError(current_url, resp.status, "Redirect missing Location header", resp.headers, None)
273
+ current_url = urllib.parse.urljoin(current_url, location)
274
+ break # Break retry loop and execute next redirect iteration
275
+
276
+ if resp.status in (429, 500, 502, 503, 504) and attempt < max_retries:
277
+ retry_after = resp.getheader("Retry-After")
278
+ conn.close()
279
+ raw_delay = float(retry_after) if retry_after and retry_after.isdigit() else (backoff_factor * (2 ** attempt))
280
+ sleep_time = min(max(0.0, raw_delay), MAX_RETRY_DELAY_SECONDS)
281
+ time.sleep(sleep_time)
282
+ continue
283
+
284
+ if resp.status >= 400:
285
+ conn.close()
286
+ raise urllib.error.HTTPError(current_url, resp.status, f"HTTP Error {resp.status}", resp.headers, None)
287
+
288
+ raw_bytes = resp.read(max_bytes + 1)
289
+ conn.close()
290
+ if len(raw_bytes) > max_bytes:
291
+ raw_bytes = raw_bytes[:max_bytes]
292
+ charset = resp.headers.get_param("charset") or "utf-8"
293
+ return raw_bytes.decode(charset, errors="replace")
294
+
295
+ except urllib.error.HTTPError as exc:
296
+ last_exc = exc
297
+ if exc.code in (301, 302, 303, 307, 308):
298
+ raise exc
299
+ if exc.code in (429, 500, 502, 503, 504) and attempt < max_retries:
300
+ time.sleep(backoff_factor * (2 ** attempt))
301
+ continue
302
+ raise exc
303
+ except (urllib.error.URLError, TimeoutError, OSError, http.client.HTTPException) as exc:
304
+ last_exc = exc
305
+ if attempt < max_retries:
306
+ time.sleep(backoff_factor * (2 ** attempt))
307
+ continue
308
+ raise exc
309
+
310
+ # If loop exited due to redirect, continue outer redirect_count loop
311
+ if 'resp' in locals() and resp.status in (301, 302, 303, 307, 308):
312
+ continue
313
+
314
+ if last_exc:
315
+ raise last_exc
316
+
317
+ raise RuntimeError(f"Maximum redirects ({max_redirects}) exceeded for {url}")
318
+
319
+
320
+ class SimpleHTMLTextExtractor(HTMLParser):
321
+ """HTML parser converting web content to readable Markdown text."""
322
+
323
+ def __init__(self):
324
+ super().__init__()
325
+ self.title = ""
326
+ self.in_title = False
327
+ self.in_script = False
328
+ self.in_style = False
329
+ self.chunks: List[str] = []
330
+ self.links: List[Tuple[str, str]] = []
331
+ self._current_href: Optional[str] = None
332
+ self._current_link_text: List[str] = []
333
+
334
+ def handle_starttag(self, tag: str, attrs: List[Tuple[str, Optional[str]]]):
335
+ tag = tag.lower()
336
+ if tag in ("script", "style", "noscript", "svg"):
337
+ self.in_script = True
338
+ elif tag == "title":
339
+ self.in_title = True
340
+ elif tag == "a":
341
+ attr_dict = dict(attrs)
342
+ self._current_href = attr_dict.get("href")
343
+ self._current_link_text = []
344
+ elif tag in ("h1", "h2", "h3", "h4", "h5", "h6"):
345
+ self.chunks.append("\n\n### ")
346
+ elif tag in ("p", "div", "section", "article", "li", "tr"):
347
+ self.chunks.append("\n")
348
+
349
+ def handle_endtag(self, tag: str):
350
+ tag = tag.lower()
351
+ if tag in ("script", "style", "noscript", "svg"):
352
+ self.in_script = False
353
+ elif tag == "title":
354
+ self.in_title = False
355
+ elif tag == "a":
356
+ if self._current_href:
357
+ link_str = "".join(self._current_link_text).strip()
358
+ if link_str and not self._current_href.startswith("javascript:"):
359
+ self.links.append((link_str, self._current_href))
360
+ self._current_href = None
361
+ self._current_link_text = []
362
+
363
+ def handle_data(self, data: str):
364
+ if self.in_script or self.in_style:
365
+ return
366
+ if self.in_title:
367
+ self.title += data
368
+ return
369
+ text = data.strip()
370
+ if text:
371
+ if self._current_href is not None:
372
+ self._current_link_text.append(text)
373
+ self.chunks.append(text + " ")
374
+
375
+ def get_markdown(self) -> str:
376
+ raw_text = "".join(self.chunks)
377
+ cleaned = re.sub(r"\n\s*\n+", "\n\n", raw_text).strip()
378
+ return cleaned
379
+
380
+
381
+ class ResearchSource:
382
+ """Abstract base class for research scrapers."""
383
+ source_type: str = "generic"
384
+
385
+ def fetch(self, target: str, timeout: int = 15, max_content_length: int = 12000) -> ResearchResult:
386
+ raise NotImplementedError
@@ -0,0 +1,129 @@
1
+ """
2
+ GitHub repository metadata, README content, and repository search scraper implementation.
3
+ """
4
+
5
+ from __future__ import annotations
6
+
7
+ import json
8
+ import re
9
+ import urllib.parse
10
+ from typing import Optional
11
+
12
+ from fable_engine.scrapers.base import ResearchResult, ResearchSource, fetch_url
13
+
14
+
15
+ class GitHubScraper(ResearchSource):
16
+ source_type = "github"
17
+
18
+ def fetch(self, target: str, timeout: int = 15, max_content_length: int = 8000) -> ResearchResult:
19
+ target = target.strip()
20
+ if not target:
21
+ return ResearchResult(
22
+ ok=False,
23
+ source_type=self.source_type,
24
+ canonical_url="",
25
+ error="Target GitHub repository or search query cannot be empty."
26
+ )
27
+
28
+ repo_path = None
29
+ if "github.com/" in target:
30
+ parsed = urllib.parse.urlparse(target)
31
+ parts = [p for p in parsed.path.strip("/").split("/") if p]
32
+ if len(parts) >= 2:
33
+ repo_path = f"{parts[0]}/{parts[1]}"
34
+ elif re.match(r"^[A-Za-z0-9_.-]+/[A-Za-z0-9_.-]+$", target):
35
+ repo_path = target
36
+
37
+ if repo_path:
38
+ api_url = f"https://api.github.com/repos/{repo_path}"
39
+ canonical_url = f"https://github.com/{repo_path}"
40
+ try:
41
+ repo_json = fetch_url(api_url, headers={"Accept": "application/vnd.github.v3+json"}, timeout=timeout)
42
+ data = json.loads(repo_json)
43
+
44
+ name = data.get("full_name", repo_path)
45
+ description = data.get("description", "No description provided.")
46
+ stars = data.get("stargazers_count", 0)
47
+ forks = data.get("forks_count", 0)
48
+ language = data.get("language", "Unknown")
49
+ default_branch = data.get("default_branch", "main")
50
+ html_url = data.get("html_url", canonical_url)
51
+
52
+ content_blocks = [
53
+ f"**Stars**: {stars} | **Forks**: {forks} | **Language**: {language} | **Branch**: {default_branch}\n",
54
+ f"**Description**: {description}\n"
55
+ ]
56
+
57
+ # Fetch README content
58
+ readme_url = f"https://raw.githubusercontent.com/{repo_path}/{default_branch}/README.md"
59
+ try:
60
+ readme_text = fetch_url(readme_url, timeout=timeout)
61
+ if len(readme_text) > max_content_length:
62
+ readme_text = readme_text[:max_content_length] + "\n\n*(README truncated for size)*"
63
+ content_blocks.append(f"## README.md\n{readme_text}")
64
+ except Exception:
65
+ content_blocks.append("*(Note: README.md could not be retrieved automatically)*")
66
+
67
+ return ResearchResult(
68
+ ok=True,
69
+ source_type=self.source_type,
70
+ canonical_url=html_url,
71
+ title=f"GitHub Repo: {name}",
72
+ author=repo_path.split("/")[0],
73
+ content="\n".join(content_blocks),
74
+ metadata={"stars": stars, "language": language}
75
+ )
76
+ except Exception as e:
77
+ return ResearchResult(
78
+ ok=False,
79
+ source_type=self.source_type,
80
+ canonical_url=canonical_url,
81
+ error=f"GitHub repository fetch failed: {str(e)}"
82
+ )
83
+
84
+ # Repository Search Fallback
85
+ query_enc = urllib.parse.quote(target)
86
+ search_url = f"https://api.github.com/search/repositories?q={query_enc}&per_page=10"
87
+ canonical_url = f"https://github.com/search?q={query_enc}"
88
+ try:
89
+ search_json = fetch_url(search_url, headers={"Accept": "application/vnd.github.v3+json"}, timeout=timeout)
90
+ data = json.loads(search_json)
91
+ items = data.get("items", [])
92
+
93
+ if not items:
94
+ return ResearchResult(
95
+ ok=True,
96
+ source_type=self.source_type,
97
+ canonical_url=canonical_url,
98
+ title=f"GitHub Search: '{target}'",
99
+ content="*(No repositories found for this query)*"
100
+ )
101
+
102
+ lines = ["## Repositories\n"]
103
+ for item in items[:10]:
104
+ full_name = item.get("full_name", "")
105
+ desc = item.get("description", "") or "No description"
106
+ stars = item.get("stargazers_count", 0)
107
+ lang = item.get("language", "N/A")
108
+ url = item.get("html_url", "")
109
+ lines.append(f"### [{full_name}]({url})")
110
+ lines.append(f"**Stars**: {stars} | **Language**: {lang}\n**Description**: {desc}\n")
111
+
112
+ return ResearchResult(
113
+ ok=True,
114
+ source_type=self.source_type,
115
+ canonical_url=canonical_url,
116
+ title=f"GitHub Repository Search: '{target}'",
117
+ content="\n".join(lines)
118
+ )
119
+ except Exception as e:
120
+ return ResearchResult(
121
+ ok=False,
122
+ source_type=self.source_type,
123
+ canonical_url=canonical_url,
124
+ error=f"GitHub repository search failed: {str(e)}"
125
+ )
126
+
127
+
128
+ def scrape_github(target: str, timeout: int = 15) -> ResearchResult:
129
+ return GitHubScraper().fetch(target, timeout=timeout)