fable-engine 1.3.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- fable_compressor.py +356 -0
- fable_engine/__init__.py +1 -0
- fable_engine/actions/__init__.py +291 -0
- fable_engine/actions/cas.py +182 -0
- fable_engine/actions/deliberation.py +523 -0
- fable_engine/actions/fleet.py +807 -0
- fable_engine/actions/lifecycle.py +298 -0
- fable_engine/actions/scrapers.py +116 -0
- fable_engine/actions/system3.py +815 -0
- fable_engine/browser.py +824 -0
- fable_engine/cas.py +974 -0
- fable_engine/fable_session.json +510 -0
- fable_engine/guards.py +283 -0
- fable_engine/schema.py +714 -0
- fable_engine/scrapers/__init__.py +32 -0
- fable_engine/scrapers/arxiv.py +115 -0
- fable_engine/scrapers/base.py +386 -0
- fable_engine/scrapers/github.py +129 -0
- fable_engine/scrapers/reddit.py +154 -0
- fable_engine/scrapers/web.py +120 -0
- fable_engine/scrapers/x.py +125 -0
- fable_engine/scrapers/youtube.py +132 -0
- fable_engine/server.py +414 -0
- fable_engine/session.py +1819 -0
- fable_engine/test_server.py +1362 -0
- fable_engine/updater.py +541 -0
- fable_engine-1.3.1.dist-info/LICENSE +22 -0
- fable_engine-1.3.1.dist-info/METADATA +173 -0
- fable_engine-1.3.1.dist-info/RECORD +104 -0
- fable_engine-1.3.1.dist-info/WHEEL +5 -0
- fable_engine-1.3.1.dist-info/entry_points.txt +5 -0
- fable_engine-1.3.1.dist-info/top_level.txt +6 -0
- fable_mode/__init__.py +3 -0
- fable_mode/__main__.py +4 -0
- fable_mode/adapters.py +1014 -0
- fable_mode/installer.py +553 -0
- fable_mode/launcher.py +437 -0
- fable_mode/manifest.py +142 -0
- fable_mode/resources.json +114 -0
- fable_mode/safety.py +103 -0
- fable_mode_entry.py +10 -0
- fable_v2/__init__.py +146 -0
- fable_v2/adapters.py +151 -0
- fable_v2/coder_fleet/__init__.py +100 -0
- fable_v2/coder_fleet/ast_tools.py +158 -0
- fable_v2/coder_fleet/compute.py +199 -0
- fable_v2/coder_fleet/design_engine.py +1316 -0
- fable_v2/coder_fleet/diagnostics.py +293 -0
- fable_v2/coder_fleet/fleet_dispatcher.py +214 -0
- fable_v2/coder_fleet/mock_auditor.py +306 -0
- fable_v2/coder_fleet/mutation.py +216 -0
- fable_v2/coder_fleet/property_oracle.py +260 -0
- fable_v2/coder_fleet/receipt_attestor.py +122 -0
- fable_v2/coder_fleet/red_team_swarm.py +908 -0
- fable_v2/coder_fleet/test_harness.py +198 -0
- fable_v2/coder_fleet/vector_engine.py +1287 -0
- fable_v2/coder_fleet/visual.py +357 -0
- fable_v2/coder_fleet/workspace.py +153 -0
- fable_v2/cortical/__init__.py +20 -0
- fable_v2/cortical/plasticity_engine.py +992 -0
- fable_v2/execution_broker.py +811 -0
- fable_v2/proof_engine.py +1141 -0
- fable_v2/protocol.py +485 -0
- fable_v2/runtime.py +1010 -0
- fable_v2/system3/__init__.py +204 -0
- fable_v2/system3/causal.py +558 -0
- fable_v2/system3/dialectical.py +577 -0
- fable_v2/system3/evolution.py +503 -0
- fable_v2/system3/executive.py +338 -0
- fable_v2/system3/free_energy.py +479 -0
- fable_v2/system3/hyperbolic.py +555 -0
- fable_v2/system3/induction.py +336 -0
- fable_v2/system3/kripke.py +548 -0
- fable_v2/system3/oracle.py +745 -0
- fable_v2/verifiers.py +72 -0
- tests/__init__.py +1 -0
- tests/test_anti_loop_circuit_breaker.py +64 -0
- tests/test_auto_updater.py +407 -0
- tests/test_coder_fleet.py +535 -0
- tests/test_delegation_compiler.py +54 -0
- tests/test_descriptor_boundaries.py +126 -0
- tests/test_design_engine.py +603 -0
- tests/test_epistemic_evidence_validator.py +66 -0
- tests/test_execution_broker.py +233 -0
- tests/test_fable_v2.py +406 -0
- tests/test_fleet_transitions.py +116 -0
- tests/test_fsm_redteam_evolution.py +406 -0
- tests/test_goal_rubric_and_pipeline.py +367 -0
- tests/test_hebbian_plasticity.py +585 -0
- tests/test_packaging_runtime.py +194 -0
- tests/test_proof_engine.py +259 -0
- tests/test_red_team_swarm.py +645 -0
- tests/test_redteam_remediation.py +169 -0
- tests/test_registration_transaction.py +375 -0
- tests/test_requested_regressions.py +467 -0
- tests/test_scrapers.py +370 -0
- tests/test_server_actions.py +93 -0
- tests/test_server_frontier_actions.py +269 -0
- tests/test_server_protocol.py +88 -0
- tests/test_stealth_browser.py +970 -0
- tests/test_system3.py +381 -0
- tests/test_system3_deep_integration.py +385 -0
- tests/test_system3_frontier.py +436 -0
- tests/test_vector_engine.py +608 -0
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Fable Engine Research Scrapers Package.
|
|
3
|
+
Exporting zero-cost, standard-library based scrapers for Web, YouTube, Reddit, X, GitHub, and arXiv.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
from fable_engine.scrapers.arxiv import ArxivScraper, scrape_arxiv
|
|
9
|
+
from fable_engine.scrapers.base import ResearchResult, ResearchSource, fetch_url
|
|
10
|
+
from fable_engine.scrapers.github import GitHubScraper, scrape_github
|
|
11
|
+
from fable_engine.scrapers.reddit import RedditScraper, scrape_reddit
|
|
12
|
+
from fable_engine.scrapers.web import WebScraper, scrape_web
|
|
13
|
+
from fable_engine.scrapers.x import XScraper, scrape_x
|
|
14
|
+
from fable_engine.scrapers.youtube import YouTubeScraper, scrape_youtube
|
|
15
|
+
|
|
16
|
+
__all__ = [
|
|
17
|
+
"ResearchResult",
|
|
18
|
+
"ResearchSource",
|
|
19
|
+
"fetch_url",
|
|
20
|
+
"WebScraper",
|
|
21
|
+
"scrape_web",
|
|
22
|
+
"YouTubeScraper",
|
|
23
|
+
"scrape_youtube",
|
|
24
|
+
"RedditScraper",
|
|
25
|
+
"scrape_reddit",
|
|
26
|
+
"XScraper",
|
|
27
|
+
"scrape_x",
|
|
28
|
+
"GitHubScraper",
|
|
29
|
+
"scrape_github",
|
|
30
|
+
"ArxivScraper",
|
|
31
|
+
"scrape_arxiv",
|
|
32
|
+
]
|
|
@@ -0,0 +1,115 @@
|
|
|
1
|
+
"""
|
|
2
|
+
arXiv research paper and search query scraper implementation using arXiv export API.
|
|
3
|
+
"""
|
|
4
|
+
|
|
5
|
+
from __future__ import annotations
|
|
6
|
+
|
|
7
|
+
import re
|
|
8
|
+
import urllib.parse
|
|
9
|
+
import xml.etree.ElementTree as ET
|
|
10
|
+
from typing import Optional
|
|
11
|
+
|
|
12
|
+
from fable_engine.scrapers.base import ResearchResult, ResearchSource, fetch_url
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class ArxivScraper(ResearchSource):
|
|
16
|
+
source_type = "arxiv"
|
|
17
|
+
|
|
18
|
+
def fetch(self, target: str, timeout: int = 15, max_content_length: int = 10000) -> ResearchResult:
|
|
19
|
+
target = target.strip()
|
|
20
|
+
if not target:
|
|
21
|
+
return ResearchResult(
|
|
22
|
+
ok=False,
|
|
23
|
+
source_type=self.source_type,
|
|
24
|
+
canonical_url="",
|
|
25
|
+
error="Target arXiv paper ID or search query cannot be empty."
|
|
26
|
+
)
|
|
27
|
+
|
|
28
|
+
paper_id = None
|
|
29
|
+
if "arxiv.org/" in target:
|
|
30
|
+
id_match = re.search(r"(?:abs|pdf)/([0-9]+\.[0-9]+(?:v[0-9]+)?)", target)
|
|
31
|
+
if id_match:
|
|
32
|
+
paper_id = id_match.group(1)
|
|
33
|
+
elif re.match(r"^[0-9]{4}\.[0-9]{4,5}(?:v[0-9]+)?$", target):
|
|
34
|
+
paper_id = target
|
|
35
|
+
|
|
36
|
+
if paper_id:
|
|
37
|
+
api_url = f"https://export.arxiv.org/api/query?id_list={paper_id}"
|
|
38
|
+
canonical_url = f"https://arxiv.org/abs/{paper_id}"
|
|
39
|
+
else:
|
|
40
|
+
query_enc = urllib.parse.quote(target)
|
|
41
|
+
api_url = f"https://export.arxiv.org/api/query?search_query=all:{query_enc}&max_results=5"
|
|
42
|
+
canonical_url = f"https://arxiv.org/search/?query={query_enc}&searchtype=all"
|
|
43
|
+
|
|
44
|
+
try:
|
|
45
|
+
xml_str = fetch_url(api_url, timeout=timeout)
|
|
46
|
+
root = ET.fromstring(xml_str)
|
|
47
|
+
ns = {"atom": "http://www.w3.org/2005/Atom", "arxiv": "http://arxiv.org/schemas/atom"}
|
|
48
|
+
|
|
49
|
+
entries = root.findall("atom:entry", ns)
|
|
50
|
+
if not entries:
|
|
51
|
+
return ResearchResult(
|
|
52
|
+
ok=True,
|
|
53
|
+
source_type=self.source_type,
|
|
54
|
+
canonical_url=canonical_url,
|
|
55
|
+
title=f"arXiv Search: '{target}'",
|
|
56
|
+
content="*(No papers found for this query)*"
|
|
57
|
+
)
|
|
58
|
+
|
|
59
|
+
blocks = []
|
|
60
|
+
first_title = ""
|
|
61
|
+
first_author = ""
|
|
62
|
+
for entry in entries:
|
|
63
|
+
title_elem = entry.find("atom:title", ns)
|
|
64
|
+
title = re.sub(r"\s+", " ", title_elem.text).strip() if title_elem is not None and title_elem.text else "Untitled"
|
|
65
|
+
if not first_title:
|
|
66
|
+
first_title = title
|
|
67
|
+
|
|
68
|
+
id_elem = entry.find("atom:id", ns)
|
|
69
|
+
entry_id = id_elem.text.strip() if id_elem is not None and id_elem.text else ""
|
|
70
|
+
|
|
71
|
+
published_elem = entry.find("atom:published", ns)
|
|
72
|
+
published = published_elem.text[:10] if published_elem is not None and published_elem.text else ""
|
|
73
|
+
|
|
74
|
+
summary_elem = entry.find("atom:summary", ns)
|
|
75
|
+
summary = re.sub(r"\s+", " ", summary_elem.text).strip() if summary_elem is not None and summary_elem.text else ""
|
|
76
|
+
|
|
77
|
+
authors = []
|
|
78
|
+
for author_elem in entry.findall("atom:author", ns):
|
|
79
|
+
name_elem = author_elem.find("atom:name", ns)
|
|
80
|
+
if name_elem is not None and name_elem.text:
|
|
81
|
+
authors.append(name_elem.text.strip())
|
|
82
|
+
if not first_author and authors:
|
|
83
|
+
first_author = ", ".join(authors)
|
|
84
|
+
|
|
85
|
+
pdf_link = entry_id.replace("/abs/", "/pdf/") + ".pdf" if "/abs/" in entry_id else entry_id
|
|
86
|
+
|
|
87
|
+
blocks.append(f"## {title}")
|
|
88
|
+
blocks.append(f"**Authors**: {', '.join(authors)}")
|
|
89
|
+
blocks.append(f"**Published**: {published} | **arXiv**: [{entry_id}]({entry_id}) | **PDF**: [{pdf_link}]({pdf_link})\n")
|
|
90
|
+
blocks.append(f"### Abstract\n{summary}\n")
|
|
91
|
+
|
|
92
|
+
full_content = "\n".join(blocks)
|
|
93
|
+
if len(full_content) > max_content_length:
|
|
94
|
+
full_content = full_content[:max_content_length] + "\n\n*(Abstracts truncated)*"
|
|
95
|
+
|
|
96
|
+
return ResearchResult(
|
|
97
|
+
ok=True,
|
|
98
|
+
source_type=self.source_type,
|
|
99
|
+
canonical_url=canonical_url,
|
|
100
|
+
title=first_title if paper_id else f"arXiv Papers for: '{target}'",
|
|
101
|
+
author=first_author,
|
|
102
|
+
content=full_content,
|
|
103
|
+
metadata={"paper_count": len(entries)}
|
|
104
|
+
)
|
|
105
|
+
except Exception as e:
|
|
106
|
+
return ResearchResult(
|
|
107
|
+
ok=False,
|
|
108
|
+
source_type=self.source_type,
|
|
109
|
+
canonical_url=canonical_url,
|
|
110
|
+
error=f"arXiv fetch failed: {str(e)}"
|
|
111
|
+
)
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def scrape_arxiv(target: str, timeout: int = 15) -> ResearchResult:
|
|
115
|
+
return ArxivScraper().fetch(target, timeout=timeout)
|
|
@@ -0,0 +1,386 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Base research scraper interfaces, structured result objects, and HTTP transport utilities.
|
|
3
|
+
Enforces standard TLS certificate verification, strict SSRF protection with alternate IP
|
|
4
|
+
canonicalization & IP pinning (preventing DNS rebinding TOCTOU window), host rate limiting,
|
|
5
|
+
clamped retry/backoff, response bounds, and explicit untrusted external content boundaries.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import datetime
|
|
11
|
+
import http.client
|
|
12
|
+
import ipaddress
|
|
13
|
+
import json
|
|
14
|
+
import logging
|
|
15
|
+
import re
|
|
16
|
+
import socket
|
|
17
|
+
import ssl
|
|
18
|
+
import threading
|
|
19
|
+
import time
|
|
20
|
+
import urllib.error
|
|
21
|
+
import urllib.parse
|
|
22
|
+
from dataclasses import dataclass, field
|
|
23
|
+
from html.parser import HTMLParser
|
|
24
|
+
from typing import Any, Dict, List, Optional, Tuple
|
|
25
|
+
|
|
26
|
+
logger = logging.getLogger("fable-engine.scrapers.base")
|
|
27
|
+
|
|
28
|
+
USER_AGENT = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/122.0.0.0 Safari/537.36 FableResearch/1.3.0"
|
|
29
|
+
|
|
30
|
+
MAX_RETRY_DELAY_SECONDS = 10.0
|
|
31
|
+
|
|
32
|
+
# Standard TLS context enforcing certificate verification
|
|
33
|
+
_SSL_CONTEXT = ssl.create_default_context()
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def is_safe_ip(ip_str: str) -> bool:
|
|
37
|
+
"""Verifies that an IP address is public and safe (not loopback, private, link-local, reserved)."""
|
|
38
|
+
try:
|
|
39
|
+
ip = ipaddress.ip_address(ip_str)
|
|
40
|
+
if (
|
|
41
|
+
ip.is_loopback
|
|
42
|
+
or ip.is_private
|
|
43
|
+
or ip.is_link_local
|
|
44
|
+
or ip.is_reserved
|
|
45
|
+
or ip.is_multicast
|
|
46
|
+
or ip.is_unspecified
|
|
47
|
+
):
|
|
48
|
+
return False
|
|
49
|
+
# Cloud metadata service check (169.254.169.254)
|
|
50
|
+
if str(ip) == "169.254.169.254":
|
|
51
|
+
return False
|
|
52
|
+
return True
|
|
53
|
+
except ValueError:
|
|
54
|
+
return False
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def parse_canonical_ip(hostname_str: str) -> Optional[str]:
|
|
58
|
+
"""Attempts to parse hostname as standard or alternate IPv4/IPv6 address (hex, octal, short IPv4)."""
|
|
59
|
+
try:
|
|
60
|
+
return str(ipaddress.ip_address(hostname_str))
|
|
61
|
+
except ValueError:
|
|
62
|
+
pass
|
|
63
|
+
try:
|
|
64
|
+
raw_bytes = socket.inet_aton(hostname_str)
|
|
65
|
+
return str(ipaddress.ip_address(raw_bytes))
|
|
66
|
+
except (socket.error, ValueError, OverflowError):
|
|
67
|
+
return None
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def validate_safe_url(url: str) -> Tuple[bool, str, Optional[str]]:
|
|
71
|
+
"""
|
|
72
|
+
Validates that a URL uses http/https scheme and its hostname resolves
|
|
73
|
+
exclusively to safe, public IP addresses (SSRF prevention).
|
|
74
|
+
Returns (is_safe, reason, canonical_resolved_ip).
|
|
75
|
+
"""
|
|
76
|
+
try:
|
|
77
|
+
parsed = urllib.parse.urlparse(url)
|
|
78
|
+
scheme = parsed.scheme.lower()
|
|
79
|
+
if scheme not in ("http", "https"):
|
|
80
|
+
return False, f"Invalid URL scheme '{parsed.scheme}'. Strictly http and https are permitted.", None
|
|
81
|
+
|
|
82
|
+
hostname = parsed.hostname
|
|
83
|
+
if not hostname:
|
|
84
|
+
return False, "URL missing valid hostname.", None
|
|
85
|
+
|
|
86
|
+
hostname_clean = hostname.strip().lower()
|
|
87
|
+
|
|
88
|
+
# Reject explicit local hostnames
|
|
89
|
+
if hostname_clean in ("localhost", "localhost.localdomain") or hostname_clean.endswith(".local"):
|
|
90
|
+
return False, f"SSRF blocked: local hostname '{hostname_clean}' is not permitted.", None
|
|
91
|
+
|
|
92
|
+
# Check if hostname is a standard or alternate numeric IP string (octal, hex, decimal)
|
|
93
|
+
canonical_ip = parse_canonical_ip(hostname_clean)
|
|
94
|
+
if canonical_ip is not None:
|
|
95
|
+
if not is_safe_ip(canonical_ip):
|
|
96
|
+
return False, f"SSRF blocked: target IP '{hostname_clean}' ({canonical_ip}) is private, loopback, or reserved.", None
|
|
97
|
+
return True, "ok", canonical_ip
|
|
98
|
+
|
|
99
|
+
# Resolve hostname DNS records
|
|
100
|
+
port = parsed.port or (443 if scheme == "https" else 80)
|
|
101
|
+
try:
|
|
102
|
+
addr_info = socket.getaddrinfo(hostname_clean, port, socket.AF_UNSPEC, socket.SOCK_STREAM)
|
|
103
|
+
except socket.gaierror as e:
|
|
104
|
+
return False, f"DNS resolution failed for host '{hostname_clean}': {e}", None
|
|
105
|
+
|
|
106
|
+
if not addr_info:
|
|
107
|
+
return False, f"Could not resolve IP addresses for host '{hostname_clean}'.", None
|
|
108
|
+
|
|
109
|
+
safe_ips = []
|
|
110
|
+
for family, _, _, _, sockaddr in addr_info:
|
|
111
|
+
ip_str = sockaddr[0]
|
|
112
|
+
if not is_safe_ip(ip_str):
|
|
113
|
+
return False, f"SSRF blocked: host '{hostname_clean}' resolved to restricted IP '{ip_str}'.", None
|
|
114
|
+
safe_ips.append(ip_str)
|
|
115
|
+
|
|
116
|
+
return True, "ok", safe_ips[0]
|
|
117
|
+
except Exception as e:
|
|
118
|
+
return False, f"URL validation error: {e}", None
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
class PinnedHTTPConnection(http.client.HTTPConnection):
|
|
122
|
+
"""HTTP connection connecting directly to a pre-validated safe IP address."""
|
|
123
|
+
|
|
124
|
+
def __init__(self, host: str, resolved_ip: str, port: int = 80, **kwargs):
|
|
125
|
+
super().__init__(host, port=port, **kwargs)
|
|
126
|
+
self.resolved_ip = resolved_ip
|
|
127
|
+
|
|
128
|
+
def connect(self):
|
|
129
|
+
self.sock = socket.create_connection((self.resolved_ip, self.port), self.timeout)
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
class PinnedHTTPSConnection(http.client.HTTPSConnection):
|
|
133
|
+
"""HTTPS connection connecting directly to a pre-validated safe IP address with TLS SNI."""
|
|
134
|
+
|
|
135
|
+
def __init__(self, host: str, resolved_ip: str, port: int = 443, **kwargs):
|
|
136
|
+
super().__init__(host, port=port, **kwargs)
|
|
137
|
+
self.resolved_ip = resolved_ip
|
|
138
|
+
|
|
139
|
+
def connect(self):
|
|
140
|
+
sock = socket.create_connection((self.resolved_ip, self.port), self.timeout)
|
|
141
|
+
self.sock = self._context.wrap_socket(sock, server_hostname=self.host)
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
class DomainRateLimiter:
|
|
145
|
+
"""Thread-safe outbound domain rate limiter ensuring minimum spacing between requests."""
|
|
146
|
+
|
|
147
|
+
def __init__(self, min_interval_seconds: float = 0.3):
|
|
148
|
+
self.min_interval = min_interval_seconds
|
|
149
|
+
self._last_request: Dict[str, float] = {}
|
|
150
|
+
self._lock = threading.Lock()
|
|
151
|
+
|
|
152
|
+
def wait_if_needed(self, url: str):
|
|
153
|
+
try:
|
|
154
|
+
hostname = urllib.parse.urlparse(url).hostname or "default"
|
|
155
|
+
domain = hostname.lower()
|
|
156
|
+
except Exception:
|
|
157
|
+
domain = "default"
|
|
158
|
+
|
|
159
|
+
with self._lock:
|
|
160
|
+
now = time.monotonic()
|
|
161
|
+
last = self._last_request.get(domain, 0.0)
|
|
162
|
+
elapsed = now - last
|
|
163
|
+
if elapsed < self.min_interval:
|
|
164
|
+
sleep_time = self.min_interval - elapsed
|
|
165
|
+
else:
|
|
166
|
+
sleep_time = 0.0
|
|
167
|
+
self._last_request[domain] = now + sleep_time
|
|
168
|
+
|
|
169
|
+
if sleep_time > 0:
|
|
170
|
+
time.sleep(sleep_time)
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
GLOBAL_RATE_LIMITER = DomainRateLimiter(min_interval_seconds=0.3)
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
@dataclass
|
|
177
|
+
class ResearchResult:
|
|
178
|
+
"""Structured research retrieval result."""
|
|
179
|
+
ok: bool
|
|
180
|
+
source_type: str
|
|
181
|
+
canonical_url: str
|
|
182
|
+
title: str = ""
|
|
183
|
+
author: str = ""
|
|
184
|
+
content: str = ""
|
|
185
|
+
error: Optional[str] = None
|
|
186
|
+
retrieved_at: str = field(default_factory=lambda: datetime.datetime.now(datetime.timezone.utc).isoformat())
|
|
187
|
+
metadata: Dict[str, Any] = field(default_factory=dict)
|
|
188
|
+
|
|
189
|
+
def __post_init__(self):
|
|
190
|
+
if "trust_level" not in self.metadata:
|
|
191
|
+
self.metadata["trust_level"] = "untrusted_external_content"
|
|
192
|
+
|
|
193
|
+
def to_markdown(self) -> str:
|
|
194
|
+
"""Formats the structured research result as readable Markdown with explicit untrusted content boundaries."""
|
|
195
|
+
if not self.ok:
|
|
196
|
+
return (
|
|
197
|
+
f"# Research Retrieval Failed: {self.source_type.title()}\n"
|
|
198
|
+
f"**Canonical URL**: {self.canonical_url or 'N/A'}\n"
|
|
199
|
+
f"**Retrieved At**: {self.retrieved_at}\n"
|
|
200
|
+
f"**Error**: {self.error or 'Unknown error'}\n"
|
|
201
|
+
)
|
|
202
|
+
md = [
|
|
203
|
+
f"# {self.source_type.title()}: {self.title or 'Untitled'}",
|
|
204
|
+
f"**Canonical URL**: {self.canonical_url}",
|
|
205
|
+
]
|
|
206
|
+
if self.author:
|
|
207
|
+
md.append(f"**Author**: {self.author}")
|
|
208
|
+
md.append(f"**Retrieved At**: {self.retrieved_at}\n")
|
|
209
|
+
if self.content:
|
|
210
|
+
md.append("[BEGIN UNTRUSTED EXTERNAL RESEARCH CONTENT]")
|
|
211
|
+
md.append(self.content)
|
|
212
|
+
md.append("[END UNTRUSTED EXTERNAL RESEARCH CONTENT]")
|
|
213
|
+
return "\n".join(md)
|
|
214
|
+
|
|
215
|
+
|
|
216
|
+
def fetch_url(
|
|
217
|
+
url: str,
|
|
218
|
+
headers: Optional[Dict[str, str]] = None,
|
|
219
|
+
timeout: int = 15,
|
|
220
|
+
max_bytes: int = 5 * 1024 * 1024,
|
|
221
|
+
max_retries: int = 2,
|
|
222
|
+
backoff_factor: float = 0.5,
|
|
223
|
+
) -> str:
|
|
224
|
+
"""
|
|
225
|
+
Fetches raw string content from URL using pinned IP transport:
|
|
226
|
+
1. Validates host & DNS records (including alternate numeric IPs) for SSRF safety
|
|
227
|
+
2. Pins connection directly to validated IP (immune to DNS rebinding)
|
|
228
|
+
3. Handles HTTP 301/302/303/307/308 redirects with re-validation on every redirect
|
|
229
|
+
4. Applies domain rate limiting
|
|
230
|
+
5. Clamps server Retry-After delays to a hard maximum ceiling (10s)
|
|
231
|
+
6. Standard TLS certificate verification & retries on 429/5xx status
|
|
232
|
+
"""
|
|
233
|
+
current_url = url
|
|
234
|
+
max_redirects = 5
|
|
235
|
+
|
|
236
|
+
for redirect_count in range(max_redirects + 1):
|
|
237
|
+
safe, reason, resolved_ip = validate_safe_url(current_url)
|
|
238
|
+
if not safe or not resolved_ip:
|
|
239
|
+
raise ValueError(f"SSRF validation failed for '{current_url}': {reason}")
|
|
240
|
+
|
|
241
|
+
GLOBAL_RATE_LIMITER.wait_if_needed(current_url)
|
|
242
|
+
|
|
243
|
+
parsed = urllib.parse.urlparse(current_url)
|
|
244
|
+
scheme = parsed.scheme.lower()
|
|
245
|
+
host = parsed.hostname
|
|
246
|
+
port = parsed.port or (443 if scheme == "https" else 80)
|
|
247
|
+
path = parsed.path or "/"
|
|
248
|
+
if parsed.query:
|
|
249
|
+
path += f"?{parsed.query}"
|
|
250
|
+
|
|
251
|
+
req_headers = {"User-Agent": USER_AGENT, "Host": host}
|
|
252
|
+
if headers:
|
|
253
|
+
req_headers.update(headers)
|
|
254
|
+
|
|
255
|
+
last_exc: Optional[Exception] = None
|
|
256
|
+
|
|
257
|
+
for attempt in range(max_retries + 1):
|
|
258
|
+
try:
|
|
259
|
+
if scheme == "https":
|
|
260
|
+
conn = PinnedHTTPSConnection(host, resolved_ip, port=port, context=_SSL_CONTEXT, timeout=timeout)
|
|
261
|
+
else:
|
|
262
|
+
conn = PinnedHTTPConnection(host, resolved_ip, port=port, timeout=timeout)
|
|
263
|
+
|
|
264
|
+
conn.request("GET", path, headers=req_headers)
|
|
265
|
+
resp = conn.getresponse()
|
|
266
|
+
|
|
267
|
+
# Handle HTTP redirects safely
|
|
268
|
+
if resp.status in (301, 302, 303, 307, 308):
|
|
269
|
+
location = resp.getheader("Location")
|
|
270
|
+
conn.close()
|
|
271
|
+
if not location:
|
|
272
|
+
raise urllib.error.HTTPError(current_url, resp.status, "Redirect missing Location header", resp.headers, None)
|
|
273
|
+
current_url = urllib.parse.urljoin(current_url, location)
|
|
274
|
+
break # Break retry loop and execute next redirect iteration
|
|
275
|
+
|
|
276
|
+
if resp.status in (429, 500, 502, 503, 504) and attempt < max_retries:
|
|
277
|
+
retry_after = resp.getheader("Retry-After")
|
|
278
|
+
conn.close()
|
|
279
|
+
raw_delay = float(retry_after) if retry_after and retry_after.isdigit() else (backoff_factor * (2 ** attempt))
|
|
280
|
+
sleep_time = min(max(0.0, raw_delay), MAX_RETRY_DELAY_SECONDS)
|
|
281
|
+
time.sleep(sleep_time)
|
|
282
|
+
continue
|
|
283
|
+
|
|
284
|
+
if resp.status >= 400:
|
|
285
|
+
conn.close()
|
|
286
|
+
raise urllib.error.HTTPError(current_url, resp.status, f"HTTP Error {resp.status}", resp.headers, None)
|
|
287
|
+
|
|
288
|
+
raw_bytes = resp.read(max_bytes + 1)
|
|
289
|
+
conn.close()
|
|
290
|
+
if len(raw_bytes) > max_bytes:
|
|
291
|
+
raw_bytes = raw_bytes[:max_bytes]
|
|
292
|
+
charset = resp.headers.get_param("charset") or "utf-8"
|
|
293
|
+
return raw_bytes.decode(charset, errors="replace")
|
|
294
|
+
|
|
295
|
+
except urllib.error.HTTPError as exc:
|
|
296
|
+
last_exc = exc
|
|
297
|
+
if exc.code in (301, 302, 303, 307, 308):
|
|
298
|
+
raise exc
|
|
299
|
+
if exc.code in (429, 500, 502, 503, 504) and attempt < max_retries:
|
|
300
|
+
time.sleep(backoff_factor * (2 ** attempt))
|
|
301
|
+
continue
|
|
302
|
+
raise exc
|
|
303
|
+
except (urllib.error.URLError, TimeoutError, OSError, http.client.HTTPException) as exc:
|
|
304
|
+
last_exc = exc
|
|
305
|
+
if attempt < max_retries:
|
|
306
|
+
time.sleep(backoff_factor * (2 ** attempt))
|
|
307
|
+
continue
|
|
308
|
+
raise exc
|
|
309
|
+
|
|
310
|
+
# If loop exited due to redirect, continue outer redirect_count loop
|
|
311
|
+
if 'resp' in locals() and resp.status in (301, 302, 303, 307, 308):
|
|
312
|
+
continue
|
|
313
|
+
|
|
314
|
+
if last_exc:
|
|
315
|
+
raise last_exc
|
|
316
|
+
|
|
317
|
+
raise RuntimeError(f"Maximum redirects ({max_redirects}) exceeded for {url}")
|
|
318
|
+
|
|
319
|
+
|
|
320
|
+
class SimpleHTMLTextExtractor(HTMLParser):
|
|
321
|
+
"""HTML parser converting web content to readable Markdown text."""
|
|
322
|
+
|
|
323
|
+
def __init__(self):
|
|
324
|
+
super().__init__()
|
|
325
|
+
self.title = ""
|
|
326
|
+
self.in_title = False
|
|
327
|
+
self.in_script = False
|
|
328
|
+
self.in_style = False
|
|
329
|
+
self.chunks: List[str] = []
|
|
330
|
+
self.links: List[Tuple[str, str]] = []
|
|
331
|
+
self._current_href: Optional[str] = None
|
|
332
|
+
self._current_link_text: List[str] = []
|
|
333
|
+
|
|
334
|
+
def handle_starttag(self, tag: str, attrs: List[Tuple[str, Optional[str]]]):
|
|
335
|
+
tag = tag.lower()
|
|
336
|
+
if tag in ("script", "style", "noscript", "svg"):
|
|
337
|
+
self.in_script = True
|
|
338
|
+
elif tag == "title":
|
|
339
|
+
self.in_title = True
|
|
340
|
+
elif tag == "a":
|
|
341
|
+
attr_dict = dict(attrs)
|
|
342
|
+
self._current_href = attr_dict.get("href")
|
|
343
|
+
self._current_link_text = []
|
|
344
|
+
elif tag in ("h1", "h2", "h3", "h4", "h5", "h6"):
|
|
345
|
+
self.chunks.append("\n\n### ")
|
|
346
|
+
elif tag in ("p", "div", "section", "article", "li", "tr"):
|
|
347
|
+
self.chunks.append("\n")
|
|
348
|
+
|
|
349
|
+
def handle_endtag(self, tag: str):
|
|
350
|
+
tag = tag.lower()
|
|
351
|
+
if tag in ("script", "style", "noscript", "svg"):
|
|
352
|
+
self.in_script = False
|
|
353
|
+
elif tag == "title":
|
|
354
|
+
self.in_title = False
|
|
355
|
+
elif tag == "a":
|
|
356
|
+
if self._current_href:
|
|
357
|
+
link_str = "".join(self._current_link_text).strip()
|
|
358
|
+
if link_str and not self._current_href.startswith("javascript:"):
|
|
359
|
+
self.links.append((link_str, self._current_href))
|
|
360
|
+
self._current_href = None
|
|
361
|
+
self._current_link_text = []
|
|
362
|
+
|
|
363
|
+
def handle_data(self, data: str):
|
|
364
|
+
if self.in_script or self.in_style:
|
|
365
|
+
return
|
|
366
|
+
if self.in_title:
|
|
367
|
+
self.title += data
|
|
368
|
+
return
|
|
369
|
+
text = data.strip()
|
|
370
|
+
if text:
|
|
371
|
+
if self._current_href is not None:
|
|
372
|
+
self._current_link_text.append(text)
|
|
373
|
+
self.chunks.append(text + " ")
|
|
374
|
+
|
|
375
|
+
def get_markdown(self) -> str:
|
|
376
|
+
raw_text = "".join(self.chunks)
|
|
377
|
+
cleaned = re.sub(r"\n\s*\n+", "\n\n", raw_text).strip()
|
|
378
|
+
return cleaned
|
|
379
|
+
|
|
380
|
+
|
|
381
|
+
class ResearchSource:
|
|
382
|
+
"""Abstract base class for research scrapers."""
|
|
383
|
+
source_type: str = "generic"
|
|
384
|
+
|
|
385
|
+
def fetch(self, target: str, timeout: int = 15, max_content_length: int = 12000) -> ResearchResult:
|
|
386
|
+
raise NotImplementedError
|
|
@@ -0,0 +1,129 @@
|
|
|
1
|
+
"""
|
|
2
|
+
GitHub repository metadata, README content, and repository search scraper implementation.
|
|
3
|
+
"""
|
|
4
|
+
|
|
5
|
+
from __future__ import annotations
|
|
6
|
+
|
|
7
|
+
import json
|
|
8
|
+
import re
|
|
9
|
+
import urllib.parse
|
|
10
|
+
from typing import Optional
|
|
11
|
+
|
|
12
|
+
from fable_engine.scrapers.base import ResearchResult, ResearchSource, fetch_url
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class GitHubScraper(ResearchSource):
|
|
16
|
+
source_type = "github"
|
|
17
|
+
|
|
18
|
+
def fetch(self, target: str, timeout: int = 15, max_content_length: int = 8000) -> ResearchResult:
|
|
19
|
+
target = target.strip()
|
|
20
|
+
if not target:
|
|
21
|
+
return ResearchResult(
|
|
22
|
+
ok=False,
|
|
23
|
+
source_type=self.source_type,
|
|
24
|
+
canonical_url="",
|
|
25
|
+
error="Target GitHub repository or search query cannot be empty."
|
|
26
|
+
)
|
|
27
|
+
|
|
28
|
+
repo_path = None
|
|
29
|
+
if "github.com/" in target:
|
|
30
|
+
parsed = urllib.parse.urlparse(target)
|
|
31
|
+
parts = [p for p in parsed.path.strip("/").split("/") if p]
|
|
32
|
+
if len(parts) >= 2:
|
|
33
|
+
repo_path = f"{parts[0]}/{parts[1]}"
|
|
34
|
+
elif re.match(r"^[A-Za-z0-9_.-]+/[A-Za-z0-9_.-]+$", target):
|
|
35
|
+
repo_path = target
|
|
36
|
+
|
|
37
|
+
if repo_path:
|
|
38
|
+
api_url = f"https://api.github.com/repos/{repo_path}"
|
|
39
|
+
canonical_url = f"https://github.com/{repo_path}"
|
|
40
|
+
try:
|
|
41
|
+
repo_json = fetch_url(api_url, headers={"Accept": "application/vnd.github.v3+json"}, timeout=timeout)
|
|
42
|
+
data = json.loads(repo_json)
|
|
43
|
+
|
|
44
|
+
name = data.get("full_name", repo_path)
|
|
45
|
+
description = data.get("description", "No description provided.")
|
|
46
|
+
stars = data.get("stargazers_count", 0)
|
|
47
|
+
forks = data.get("forks_count", 0)
|
|
48
|
+
language = data.get("language", "Unknown")
|
|
49
|
+
default_branch = data.get("default_branch", "main")
|
|
50
|
+
html_url = data.get("html_url", canonical_url)
|
|
51
|
+
|
|
52
|
+
content_blocks = [
|
|
53
|
+
f"**Stars**: {stars} | **Forks**: {forks} | **Language**: {language} | **Branch**: {default_branch}\n",
|
|
54
|
+
f"**Description**: {description}\n"
|
|
55
|
+
]
|
|
56
|
+
|
|
57
|
+
# Fetch README content
|
|
58
|
+
readme_url = f"https://raw.githubusercontent.com/{repo_path}/{default_branch}/README.md"
|
|
59
|
+
try:
|
|
60
|
+
readme_text = fetch_url(readme_url, timeout=timeout)
|
|
61
|
+
if len(readme_text) > max_content_length:
|
|
62
|
+
readme_text = readme_text[:max_content_length] + "\n\n*(README truncated for size)*"
|
|
63
|
+
content_blocks.append(f"## README.md\n{readme_text}")
|
|
64
|
+
except Exception:
|
|
65
|
+
content_blocks.append("*(Note: README.md could not be retrieved automatically)*")
|
|
66
|
+
|
|
67
|
+
return ResearchResult(
|
|
68
|
+
ok=True,
|
|
69
|
+
source_type=self.source_type,
|
|
70
|
+
canonical_url=html_url,
|
|
71
|
+
title=f"GitHub Repo: {name}",
|
|
72
|
+
author=repo_path.split("/")[0],
|
|
73
|
+
content="\n".join(content_blocks),
|
|
74
|
+
metadata={"stars": stars, "language": language}
|
|
75
|
+
)
|
|
76
|
+
except Exception as e:
|
|
77
|
+
return ResearchResult(
|
|
78
|
+
ok=False,
|
|
79
|
+
source_type=self.source_type,
|
|
80
|
+
canonical_url=canonical_url,
|
|
81
|
+
error=f"GitHub repository fetch failed: {str(e)}"
|
|
82
|
+
)
|
|
83
|
+
|
|
84
|
+
# Repository Search Fallback
|
|
85
|
+
query_enc = urllib.parse.quote(target)
|
|
86
|
+
search_url = f"https://api.github.com/search/repositories?q={query_enc}&per_page=10"
|
|
87
|
+
canonical_url = f"https://github.com/search?q={query_enc}"
|
|
88
|
+
try:
|
|
89
|
+
search_json = fetch_url(search_url, headers={"Accept": "application/vnd.github.v3+json"}, timeout=timeout)
|
|
90
|
+
data = json.loads(search_json)
|
|
91
|
+
items = data.get("items", [])
|
|
92
|
+
|
|
93
|
+
if not items:
|
|
94
|
+
return ResearchResult(
|
|
95
|
+
ok=True,
|
|
96
|
+
source_type=self.source_type,
|
|
97
|
+
canonical_url=canonical_url,
|
|
98
|
+
title=f"GitHub Search: '{target}'",
|
|
99
|
+
content="*(No repositories found for this query)*"
|
|
100
|
+
)
|
|
101
|
+
|
|
102
|
+
lines = ["## Repositories\n"]
|
|
103
|
+
for item in items[:10]:
|
|
104
|
+
full_name = item.get("full_name", "")
|
|
105
|
+
desc = item.get("description", "") or "No description"
|
|
106
|
+
stars = item.get("stargazers_count", 0)
|
|
107
|
+
lang = item.get("language", "N/A")
|
|
108
|
+
url = item.get("html_url", "")
|
|
109
|
+
lines.append(f"### [{full_name}]({url})")
|
|
110
|
+
lines.append(f"**Stars**: {stars} | **Language**: {lang}\n**Description**: {desc}\n")
|
|
111
|
+
|
|
112
|
+
return ResearchResult(
|
|
113
|
+
ok=True,
|
|
114
|
+
source_type=self.source_type,
|
|
115
|
+
canonical_url=canonical_url,
|
|
116
|
+
title=f"GitHub Repository Search: '{target}'",
|
|
117
|
+
content="\n".join(lines)
|
|
118
|
+
)
|
|
119
|
+
except Exception as e:
|
|
120
|
+
return ResearchResult(
|
|
121
|
+
ok=False,
|
|
122
|
+
source_type=self.source_type,
|
|
123
|
+
canonical_url=canonical_url,
|
|
124
|
+
error=f"GitHub repository search failed: {str(e)}"
|
|
125
|
+
)
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def scrape_github(target: str, timeout: int = 15) -> ResearchResult:
|
|
129
|
+
return GitHubScraper().fetch(target, timeout=timeout)
|