fable-engine 1.3.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (104) hide show
  1. fable_compressor.py +356 -0
  2. fable_engine/__init__.py +1 -0
  3. fable_engine/actions/__init__.py +291 -0
  4. fable_engine/actions/cas.py +182 -0
  5. fable_engine/actions/deliberation.py +523 -0
  6. fable_engine/actions/fleet.py +807 -0
  7. fable_engine/actions/lifecycle.py +298 -0
  8. fable_engine/actions/scrapers.py +116 -0
  9. fable_engine/actions/system3.py +815 -0
  10. fable_engine/browser.py +824 -0
  11. fable_engine/cas.py +974 -0
  12. fable_engine/fable_session.json +510 -0
  13. fable_engine/guards.py +283 -0
  14. fable_engine/schema.py +714 -0
  15. fable_engine/scrapers/__init__.py +32 -0
  16. fable_engine/scrapers/arxiv.py +115 -0
  17. fable_engine/scrapers/base.py +386 -0
  18. fable_engine/scrapers/github.py +129 -0
  19. fable_engine/scrapers/reddit.py +154 -0
  20. fable_engine/scrapers/web.py +120 -0
  21. fable_engine/scrapers/x.py +125 -0
  22. fable_engine/scrapers/youtube.py +132 -0
  23. fable_engine/server.py +414 -0
  24. fable_engine/session.py +1819 -0
  25. fable_engine/test_server.py +1362 -0
  26. fable_engine/updater.py +541 -0
  27. fable_engine-1.3.1.dist-info/LICENSE +22 -0
  28. fable_engine-1.3.1.dist-info/METADATA +173 -0
  29. fable_engine-1.3.1.dist-info/RECORD +104 -0
  30. fable_engine-1.3.1.dist-info/WHEEL +5 -0
  31. fable_engine-1.3.1.dist-info/entry_points.txt +5 -0
  32. fable_engine-1.3.1.dist-info/top_level.txt +6 -0
  33. fable_mode/__init__.py +3 -0
  34. fable_mode/__main__.py +4 -0
  35. fable_mode/adapters.py +1014 -0
  36. fable_mode/installer.py +553 -0
  37. fable_mode/launcher.py +437 -0
  38. fable_mode/manifest.py +142 -0
  39. fable_mode/resources.json +114 -0
  40. fable_mode/safety.py +103 -0
  41. fable_mode_entry.py +10 -0
  42. fable_v2/__init__.py +146 -0
  43. fable_v2/adapters.py +151 -0
  44. fable_v2/coder_fleet/__init__.py +100 -0
  45. fable_v2/coder_fleet/ast_tools.py +158 -0
  46. fable_v2/coder_fleet/compute.py +199 -0
  47. fable_v2/coder_fleet/design_engine.py +1316 -0
  48. fable_v2/coder_fleet/diagnostics.py +293 -0
  49. fable_v2/coder_fleet/fleet_dispatcher.py +214 -0
  50. fable_v2/coder_fleet/mock_auditor.py +306 -0
  51. fable_v2/coder_fleet/mutation.py +216 -0
  52. fable_v2/coder_fleet/property_oracle.py +260 -0
  53. fable_v2/coder_fleet/receipt_attestor.py +122 -0
  54. fable_v2/coder_fleet/red_team_swarm.py +908 -0
  55. fable_v2/coder_fleet/test_harness.py +198 -0
  56. fable_v2/coder_fleet/vector_engine.py +1287 -0
  57. fable_v2/coder_fleet/visual.py +357 -0
  58. fable_v2/coder_fleet/workspace.py +153 -0
  59. fable_v2/cortical/__init__.py +20 -0
  60. fable_v2/cortical/plasticity_engine.py +992 -0
  61. fable_v2/execution_broker.py +811 -0
  62. fable_v2/proof_engine.py +1141 -0
  63. fable_v2/protocol.py +485 -0
  64. fable_v2/runtime.py +1010 -0
  65. fable_v2/system3/__init__.py +204 -0
  66. fable_v2/system3/causal.py +558 -0
  67. fable_v2/system3/dialectical.py +577 -0
  68. fable_v2/system3/evolution.py +503 -0
  69. fable_v2/system3/executive.py +338 -0
  70. fable_v2/system3/free_energy.py +479 -0
  71. fable_v2/system3/hyperbolic.py +555 -0
  72. fable_v2/system3/induction.py +336 -0
  73. fable_v2/system3/kripke.py +548 -0
  74. fable_v2/system3/oracle.py +745 -0
  75. fable_v2/verifiers.py +72 -0
  76. tests/__init__.py +1 -0
  77. tests/test_anti_loop_circuit_breaker.py +64 -0
  78. tests/test_auto_updater.py +407 -0
  79. tests/test_coder_fleet.py +535 -0
  80. tests/test_delegation_compiler.py +54 -0
  81. tests/test_descriptor_boundaries.py +126 -0
  82. tests/test_design_engine.py +603 -0
  83. tests/test_epistemic_evidence_validator.py +66 -0
  84. tests/test_execution_broker.py +233 -0
  85. tests/test_fable_v2.py +406 -0
  86. tests/test_fleet_transitions.py +116 -0
  87. tests/test_fsm_redteam_evolution.py +406 -0
  88. tests/test_goal_rubric_and_pipeline.py +367 -0
  89. tests/test_hebbian_plasticity.py +585 -0
  90. tests/test_packaging_runtime.py +194 -0
  91. tests/test_proof_engine.py +259 -0
  92. tests/test_red_team_swarm.py +645 -0
  93. tests/test_redteam_remediation.py +169 -0
  94. tests/test_registration_transaction.py +375 -0
  95. tests/test_requested_regressions.py +467 -0
  96. tests/test_scrapers.py +370 -0
  97. tests/test_server_actions.py +93 -0
  98. tests/test_server_frontier_actions.py +269 -0
  99. tests/test_server_protocol.py +88 -0
  100. tests/test_stealth_browser.py +970 -0
  101. tests/test_system3.py +381 -0
  102. tests/test_system3_deep_integration.py +385 -0
  103. tests/test_system3_frontier.py +436 -0
  104. tests/test_vector_engine.py +608 -0
@@ -0,0 +1,154 @@
1
+ """
2
+ Reddit threads and subreddits scraper implementation using Reddit's public JSON interface (.json).
3
+ """
4
+
5
+ from __future__ import annotations
6
+
7
+ import json
8
+ import urllib.parse
9
+ from typing import Optional
10
+
11
+ from fable_engine.scrapers.base import ResearchResult, ResearchSource, fetch_url
12
+
13
+
14
+ class RedditScraper(ResearchSource):
15
+ source_type = "reddit"
16
+
17
+ def fetch(self, target: str, timeout: int = 15, max_content_length: int = 10000) -> ResearchResult:
18
+ target = target.strip()
19
+ if not target:
20
+ return ResearchResult(
21
+ ok=False,
22
+ source_type=self.source_type,
23
+ canonical_url="",
24
+ error="Target Reddit URL, subreddit, or search query cannot be empty."
25
+ )
26
+
27
+ if target.startswith("http://") or target.startswith("https://"):
28
+ clean_url = target.split("?")[0].rstrip("/")
29
+ url = clean_url + ".json"
30
+ canonical_url = clean_url
31
+ elif target.startswith("r/") or target.startswith("/r/"):
32
+ sub_path = target.lstrip("/").rstrip("/")
33
+ url = f"https://www.reddit.com/{sub_path}/hot.json?limit=10"
34
+ canonical_url = f"https://www.reddit.com/{sub_path}"
35
+ else:
36
+ query_enc = urllib.parse.quote(target)
37
+ url = f"https://www.reddit.com/search.json?q={query_enc}&limit=10"
38
+ canonical_url = f"https://www.reddit.com/search?q={query_enc}"
39
+
40
+ try:
41
+ json_str = fetch_url(url, timeout=timeout)
42
+ data = json.loads(json_str)
43
+
44
+ # Handle single thread post response (list of listings)
45
+ if isinstance(data, list) and len(data) >= 1 and isinstance(data[0], dict):
46
+ children = data[0].get("data", {}).get("children", [])
47
+ if not children or not isinstance(children[0], dict):
48
+ return ResearchResult(
49
+ ok=False,
50
+ source_type=self.source_type,
51
+ canonical_url=canonical_url,
52
+ error="Reddit post data structure was empty or unexpected."
53
+ )
54
+
55
+ post_data = children[0].get("data", {})
56
+ title = post_data.get("title", "Reddit Post")
57
+ subreddit = post_data.get("subreddit", "")
58
+ author = post_data.get("author", "[deleted]")
59
+ score = post_data.get("score", 0)
60
+ selftext = post_data.get("selftext", "")
61
+ permalink = post_data.get("permalink", "")
62
+ post_url = f"https://www.reddit.com{permalink}" if permalink else canonical_url
63
+
64
+ content_blocks = [
65
+ f"**Subreddit**: r/{subreddit} | **Author**: u/{author} | **Score**: {score}\n"
66
+ ]
67
+ if selftext:
68
+ content_blocks.append(f"## Post Text\n{selftext}\n")
69
+
70
+ if len(data) >= 2 and isinstance(data[1], dict):
71
+ comments = data[1].get("data", {}).get("children", [])
72
+ comment_lines = ["## Top Comments\n"]
73
+ count = 0
74
+ for c in comments:
75
+ if count >= 10:
76
+ break
77
+ if isinstance(c, dict):
78
+ cdata = c.get("data", {})
79
+ c_author = cdata.get("author", "")
80
+ c_body = cdata.get("body", "")
81
+ c_score = cdata.get("score", 0)
82
+ if c_body:
83
+ comment_lines.append(f"### u/{c_author} ({c_score} points)")
84
+ comment_lines.append(f"{c_body}\n")
85
+ count += 1
86
+ if count > 0:
87
+ content_blocks.extend(comment_lines)
88
+
89
+ full_content = "\n".join(content_blocks)
90
+ if len(full_content) > max_content_length:
91
+ full_content = full_content[:max_content_length] + "\n\n*(Content truncated)*"
92
+
93
+ return ResearchResult(
94
+ ok=True,
95
+ source_type=self.source_type,
96
+ canonical_url=post_url,
97
+ title=title,
98
+ author=f"u/{author}",
99
+ content=full_content,
100
+ metadata={"subreddit": subreddit, "score": score}
101
+ )
102
+
103
+ # Handle subreddit listing or search results
104
+ elif isinstance(data, dict) and "data" in data:
105
+ children = data.get("data", {}).get("children", [])
106
+ if not children:
107
+ return ResearchResult(
108
+ ok=True,
109
+ source_type=self.source_type,
110
+ canonical_url=canonical_url,
111
+ title=f"Reddit Listing: '{target}'",
112
+ content="*(No posts found for this subreddit or query)*"
113
+ )
114
+
115
+ items = ["## Posts\n"]
116
+ for child in children[:10]:
117
+ if isinstance(child, dict):
118
+ p = child.get("data", {})
119
+ p_title = p.get("title", "")
120
+ p_sub = p.get("subreddit", "")
121
+ p_author = p.get("author", "")
122
+ p_score = p.get("score", 0)
123
+ p_comments = p.get("num_comments", 0)
124
+ p_permalink = f"https://www.reddit.com{p.get('permalink', '')}"
125
+ p_text = p.get("selftext", "")[:250]
126
+ text_str = f"\n _{p_text.strip()}..._" if p_text else ""
127
+ items.append(f"- **[{p_title}]({p_permalink})**")
128
+ items.append(f" *r/{p_sub} | u/{p_author} | Score: {p_score} | Comments: {p_comments}*{text_str}\n")
129
+
130
+ return ResearchResult(
131
+ ok=True,
132
+ source_type=self.source_type,
133
+ canonical_url=canonical_url,
134
+ title=f"Reddit Listing: '{target}'",
135
+ content="\n".join(items)
136
+ )
137
+
138
+ return ResearchResult(
139
+ ok=False,
140
+ source_type=self.source_type,
141
+ canonical_url=canonical_url,
142
+ error="Unexpected JSON response format from Reddit API."
143
+ )
144
+ except Exception as e:
145
+ return ResearchResult(
146
+ ok=False,
147
+ source_type=self.source_type,
148
+ canonical_url=canonical_url,
149
+ error=f"Reddit fetch failed: {str(e)}"
150
+ )
151
+
152
+
153
+ def scrape_reddit(target: str, timeout: int = 15) -> ResearchResult:
154
+ return RedditScraper().fetch(target, timeout=timeout)
@@ -0,0 +1,120 @@
1
+ """
2
+ Web page and search query scraper implementation.
3
+ """
4
+
5
+ from __future__ import annotations
6
+
7
+ import re
8
+ import urllib.parse
9
+ from typing import Optional
10
+
11
+ from fable_engine.scrapers.base import ResearchResult, ResearchSource, SimpleHTMLTextExtractor, fetch_url
12
+
13
+
14
+ class WebScraper(ResearchSource):
15
+ source_type = "web"
16
+
17
+ def fetch(self, target: str, timeout: int = 15, max_content_length: int = 12000) -> ResearchResult:
18
+ target = target.strip()
19
+ if not target:
20
+ return ResearchResult(
21
+ ok=False,
22
+ source_type=self.source_type,
23
+ canonical_url="",
24
+ error="Target URL or query string cannot be empty."
25
+ )
26
+
27
+ if target.startswith("http://") or target.startswith("https://"):
28
+ try:
29
+ html = fetch_url(target, timeout=timeout)
30
+ parser = SimpleHTMLTextExtractor()
31
+ parser.feed(html)
32
+ title = parser.title.strip() or target
33
+ content = parser.get_markdown()
34
+ if len(content) > max_content_length:
35
+ content = content[:max_content_length] + "\n\n*(Content truncated for size)*"
36
+
37
+ links_md = ""
38
+ if parser.links:
39
+ seen = set()
40
+ link_lines = []
41
+ for text, href in parser.links[:15]:
42
+ full_url = urllib.parse.urljoin(target, href)
43
+ if full_url not in seen and text:
44
+ seen.add(full_url)
45
+ link_lines.append(f"- [{text}]({full_url})")
46
+ if link_lines:
47
+ links_md = "\n\n## Key Links\n" + "\n".join(link_lines)
48
+
49
+ return ResearchResult(
50
+ ok=True,
51
+ source_type=self.source_type,
52
+ canonical_url=target,
53
+ title=title,
54
+ content=f"## Web Page Content\n{content}{links_md}"
55
+ )
56
+ except Exception as e:
57
+ return ResearchResult(
58
+ ok=False,
59
+ source_type=self.source_type,
60
+ canonical_url=target,
61
+ error=f"Web page fetch failed: {str(e)}"
62
+ )
63
+ else:
64
+ # DuckDuckGo HTML Search
65
+ try:
66
+ query_enc = urllib.parse.quote(target)
67
+ ddg_url = f"https://html.duckduckgo.com/html/?q={query_enc}"
68
+ html = fetch_url(ddg_url, timeout=timeout)
69
+
70
+ results = []
71
+ result_blocks = re.findall(
72
+ r'<a class="result__url" href="([^"]+)".*?</a>.*?<a class="result__snippet[^"]*">(.*?)</a>',
73
+ html,
74
+ re.DOTALL
75
+ )
76
+ if not result_blocks:
77
+ urls = re.findall(r'class="result__a" href="([^"]+)">([^<]+)</a>', html)
78
+ snippets = re.findall(r'class="result__snippet[^"]*">(.*?)</span>', html)
79
+ for i in range(min(len(urls), 10)):
80
+ raw_url, title = urls[i]
81
+ parsed_url = urllib.parse.parse_qs(urllib.parse.urlparse(raw_url).query).get("uddg", [raw_url])[0]
82
+ snip = snippets[i] if i < len(snippets) else ""
83
+ snip_clean = re.sub(r"<[^>]+>", "", snip).strip()
84
+ title_clean = re.sub(r"<[^>]+>", "", title).strip()
85
+ results.append(f"### [{title_clean}]({parsed_url})\n**Snippet**: {snip_clean}\n")
86
+ else:
87
+ for raw_url, snip in result_blocks[:10]:
88
+ snip_clean = re.sub(r"<[^>]+>", "", snip).strip()
89
+ results.append(f"### [Result]({raw_url})\n**Snippet**: {snip_clean}\n")
90
+
91
+ if not results:
92
+ parser = SimpleHTMLTextExtractor()
93
+ parser.feed(html)
94
+ content = parser.get_markdown()[:max_content_length]
95
+ return ResearchResult(
96
+ ok=True,
97
+ source_type=self.source_type,
98
+ canonical_url=ddg_url,
99
+ title=f"Web Search: '{target}'",
100
+ content=content
101
+ )
102
+
103
+ return ResearchResult(
104
+ ok=True,
105
+ source_type=self.source_type,
106
+ canonical_url=ddg_url,
107
+ title=f"Web Search Results for: '{target}'",
108
+ content="\n".join(results)
109
+ )
110
+ except Exception as e:
111
+ return ResearchResult(
112
+ ok=False,
113
+ source_type=self.source_type,
114
+ canonical_url=f"https://duckduckgo.com/html/?q={urllib.parse.quote(target)}",
115
+ error=f"Web search failed: {str(e)}"
116
+ )
117
+
118
+
119
+ def scrape_web(target: str, timeout: int = 15) -> ResearchResult:
120
+ return WebScraper().fetch(target, timeout=timeout)
@@ -0,0 +1,125 @@
1
+ """
2
+ X (Twitter) tweet syndication and oembed lookup scraper implementation.
3
+ Exact Tweet URLs utilize Twitter syndication API; handles/profiles use oEmbed embed summaries.
4
+ """
5
+
6
+ from __future__ import annotations
7
+
8
+ import json
9
+ import re
10
+ import urllib.parse
11
+ from typing import Optional
12
+
13
+ from fable_engine.scrapers.base import ResearchResult, ResearchSource, SimpleHTMLTextExtractor, fetch_url
14
+
15
+
16
+ class XScraper(ResearchSource):
17
+ source_type = "x"
18
+
19
+ def fetch(self, target: str, timeout: int = 15, max_content_length: int = 8000) -> ResearchResult:
20
+ target = target.strip()
21
+ if not target:
22
+ return ResearchResult(
23
+ ok=False,
24
+ source_type=self.source_type,
25
+ canonical_url="",
26
+ error="Target Tweet URL or handle cannot be empty."
27
+ )
28
+
29
+ tweet_id = None
30
+ tweet_id_match = re.search(r"status(?:es)?/(\d+)", target)
31
+ if tweet_id_match:
32
+ tweet_id = tweet_id_match.group(1)
33
+
34
+ if tweet_id:
35
+ # Use Twitter Syndication JSON API (free endpoint for individual tweets)
36
+ synd_url = f"https://cdn.syndication.twimg.com/tweet-result?id={tweet_id}&token=x"
37
+ canonical_url = f"https://x.com/i/status/{tweet_id}"
38
+ try:
39
+ json_str = fetch_url(synd_url, timeout=timeout)
40
+ data = json.loads(json_str)
41
+
42
+ text = data.get("text", "")
43
+ user = data.get("user", {})
44
+ name = user.get("name", "User")
45
+ screen_name = user.get("screen_name", "unknown")
46
+ created_at = data.get("created_at", "")
47
+ likes = data.get("favorite_count", 0)
48
+ retweets = data.get("retweet_count", 0)
49
+ full_url = f"https://x.com/{screen_name}/status/{tweet_id}"
50
+
51
+ content = (
52
+ f"**Date**: {created_at} | **Likes**: {likes} | **Retweets**: {retweets}\n\n"
53
+ f"## Tweet\n{text}"
54
+ )
55
+
56
+ return ResearchResult(
57
+ ok=True,
58
+ source_type=self.source_type,
59
+ canonical_url=full_url,
60
+ title=f"Tweet by @{screen_name} ({name})",
61
+ author=f"@{screen_name}",
62
+ content=content,
63
+ metadata={"tweet_id": tweet_id, "likes": likes, "retweets": retweets}
64
+ )
65
+ except Exception as e:
66
+ # Fallback to Twitter oEmbed API
67
+ try:
68
+ oembed_url = f"https://publish.twitter.com/oembed?url=https://twitter.com/i/status/{tweet_id}"
69
+ json_str = fetch_url(oembed_url, timeout=timeout)
70
+ data = json.loads(json_str)
71
+ author = data.get("author_name", "User")
72
+ html_embed = data.get("html", "")
73
+ parser = SimpleHTMLTextExtractor()
74
+ parser.feed(html_embed)
75
+ text = parser.get_markdown()
76
+ return ResearchResult(
77
+ ok=True,
78
+ source_type=self.source_type,
79
+ canonical_url=canonical_url,
80
+ title=f"Tweet by {author}",
81
+ author=author,
82
+ content=f"## Content\n{text}"
83
+ )
84
+ except Exception:
85
+ return ResearchResult(
86
+ ok=False,
87
+ source_type=self.source_type,
88
+ canonical_url=canonical_url,
89
+ error=f"X/Twitter tweet scrape failed for status ID {tweet_id}: {str(e)}"
90
+ )
91
+
92
+ # Handle user profile / handle lookup via oembed
93
+ handle = target.lstrip("@").strip()
94
+ canonical_url = f"https://x.com/{handle}"
95
+ try:
96
+ oembed_url = f"https://publish.twitter.com/oembed?url=https://twitter.com/{handle}"
97
+ json_str = fetch_url(oembed_url, timeout=timeout)
98
+ data = json.loads(json_str)
99
+ author = data.get("author_name", handle)
100
+ html_embed = data.get("html", "")
101
+ parser = SimpleHTMLTextExtractor()
102
+ parser.feed(html_embed)
103
+ text = parser.get_markdown()
104
+ return ResearchResult(
105
+ ok=True,
106
+ source_type=self.source_type,
107
+ canonical_url=canonical_url,
108
+ title=f"X Profile / Handle: @{handle}",
109
+ author=author,
110
+ content=f"## Summary\n{text}"
111
+ )
112
+ except Exception as e:
113
+ return ResearchResult(
114
+ ok=False,
115
+ source_type=self.source_type,
116
+ canonical_url=canonical_url,
117
+ error=(
118
+ f"X handle/profile lookup failed for '@{handle}': {str(e)}. "
119
+ "Note: For exact tweet contents, provide a direct Tweet status URL."
120
+ )
121
+ )
122
+
123
+
124
+ def scrape_x(target: str, timeout: int = 15) -> ResearchResult:
125
+ return XScraper().fetch(target, timeout=timeout)
@@ -0,0 +1,132 @@
1
+ """
2
+ Best-effort YouTube metadata, search, and caption track scraper implementation.
3
+ Uses heuristic page-structure parsing for public watch page metadata and automated caption tracks.
4
+ """
5
+
6
+ from __future__ import annotations
7
+
8
+ import json
9
+ import re
10
+ import urllib.parse
11
+ import xml.etree.ElementTree as ET
12
+ from typing import Optional
13
+
14
+ from fable_engine.scrapers.base import ResearchResult, ResearchSource, fetch_url
15
+
16
+
17
+ class YouTubeScraper(ResearchSource):
18
+ """
19
+ Best-effort scraper for YouTube public metadata and caption tracks.
20
+ Heuristic page parsing is subject to upstream YouTube markup variations.
21
+ """
22
+ source_type = "youtube"
23
+
24
+ def fetch(self, target: str, timeout: int = 15, max_content_length: int = 8000) -> ResearchResult:
25
+ target = target.strip()
26
+ video_id = None
27
+
28
+ if "youtube.com" in target or "youtu.be" in target:
29
+ parsed = urllib.parse.urlparse(target)
30
+ if "youtu.be" in target:
31
+ video_id = parsed.path.lstrip("/")
32
+ else:
33
+ qs = urllib.parse.parse_qs(parsed.query)
34
+ video_id = qs.get("v", [None])[0]
35
+ elif len(target) == 11 and re.match(r"^[A-Za-z0-9_-]{11}$", target):
36
+ video_id = target
37
+
38
+ if not video_id:
39
+ # Best-effort search resolution
40
+ try:
41
+ query_enc = urllib.parse.quote(target)
42
+ search_url = f"https://www.youtube.com/results?search_query={query_enc}"
43
+ html = fetch_url(search_url, timeout=timeout)
44
+ vids = re.findall(r'"videoId":"([A-Za-z0-9_-]{11})"', html)
45
+ if vids:
46
+ video_id = vids[0]
47
+ else:
48
+ return ResearchResult(
49
+ ok=False,
50
+ source_type=self.source_type,
51
+ canonical_url=search_url,
52
+ error=f"Best-effort YouTube search could not resolve video ID for '{target}'."
53
+ )
54
+ except Exception as e:
55
+ return ResearchResult(
56
+ ok=False,
57
+ source_type=self.source_type,
58
+ canonical_url=f"https://www.youtube.com/results?search_query={urllib.parse.quote(target)}",
59
+ error=f"Best-effort YouTube search resolution failed: {str(e)}"
60
+ )
61
+
62
+ canonical_url = f"https://www.youtube.com/watch?v={video_id}"
63
+ try:
64
+ html = fetch_url(canonical_url, timeout=timeout)
65
+
66
+ title_match = re.search(r'<meta property="og:title" content="([^"]+)"', html) or re.search(r'"title":"([^"]+)"', html)
67
+ title = title_match.group(1) if title_match else f"YouTube Video ({video_id})"
68
+ title = re.sub(r"\\u0026", "&", title)
69
+
70
+ channel_match = re.search(r'<link itemprop="name" content="([^"]+)"', html) or re.search(r'"author":"([^"]+)"', html)
71
+ channel = channel_match.group(1) if channel_match else "Unknown Channel"
72
+
73
+ desc_match = re.search(r'<meta property="og:description" content="([^"]+)"', html)
74
+ description = desc_match.group(1) if desc_match else ""
75
+
76
+ transcript_text = ""
77
+ transcript_status = "not_found"
78
+ captions_match = re.search(r'"captionTracks":\[(.*?)\]', html)
79
+ if captions_match:
80
+ try:
81
+ tracks_json = json.loads(f"[{captions_match.group(1)}]")
82
+ if tracks_json and isinstance(tracks_json[0], dict) and "baseUrl" in tracks_json[0]:
83
+ caption_url = tracks_json[0]["baseUrl"]
84
+ caption_xml = fetch_url(caption_url, timeout=timeout)
85
+ root = ET.fromstring(caption_xml)
86
+ lines = []
87
+ for child in root.findall("text"):
88
+ txt = child.text or ""
89
+ txt_clean = re.sub(r"&amp;", "&", txt)
90
+ txt_clean = re.sub(r"&#39;", "'", txt_clean)
91
+ txt_clean = re.sub(r"&quot;", '"', txt_clean)
92
+ if txt_clean.strip():
93
+ lines.append(txt_clean.strip())
94
+ transcript_text = " ".join(lines)
95
+ transcript_status = "retrieved" if transcript_text else "empty"
96
+ except Exception as exc:
97
+ transcript_status = f"parse_error: {exc}"
98
+
99
+ content_blocks = [f"**Channel**: {channel}\n"]
100
+ if description:
101
+ content_blocks.append(f"## Description\n{description[:1000]}\n")
102
+ if transcript_text:
103
+ if len(transcript_text) > max_content_length:
104
+ transcript_text = transcript_text[:max_content_length] + "...\n*(Transcript truncated)*"
105
+ content_blocks.append(f"## Transcript / Captions\n{transcript_text}")
106
+ else:
107
+ content_blocks.append(f"*(Note: Best-effort automated transcript track status: {transcript_status})*")
108
+
109
+ return ResearchResult(
110
+ ok=True,
111
+ source_type=self.source_type,
112
+ canonical_url=canonical_url,
113
+ title=title,
114
+ author=channel,
115
+ content="\n".join(content_blocks),
116
+ metadata={
117
+ "video_id": video_id,
118
+ "extraction_mode": "best_effort_heuristic",
119
+ "transcript_status": transcript_status,
120
+ }
121
+ )
122
+ except Exception as e:
123
+ return ResearchResult(
124
+ ok=False,
125
+ source_type=self.source_type,
126
+ canonical_url=canonical_url,
127
+ error=f"Best-effort YouTube video scrape failed: {str(e)}"
128
+ )
129
+
130
+
131
+ def scrape_youtube(target: str, timeout: int = 15) -> ResearchResult:
132
+ return YouTubeScraper().fetch(target, timeout=timeout)