advanced-keyword-scraper 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,22 @@
1
+ Metadata-Version: 2.4
2
+ Name: advanced-keyword-scraper
3
+ Version: 0.1.0
4
+ Summary: A multithreaded Google autocomplete keyword scraper with intent classification and Excel export.
5
+ Author-email: Your Name <your.email@example.com>
6
+ License: MIT
7
+ Requires-Python: >=3.8
8
+ Description-Content-Type: text/markdown
9
+ Requires-Dist: requests>=2.28.0
10
+ Requires-Dist: pandas>=2.0.0
11
+ Requires-Dist: openpyxl>=3.1.0
12
+
13
+ # Advanced Keyword Scraper
14
+
15
+ A multithreaded Google autocomplete keyword scraper with intent classification and Excel export.
16
+
17
+ ## Installation
18
+ ```bash
19
+ pip install advanced-keyword-scraper
20
+
21
+ ## Author
22
+ Developed with 🔥 by **LORD SRIJWN**
@@ -0,0 +1,10 @@
1
+ # Advanced Keyword Scraper
2
+
3
+ A multithreaded Google autocomplete keyword scraper with intent classification and Excel export.
4
+
5
+ ## Installation
6
+ ```bash
7
+ pip install advanced-keyword-scraper
8
+
9
+ ## Author
10
+ Developed with 🔥 by **LORD SRIJWN**
@@ -0,0 +1,22 @@
1
+ [build-system]
2
+ requires = ["setuptools>=61.0"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "advanced-keyword-scraper"
7
+ version = "0.1.0"
8
+ description = "A multithreaded Google autocomplete keyword scraper with intent classification and Excel export."
9
+ readme = "README.md"
10
+ requires-python = ">=3.8"
11
+ license = {text = "MIT"}
12
+ authors = [
13
+ { name = "Your Name", email = "your.email@example.com" }
14
+ ]
15
+ dependencies = [
16
+ "requests>=2.28.0",
17
+ "pandas>=2.0.0",
18
+ "openpyxl>=3.1.0"
19
+ ]
20
+
21
+ [project.scripts]
22
+ keyword-scraper = "keyword_scraper.cli:main"
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,22 @@
1
+ Metadata-Version: 2.4
2
+ Name: advanced-keyword-scraper
3
+ Version: 0.1.0
4
+ Summary: A multithreaded Google autocomplete keyword scraper with intent classification and Excel export.
5
+ Author-email: Your Name <your.email@example.com>
6
+ License: MIT
7
+ Requires-Python: >=3.8
8
+ Description-Content-Type: text/markdown
9
+ Requires-Dist: requests>=2.28.0
10
+ Requires-Dist: pandas>=2.0.0
11
+ Requires-Dist: openpyxl>=3.1.0
12
+
13
+ # Advanced Keyword Scraper
14
+
15
+ A multithreaded Google autocomplete keyword scraper with intent classification and Excel export.
16
+
17
+ ## Installation
18
+ ```bash
19
+ pip install advanced-keyword-scraper
20
+
21
+ ## Author
22
+ Developed with 🔥 by **LORD SRIJWN**
@@ -0,0 +1,10 @@
1
+ README.md
2
+ pyproject.toml
3
+ src/advanced_keyword_scraper.egg-info/PKG-INFO
4
+ src/advanced_keyword_scraper.egg-info/SOURCES.txt
5
+ src/advanced_keyword_scraper.egg-info/dependency_links.txt
6
+ src/advanced_keyword_scraper.egg-info/entry_points.txt
7
+ src/advanced_keyword_scraper.egg-info/requires.txt
8
+ src/advanced_keyword_scraper.egg-info/top_level.txt
9
+ src/keyword_scraper/cli.py
10
+ src/keyword_scraper/init.py
@@ -0,0 +1,2 @@
1
+ [console_scripts]
2
+ keyword-scraper = keyword_scraper.cli:main
@@ -0,0 +1,3 @@
1
+ requests>=2.28.0
2
+ pandas>=2.0.0
3
+ openpyxl>=3.1.0
@@ -0,0 +1,126 @@
1
+ import argparse
2
+ import concurrent.futures
3
+ from datetime import datetime
4
+ import json
5
+ import os
6
+ import random
7
+ import string
8
+ import time
9
+ import pandas as pd
10
+ import requests
11
+
12
+ USER_AGENTS = [
13
+ "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/122.0.0.0 Safari/537.36",
14
+ "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/121.0.0.0 Safari/537.36",
15
+ "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/122.0.0.0 Safari/537.36",
16
+ ]
17
+
18
+ def get_suggestions(keyword, country, lang):
19
+ url = f"http://suggestqueries.google.com/complete/search?client=chrome&gl={country}&hl={lang}&q={keyword}"
20
+ headers = {"User-Agent": random.choice(USER_AGENTS)}
21
+ try:
22
+ response = requests.get(url, headers=headers, timeout=5)
23
+ if response.status_code == 200:
24
+ data = json.loads(response.text)
25
+ return data[1] if len(data) > 1 else []
26
+ except Exception:
27
+ pass
28
+ return []
29
+
30
+ def process_task(task, country, lang):
31
+ base_kw, pattern, query = task
32
+ suggestions = get_suggestions(query, country, lang)
33
+ time.sleep(random.uniform(0.05, 0.15))
34
+ return base_kw, pattern, suggestions
35
+
36
+ def classify_intent(keyword):
37
+ kw = keyword.lower()
38
+ transactional = ["buy", "price", "cheap", "cost", "under", "online", "shop", "order", "deal", "discount", "store", "sale", "rate"]
39
+ commercial = ["best", "top", "review", "vs", "compare", "difference", "brand", "latest", "new"]
40
+ informational = ["how", "what", "where", "why", "which", "can", "guide", "tips", "ideas", "tutorial", "design", "style", "pattern"]
41
+
42
+ if any(w in kw for w in transactional):
43
+ return "Transactional (Buying)"
44
+ elif any(w in kw for w in commercial):
45
+ return "Commercial (Research)"
46
+ elif any(w in kw for w in informational):
47
+ return "Informational"
48
+ else:
49
+ return "General / Long-Tail"
50
+
51
+ def run_scraper(keywords, country, lang, max_threads, output_folder):
52
+ if not os.path.exists(output_folder):
53
+ os.makedirs(output_folder)
54
+
55
+ timestamp = datetime.now().strftime("%Y%m%d_%H%M%S")
56
+ filename = os.path.join(output_folder, f"keyword_analysis_{timestamp}.xlsx")
57
+
58
+ questions = ["how", "what", "where", "why", "which", "can"]
59
+ prepositions = ["for", "with", "under", "near", "vs", "without"]
60
+ buying_intent = ["best", "buy", "price", "online", "top", "cheap", "shop"]
61
+
62
+ tasks = []
63
+ for base in keywords:
64
+ tasks.append((base, "Normal Search", base))
65
+ for letter in string.ascii_lowercase:
66
+ tasks.append((base, f"Alpha ({letter})", f"{base} {letter}"))
67
+ for num in string.digits:
68
+ tasks.append((base, f"Number ({num})", f"{base} {num}"))
69
+ for q in questions:
70
+ tasks.append((base, f"Question ({q})", f"{q} {base}"))
71
+ tasks.append((base, f"Question ({q})", f"{base} {q}"))
72
+ for prep in prepositions:
73
+ tasks.append((base, f"Preposition ({prep})", f"{base} {prep}"))
74
+ for intent in buying_intent:
75
+ tasks.append((base, f"Intent ({intent})", f"{intent} {base}"))
76
+
77
+ print(f"Total Queries Generated: {len(tasks)}")
78
+ print(f"Scraping started (Country: {country}, Threads: {max_threads})...\n")
79
+
80
+ all_keywords = set()
81
+ results = []
82
+ completed = 0
83
+
84
+ with concurrent.futures.ThreadPoolExecutor(max_workers=max_threads) as executor:
85
+ futures = [executor.submit(process_task, task, country, lang) for task in tasks]
86
+ for future in concurrent.futures.as_completed(futures):
87
+ base_kw, pattern, suggestions = future.result()
88
+ completed += 1
89
+
90
+ for s in suggestions:
91
+ if s not in all_keywords:
92
+ all_keywords.add(s)
93
+ intent = classify_intent(s)
94
+ results.append([base_kw, pattern, s, intent])
95
+
96
+ if completed % 100 == 0 or completed == len(tasks):
97
+ print(f"Progress: {completed}/{len(tasks)} queries completed...")
98
+
99
+ df = pd.DataFrame(results, columns=["Base Keyword", "Search Pattern", "Long-Tail Keyword", "Search Intent"])
100
+
101
+ print("\nGenerating Excel report...")
102
+ with pd.ExcelWriter(filename, engine="openpyxl") as writer:
103
+ df.to_excel(writer, sheet_name="All Keywords", index=False)
104
+ for intent_name, group_df in df.groupby("Search Intent"):
105
+ sheet_title = intent_name.split()[0]
106
+ group_df.to_excel(writer, sheet_name=sheet_title, index=False)
107
+
108
+ summary_df = df.groupby(["Base Keyword", "Search Intent"]).size().unstack(fill_value=0)
109
+ summary_df.to_excel(writer, sheet_name="Intent Summary")
110
+
111
+ print(f"\nCompleted! Total {len(df)} unique keywords fetched.")
112
+ print(f"Saved File: {os.path.abspath(filename)}")
113
+
114
+ def main():
115
+ parser = argparse.ArgumentParser(description="Advanced Google Autocomplete Keyword Scraper CLI Tool")
116
+ parser.add_argument("keywords", nargs="+", help="Base keywords to scrape (e.g. 'cotton kurti' 'western gown')")
117
+ parser.add_argument("--country", default="in", help="Target country code (e.g. 'in' for India, 'bd' for Bangladesh)")
118
+ parser.add_argument("--lang", default="en", help="Language code (e.g. 'en', 'bn')")
119
+ parser.add_argument("--threads", type=int, default=12, help="Number of concurrent threads")
120
+ parser.add_argument("--output", default="SEO_Keywords", help="Folder name to save results")
121
+
122
+ args = parser.parse_args()
123
+ run_scraper(args.keywords, args.country, args.lang, args.threads, args.output)
124
+
125
+ if __name__ == "__main__":
126
+ main()
@@ -0,0 +1,2 @@
1
+ __version__ = "0.1.0"
2
+ __author__ = "LORD SRIJWN"