txt2phrases 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- txt2phrases-0.1.0/PKG-INFO +134 -0
- txt2phrases-0.1.0/README.md +114 -0
- txt2phrases-0.1.0/pyproject.toml +3 -0
- txt2phrases-0.1.0/setup.cfg +36 -0
- txt2phrases-0.1.0/txt2phrases/__init__.py +9 -0
- txt2phrases-0.1.0/txt2phrases/classify_specific.py +122 -0
- txt2phrases-0.1.0/txt2phrases/html2txt.py +46 -0
- txt2phrases-0.1.0/txt2phrases/keyword.py +160 -0
- txt2phrases-0.1.0/txt2phrases.egg-info/PKG-INFO +134 -0
- txt2phrases-0.1.0/txt2phrases.egg-info/SOURCES.txt +13 -0
- txt2phrases-0.1.0/txt2phrases.egg-info/dependency_links.txt +1 -0
- txt2phrases-0.1.0/txt2phrases.egg-info/entry_points.txt +4 -0
- txt2phrases-0.1.0/txt2phrases.egg-info/requires.txt +5 -0
- txt2phrases-0.1.0/txt2phrases.egg-info/top_level.txt +1 -0
|
@@ -0,0 +1,134 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: txt2phrases
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: A Python library for HTML to TXT conversion, keyword extraction, and TF-IDF-based per-chapter classification.
|
|
5
|
+
Home-page: https://github.com/semanticClimate/encyclopedia/tree/main/txt2phrases
|
|
6
|
+
Author: Udita Agawal
|
|
7
|
+
Author-email: udita20agarwal@example.com
|
|
8
|
+
Maintainer: Renu Kumari
|
|
9
|
+
Maintainer-email: rk_2013@nipgr.ac.in
|
|
10
|
+
Classifier: Programming Language :: Python :: 3
|
|
11
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
12
|
+
Classifier: Operating System :: OS Independent
|
|
13
|
+
Requires-Python: >=3.8
|
|
14
|
+
Description-Content-Type: text/markdown
|
|
15
|
+
Requires-Dist: beautifulsoup4
|
|
16
|
+
Requires-Dist: pandas
|
|
17
|
+
Requires-Dist: tqdm
|
|
18
|
+
Requires-Dist: transformers
|
|
19
|
+
Requires-Dist: scikit-learn
|
|
20
|
+
|
|
21
|
+
```
|
|
22
|
+
## txt2phrases
|
|
23
|
+
|
|
24
|
+
A Python library for:
|
|
25
|
+
|
|
26
|
+
1. **HTML to TXT conversion**
|
|
27
|
+
2. **Keyword extraction using Hugging Face Transformers**
|
|
28
|
+
3. **Per-chapter TF-IDF-based specific keyword classification**
|
|
29
|
+
|
|
30
|
+
---
|
|
31
|
+
|
|
32
|
+
## Installation
|
|
33
|
+
|
|
34
|
+
You can install `txt2phrases` directly from PyPI:
|
|
35
|
+
|
|
36
|
+
```bash
|
|
37
|
+
pip install txt2phrases
|
|
38
|
+
```
|
|
39
|
+
|
|
40
|
+
---
|
|
41
|
+
|
|
42
|
+
## CLI Usage
|
|
43
|
+
|
|
44
|
+
## Convert HTML → TXT
|
|
45
|
+
|
|
46
|
+
Convert all HTML files in a folder to plain text:
|
|
47
|
+
|
|
48
|
+
```bash
|
|
49
|
+
html2text -i path/to/html_folder -o path/to/output_folder
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
- **-i / --input** : Path to the folder containing HTML files
|
|
53
|
+
- **-o / --output** : Path to the folder where TXT files will be saved
|
|
54
|
+
|
|
55
|
+
## Extract keywords from TXT files
|
|
56
|
+
|
|
57
|
+
Extract top keywords from all TXT files in a folder:
|
|
58
|
+
|
|
59
|
+
```bash
|
|
60
|
+
extract_keywords -i path/to/txt_folder -o path/to/output_folder -n 3500
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
- **-i / --input_folder** : Folder containing TXT files
|
|
64
|
+
- **-o / --output_folder** : Folder to save keyword CSVs
|
|
65
|
+
- **-n / --top_n** : Number of top keywords to extract (default: 3500)
|
|
66
|
+
|
|
67
|
+
## Generate per-chapter specific keywords (TF-IDF)
|
|
68
|
+
|
|
69
|
+
Create per-chapter CSVs listing keywords specific to each chapter:
|
|
70
|
+
|
|
71
|
+
```bash
|
|
72
|
+
specific_keywords -i path/to/csv_folder -o path/to/output_folder -t 0.6 -f 5
|
|
73
|
+
```
|
|
74
|
+
|
|
75
|
+
- **-i / --input_dir** : Folder with per-chapter CSV files containing `keyword,count`
|
|
76
|
+
- **-o / --output_dir** : Folder to save per-chapter specific keyword CSVs
|
|
77
|
+
- **-t / --threshold** : TF-IDF threshold for a keyword to be considered specific (default: 0.6)
|
|
78
|
+
- **-f / --min_freq** : Minimum frequency of a keyword to consider (default: 5)
|
|
79
|
+
|
|
80
|
+
---
|
|
81
|
+
|
|
82
|
+
## Python Usage
|
|
83
|
+
|
|
84
|
+
## Convert HTML → TXT
|
|
85
|
+
|
|
86
|
+
```python
|
|
87
|
+
from txt2phrases.html2text import html_to_txt_folder
|
|
88
|
+
|
|
89
|
+
html_to_txt_folder("path/to/html_folder", "path/to/output_folder")
|
|
90
|
+
```
|
|
91
|
+
|
|
92
|
+
## Extract Keywords
|
|
93
|
+
|
|
94
|
+
```python
|
|
95
|
+
from txt2phrases.keyword import KeywordExtraction
|
|
96
|
+
|
|
97
|
+
extractor = KeywordExtraction(
|
|
98
|
+
textfile="path/to/file.txt",
|
|
99
|
+
saving_path="path/to/output_folder",
|
|
100
|
+
output_filename="keywords.csv",
|
|
101
|
+
top_n=1000
|
|
102
|
+
)
|
|
103
|
+
|
|
104
|
+
top_keywords = extractor.extract_keywords()
|
|
105
|
+
```
|
|
106
|
+
|
|
107
|
+
## Per-Chapter Specific Keywords
|
|
108
|
+
|
|
109
|
+
```python
|
|
110
|
+
from txt2phrases.classify_specific import classify_keywords_split_files
|
|
111
|
+
|
|
112
|
+
classify_keywords_split_files(
|
|
113
|
+
input_dir="path/to/chapter_csv_folder",
|
|
114
|
+
output_dir="path/to/output_folder",
|
|
115
|
+
threshold=0.6,
|
|
116
|
+
min_freq=5
|
|
117
|
+
)
|
|
118
|
+
```
|
|
119
|
+
|
|
120
|
+
---
|
|
121
|
+
|
|
122
|
+
## Requirements
|
|
123
|
+
|
|
124
|
+
- Python 3.8+
|
|
125
|
+
- `beautifulsoup4`
|
|
126
|
+
- `pandas`
|
|
127
|
+
- `tqdm`
|
|
128
|
+
- `transformers`
|
|
129
|
+
- `scikit-learn`
|
|
130
|
+
|
|
131
|
+
---
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
```
|
|
@@ -0,0 +1,114 @@
|
|
|
1
|
+
```
|
|
2
|
+
## txt2phrases
|
|
3
|
+
|
|
4
|
+
A Python library for:
|
|
5
|
+
|
|
6
|
+
1. **HTML to TXT conversion**
|
|
7
|
+
2. **Keyword extraction using Hugging Face Transformers**
|
|
8
|
+
3. **Per-chapter TF-IDF-based specific keyword classification**
|
|
9
|
+
|
|
10
|
+
---
|
|
11
|
+
|
|
12
|
+
## Installation
|
|
13
|
+
|
|
14
|
+
You can install `txt2phrases` directly from PyPI:
|
|
15
|
+
|
|
16
|
+
```bash
|
|
17
|
+
pip install txt2phrases
|
|
18
|
+
```
|
|
19
|
+
|
|
20
|
+
---
|
|
21
|
+
|
|
22
|
+
## CLI Usage
|
|
23
|
+
|
|
24
|
+
## Convert HTML → TXT
|
|
25
|
+
|
|
26
|
+
Convert all HTML files in a folder to plain text:
|
|
27
|
+
|
|
28
|
+
```bash
|
|
29
|
+
html2text -i path/to/html_folder -o path/to/output_folder
|
|
30
|
+
```
|
|
31
|
+
|
|
32
|
+
- **-i / --input** : Path to the folder containing HTML files
|
|
33
|
+
- **-o / --output** : Path to the folder where TXT files will be saved
|
|
34
|
+
|
|
35
|
+
## Extract keywords from TXT files
|
|
36
|
+
|
|
37
|
+
Extract top keywords from all TXT files in a folder:
|
|
38
|
+
|
|
39
|
+
```bash
|
|
40
|
+
extract_keywords -i path/to/txt_folder -o path/to/output_folder -n 3500
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
- **-i / --input_folder** : Folder containing TXT files
|
|
44
|
+
- **-o / --output_folder** : Folder to save keyword CSVs
|
|
45
|
+
- **-n / --top_n** : Number of top keywords to extract (default: 3500)
|
|
46
|
+
|
|
47
|
+
## Generate per-chapter specific keywords (TF-IDF)
|
|
48
|
+
|
|
49
|
+
Create per-chapter CSVs listing keywords specific to each chapter:
|
|
50
|
+
|
|
51
|
+
```bash
|
|
52
|
+
specific_keywords -i path/to/csv_folder -o path/to/output_folder -t 0.6 -f 5
|
|
53
|
+
```
|
|
54
|
+
|
|
55
|
+
- **-i / --input_dir** : Folder with per-chapter CSV files containing `keyword,count`
|
|
56
|
+
- **-o / --output_dir** : Folder to save per-chapter specific keyword CSVs
|
|
57
|
+
- **-t / --threshold** : TF-IDF threshold for a keyword to be considered specific (default: 0.6)
|
|
58
|
+
- **-f / --min_freq** : Minimum frequency of a keyword to consider (default: 5)
|
|
59
|
+
|
|
60
|
+
---
|
|
61
|
+
|
|
62
|
+
## Python Usage
|
|
63
|
+
|
|
64
|
+
## Convert HTML → TXT
|
|
65
|
+
|
|
66
|
+
```python
|
|
67
|
+
from txt2phrases.html2text import html_to_txt_folder
|
|
68
|
+
|
|
69
|
+
html_to_txt_folder("path/to/html_folder", "path/to/output_folder")
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
## Extract Keywords
|
|
73
|
+
|
|
74
|
+
```python
|
|
75
|
+
from txt2phrases.keyword import KeywordExtraction
|
|
76
|
+
|
|
77
|
+
extractor = KeywordExtraction(
|
|
78
|
+
textfile="path/to/file.txt",
|
|
79
|
+
saving_path="path/to/output_folder",
|
|
80
|
+
output_filename="keywords.csv",
|
|
81
|
+
top_n=1000
|
|
82
|
+
)
|
|
83
|
+
|
|
84
|
+
top_keywords = extractor.extract_keywords()
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
## Per-Chapter Specific Keywords
|
|
88
|
+
|
|
89
|
+
```python
|
|
90
|
+
from txt2phrases.classify_specific import classify_keywords_split_files
|
|
91
|
+
|
|
92
|
+
classify_keywords_split_files(
|
|
93
|
+
input_dir="path/to/chapter_csv_folder",
|
|
94
|
+
output_dir="path/to/output_folder",
|
|
95
|
+
threshold=0.6,
|
|
96
|
+
min_freq=5
|
|
97
|
+
)
|
|
98
|
+
```
|
|
99
|
+
|
|
100
|
+
---
|
|
101
|
+
|
|
102
|
+
## Requirements
|
|
103
|
+
|
|
104
|
+
- Python 3.8+
|
|
105
|
+
- `beautifulsoup4`
|
|
106
|
+
- `pandas`
|
|
107
|
+
- `tqdm`
|
|
108
|
+
- `transformers`
|
|
109
|
+
- `scikit-learn`
|
|
110
|
+
|
|
111
|
+
---
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
```
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
[metadata]
|
|
2
|
+
name = txt2phrases
|
|
3
|
+
version = 0.1.0
|
|
4
|
+
author = Udita Agawal
|
|
5
|
+
author_email = udita20agarwal@example.com
|
|
6
|
+
maintainer = Renu Kumari
|
|
7
|
+
maintainer_email = rk_2013@nipgr.ac.in
|
|
8
|
+
description = A Python library for HTML to TXT conversion, keyword extraction, and TF-IDF-based per-chapter classification.
|
|
9
|
+
long_description = file: README.md
|
|
10
|
+
long_description_content_type = text/markdown
|
|
11
|
+
url = https://github.com/semanticClimate/encyclopedia/tree/main/txt2phrases
|
|
12
|
+
classifiers =
|
|
13
|
+
Programming Language :: Python :: 3
|
|
14
|
+
License :: OSI Approved :: MIT License
|
|
15
|
+
Operating System :: OS Independent
|
|
16
|
+
|
|
17
|
+
[options]
|
|
18
|
+
packages = find:
|
|
19
|
+
python_requires = >=3.8
|
|
20
|
+
install_requires =
|
|
21
|
+
beautifulsoup4
|
|
22
|
+
pandas
|
|
23
|
+
tqdm
|
|
24
|
+
transformers
|
|
25
|
+
scikit-learn
|
|
26
|
+
|
|
27
|
+
[options.entry_points]
|
|
28
|
+
console_scripts =
|
|
29
|
+
html2txt = txt2phrases.html2txt:main
|
|
30
|
+
extract_keywords = txt2phrases.keyword:main
|
|
31
|
+
specific_keywords = txt2phrases.classify_specific:main
|
|
32
|
+
|
|
33
|
+
[egg_info]
|
|
34
|
+
tag_build =
|
|
35
|
+
tag_date = 0
|
|
36
|
+
|
|
@@ -0,0 +1,122 @@
|
|
|
1
|
+
import os
|
|
2
|
+
import pandas as pd
|
|
3
|
+
from collections import defaultdict
|
|
4
|
+
from argparse import ArgumentParser
|
|
5
|
+
from sklearn.feature_extraction.text import TfidfTransformer
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def classify_keywords_split_files(input_dir, output_dir, threshold=0.6, min_freq=5):
|
|
9
|
+
"""
|
|
10
|
+
For each chapter, create a separate CSV listing keywords that are 'specific' to it
|
|
11
|
+
according to TF-IDF >= threshold.
|
|
12
|
+
|
|
13
|
+
Input: multiple chapter CSVs in `input_dir`, each with columns: keyword, count
|
|
14
|
+
Output: one file per chapter in `output_dir`: <chapter>_specific_keywords.csv
|
|
15
|
+
columns -> keyword, tfidf, count
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
os.makedirs(output_dir, exist_ok=True)
|
|
19
|
+
|
|
20
|
+
# --- Load per-chapter frequencies (with min_freq filtering) ---
|
|
21
|
+
chapter_files = [f for f in os.listdir(input_dir) if f.endswith(".csv")]
|
|
22
|
+
if not chapter_files:
|
|
23
|
+
print(f"No CSV files found in {input_dir}")
|
|
24
|
+
return
|
|
25
|
+
|
|
26
|
+
keyword_chapter_freq = defaultdict(dict) # {keyword: {chapter: count}}
|
|
27
|
+
chapters = []
|
|
28
|
+
|
|
29
|
+
for file in chapter_files:
|
|
30
|
+
chapter = os.path.splitext(file)[0]
|
|
31
|
+
chapters.append(chapter)
|
|
32
|
+
df = pd.read_csv(os.path.join(input_dir, file))
|
|
33
|
+
# Ensure required columns exist
|
|
34
|
+
if not {"keyword", "count"}.issubset(df.columns):
|
|
35
|
+
raise ValueError(f"{file} must contain 'keyword' and 'count' columns")
|
|
36
|
+
|
|
37
|
+
# Filter by min frequency
|
|
38
|
+
df = df[df["count"] >= min_freq]
|
|
39
|
+
|
|
40
|
+
for _, row in df.iterrows():
|
|
41
|
+
keyword = str(row["keyword"])
|
|
42
|
+
freq = int(row["count"])
|
|
43
|
+
keyword_chapter_freq[keyword][chapter] = freq
|
|
44
|
+
|
|
45
|
+
chapters = sorted(chapters)
|
|
46
|
+
keywords = list(keyword_chapter_freq.keys())
|
|
47
|
+
|
|
48
|
+
# Handle edge case: no keywords survive min_freq
|
|
49
|
+
if not keywords:
|
|
50
|
+
# still create empty files per chapter with headers
|
|
51
|
+
for chapter in chapters:
|
|
52
|
+
out_path = os.path.join(output_dir, f"{chapter}_specific_keywords.csv")
|
|
53
|
+
pd.DataFrame(columns=["keyword", "tfidf", "count"]).to_csv(out_path, index=False)
|
|
54
|
+
print(f"Saved (empty) {out_path}")
|
|
55
|
+
return
|
|
56
|
+
|
|
57
|
+
# --- Build keyword x chapter matrix ---
|
|
58
|
+
data = []
|
|
59
|
+
for kw in keywords:
|
|
60
|
+
row = [keyword_chapter_freq[kw].get(ch, 0) for ch in chapters]
|
|
61
|
+
data.append(row)
|
|
62
|
+
df_matrix = pd.DataFrame(data, index=keywords, columns=chapters)
|
|
63
|
+
|
|
64
|
+
# --- TF-IDF over (keywords x chapters) ---
|
|
65
|
+
transformer = TfidfTransformer()
|
|
66
|
+
tfidf_matrix = transformer.fit_transform(df_matrix) # sparse
|
|
67
|
+
|
|
68
|
+
# --- For each chapter, collect keywords specific to it ---
|
|
69
|
+
for j, chapter in enumerate(chapters):
|
|
70
|
+
rows = []
|
|
71
|
+
for i, keyword in enumerate(df_matrix.index):
|
|
72
|
+
score = tfidf_matrix[i, j]
|
|
73
|
+
if score >= threshold:
|
|
74
|
+
freq_in_chapter = int(df_matrix.iloc[i, j])
|
|
75
|
+
# Optionally skip if freq is zero (rare when score >= threshold, but safe):
|
|
76
|
+
if freq_in_chapter > 0:
|
|
77
|
+
rows.append((keyword, float(score), freq_in_chapter))
|
|
78
|
+
|
|
79
|
+
# Sort by TF-IDF descending, then by frequency descending
|
|
80
|
+
rows.sort(key=lambda x: (x[1], x[2]), reverse=True)
|
|
81
|
+
|
|
82
|
+
out_df = pd.DataFrame(rows, columns=["keyword", "tfidf", "count"])
|
|
83
|
+
out_path = os.path.join(output_dir, f"{chapter}_specific_keywords.csv")
|
|
84
|
+
out_df.to_csv(out_path, index=False)
|
|
85
|
+
print(f"Saved {out_path} ({len(out_df)} keywords)")
|
|
86
|
+
|
|
87
|
+
def main():
|
|
88
|
+
import argparse
|
|
89
|
+
from .classify_specific import classify_keywords_split_files
|
|
90
|
+
|
|
91
|
+
parser = argparse.ArgumentParser(
|
|
92
|
+
description="Create per-chapter files of TF-IDF-specific keywords"
|
|
93
|
+
)
|
|
94
|
+
parser.add_argument(
|
|
95
|
+
"-i", "--input_dir", required=True,
|
|
96
|
+
help="Path to folder with chapter CSVs (keyword,count)"
|
|
97
|
+
)
|
|
98
|
+
parser.add_argument(
|
|
99
|
+
"-o", "--output_dir", required=True,
|
|
100
|
+
help="Path to save per-chapter specific keyword files"
|
|
101
|
+
)
|
|
102
|
+
parser.add_argument(
|
|
103
|
+
"-t", "--threshold", type=float, default=0.6,
|
|
104
|
+
help="TF-IDF threshold for a keyword to be considered specific"
|
|
105
|
+
)
|
|
106
|
+
parser.add_argument(
|
|
107
|
+
"-f", "--min_freq", type=int, default=5,
|
|
108
|
+
help="Minimum frequency of a keyword to consider"
|
|
109
|
+
)
|
|
110
|
+
|
|
111
|
+
args = parser.parse_args()
|
|
112
|
+
|
|
113
|
+
classify_keywords_split_files(
|
|
114
|
+
input_dir=args.input_dir,
|
|
115
|
+
output_dir=args.output_dir,
|
|
116
|
+
threshold=args.threshold,
|
|
117
|
+
min_freq=args.min_freq
|
|
118
|
+
)
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
if __name__ == "__main__":
|
|
122
|
+
main()
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
import os
|
|
2
|
+
import argparse
|
|
3
|
+
from bs4 import BeautifulSoup
|
|
4
|
+
|
|
5
|
+
def html_to_txt_folder(input_folder, output_folder):
|
|
6
|
+
# Create output folder if it doesn't exist
|
|
7
|
+
os.makedirs(output_folder, exist_ok=True)
|
|
8
|
+
|
|
9
|
+
# Loop through all files in the input folder
|
|
10
|
+
for filename in os.listdir(input_folder):
|
|
11
|
+
if filename.endswith(".html"):
|
|
12
|
+
input_path = os.path.join(input_folder, filename)
|
|
13
|
+
output_filename = os.path.splitext(filename)[0] + ".txt"
|
|
14
|
+
output_path = os.path.join(output_folder, output_filename)
|
|
15
|
+
|
|
16
|
+
# Read and parse HTML
|
|
17
|
+
with open(input_path, "r", encoding="utf-8") as f:
|
|
18
|
+
soup = BeautifulSoup(f, "html.parser")
|
|
19
|
+
text = soup.get_text()
|
|
20
|
+
|
|
21
|
+
# Write plain text to new file
|
|
22
|
+
with open(output_path, "w", encoding="utf-8") as f:
|
|
23
|
+
f.write(text)
|
|
24
|
+
|
|
25
|
+
print(f"Converted: {filename} → {output_filename}")
|
|
26
|
+
|
|
27
|
+
def main():
|
|
28
|
+
import argparse
|
|
29
|
+
from .html2txt import html_to_txt_folder # your existing function
|
|
30
|
+
|
|
31
|
+
parser = argparse.ArgumentParser(
|
|
32
|
+
description="Convert HTML files to TXT files"
|
|
33
|
+
)
|
|
34
|
+
parser.add_argument(
|
|
35
|
+
"-i", "--input", required=True, help="Input folder containing HTML files"
|
|
36
|
+
)
|
|
37
|
+
parser.add_argument(
|
|
38
|
+
"-o", "--output", required=True, help="Output folder for TXT files"
|
|
39
|
+
)
|
|
40
|
+
|
|
41
|
+
args = parser.parse_args()
|
|
42
|
+
html_to_txt_folder(args.input, args.output)
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
if __name__ == "__main__":
|
|
46
|
+
main()
|
|
@@ -0,0 +1,160 @@
|
|
|
1
|
+
import os
|
|
2
|
+
import re
|
|
3
|
+
from collections import Counter
|
|
4
|
+
import pandas as pd
|
|
5
|
+
from tqdm import tqdm
|
|
6
|
+
from argparse import ArgumentParser
|
|
7
|
+
from transformers import (
|
|
8
|
+
TokenClassificationPipeline,
|
|
9
|
+
AutoModelForTokenClassification,
|
|
10
|
+
AutoTokenizer,
|
|
11
|
+
)
|
|
12
|
+
from transformers.pipelines import AggregationStrategy
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
# -----------------------------
|
|
16
|
+
# Keyphrase Extraction Pipeline
|
|
17
|
+
# -----------------------------
|
|
18
|
+
class KeyphraseExtractionPipeline(TokenClassificationPipeline):
|
|
19
|
+
"""
|
|
20
|
+
A customized Hugging Face TokenClassificationPipeline
|
|
21
|
+
for extracting keywords/keyphrases.
|
|
22
|
+
"""
|
|
23
|
+
|
|
24
|
+
def __init__(self, model_name, *args, **kwargs):
|
|
25
|
+
super().__init__(
|
|
26
|
+
model=AutoModelForTokenClassification.from_pretrained(model_name),
|
|
27
|
+
tokenizer=AutoTokenizer.from_pretrained(model_name),
|
|
28
|
+
*args,
|
|
29
|
+
**kwargs
|
|
30
|
+
)
|
|
31
|
+
|
|
32
|
+
def postprocess(self, *args, **kwargs):
|
|
33
|
+
results = super().postprocess(
|
|
34
|
+
*args,
|
|
35
|
+
aggregation_strategy=AggregationStrategy.SIMPLE,
|
|
36
|
+
**kwargs
|
|
37
|
+
)
|
|
38
|
+
return [result.get("word").strip() for result in results if result.get("word")]
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
# -----------------------------
|
|
42
|
+
# Keyword Extraction Class
|
|
43
|
+
# -----------------------------
|
|
44
|
+
class KeywordExtraction:
|
|
45
|
+
"""
|
|
46
|
+
Extracts the most important keywords from a text file using a model.
|
|
47
|
+
"""
|
|
48
|
+
|
|
49
|
+
def __init__(self, textfile, saving_path, output_filename="keywords.csv", top_n=1000):
|
|
50
|
+
self.text = []
|
|
51
|
+
self.keyphrases = []
|
|
52
|
+
self.output_filename = output_filename
|
|
53
|
+
self.top_n = top_n
|
|
54
|
+
|
|
55
|
+
if textfile and os.path.isfile(textfile) and textfile.endswith(".txt"):
|
|
56
|
+
self.textfile = textfile
|
|
57
|
+
else:
|
|
58
|
+
raise ValueError('Please provide a valid text file path ending with ".txt"')
|
|
59
|
+
|
|
60
|
+
if os.path.isdir(saving_path):
|
|
61
|
+
self.saving_path = saving_path
|
|
62
|
+
else:
|
|
63
|
+
raise ValueError('Please provide a valid saving path')
|
|
64
|
+
|
|
65
|
+
# -----------------------------
|
|
66
|
+
# Read and split text file
|
|
67
|
+
# -----------------------------
|
|
68
|
+
def read_from_text_file(self, method="sentence"):
|
|
69
|
+
with open(self.textfile, encoding="utf-8") as f:
|
|
70
|
+
full_text = f.read().strip()
|
|
71
|
+
|
|
72
|
+
if method == "sentence":
|
|
73
|
+
self.text = re.split(r'(?<=[.!?])\s+', full_text)
|
|
74
|
+
elif method == "chunk":
|
|
75
|
+
words = full_text.split()
|
|
76
|
+
chunk_size = 300
|
|
77
|
+
self.text = [" ".join(words[i:i+chunk_size]) for i in range(0, len(words), chunk_size)]
|
|
78
|
+
else:
|
|
79
|
+
self.text = [full_text]
|
|
80
|
+
|
|
81
|
+
print(f"Total text chunks to process: {len(self.text)}")
|
|
82
|
+
print(f"First chunk preview: {self.text[0][:200]}...\n")
|
|
83
|
+
|
|
84
|
+
# -----------------------------
|
|
85
|
+
# Extract Keywords in batches
|
|
86
|
+
# -----------------------------
|
|
87
|
+
def extract_keywords(self, batch_size=16):
|
|
88
|
+
self.read_from_text_file(method="sentence")
|
|
89
|
+
|
|
90
|
+
model_name = "ml6team/keyphrase-extraction-kbir-inspec"
|
|
91
|
+
extractor = KeyphraseExtractionPipeline(model_name=model_name)
|
|
92
|
+
|
|
93
|
+
for i in tqdm(range(0, len(self.text), batch_size), desc="Extracting keywords"):
|
|
94
|
+
batch_lines = self.text[i:i + batch_size]
|
|
95
|
+
batch_keyphrases_list = extractor(batch_lines)
|
|
96
|
+
for keyphrases in batch_keyphrases_list:
|
|
97
|
+
self.keyphrases.extend(keyphrases)
|
|
98
|
+
|
|
99
|
+
# Count keywords
|
|
100
|
+
self.keyphrase_counts = Counter(self.keyphrases)
|
|
101
|
+
self.keyphrases = list(set(self.keyphrases))
|
|
102
|
+
|
|
103
|
+
# Top N
|
|
104
|
+
top_keywords = [kw for kw, _ in self.keyphrase_counts.most_common(self.top_n)]
|
|
105
|
+
|
|
106
|
+
# Save CSV
|
|
107
|
+
os.makedirs(self.saving_path, exist_ok=True)
|
|
108
|
+
output_file = os.path.join(self.saving_path, self.output_filename)
|
|
109
|
+
df = pd.DataFrame(self.keyphrase_counts.most_common(self.top_n), columns=["keyword", "count"])
|
|
110
|
+
df.to_csv(output_file, index=False)
|
|
111
|
+
print(f"\nCSV saved successfully: {output_file}")
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
print(f"Total unique keywords: {len(self.keyphrases)}")
|
|
115
|
+
print(f"Top 10 keywords: {self.keyphrase_counts.most_common(10)}")
|
|
116
|
+
return top_keywords
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
# -----------------------------
|
|
120
|
+
# CLI
|
|
121
|
+
# -----------------------------
|
|
122
|
+
def main():
|
|
123
|
+
import argparse
|
|
124
|
+
from .keyword import KeywordExtraction
|
|
125
|
+
import os
|
|
126
|
+
|
|
127
|
+
parser = argparse.ArgumentParser(
|
|
128
|
+
description="Extract keywords from all TXT files in a folder"
|
|
129
|
+
)
|
|
130
|
+
parser.add_argument(
|
|
131
|
+
"-i", "--input_folder", required=True, help="Folder containing TXT files"
|
|
132
|
+
)
|
|
133
|
+
parser.add_argument(
|
|
134
|
+
"-o", "--output_folder", required=True, help="Folder to save keyword CSVs"
|
|
135
|
+
)
|
|
136
|
+
parser.add_argument(
|
|
137
|
+
"-n", "--top_n", type=int, default=3500, help="Number of top keywords to extract"
|
|
138
|
+
)
|
|
139
|
+
|
|
140
|
+
args = parser.parse_args()
|
|
141
|
+
|
|
142
|
+
os.makedirs(args.output_folder, exist_ok=True)
|
|
143
|
+
txt_files = [f for f in os.listdir(args.input_folder) if f.endswith(".txt")]
|
|
144
|
+
|
|
145
|
+
for txt_file in txt_files:
|
|
146
|
+
input_path = os.path.join(args.input_folder, txt_file)
|
|
147
|
+
base_name = os.path.splitext(txt_file)[0]
|
|
148
|
+
output_filename = base_name + "_keywords.csv"
|
|
149
|
+
|
|
150
|
+
extractor = KeywordExtraction(
|
|
151
|
+
textfile=input_path,
|
|
152
|
+
saving_path=args.output_folder,
|
|
153
|
+
output_filename=output_filename,
|
|
154
|
+
top_n=args.top_n
|
|
155
|
+
)
|
|
156
|
+
extractor.extract_keywords()
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
if __name__ == "__main__":
|
|
160
|
+
main()
|
|
@@ -0,0 +1,134 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: txt2phrases
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: A Python library for HTML to TXT conversion, keyword extraction, and TF-IDF-based per-chapter classification.
|
|
5
|
+
Home-page: https://github.com/semanticClimate/encyclopedia/tree/main/txt2phrases
|
|
6
|
+
Author: Udita Agawal
|
|
7
|
+
Author-email: udita20agarwal@example.com
|
|
8
|
+
Maintainer: Renu Kumari
|
|
9
|
+
Maintainer-email: rk_2013@nipgr.ac.in
|
|
10
|
+
Classifier: Programming Language :: Python :: 3
|
|
11
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
12
|
+
Classifier: Operating System :: OS Independent
|
|
13
|
+
Requires-Python: >=3.8
|
|
14
|
+
Description-Content-Type: text/markdown
|
|
15
|
+
Requires-Dist: beautifulsoup4
|
|
16
|
+
Requires-Dist: pandas
|
|
17
|
+
Requires-Dist: tqdm
|
|
18
|
+
Requires-Dist: transformers
|
|
19
|
+
Requires-Dist: scikit-learn
|
|
20
|
+
|
|
21
|
+
```
|
|
22
|
+
## txt2phrases
|
|
23
|
+
|
|
24
|
+
A Python library for:
|
|
25
|
+
|
|
26
|
+
1. **HTML to TXT conversion**
|
|
27
|
+
2. **Keyword extraction using Hugging Face Transformers**
|
|
28
|
+
3. **Per-chapter TF-IDF-based specific keyword classification**
|
|
29
|
+
|
|
30
|
+
---
|
|
31
|
+
|
|
32
|
+
## Installation
|
|
33
|
+
|
|
34
|
+
You can install `txt2phrases` directly from PyPI:
|
|
35
|
+
|
|
36
|
+
```bash
|
|
37
|
+
pip install txt2phrases
|
|
38
|
+
```
|
|
39
|
+
|
|
40
|
+
---
|
|
41
|
+
|
|
42
|
+
## CLI Usage
|
|
43
|
+
|
|
44
|
+
## Convert HTML → TXT
|
|
45
|
+
|
|
46
|
+
Convert all HTML files in a folder to plain text:
|
|
47
|
+
|
|
48
|
+
```bash
|
|
49
|
+
html2text -i path/to/html_folder -o path/to/output_folder
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
- **-i / --input** : Path to the folder containing HTML files
|
|
53
|
+
- **-o / --output** : Path to the folder where TXT files will be saved
|
|
54
|
+
|
|
55
|
+
## Extract keywords from TXT files
|
|
56
|
+
|
|
57
|
+
Extract top keywords from all TXT files in a folder:
|
|
58
|
+
|
|
59
|
+
```bash
|
|
60
|
+
extract_keywords -i path/to/txt_folder -o path/to/output_folder -n 3500
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
- **-i / --input_folder** : Folder containing TXT files
|
|
64
|
+
- **-o / --output_folder** : Folder to save keyword CSVs
|
|
65
|
+
- **-n / --top_n** : Number of top keywords to extract (default: 3500)
|
|
66
|
+
|
|
67
|
+
## Generate per-chapter specific keywords (TF-IDF)
|
|
68
|
+
|
|
69
|
+
Create per-chapter CSVs listing keywords specific to each chapter:
|
|
70
|
+
|
|
71
|
+
```bash
|
|
72
|
+
specific_keywords -i path/to/csv_folder -o path/to/output_folder -t 0.6 -f 5
|
|
73
|
+
```
|
|
74
|
+
|
|
75
|
+
- **-i / --input_dir** : Folder with per-chapter CSV files containing `keyword,count`
|
|
76
|
+
- **-o / --output_dir** : Folder to save per-chapter specific keyword CSVs
|
|
77
|
+
- **-t / --threshold** : TF-IDF threshold for a keyword to be considered specific (default: 0.6)
|
|
78
|
+
- **-f / --min_freq** : Minimum frequency of a keyword to consider (default: 5)
|
|
79
|
+
|
|
80
|
+
---
|
|
81
|
+
|
|
82
|
+
## Python Usage
|
|
83
|
+
|
|
84
|
+
## Convert HTML → TXT
|
|
85
|
+
|
|
86
|
+
```python
|
|
87
|
+
from txt2phrases.html2text import html_to_txt_folder
|
|
88
|
+
|
|
89
|
+
html_to_txt_folder("path/to/html_folder", "path/to/output_folder")
|
|
90
|
+
```
|
|
91
|
+
|
|
92
|
+
## Extract Keywords
|
|
93
|
+
|
|
94
|
+
```python
|
|
95
|
+
from txt2phrases.keyword import KeywordExtraction
|
|
96
|
+
|
|
97
|
+
extractor = KeywordExtraction(
|
|
98
|
+
textfile="path/to/file.txt",
|
|
99
|
+
saving_path="path/to/output_folder",
|
|
100
|
+
output_filename="keywords.csv",
|
|
101
|
+
top_n=1000
|
|
102
|
+
)
|
|
103
|
+
|
|
104
|
+
top_keywords = extractor.extract_keywords()
|
|
105
|
+
```
|
|
106
|
+
|
|
107
|
+
## Per-Chapter Specific Keywords
|
|
108
|
+
|
|
109
|
+
```python
|
|
110
|
+
from txt2phrases.classify_specific import classify_keywords_split_files
|
|
111
|
+
|
|
112
|
+
classify_keywords_split_files(
|
|
113
|
+
input_dir="path/to/chapter_csv_folder",
|
|
114
|
+
output_dir="path/to/output_folder",
|
|
115
|
+
threshold=0.6,
|
|
116
|
+
min_freq=5
|
|
117
|
+
)
|
|
118
|
+
```
|
|
119
|
+
|
|
120
|
+
---
|
|
121
|
+
|
|
122
|
+
## Requirements
|
|
123
|
+
|
|
124
|
+
- Python 3.8+
|
|
125
|
+
- `beautifulsoup4`
|
|
126
|
+
- `pandas`
|
|
127
|
+
- `tqdm`
|
|
128
|
+
- `transformers`
|
|
129
|
+
- `scikit-learn`
|
|
130
|
+
|
|
131
|
+
---
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
```
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
README.md
|
|
2
|
+
pyproject.toml
|
|
3
|
+
setup.cfg
|
|
4
|
+
txt2phrases/__init__.py
|
|
5
|
+
txt2phrases/classify_specific.py
|
|
6
|
+
txt2phrases/html2txt.py
|
|
7
|
+
txt2phrases/keyword.py
|
|
8
|
+
txt2phrases.egg-info/PKG-INFO
|
|
9
|
+
txt2phrases.egg-info/SOURCES.txt
|
|
10
|
+
txt2phrases.egg-info/dependency_links.txt
|
|
11
|
+
txt2phrases.egg-info/entry_points.txt
|
|
12
|
+
txt2phrases.egg-info/requires.txt
|
|
13
|
+
txt2phrases.egg-info/top_level.txt
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
txt2phrases
|