imbed_data_prep 0.1.1__tar.gz → 0.1.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/PKG-INFO +2 -2
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/epstein_files/README.md +2 -2
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/lmsys_ai_conversations/__init__.py +0 -19
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/trump_vs_zelenskyy/trump_vs_zelensky.md +47 -55
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/pyproject.toml +1 -1
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/.gitattributes +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/.github/workflows/ci.yml +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/.gitignore +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/LICENSE +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/README.md +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/__init__.py +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/arxiv/README.md +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/arxiv/__init__.py +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/embeddings_of_aggregations/README.md +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/embeddings_of_aggregations/__init__.py +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/embeddings_of_aggregations/embeddings_and_order.ipynb +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/epstein_files/__init__.py +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/epstein_files/epstein_files.ipynb +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/epstein_files/epstein_files_tables_info.json +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/epstein_files/epstein_files_tables_info.pickle +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/eurovis/README.md +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/eurovis/__init__.py +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/eurovis/eurovis.ipynb +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/github_repos/README.md +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/github_repos/__init__.py +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/github_repos/github_repos.ipynb +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/hcp/README.md +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/hcp/__init__.py +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/hcp/hcp_analysis.ipynb +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/jersey_laws/README.md +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/jersey_laws/__init__.py +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/jersey_laws/jersey_laws.ipynb +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/lmsys_ai_conversations/README.md +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/mcdonalds_reviews/README.md +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/mcdonalds_reviews/__init__.py +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/mcdonalds_reviews/mcdonalds_reviews_dacc.ipynb +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/prompt_injections/README.md +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/prompt_injections/__init__.py +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/prompt_injections/prompt_injection_w_umap_embeddings.tsv +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/trump_vs_zelenskyy/README.md +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/trump_vs_zelenskyy/__init__.py +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/trump_vs_zelenskyy/trump_vs_zelenskyy.ipynb +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/trump_vs_zelenskyy/trump_vs_zelenskyy_embeddings.parquet +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/trump_vs_zelenskyy/trump_vs_zelenskyy_transcript.parquet +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/twitter_sentiment/README.md +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/twitter_sentiment/__init__.py +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/twitter_sentiment/twitter_sentiment.ipynb +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/ultra_chat/README.md +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/ultra_chat/__init__.py +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/ultra_chat/ultra_chat.ipynb +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/wildchat/README.md +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/wildchat/__init__.py +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/wildchat/wildchat.ipynb +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/wordnet_words/README.md +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/wordnet_words/__init__.py +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/wordnet_words/test_synset_refactor.ipynb +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/wordnet_words/wordnet_words.ipynb +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/misc/REFACTOR_SUMMARY.md +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/misc/ai_prompts.ipynb +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/misc/data/cheat_sheets/Cheat Sheet for Python Machine Learning and Data Science.pdf +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/misc/data/cheat_sheets/Data Science Cheat Sheets.pdf +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/misc/data/cheat_sheets/Data Science Cheatsheet.pdf +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/misc/data/cheat_sheets/ML Cheatsheet Documentation.pdf +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/misc/data/cheat_sheets/Machine Learning Cheat Sheet.pdf +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/misc/data/cheat_sheets/Machine Learning Interview Cheat Sheets.pdf +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/misc/data/cheat_sheets/Machine Learning and Data Science Cheat Sheet.pdf +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/misc/data/cheat_sheets/Scikit-Learn Cheat Sheet for Machine Learning.pdf +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/misc/data/cheat_sheets/Scikit-Learn Cheat Sheet: Python Machine Learning.pdf +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/misc/data/cheat_sheets/Scikit-Learn CheatSheet: Python Machine Learning Tutorial.pdf +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/misc/data/cheat_sheets/The Complete Collection of Data Science Cheat Sheets.pdf +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/misc/data/tmp.csv +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/misc/eurovis copy 2.ipynb +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/misc/eurovis copy.ipynb +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/misc/explore_refactored_data.py +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/misc/oa_embeddings_sentiment_models.pickle +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/misc/quick_test_refactor.py +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/misc/using_ai_to_get_data_descriptions.ipynb +0 -0
- {imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/misc/various_data_preps.ipynb +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
Metadata-Version: 2.
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
2
|
Name: imbed_data_prep
|
|
3
|
-
Version: 0.1.
|
|
3
|
+
Version: 0.1.2
|
|
4
4
|
Summary: Modules to acquire and prepare data for the imbed package.
|
|
5
5
|
Project-URL: Homepage, https://github.com/thorwhalen/imbed_data_prep
|
|
6
6
|
Project-URL: Repository, https://github.com/thorwhalen/imbed_data_prep
|
|
@@ -146,8 +146,8 @@ Transforms the raw parquet tables into Cosmograph-ready DataFrames:
|
|
|
146
146
|
from imbed_data_prep.epstein_files import prepare_cosmograph_data
|
|
147
147
|
|
|
148
148
|
cosmo_data = prepare_cosmograph_data(data_dir="path/to/parquets")
|
|
149
|
-
points = cosmo_data[
|
|
150
|
-
links
|
|
149
|
+
points = cosmo_data["points"] # canonical_name, degree, hop_distance
|
|
150
|
+
links = cosmo_data["links"] # source, target, weight, action, ...
|
|
151
151
|
```
|
|
152
152
|
|
|
153
153
|
The resulting DataFrames can be passed directly to Cosmograph:
|
{imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/lmsys_ai_conversations/__init__.py
RENAMED
|
@@ -765,22 +765,3 @@ def compute_and_save_kmeans(
|
|
|
765
765
|
dacc.saves[data_name] = kmeans_clusters
|
|
766
766
|
|
|
767
767
|
return kmeans_clusters
|
|
768
|
-
|
|
769
|
-
|
|
770
|
-
if __name__ == "__main__":
|
|
771
|
-
from argh import dispatch_commands
|
|
772
|
-
|
|
773
|
-
dispatch_commands(
|
|
774
|
-
[
|
|
775
|
-
compute_and_save_embeddings,
|
|
776
|
-
compute_and_save_incremental_pca,
|
|
777
|
-
compute_and_save_planar_embeddings,
|
|
778
|
-
compute_and_save_planar_embeddings_with_incremental_pca,
|
|
779
|
-
compute_and_save_planar_embeddings_light,
|
|
780
|
-
compute_and_save_grouped_embeddings,
|
|
781
|
-
compute_and_save_embeddings_pca,
|
|
782
|
-
compute_and_save_dbscan,
|
|
783
|
-
compute_and_save_kmeans,
|
|
784
|
-
compute_and_save_ncvis_planar_embeddings,
|
|
785
|
-
]
|
|
786
|
-
)
|
|
@@ -5,7 +5,7 @@
|
|
|
5
5
|
```python
|
|
6
6
|
import os
|
|
7
7
|
import re
|
|
8
|
-
import pandas as pd
|
|
8
|
+
import pandas as pd
|
|
9
9
|
import numpy as np
|
|
10
10
|
import requests
|
|
11
11
|
|
|
@@ -15,13 +15,13 @@ import tabled
|
|
|
15
15
|
|
|
16
16
|
```python
|
|
17
17
|
# settings
|
|
18
|
-
raw_src_url =
|
|
18
|
+
raw_src_url = "https://raw.githubusercontent.com/thorwhalen/content/refs/heads/master/text/trump-zelensky-2025-03-01--with_speakers.txt"
|
|
19
19
|
|
|
20
|
-
rootdir =
|
|
20
|
+
rootdir = "." # NOTE: Put your own rootdir here
|
|
21
21
|
|
|
22
22
|
# save keys (e.g. relative paths)
|
|
23
|
-
embeddings_save_key =
|
|
24
|
-
transcript_save_key =
|
|
23
|
+
embeddings_save_key = "data/trump_vs_zelenskyy_embeddings.parquet"
|
|
24
|
+
transcript_save_key = "data/trump_vs_zelenskyy_transcript.parquet"
|
|
25
25
|
```
|
|
26
26
|
|
|
27
27
|
|
|
@@ -40,12 +40,11 @@ transcript_text = requests.get(raw_src_url).text
|
|
|
40
40
|
# Every line of transcript_text starts with [speaker]: [text]
|
|
41
41
|
# Let's parse the transcript_text to get a list of (speaker, text) dicts
|
|
42
42
|
# Define a regex pattern to match the speaker and text
|
|
43
|
-
pattern = re.compile(r
|
|
43
|
+
pattern = re.compile(r"\[(?P<speaker>[^\]]+)\]: (?P<text>.*)")
|
|
44
44
|
|
|
45
45
|
# Parse the transcript_text to get a list of (speaker, text) dicts
|
|
46
46
|
transcript_dict_list = [
|
|
47
|
-
match.groupdict()
|
|
48
|
-
for match in pattern.finditer(transcript_text)
|
|
47
|
+
match.groupdict() for match in pattern.finditer(transcript_text)
|
|
49
48
|
]
|
|
50
49
|
|
|
51
50
|
transcript_df = pd.DataFrame(transcript_dict_list)
|
|
@@ -58,7 +57,7 @@ transcript_df
|
|
|
58
57
|
|
|
59
58
|
|
|
60
59
|
```python
|
|
61
|
-
t = transcript_df[
|
|
60
|
+
t = transcript_df["speaker"].value_counts()
|
|
62
61
|
n_top_speakers = 4
|
|
63
62
|
print(f"Unique speakers: {len(t)}")
|
|
64
63
|
print(f"Top 5 speakers: {t.head(n_top_speakers)}")
|
|
@@ -76,11 +75,11 @@ print(f"Top 5 speakers: {t.head(n_top_speakers)}")
|
|
|
76
75
|
|
|
77
76
|
```python
|
|
78
77
|
# replace all speakers that are not the top 3 with 'Other'
|
|
79
|
-
top_speakers = transcript_df[
|
|
80
|
-
transcript_df[
|
|
81
|
-
lambda speaker: speaker if speaker in top_speakers else
|
|
78
|
+
top_speakers = transcript_df["speaker"].value_counts().head(3).index
|
|
79
|
+
transcript_df["speaker"] = transcript_df["speaker"].apply(
|
|
80
|
+
lambda speaker: speaker if speaker in top_speakers else "Other"
|
|
82
81
|
)
|
|
83
|
-
transcript_df[
|
|
82
|
+
transcript_df["speaker"].value_counts() # only 4 unique speakers now
|
|
84
83
|
```
|
|
85
84
|
|
|
86
85
|
|
|
@@ -100,24 +99,24 @@ transcript_df['speaker'].value_counts() # only 4 unique speakers now
|
|
|
100
99
|
if embeddings_save_key not in store:
|
|
101
100
|
# compute the embeddings
|
|
102
101
|
import oa
|
|
103
|
-
|
|
104
|
-
|
|
102
|
+
|
|
103
|
+
embeddings_vectors = oa.embeddings(transcript_df["text"])
|
|
104
|
+
embeddings_df = pd.DataFrame({"embeddings": embeddings_vectors})
|
|
105
105
|
store[embeddings_save_key] = embeddings_df
|
|
106
106
|
else:
|
|
107
107
|
embeddings_df = store[embeddings_save_key]
|
|
108
|
-
embeddings_vectors = np.vstack(embeddings_df[
|
|
108
|
+
embeddings_vectors = np.vstack(embeddings_df["embeddings"])
|
|
109
109
|
```
|
|
110
110
|
|
|
111
111
|
|
|
112
112
|
```python
|
|
113
113
|
# project embeddings to plane using TSNE
|
|
114
|
-
if
|
|
115
|
-
|
|
114
|
+
if "tsne_x" not in transcript_df.columns:
|
|
116
115
|
from sklearn.manifold import TSNE
|
|
117
|
-
|
|
116
|
+
|
|
118
117
|
tsne_vectors = TSNE(n_components=2).fit_transform(embeddings_vectors)
|
|
119
118
|
|
|
120
|
-
t = pd.DataFrame(tsne_vectors, columns=[
|
|
119
|
+
t = pd.DataFrame(tsne_vectors, columns=["tsne_x", "tsne_y"])
|
|
121
120
|
transcript_df = pd.concat([transcript_df, t], axis=1)
|
|
122
121
|
|
|
123
122
|
print(f"{transcript_df.shape=}")
|
|
@@ -141,23 +140,19 @@ transcript_df.iloc[0]
|
|
|
141
140
|
|
|
142
141
|
```python
|
|
143
142
|
# project embeddings to plane using linear discriminant analysis on speakers
|
|
144
|
-
if
|
|
145
|
-
|
|
146
|
-
from sklearn.discriminant_analysis import LinearDiscriminantAnalysis as LDA
|
|
143
|
+
if "lda_x" not in transcript_df.columns:
|
|
144
|
+
from sklearn.discriminant_analysis import LinearDiscriminantAnalysis as LDA
|
|
147
145
|
from sklearn.pipeline import Pipeline
|
|
148
146
|
from sklearn.decomposition import PCA
|
|
149
147
|
|
|
150
|
-
speakers = transcript_df[
|
|
148
|
+
speakers = transcript_df["speaker"]
|
|
151
149
|
|
|
152
|
-
pipeline = Pipeline([
|
|
153
|
-
('pca', PCA(n_components=50)),
|
|
154
|
-
('lda', LDA(n_components=2))
|
|
155
|
-
])
|
|
150
|
+
pipeline = Pipeline([("pca", PCA(n_components=50)), ("lda", LDA(n_components=2))])
|
|
156
151
|
|
|
157
152
|
pipeline.fit(embeddings_vectors, y=speakers)
|
|
158
153
|
lda_vectors = pipeline.transform(embeddings_vectors)
|
|
159
154
|
|
|
160
|
-
t = pd.DataFrame(lda_vectors, columns=[
|
|
155
|
+
t = pd.DataFrame(lda_vectors, columns=["lda_x", "lda_y"])
|
|
161
156
|
transcript_df = pd.concat([transcript_df, t], axis=1)
|
|
162
157
|
|
|
163
158
|
print(f"{transcript_df.shape=}")
|
|
@@ -183,23 +178,19 @@ transcript_df.iloc[0]
|
|
|
183
178
|
|
|
184
179
|
```python
|
|
185
180
|
# project embeddings to plane using linear discriminant analysis on speakers
|
|
186
|
-
if
|
|
187
|
-
|
|
188
|
-
from sklearn.discriminant_analysis import LinearDiscriminantAnalysis as LDA
|
|
181
|
+
if "single_speaker_lda" not in transcript_df.columns:
|
|
182
|
+
from sklearn.discriminant_analysis import LinearDiscriminantAnalysis as LDA
|
|
189
183
|
from sklearn.pipeline import Pipeline
|
|
190
184
|
from sklearn.decomposition import PCA
|
|
191
185
|
|
|
192
|
-
speakers = transcript_df[
|
|
186
|
+
speakers = transcript_df["speaker"]
|
|
193
187
|
|
|
194
|
-
pipeline = Pipeline([
|
|
195
|
-
('pca', PCA(n_components=50)),
|
|
196
|
-
('lda', LDA(n_components=1))
|
|
197
|
-
])
|
|
188
|
+
pipeline = Pipeline([("pca", PCA(n_components=50)), ("lda", LDA(n_components=1))])
|
|
198
189
|
|
|
199
190
|
pipeline.fit(embeddings_vectors, y=speakers)
|
|
200
191
|
lda_vectors = pipeline.transform(embeddings_vectors)
|
|
201
192
|
|
|
202
|
-
t = pd.DataFrame(lda_vectors, columns=[
|
|
193
|
+
t = pd.DataFrame(lda_vectors, columns=["single_speaker_lda"])
|
|
203
194
|
transcript_df = pd.concat([transcript_df, t], axis=1)
|
|
204
195
|
|
|
205
196
|
print(f"{transcript_df.shape=}")
|
|
@@ -237,19 +228,20 @@ transcript_df.iloc[0]
|
|
|
237
228
|
|
|
238
229
|
|
|
239
230
|
```python
|
|
240
|
-
if
|
|
231
|
+
if "pca_1" not in transcript_df.columns:
|
|
241
232
|
from sklearn.decomposition import PCA
|
|
233
|
+
|
|
242
234
|
pca = PCA(n_components=2)
|
|
243
235
|
|
|
244
236
|
pca_vectors = pca.fit_transform(embeddings_vectors)
|
|
245
237
|
|
|
246
|
-
t = pd.DataFrame(pca_vectors, columns=[
|
|
238
|
+
t = pd.DataFrame(pca_vectors, columns=["pca_1", "pca_2"])
|
|
247
239
|
transcript_df = pd.concat([transcript_df, t], axis=1)
|
|
248
240
|
```
|
|
249
241
|
|
|
250
242
|
|
|
251
243
|
```python
|
|
252
|
-
ww = list(map(sentiment_score, transcript_df.iloc[:3][
|
|
244
|
+
ww = list(map(sentiment_score, transcript_df.iloc[:3]["text"]))
|
|
253
245
|
ww
|
|
254
246
|
```
|
|
255
247
|
|
|
@@ -262,10 +254,12 @@ ww
|
|
|
262
254
|
|
|
263
255
|
|
|
264
256
|
```python
|
|
265
|
-
if
|
|
257
|
+
if "sentiment_f" not in transcript_df.columns:
|
|
266
258
|
from mood.sentiment import flair_sentiment_score
|
|
267
259
|
|
|
268
|
-
t = pd.DataFrame(
|
|
260
|
+
t = pd.DataFrame(
|
|
261
|
+
list(map(flair_sentiment_score, transcript_df["text"])), columns=["sentiment_f"]
|
|
262
|
+
)
|
|
269
263
|
transcript_df = pd.concat([transcript_df, t], axis=1)
|
|
270
264
|
transcript_df.iloc[0]
|
|
271
265
|
```
|
|
@@ -300,20 +294,19 @@ transcript_df.iloc[0]
|
|
|
300
294
|
```python
|
|
301
295
|
import dol
|
|
302
296
|
|
|
303
|
-
pickle_store = dol.PickleFiles(
|
|
304
|
-
models = pickle_store[
|
|
297
|
+
pickle_store = dol.PickleFiles(".")
|
|
298
|
+
models = pickle_store["oa_embeddings_sentiment_models.pickle"]
|
|
305
299
|
|
|
306
300
|
print(f"{list(models)=}")
|
|
307
301
|
|
|
308
|
-
label =
|
|
302
|
+
label = "anger"
|
|
309
303
|
print(f"{list(models[label])=}")
|
|
310
304
|
print(f"{models[label]['stats']}")
|
|
311
305
|
|
|
312
|
-
if
|
|
306
|
+
if "disgust" not in transcript_df.columns:
|
|
313
307
|
for sentiment, d in models.items():
|
|
314
|
-
model = d[
|
|
308
|
+
model = d["model"]
|
|
315
309
|
model.predict()
|
|
316
|
-
|
|
317
310
|
```
|
|
318
311
|
|
|
319
312
|
list(models)=['anger', 'sadness', 'surprise', 'disgust', 'fear']
|
|
@@ -349,11 +342,11 @@ if 'disgust' not in transcript_df.columns:
|
|
|
349
342
|
|
|
350
343
|
|
|
351
344
|
```python
|
|
352
|
-
if
|
|
345
|
+
if "compound" not in transcript_df.columns:
|
|
353
346
|
from vaderSentiment.vaderSentiment import SentimentIntensityAnalyzer
|
|
354
347
|
|
|
355
348
|
analyzer = SentimentIntensityAnalyzer()
|
|
356
|
-
t = pd.DataFrame(list(map(analyzer.polarity_scores, transcript_df[
|
|
349
|
+
t = pd.DataFrame(list(map(analyzer.polarity_scores, transcript_df["text"])))
|
|
357
350
|
transcript_df = pd.concat([transcript_df, t], axis=1)
|
|
358
351
|
|
|
359
352
|
print(f"{transcript_df.shape=}")
|
|
@@ -384,12 +377,11 @@ transcript_df.iloc[0]
|
|
|
384
377
|
|
|
385
378
|
|
|
386
379
|
```python
|
|
387
|
-
if
|
|
380
|
+
if "Happy" not in transcript_df.columns:
|
|
388
381
|
import text2emotion as te
|
|
389
382
|
|
|
390
|
-
t = pd.DataFrame(list(map(te.get_emotion, transcript_df[
|
|
383
|
+
t = pd.DataFrame(list(map(te.get_emotion, transcript_df["text"])))
|
|
391
384
|
transcript_df = pd.concat([transcript_df, t], axis=1)
|
|
392
|
-
|
|
393
385
|
```
|
|
394
386
|
|
|
395
387
|
[nltk_data] Downloading package stopwords to
|
|
@@ -405,7 +397,7 @@ if 'Happy' not in transcript_df.columns:
|
|
|
405
397
|
|
|
406
398
|
|
|
407
399
|
```python
|
|
408
|
-
transcript_df[
|
|
400
|
+
transcript_df["turn"] = range(len(transcript_df))
|
|
409
401
|
```
|
|
410
402
|
|
|
411
403
|
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/embeddings_of_aggregations/README.md
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/epstein_files/epstein_files.ipynb
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/github_repos/github_repos.ipynb
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/jersey_laws/jersey_laws.ipynb
RENAMED
|
File without changes
|
{imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/lmsys_ai_conversations/README.md
RENAMED
|
File without changes
|
|
File without changes
|
{imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/mcdonalds_reviews/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/prompt_injections/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
{imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/trump_vs_zelenskyy/README.md
RENAMED
|
File without changes
|
{imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/trump_vs_zelenskyy/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/twitter_sentiment/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/imbed_data_prep/wordnet_words/wordnet_words.ipynb
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/misc/data/cheat_sheets/Data Science Cheat Sheets.pdf
RENAMED
|
File without changes
|
{imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/misc/data/cheat_sheets/Data Science Cheatsheet.pdf
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{imbed_data_prep-0.1.1 → imbed_data_prep-0.1.2}/misc/using_ai_to_get_data_descriptions.ipynb
RENAMED
|
File without changes
|
|
File without changes
|