wordtangible 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- wordtangible/__init__.py +5 -0
- wordtangible/concrete.py +156 -0
- wordtangible/imageable.py +0 -0
- wordtangible/resources/__init__.py +0 -0
- wordtangible/resources/concreteness_ratings.csv +40311 -0
- wordtangible-0.1.0.dist-info/LICENSE +21 -0
- wordtangible-0.1.0.dist-info/METADATA +75 -0
- wordtangible-0.1.0.dist-info/RECORD +9 -0
- wordtangible-0.1.0.dist-info/WHEEL +4 -0
wordtangible/__init__.py
ADDED
wordtangible/concrete.py
ADDED
|
@@ -0,0 +1,156 @@
|
|
|
1
|
+
import csv
|
|
2
|
+
from pathlib import Path
|
|
3
|
+
import nltk
|
|
4
|
+
from nltk.tokenize import word_tokenize
|
|
5
|
+
from nltk.corpus import stopwords
|
|
6
|
+
from importlib import resources
|
|
7
|
+
|
|
8
|
+
nltk.download("punkt", quiet=True)
|
|
9
|
+
nltk.download("stopwords", quiet=True)
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def _load_concreteness_ratings() -> dict[str, float]:
|
|
13
|
+
concreteness_dict = {}
|
|
14
|
+
|
|
15
|
+
# Use importlib.resources to access the CSV file
|
|
16
|
+
with resources.open_text(
|
|
17
|
+
"wordtangible.resources", "concreteness_ratings.csv"
|
|
18
|
+
) as csvfile:
|
|
19
|
+
reader = csv.DictReader(csvfile)
|
|
20
|
+
for row in reader:
|
|
21
|
+
concreteness_dict[row["Word"]] = float(row["Concreteness"])
|
|
22
|
+
|
|
23
|
+
return concreteness_dict
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
CONCRETENESS_RATINGS = _load_concreteness_ratings()
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def word_concreteness(word: str) -> float | None:
|
|
30
|
+
"""
|
|
31
|
+
Get the concreteness rating for a given word.
|
|
32
|
+
|
|
33
|
+
The concreteness ratings are derived from three sources:
|
|
34
|
+
1. MRC Psycholinguistic Database
|
|
35
|
+
2. Brysbaert et al. concreteness ratings
|
|
36
|
+
3. Glasgow concreteness ratings
|
|
37
|
+
|
|
38
|
+
All ratings were normalized to a 1-5 scale, where:
|
|
39
|
+
- 1 represents highly abstract words
|
|
40
|
+
- 5 represents highly concrete words
|
|
41
|
+
|
|
42
|
+
If a word was rated in only one list, that list's rating was used.
|
|
43
|
+
If a word was rated in multiple lists, the average of those ratings was used.
|
|
44
|
+
|
|
45
|
+
Args:
|
|
46
|
+
word (str): The word to look up.
|
|
47
|
+
|
|
48
|
+
Returns:
|
|
49
|
+
float | None: The concreteness rating of the word if available, None otherwise.
|
|
50
|
+
"""
|
|
51
|
+
return CONCRETENESS_RATINGS.get(word, None)
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def avg_text_concreteness(
|
|
55
|
+
text: str, include_stopwords: bool = False, only_rated_words: bool = True
|
|
56
|
+
) -> float:
|
|
57
|
+
"""
|
|
58
|
+
Calculate the average concreteness rating for a given text.
|
|
59
|
+
|
|
60
|
+
This function tokenizes the input text, retrieves concreteness ratings for each token,
|
|
61
|
+
and calculates the average concreteness score.
|
|
62
|
+
|
|
63
|
+
Args:
|
|
64
|
+
text (str): The input text to analyze.
|
|
65
|
+
include_stopwords (bool, optional): Whether to include stopwords in the analysis.
|
|
66
|
+
Defaults to False.
|
|
67
|
+
only_rated_words (bool, optional): Whether to only consider words with known
|
|
68
|
+
concreteness ratings in the average calculation. Defaults to True.
|
|
69
|
+
|
|
70
|
+
Returns:
|
|
71
|
+
float: The average concreteness rating of the text. Returns 0.0 if no words
|
|
72
|
+
are found or if no words have concreteness ratings.
|
|
73
|
+
|
|
74
|
+
Note:
|
|
75
|
+
- Concreteness ratings range from 1 (highly abstract) to 5 (highly concrete).
|
|
76
|
+
- If only_rated_words is True, words without concreteness ratings are excluded
|
|
77
|
+
from both the numerator and denominator of the average calculation.
|
|
78
|
+
- If only_rated_words is False, all words are included in the denominator,
|
|
79
|
+
but only rated words contribute to the numerator.
|
|
80
|
+
"""
|
|
81
|
+
tokens = _get_tokens(text, include_stopwords)
|
|
82
|
+
|
|
83
|
+
if len(tokens) == 0:
|
|
84
|
+
return 0.0
|
|
85
|
+
|
|
86
|
+
concreteness_ratings = [
|
|
87
|
+
concreteness
|
|
88
|
+
for token in tokens
|
|
89
|
+
if (concreteness := word_concreteness(token)) is not None
|
|
90
|
+
]
|
|
91
|
+
num_tokens = len(concreteness_ratings if only_rated_words else tokens)
|
|
92
|
+
total_concreteness = sum(concreteness_ratings)
|
|
93
|
+
|
|
94
|
+
return (total_concreteness / num_tokens) if num_tokens > 0 else 0.0
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def concrete_abstract_ratio(
|
|
98
|
+
text: str,
|
|
99
|
+
include_stopwords: bool = False,
|
|
100
|
+
very_concrete_threshold: float = 4.0,
|
|
101
|
+
very_abstract_threshold: float = 2.0,
|
|
102
|
+
) -> float:
|
|
103
|
+
"""
|
|
104
|
+
Calculate the ratio of very concrete words to very abstract words in a given text.
|
|
105
|
+
|
|
106
|
+
This function tokenizes the input text, determines the concreteness of each word,
|
|
107
|
+
and calculates the ratio of words that are considered very concrete to those
|
|
108
|
+
considered very abstract based on the provided thresholds.
|
|
109
|
+
|
|
110
|
+
Args:
|
|
111
|
+
text (str): The input text to analyze.
|
|
112
|
+
include_stopwords (bool, optional): Whether to include stopwords in the analysis.
|
|
113
|
+
Defaults to False.
|
|
114
|
+
very_concrete_threshold (float, optional): The concreteness rating threshold
|
|
115
|
+
for a word to be considered very concrete. Defaults to 4.0.
|
|
116
|
+
very_abstract_threshold (float, optional): The concreteness rating threshold
|
|
117
|
+
for a word to be considered very abstract. Defaults to 2.0.
|
|
118
|
+
|
|
119
|
+
Returns:
|
|
120
|
+
float: The ratio of very concrete words to very abstract words.
|
|
121
|
+
Returns float('inf') if there are concrete words but no abstract words.
|
|
122
|
+
Returns 0.0 if there are no concrete words or if the text is empty.
|
|
123
|
+
|
|
124
|
+
Note:
|
|
125
|
+
- Concreteness ratings range from 1 (highly abstract) to 5 (highly concrete).
|
|
126
|
+
- Words with concreteness ratings between the two thresholds are not counted
|
|
127
|
+
in either category.
|
|
128
|
+
- Words without known concreteness ratings are ignored.
|
|
129
|
+
"""
|
|
130
|
+
tokens = _get_tokens(text, include_stopwords)
|
|
131
|
+
|
|
132
|
+
concrete_words = 0
|
|
133
|
+
abstract_words = 0
|
|
134
|
+
|
|
135
|
+
for token in tokens:
|
|
136
|
+
concreteness = word_concreteness(token)
|
|
137
|
+
if concreteness is not None:
|
|
138
|
+
if concreteness >= very_concrete_threshold:
|
|
139
|
+
concrete_words += 1
|
|
140
|
+
elif concreteness <= very_abstract_threshold:
|
|
141
|
+
abstract_words += 1
|
|
142
|
+
|
|
143
|
+
if abstract_words == 0:
|
|
144
|
+
return float("inf") if concrete_words > 0 else 0.0
|
|
145
|
+
|
|
146
|
+
return concrete_words / abstract_words
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
def _get_tokens(text: str, include_stopwords: bool = False):
|
|
150
|
+
tokens = [token for token in word_tokenize(text.lower()) if token.isalpha()]
|
|
151
|
+
|
|
152
|
+
if not include_stopwords:
|
|
153
|
+
stop_words = set(stopwords.words("english"))
|
|
154
|
+
tokens = [token for token in tokens if token not in stop_words]
|
|
155
|
+
|
|
156
|
+
return tokens
|
|
File without changes
|
|
File without changes
|