wordtangible 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,5 @@
1
+ from .concrete import word_concreteness
2
+ from .concrete import concrete_abstract_ratio
3
+ from .concrete import avg_text_concreteness
4
+
5
+ __all__ = ["word_concreteness", "concrete_abstract_ratio", "avg_text_concreteness"]
@@ -0,0 +1,156 @@
1
+ import csv
2
+ from pathlib import Path
3
+ import nltk
4
+ from nltk.tokenize import word_tokenize
5
+ from nltk.corpus import stopwords
6
+ from importlib import resources
7
+
8
+ nltk.download("punkt", quiet=True)
9
+ nltk.download("stopwords", quiet=True)
10
+
11
+
12
+ def _load_concreteness_ratings() -> dict[str, float]:
13
+ concreteness_dict = {}
14
+
15
+ # Use importlib.resources to access the CSV file
16
+ with resources.open_text(
17
+ "wordtangible.resources", "concreteness_ratings.csv"
18
+ ) as csvfile:
19
+ reader = csv.DictReader(csvfile)
20
+ for row in reader:
21
+ concreteness_dict[row["Word"]] = float(row["Concreteness"])
22
+
23
+ return concreteness_dict
24
+
25
+
26
+ CONCRETENESS_RATINGS = _load_concreteness_ratings()
27
+
28
+
29
+ def word_concreteness(word: str) -> float | None:
30
+ """
31
+ Get the concreteness rating for a given word.
32
+
33
+ The concreteness ratings are derived from three sources:
34
+ 1. MRC Psycholinguistic Database
35
+ 2. Brysbaert et al. concreteness ratings
36
+ 3. Glasgow concreteness ratings
37
+
38
+ All ratings were normalized to a 1-5 scale, where:
39
+ - 1 represents highly abstract words
40
+ - 5 represents highly concrete words
41
+
42
+ If a word was rated in only one list, that list's rating was used.
43
+ If a word was rated in multiple lists, the average of those ratings was used.
44
+
45
+ Args:
46
+ word (str): The word to look up.
47
+
48
+ Returns:
49
+ float | None: The concreteness rating of the word if available, None otherwise.
50
+ """
51
+ return CONCRETENESS_RATINGS.get(word, None)
52
+
53
+
54
+ def avg_text_concreteness(
55
+ text: str, include_stopwords: bool = False, only_rated_words: bool = True
56
+ ) -> float:
57
+ """
58
+ Calculate the average concreteness rating for a given text.
59
+
60
+ This function tokenizes the input text, retrieves concreteness ratings for each token,
61
+ and calculates the average concreteness score.
62
+
63
+ Args:
64
+ text (str): The input text to analyze.
65
+ include_stopwords (bool, optional): Whether to include stopwords in the analysis.
66
+ Defaults to False.
67
+ only_rated_words (bool, optional): Whether to only consider words with known
68
+ concreteness ratings in the average calculation. Defaults to True.
69
+
70
+ Returns:
71
+ float: The average concreteness rating of the text. Returns 0.0 if no words
72
+ are found or if no words have concreteness ratings.
73
+
74
+ Note:
75
+ - Concreteness ratings range from 1 (highly abstract) to 5 (highly concrete).
76
+ - If only_rated_words is True, words without concreteness ratings are excluded
77
+ from both the numerator and denominator of the average calculation.
78
+ - If only_rated_words is False, all words are included in the denominator,
79
+ but only rated words contribute to the numerator.
80
+ """
81
+ tokens = _get_tokens(text, include_stopwords)
82
+
83
+ if len(tokens) == 0:
84
+ return 0.0
85
+
86
+ concreteness_ratings = [
87
+ concreteness
88
+ for token in tokens
89
+ if (concreteness := word_concreteness(token)) is not None
90
+ ]
91
+ num_tokens = len(concreteness_ratings if only_rated_words else tokens)
92
+ total_concreteness = sum(concreteness_ratings)
93
+
94
+ return (total_concreteness / num_tokens) if num_tokens > 0 else 0.0
95
+
96
+
97
+ def concrete_abstract_ratio(
98
+ text: str,
99
+ include_stopwords: bool = False,
100
+ very_concrete_threshold: float = 4.0,
101
+ very_abstract_threshold: float = 2.0,
102
+ ) -> float:
103
+ """
104
+ Calculate the ratio of very concrete words to very abstract words in a given text.
105
+
106
+ This function tokenizes the input text, determines the concreteness of each word,
107
+ and calculates the ratio of words that are considered very concrete to those
108
+ considered very abstract based on the provided thresholds.
109
+
110
+ Args:
111
+ text (str): The input text to analyze.
112
+ include_stopwords (bool, optional): Whether to include stopwords in the analysis.
113
+ Defaults to False.
114
+ very_concrete_threshold (float, optional): The concreteness rating threshold
115
+ for a word to be considered very concrete. Defaults to 4.0.
116
+ very_abstract_threshold (float, optional): The concreteness rating threshold
117
+ for a word to be considered very abstract. Defaults to 2.0.
118
+
119
+ Returns:
120
+ float: The ratio of very concrete words to very abstract words.
121
+ Returns float('inf') if there are concrete words but no abstract words.
122
+ Returns 0.0 if there are no concrete words or if the text is empty.
123
+
124
+ Note:
125
+ - Concreteness ratings range from 1 (highly abstract) to 5 (highly concrete).
126
+ - Words with concreteness ratings between the two thresholds are not counted
127
+ in either category.
128
+ - Words without known concreteness ratings are ignored.
129
+ """
130
+ tokens = _get_tokens(text, include_stopwords)
131
+
132
+ concrete_words = 0
133
+ abstract_words = 0
134
+
135
+ for token in tokens:
136
+ concreteness = word_concreteness(token)
137
+ if concreteness is not None:
138
+ if concreteness >= very_concrete_threshold:
139
+ concrete_words += 1
140
+ elif concreteness <= very_abstract_threshold:
141
+ abstract_words += 1
142
+
143
+ if abstract_words == 0:
144
+ return float("inf") if concrete_words > 0 else 0.0
145
+
146
+ return concrete_words / abstract_words
147
+
148
+
149
+ def _get_tokens(text: str, include_stopwords: bool = False):
150
+ tokens = [token for token in word_tokenize(text.lower()) if token.isalpha()]
151
+
152
+ if not include_stopwords:
153
+ stop_words = set(stopwords.words("english"))
154
+ tokens = [token for token in tokens if token not in stop_words]
155
+
156
+ return tokens
File without changes
File without changes