sentix 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sentix/__init__.py +18 -0
- sentix/analyzer.py +224 -0
- sentix/certainty.py +45 -0
- sentix/confidence.py +53 -0
- sentix/explanation.py +404 -0
- sentix/lexicon/__init__.py +0 -0
- sentix/lexicon/aspects.py +7 -0
- sentix/lexicon/emoji.py +24 -0
- sentix/lexicon/emotion.py +36 -0
- sentix/lexicon/phrases.py +9 -0
- sentix/lexicon/sentiment.py +60 -0
- sentix/normalize.py +234 -0
- sentix/result.py +103 -0
- sentix/rules/aspect.py +99 -0
- sentix/rules/conditional.py +411 -0
- sentix/rules/context.py +249 -0
- sentix/rules/contradiction.py +587 -0
- sentix/rules/contrast.py +1081 -0
- sentix/rules/emotion.py +31 -0
- sentix/rules/intensity.py +31 -0
- sentix/rules/negation.py +274 -0
- sentix/rules/normalize.py +5 -0
- sentix/rules/phrases.py +30 -0
- sentix/rules/repetition.py +16 -0
- sentix/rules/reversal.py +312 -0
- sentix/scorer.py +542 -0
- sentix/tokenizer.py +16 -0
- sentix-1.0.0.dist-info/METADATA +342 -0
- sentix-1.0.0.dist-info/RECORD +32 -0
- sentix-1.0.0.dist-info/WHEEL +5 -0
- sentix-1.0.0.dist-info/licenses/LICENSE +21 -0
- sentix-1.0.0.dist-info/top_level.txt +1 -0
sentix/__init__.py
ADDED
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Sentix
|
|
3
|
+
======
|
|
4
|
+
|
|
5
|
+
A lightweight rule-based sentiment analysis library for Python.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from .analyzer import SentimentAnalyzer
|
|
9
|
+
from .result import SentimentResult
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
__version__ = "1.0.0"
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
__all__ = [
|
|
16
|
+
"SentimentAnalyzer",
|
|
17
|
+
"SentimentResult",
|
|
18
|
+
]
|
sentix/analyzer.py
ADDED
|
@@ -0,0 +1,224 @@
|
|
|
1
|
+
from .tokenizer import tokenize
|
|
2
|
+
|
|
3
|
+
from .scorer import (
|
|
4
|
+
score_tokens_with_evidence,
|
|
5
|
+
punctuation_modifier,
|
|
6
|
+
)
|
|
7
|
+
|
|
8
|
+
from .confidence import calculate_confidence
|
|
9
|
+
from .result import SentimentResult
|
|
10
|
+
|
|
11
|
+
from .rules.emotion import detect_emotions
|
|
12
|
+
from .rules.aspect import detect_aspects
|
|
13
|
+
from .rules.context import analyze_context
|
|
14
|
+
from .rules.contrast import analyze_contrast
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def normalize_scores(
|
|
18
|
+
positive: float,
|
|
19
|
+
negative: float,
|
|
20
|
+
) -> tuple[float, float, float]:
|
|
21
|
+
"""
|
|
22
|
+
Convert positive / negative evidence into
|
|
23
|
+
normalized probabilities.
|
|
24
|
+
|
|
25
|
+
Neutral receives the remaining probability.
|
|
26
|
+
"""
|
|
27
|
+
|
|
28
|
+
total = positive + negative
|
|
29
|
+
|
|
30
|
+
if total == 0:
|
|
31
|
+
return 0.0, 0.0, 1.0
|
|
32
|
+
|
|
33
|
+
positive_ratio = positive / total
|
|
34
|
+
negative_ratio = negative / total
|
|
35
|
+
|
|
36
|
+
return (
|
|
37
|
+
positive_ratio,
|
|
38
|
+
negative_ratio,
|
|
39
|
+
0.0,
|
|
40
|
+
)
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
class SentimentAnalyzer:
|
|
44
|
+
|
|
45
|
+
def predict(
|
|
46
|
+
self,
|
|
47
|
+
text: str,
|
|
48
|
+
) -> SentimentResult:
|
|
49
|
+
|
|
50
|
+
# ====================================================
|
|
51
|
+
# TOKENIZATION
|
|
52
|
+
# ====================================================
|
|
53
|
+
|
|
54
|
+
tokens = tokenize(text)
|
|
55
|
+
|
|
56
|
+
# ====================================================
|
|
57
|
+
# BASE SENTIMENT SCORING
|
|
58
|
+
# ====================================================
|
|
59
|
+
|
|
60
|
+
(
|
|
61
|
+
score,
|
|
62
|
+
positive,
|
|
63
|
+
negative,
|
|
64
|
+
evidence,
|
|
65
|
+
) = score_tokens_with_evidence(
|
|
66
|
+
tokens,
|
|
67
|
+
text,
|
|
68
|
+
)
|
|
69
|
+
|
|
70
|
+
# ====================================================
|
|
71
|
+
# PUNCTUATION
|
|
72
|
+
# ====================================================
|
|
73
|
+
|
|
74
|
+
punctuation_factor = (
|
|
75
|
+
punctuation_modifier(text)
|
|
76
|
+
)
|
|
77
|
+
|
|
78
|
+
if punctuation_factor != 1.0:
|
|
79
|
+
raw_score = (
|
|
80
|
+
score / punctuation_factor
|
|
81
|
+
)
|
|
82
|
+
else:
|
|
83
|
+
raw_score = score
|
|
84
|
+
|
|
85
|
+
# ====================================================
|
|
86
|
+
# EMOTIONS
|
|
87
|
+
# ====================================================
|
|
88
|
+
|
|
89
|
+
emotions = detect_emotions(
|
|
90
|
+
tokens
|
|
91
|
+
)
|
|
92
|
+
|
|
93
|
+
# ====================================================
|
|
94
|
+
# CONTEXT
|
|
95
|
+
# ====================================================
|
|
96
|
+
|
|
97
|
+
context = analyze_context(
|
|
98
|
+
tokens
|
|
99
|
+
)
|
|
100
|
+
|
|
101
|
+
# ====================================================
|
|
102
|
+
# CONTRAST
|
|
103
|
+
# ====================================================
|
|
104
|
+
|
|
105
|
+
contrast = analyze_contrast(
|
|
106
|
+
tokens
|
|
107
|
+
)
|
|
108
|
+
|
|
109
|
+
# ====================================================
|
|
110
|
+
# NORMALIZED PROBABILITIES
|
|
111
|
+
# ====================================================
|
|
112
|
+
|
|
113
|
+
(
|
|
114
|
+
positive_prob,
|
|
115
|
+
negative_prob,
|
|
116
|
+
neutral_prob,
|
|
117
|
+
) = normalize_scores(
|
|
118
|
+
positive,
|
|
119
|
+
negative,
|
|
120
|
+
)
|
|
121
|
+
|
|
122
|
+
# ====================================================
|
|
123
|
+
# LABEL DECISION
|
|
124
|
+
# ====================================================
|
|
125
|
+
|
|
126
|
+
label = None
|
|
127
|
+
|
|
128
|
+
# ----------------------------------------------------
|
|
129
|
+
# CONTRAST HAS HIGH PRIORITY
|
|
130
|
+
#
|
|
131
|
+
# If the contrast engine has identified a semantic
|
|
132
|
+
# relationship, use its decision before falling back
|
|
133
|
+
# to the global sentiment score.
|
|
134
|
+
# ----------------------------------------------------
|
|
135
|
+
|
|
136
|
+
if (
|
|
137
|
+
contrast.get("has_contrast")
|
|
138
|
+
and contrast.get("label") is not None
|
|
139
|
+
):
|
|
140
|
+
label = contrast["label"]
|
|
141
|
+
|
|
142
|
+
# ----------------------------------------------------
|
|
143
|
+
# DIFFERENT SENTENCE TARGETS
|
|
144
|
+
#
|
|
145
|
+
# Example:
|
|
146
|
+
#
|
|
147
|
+
# The camera is terrible.
|
|
148
|
+
# The screen is amazing.
|
|
149
|
+
#
|
|
150
|
+
# -> mixed
|
|
151
|
+
# ----------------------------------------------------
|
|
152
|
+
|
|
153
|
+
elif (
|
|
154
|
+
context.get(
|
|
155
|
+
"different_sentence_targets",
|
|
156
|
+
False,
|
|
157
|
+
)
|
|
158
|
+
and positive > 0.5
|
|
159
|
+
and negative > 0.5
|
|
160
|
+
):
|
|
161
|
+
label = "mixed"
|
|
162
|
+
|
|
163
|
+
# ----------------------------------------------------
|
|
164
|
+
# GENERAL MIXED SENTIMENT
|
|
165
|
+
# ----------------------------------------------------
|
|
166
|
+
|
|
167
|
+
elif (
|
|
168
|
+
positive > 0.5
|
|
169
|
+
and negative > 0.5
|
|
170
|
+
and abs(
|
|
171
|
+
positive - negative
|
|
172
|
+
) < 0.75
|
|
173
|
+
):
|
|
174
|
+
label = "mixed"
|
|
175
|
+
|
|
176
|
+
# ----------------------------------------------------
|
|
177
|
+
# POSITIVE
|
|
178
|
+
# ----------------------------------------------------
|
|
179
|
+
|
|
180
|
+
elif score > 0.05:
|
|
181
|
+
label = "positive"
|
|
182
|
+
|
|
183
|
+
# ----------------------------------------------------
|
|
184
|
+
# NEGATIVE
|
|
185
|
+
# ----------------------------------------------------
|
|
186
|
+
|
|
187
|
+
elif score < -0.05:
|
|
188
|
+
label = "negative"
|
|
189
|
+
|
|
190
|
+
# ----------------------------------------------------
|
|
191
|
+
# NEUTRAL
|
|
192
|
+
# ----------------------------------------------------
|
|
193
|
+
|
|
194
|
+
else:
|
|
195
|
+
label = "neutral"
|
|
196
|
+
|
|
197
|
+
# ====================================================
|
|
198
|
+
# CONFIDENCE
|
|
199
|
+
# ====================================================
|
|
200
|
+
|
|
201
|
+
confidence = calculate_confidence(
|
|
202
|
+
positive,
|
|
203
|
+
negative,
|
|
204
|
+
label,
|
|
205
|
+
tokens,
|
|
206
|
+
)
|
|
207
|
+
|
|
208
|
+
# ====================================================
|
|
209
|
+
# RESULT
|
|
210
|
+
# ====================================================
|
|
211
|
+
|
|
212
|
+
return SentimentResult(
|
|
213
|
+
label=label,
|
|
214
|
+
score=score,
|
|
215
|
+
positive=positive_prob,
|
|
216
|
+
negative=negative_prob,
|
|
217
|
+
neutral=neutral_prob,
|
|
218
|
+
confidence=confidence,
|
|
219
|
+
emotions=emotions,
|
|
220
|
+
aspects=detect_aspects(tokens),
|
|
221
|
+
evidence=evidence,
|
|
222
|
+
raw_score=raw_score,
|
|
223
|
+
punctuation_modifier=punctuation_factor,
|
|
224
|
+
)
|
sentix/certainty.py
ADDED
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
UNCERTAINTY_WORDS = {
|
|
2
|
+
"maybe": 0.75,
|
|
3
|
+
"perhaps": 0.75,
|
|
4
|
+
"possibly": 0.75,
|
|
5
|
+
"might": 0.75,
|
|
6
|
+
"could": 0.80,
|
|
7
|
+
"probably": 0.85,
|
|
8
|
+
"possibly": 0.75,
|
|
9
|
+
"seems": 0.85,
|
|
10
|
+
"seemingly": 0.85,
|
|
11
|
+
"guess": 0.80,
|
|
12
|
+
"think": 0.90,
|
|
13
|
+
"believe": 0.90,
|
|
14
|
+
}
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
CERTAINTY_WORDS = {
|
|
18
|
+
"definitely": 1.15,
|
|
19
|
+
"certainly": 1.15,
|
|
20
|
+
"clearly": 1.10,
|
|
21
|
+
"undoubtedly": 1.20,
|
|
22
|
+
"surely": 1.10,
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def get_certainty_modifier(tokens: list[str]) -> float:
|
|
27
|
+
"""
|
|
28
|
+
Calculate a confidence modifier from certainty/uncertainty words.
|
|
29
|
+
|
|
30
|
+
Values below 1.0 reduce confidence.
|
|
31
|
+
Values above 1.0 increase confidence.
|
|
32
|
+
"""
|
|
33
|
+
|
|
34
|
+
modifier = 1.0
|
|
35
|
+
|
|
36
|
+
for token in tokens:
|
|
37
|
+
word = token.lower()
|
|
38
|
+
|
|
39
|
+
if word in UNCERTAINTY_WORDS:
|
|
40
|
+
modifier *= UNCERTAINTY_WORDS[word]
|
|
41
|
+
|
|
42
|
+
elif word in CERTAINTY_WORDS:
|
|
43
|
+
modifier *= CERTAINTY_WORDS[word]
|
|
44
|
+
|
|
45
|
+
return modifier
|
sentix/confidence.py
ADDED
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
import math
|
|
2
|
+
|
|
3
|
+
from .certainty import get_certainty_modifier
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
def calculate_confidence(
|
|
7
|
+
positive: float,
|
|
8
|
+
negative: float,
|
|
9
|
+
label: str,
|
|
10
|
+
tokens: list[str] | None = None,
|
|
11
|
+
) -> float:
|
|
12
|
+
"""
|
|
13
|
+
Calculate confidence in the predicted sentiment.
|
|
14
|
+
|
|
15
|
+
Confidence is based on:
|
|
16
|
+
|
|
17
|
+
1. Amount of sentiment evidence.
|
|
18
|
+
2. Dominance of the predicted sentiment.
|
|
19
|
+
3. Certainty or uncertainty expressed in the text.
|
|
20
|
+
|
|
21
|
+
Returns a value between 0.0 and 1.0.
|
|
22
|
+
"""
|
|
23
|
+
|
|
24
|
+
evidence = positive + negative
|
|
25
|
+
|
|
26
|
+
# No sentiment evidence.
|
|
27
|
+
if evidence == 0:
|
|
28
|
+
return 0.5 if label == "neutral" else 0.0
|
|
29
|
+
|
|
30
|
+
# Calculate polarity dominance.
|
|
31
|
+
if label == "positive":
|
|
32
|
+
dominance = positive / evidence
|
|
33
|
+
|
|
34
|
+
elif label == "negative":
|
|
35
|
+
dominance = negative / evidence
|
|
36
|
+
|
|
37
|
+
elif label == "mixed":
|
|
38
|
+
dominance = 1.0 - abs(positive - negative) / evidence
|
|
39
|
+
|
|
40
|
+
else:
|
|
41
|
+
dominance = 0.0
|
|
42
|
+
|
|
43
|
+
# Convert evidence strength into a bounded value.
|
|
44
|
+
evidence_strength = 1.0 - math.exp(-evidence)
|
|
45
|
+
|
|
46
|
+
confidence = dominance * evidence_strength
|
|
47
|
+
|
|
48
|
+
# Apply certainty / uncertainty information.
|
|
49
|
+
if tokens is not None:
|
|
50
|
+
certainty_modifier = get_certainty_modifier(tokens)
|
|
51
|
+
confidence *= certainty_modifier
|
|
52
|
+
|
|
53
|
+
return min(max(confidence, 0.0), 1.0)
|