psymarkers 0.1.11__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- psymarkers-0.1.11/PKG-INFO +105 -0
- psymarkers-0.1.11/README.md +81 -0
- psymarkers-0.1.11/pyproject.toml +100 -0
- psymarkers-0.1.11/pyproject.toml.orig +91 -0
- psymarkers-0.1.11/src/psymarkers/__init__.py +1 -0
- psymarkers-0.1.11/src/psymarkers/markers.py +235 -0
- psymarkers-0.1.11/src/psymarkers/py.typed +0 -0
- psymarkers-0.1.11/src/psymarkers/resources.py +23 -0
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
Metadata-Version: 2.3
|
|
2
|
+
Name: psymarkers
|
|
3
|
+
Version: 0.1.11
|
|
4
|
+
Summary: TBD
|
|
5
|
+
Keywords: nlp,text-analysis,psychology,mental-health,sentiment-analysis,linguistics,feature-extraction
|
|
6
|
+
Author: blazaid
|
|
7
|
+
Author-email: blazaid <alber.diaz@pm.me>
|
|
8
|
+
License: GPL-3.0-or-later
|
|
9
|
+
Classifier: Development Status :: 4 - Beta
|
|
10
|
+
Classifier: Intended Audience :: Science/Research
|
|
11
|
+
Classifier: Intended Audience :: Healthcare Industry
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: Topic :: Text Processing :: Linguistic
|
|
14
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
15
|
+
Classifier: Topic :: Scientific/Engineering :: Medical Science Apps.
|
|
16
|
+
Classifier: Programming Language :: Python :: 3
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
19
|
+
Requires-Dist: nltk>=3.10.3
|
|
20
|
+
Requires-Dist: textstat>=0.7.13
|
|
21
|
+
Requires-Python: >=3.13
|
|
22
|
+
Project-URL: Repository, https://codeberg.org/knodis/psymarkers
|
|
23
|
+
Description-Content-Type: text/markdown
|
|
24
|
+
|
|
25
|
+
# psymarkers
|
|
26
|
+
|
|
27
|
+
[](https://www.gnu.org/licenses/gpl-3.0)
|
|
28
|
+
[](https://www.python.org/)
|
|
29
|
+
|
|
30
|
+
Biblioteca de Python rápida y eficiente para extraer marcadores psicológicos y lingüísticos de textos. Está basada en
|
|
31
|
+
literatura clínica y psicológica, y cuantifica patrones del lenguaje comúnmente asociados con estados de depresión,
|
|
32
|
+
ansiedad, rumiación cognitiva o ideación suicida.
|
|
33
|
+
|
|
34
|
+
## Instalación
|
|
35
|
+
|
|
36
|
+
Puedes instalar la biblioteca directamente desde [PyPi](https://pypi.org/):
|
|
37
|
+
|
|
38
|
+
```text
|
|
39
|
+
pip install psymarkers
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
**Nota**: Requiere la descarga previa de los diccionarios de `nltk` si no los tienes en tu entorno:
|
|
43
|
+
|
|
44
|
+
- `nltk.download('punkt')`
|
|
45
|
+
- `nltk.download('vader_lexicon')`
|
|
46
|
+
- `nltk.download('averaged_perceptron_tagger')`
|
|
47
|
+
|
|
48
|
+
## Indicadores disponibles
|
|
49
|
+
|
|
50
|
+
Los indicadores actuales incluyen:
|
|
51
|
+
|
|
52
|
+
- `absolutist_ratio`: Proporción de palabras absolutistas (always, never, completely), marcador de ansiedad y depresión.
|
|
53
|
+
- `pronouns_ratio`: Proporción de pronombres singulares de primera persona (I, me, my), asociado a focalización en uno
|
|
54
|
+
mismo.
|
|
55
|
+
- `pronouns_plural_ratio`: Proporción de pronombres plurales (we, us). Su disminución es un indicador de aislamiento
|
|
56
|
+
social.
|
|
57
|
+
- `somatic_ratio`: Uso de palabras relacionadas con el cuerpo, dolor y fatiga.
|
|
58
|
+
- `hopelessness_ratio`: Proporción de léxico que expresa desesperanza y falta de futuro.
|
|
59
|
+
- `negative_sentiment` / `compound_sentiment`: Valencia emocional negativa y polaridad afectiva general (mediante
|
|
60
|
+
VADER).
|
|
61
|
+
- `type_token_ratio`: Diversidad léxica. Un ratio bajo indica vocabulario repetitivo (rumiación cognitiva).
|
|
62
|
+
- `temporal_myopia_index`: Ratio de verbos en pasado frente a verbos modales futuros, indicando estrechamiento temporal
|
|
63
|
+
o incapacidad de proyectarse al futuro.
|
|
64
|
+
- `readability_score`: Nivel de legibilidad de Flesch-Kincaid. Textos muy desorganizados se asocian a deterioro
|
|
65
|
+
cognitivo o psicosis.
|
|
66
|
+
- `insomnia_post`: Validador de metadatos que retorna 1.0 si el texto fue publicado en horario nocturno (2 AM - 5 AM).
|
|
67
|
+
|
|
68
|
+
## Ejemplos de uso
|
|
69
|
+
|
|
70
|
+
### Como _pipeline_ en aprendizaje automático
|
|
71
|
+
|
|
72
|
+
Ya que (intentamos) que todos los indicadores compartan una interfaz común, se puede usar para crear vectores de
|
|
73
|
+
características limpios y listos para entrenar modelos:
|
|
74
|
+
|
|
75
|
+
```python
|
|
76
|
+
from psymarkers import markers
|
|
77
|
+
|
|
78
|
+
# 1. Defines tu pipeline de extracción
|
|
79
|
+
metrics_pipeline = [
|
|
80
|
+
markers.absolutist_ratio,
|
|
81
|
+
markers.pronouns_ratio,
|
|
82
|
+
markers.hopelessness_ratio,
|
|
83
|
+
markers.negative_sentiment,
|
|
84
|
+
markers.temporal_myopia_index,
|
|
85
|
+
markers.type_token_ratio
|
|
86
|
+
]
|
|
87
|
+
|
|
88
|
+
text = "I am completely exhausted and have no future."
|
|
89
|
+
date = "2023-10-27T03:15:00Z"
|
|
90
|
+
|
|
91
|
+
# 2. Extraes todas las características en un diccionario en una sola pasada
|
|
92
|
+
features = {
|
|
93
|
+
metric.__class__.__name__: metric(text, date=date)
|
|
94
|
+
for metric in metrics_pipeline
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
# 3. Haces lo que quireas con los datos
|
|
98
|
+
print(features) # Salida esperada: {'RegexRatioMetric': 0.11, 'SentimentMetric': 0.8, ...}
|
|
99
|
+
```
|
|
100
|
+
|
|
101
|
+
## Licencia
|
|
102
|
+
|
|
103
|
+
Este proyecto está licenciado bajo los términos de la **GNU General Public License v3.0 (GPLv3)**. Eres libre de usar,
|
|
104
|
+
modificar y distribuir este software, siempre y cuando las obras derivadas se distribuyan bajo la misma licencia
|
|
105
|
+
abierta. Para más información, por favor consulta el fichero [`LICENSE`](./LICENSE).
|
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
# psymarkers
|
|
2
|
+
|
|
3
|
+
[](https://www.gnu.org/licenses/gpl-3.0)
|
|
4
|
+
[](https://www.python.org/)
|
|
5
|
+
|
|
6
|
+
Biblioteca de Python rápida y eficiente para extraer marcadores psicológicos y lingüísticos de textos. Está basada en
|
|
7
|
+
literatura clínica y psicológica, y cuantifica patrones del lenguaje comúnmente asociados con estados de depresión,
|
|
8
|
+
ansiedad, rumiación cognitiva o ideación suicida.
|
|
9
|
+
|
|
10
|
+
## Instalación
|
|
11
|
+
|
|
12
|
+
Puedes instalar la biblioteca directamente desde [PyPi](https://pypi.org/):
|
|
13
|
+
|
|
14
|
+
```text
|
|
15
|
+
pip install psymarkers
|
|
16
|
+
```
|
|
17
|
+
|
|
18
|
+
**Nota**: Requiere la descarga previa de los diccionarios de `nltk` si no los tienes en tu entorno:
|
|
19
|
+
|
|
20
|
+
- `nltk.download('punkt')`
|
|
21
|
+
- `nltk.download('vader_lexicon')`
|
|
22
|
+
- `nltk.download('averaged_perceptron_tagger')`
|
|
23
|
+
|
|
24
|
+
## Indicadores disponibles
|
|
25
|
+
|
|
26
|
+
Los indicadores actuales incluyen:
|
|
27
|
+
|
|
28
|
+
- `absolutist_ratio`: Proporción de palabras absolutistas (always, never, completely), marcador de ansiedad y depresión.
|
|
29
|
+
- `pronouns_ratio`: Proporción de pronombres singulares de primera persona (I, me, my), asociado a focalización en uno
|
|
30
|
+
mismo.
|
|
31
|
+
- `pronouns_plural_ratio`: Proporción de pronombres plurales (we, us). Su disminución es un indicador de aislamiento
|
|
32
|
+
social.
|
|
33
|
+
- `somatic_ratio`: Uso de palabras relacionadas con el cuerpo, dolor y fatiga.
|
|
34
|
+
- `hopelessness_ratio`: Proporción de léxico que expresa desesperanza y falta de futuro.
|
|
35
|
+
- `negative_sentiment` / `compound_sentiment`: Valencia emocional negativa y polaridad afectiva general (mediante
|
|
36
|
+
VADER).
|
|
37
|
+
- `type_token_ratio`: Diversidad léxica. Un ratio bajo indica vocabulario repetitivo (rumiación cognitiva).
|
|
38
|
+
- `temporal_myopia_index`: Ratio de verbos en pasado frente a verbos modales futuros, indicando estrechamiento temporal
|
|
39
|
+
o incapacidad de proyectarse al futuro.
|
|
40
|
+
- `readability_score`: Nivel de legibilidad de Flesch-Kincaid. Textos muy desorganizados se asocian a deterioro
|
|
41
|
+
cognitivo o psicosis.
|
|
42
|
+
- `insomnia_post`: Validador de metadatos que retorna 1.0 si el texto fue publicado en horario nocturno (2 AM - 5 AM).
|
|
43
|
+
|
|
44
|
+
## Ejemplos de uso
|
|
45
|
+
|
|
46
|
+
### Como _pipeline_ en aprendizaje automático
|
|
47
|
+
|
|
48
|
+
Ya que (intentamos) que todos los indicadores compartan una interfaz común, se puede usar para crear vectores de
|
|
49
|
+
características limpios y listos para entrenar modelos:
|
|
50
|
+
|
|
51
|
+
```python
|
|
52
|
+
from psymarkers import markers
|
|
53
|
+
|
|
54
|
+
# 1. Defines tu pipeline de extracción
|
|
55
|
+
metrics_pipeline = [
|
|
56
|
+
markers.absolutist_ratio,
|
|
57
|
+
markers.pronouns_ratio,
|
|
58
|
+
markers.hopelessness_ratio,
|
|
59
|
+
markers.negative_sentiment,
|
|
60
|
+
markers.temporal_myopia_index,
|
|
61
|
+
markers.type_token_ratio
|
|
62
|
+
]
|
|
63
|
+
|
|
64
|
+
text = "I am completely exhausted and have no future."
|
|
65
|
+
date = "2023-10-27T03:15:00Z"
|
|
66
|
+
|
|
67
|
+
# 2. Extraes todas las características en un diccionario en una sola pasada
|
|
68
|
+
features = {
|
|
69
|
+
metric.__class__.__name__: metric(text, date=date)
|
|
70
|
+
for metric in metrics_pipeline
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
# 3. Haces lo que quireas con los datos
|
|
74
|
+
print(features) # Salida esperada: {'RegexRatioMetric': 0.11, 'SentimentMetric': 0.8, ...}
|
|
75
|
+
```
|
|
76
|
+
|
|
77
|
+
## Licencia
|
|
78
|
+
|
|
79
|
+
Este proyecto está licenciado bajo los términos de la **GNU General Public License v3.0 (GPLv3)**. Eres libre de usar,
|
|
80
|
+
modificar y distribuir este software, siempre y cuando las obras derivadas se distribuyan bajo la misma licencia
|
|
81
|
+
abierta. Para más información, por favor consulta el fichero [`LICENSE`](./LICENSE).
|
|
@@ -0,0 +1,100 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "psymarkers"
|
|
3
|
+
version = "0.1.11"
|
|
4
|
+
description = "TBD"
|
|
5
|
+
readme = "README.md"
|
|
6
|
+
requires-python = ">=3.13"
|
|
7
|
+
keywords = [
|
|
8
|
+
"nlp",
|
|
9
|
+
"text-analysis",
|
|
10
|
+
"psychology",
|
|
11
|
+
"mental-health",
|
|
12
|
+
"sentiment-analysis",
|
|
13
|
+
"linguistics",
|
|
14
|
+
"feature-extraction",
|
|
15
|
+
]
|
|
16
|
+
classifiers = [
|
|
17
|
+
"Development Status :: 4 - Beta",
|
|
18
|
+
"Intended Audience :: Science/Research",
|
|
19
|
+
"Intended Audience :: Healthcare Industry",
|
|
20
|
+
"Intended Audience :: Developers",
|
|
21
|
+
"Topic :: Text Processing :: Linguistic",
|
|
22
|
+
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
23
|
+
"Topic :: Scientific/Engineering :: Medical Science Apps.",
|
|
24
|
+
"Programming Language :: Python :: 3",
|
|
25
|
+
"Programming Language :: Python :: 3.13",
|
|
26
|
+
"Programming Language :: Python :: 3.14",
|
|
27
|
+
]
|
|
28
|
+
dependencies = [
|
|
29
|
+
"nltk>=3.10.3",
|
|
30
|
+
"textstat>=0.7.13",
|
|
31
|
+
]
|
|
32
|
+
|
|
33
|
+
[project.license]
|
|
34
|
+
text = "GPL-3.0-or-later"
|
|
35
|
+
|
|
36
|
+
[[project.authors]]
|
|
37
|
+
name = "blazaid"
|
|
38
|
+
email = "alber.diaz@pm.me"
|
|
39
|
+
|
|
40
|
+
[project.urls]
|
|
41
|
+
Repository = "https://codeberg.org/knodis/psymarkers"
|
|
42
|
+
|
|
43
|
+
[build-system]
|
|
44
|
+
requires = ["uv_build>=0.12.23"]
|
|
45
|
+
build-backend = "uv_build"
|
|
46
|
+
|
|
47
|
+
[dependency-groups]
|
|
48
|
+
dev = [
|
|
49
|
+
"bump-my-version>=1.5.1",
|
|
50
|
+
"mypy>=2.4.0",
|
|
51
|
+
"pre-commit>=4.6.2",
|
|
52
|
+
"pytest>=9.1.1",
|
|
53
|
+
"ruff>=0.16.10",
|
|
54
|
+
]
|
|
55
|
+
|
|
56
|
+
[tool.mypy]
|
|
57
|
+
strict = true
|
|
58
|
+
ignore_missing_imports = true
|
|
59
|
+
|
|
60
|
+
[tool.pytest.ini_options]
|
|
61
|
+
pythonpath = ["src"]
|
|
62
|
+
testpaths = ["tests"]
|
|
63
|
+
|
|
64
|
+
[tool.ruff]
|
|
65
|
+
line-length = 80
|
|
66
|
+
target-version = "py313"
|
|
67
|
+
|
|
68
|
+
[tool.ruff.format]
|
|
69
|
+
quote-style = "double"
|
|
70
|
+
indent-style = "space"
|
|
71
|
+
|
|
72
|
+
[tool.ruff.lint]
|
|
73
|
+
select = [
|
|
74
|
+
"E",
|
|
75
|
+
"F",
|
|
76
|
+
"I",
|
|
77
|
+
"W",
|
|
78
|
+
"UP",
|
|
79
|
+
"B",
|
|
80
|
+
]
|
|
81
|
+
ignore = []
|
|
82
|
+
|
|
83
|
+
[tool.ruff.lint.pycodestyle]
|
|
84
|
+
max-doc-length = 72
|
|
85
|
+
|
|
86
|
+
[tool.bumpversion]
|
|
87
|
+
current_version = "0.1.11"
|
|
88
|
+
commit = true
|
|
89
|
+
tag = true
|
|
90
|
+
tag_name = "v{new_version}"
|
|
91
|
+
|
|
92
|
+
[[tool.bumpversion.files]]
|
|
93
|
+
filename = "pyproject.toml"
|
|
94
|
+
search = 'version = "{current_version}"'
|
|
95
|
+
replace = 'version = "{new_version}"'
|
|
96
|
+
|
|
97
|
+
[[tool.bumpversion.files]]
|
|
98
|
+
filename = "src/psymarkers/__init__.py"
|
|
99
|
+
search = '__version__ = "{current_version}"'
|
|
100
|
+
replace = '__version__ = "{new_version}"'
|
|
@@ -0,0 +1,91 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "psymarkers"
|
|
3
|
+
version = "0.1.11"
|
|
4
|
+
description = "TBD"
|
|
5
|
+
readme = "README.md"
|
|
6
|
+
requires-python = ">=3.13"
|
|
7
|
+
license = { text = "GPL-3.0-or-later" }
|
|
8
|
+
authors = [
|
|
9
|
+
{ name = "blazaid", email = "alber.diaz@pm.me" }
|
|
10
|
+
]
|
|
11
|
+
keywords = [
|
|
12
|
+
"nlp",
|
|
13
|
+
"text-analysis",
|
|
14
|
+
"psychology",
|
|
15
|
+
"mental-health",
|
|
16
|
+
"sentiment-analysis",
|
|
17
|
+
"linguistics",
|
|
18
|
+
"feature-extraction"
|
|
19
|
+
]
|
|
20
|
+
classifiers = [
|
|
21
|
+
"Development Status :: 4 - Beta",
|
|
22
|
+
"Intended Audience :: Science/Research",
|
|
23
|
+
"Intended Audience :: Healthcare Industry",
|
|
24
|
+
"Intended Audience :: Developers",
|
|
25
|
+
"Topic :: Text Processing :: Linguistic",
|
|
26
|
+
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
27
|
+
"Topic :: Scientific/Engineering :: Medical Science Apps.",
|
|
28
|
+
"Programming Language :: Python :: 3",
|
|
29
|
+
"Programming Language :: Python :: 3.13",
|
|
30
|
+
"Programming Language :: Python :: 3.14",
|
|
31
|
+
]
|
|
32
|
+
|
|
33
|
+
dependencies = [
|
|
34
|
+
"nltk>=3.10.3",
|
|
35
|
+
"textstat>=0.7.13",
|
|
36
|
+
]
|
|
37
|
+
|
|
38
|
+
[project.urls]
|
|
39
|
+
Repository = "https://codeberg.org/knodis/psymarkers"
|
|
40
|
+
|
|
41
|
+
[build-system]
|
|
42
|
+
requires = ["uv_build>=0.12.23"]
|
|
43
|
+
build-backend = "uv_build"
|
|
44
|
+
|
|
45
|
+
[dependency-groups]
|
|
46
|
+
dev = [
|
|
47
|
+
"bump-my-version>=1.5.1",
|
|
48
|
+
"mypy>=2.4.0",
|
|
49
|
+
"pre-commit>=4.6.2",
|
|
50
|
+
"pytest>=9.1.1",
|
|
51
|
+
"ruff>=0.16.10",
|
|
52
|
+
]
|
|
53
|
+
|
|
54
|
+
[tool.mypy]
|
|
55
|
+
strict = true
|
|
56
|
+
ignore_missing_imports = true
|
|
57
|
+
|
|
58
|
+
[tool.pytest.ini_options]
|
|
59
|
+
pythonpath = ["src"]
|
|
60
|
+
testpaths = ["tests"]
|
|
61
|
+
|
|
62
|
+
[tool.ruff]
|
|
63
|
+
line-length = 80
|
|
64
|
+
target-version = "py313"
|
|
65
|
+
|
|
66
|
+
[tool.ruff.lint.pycodestyle]
|
|
67
|
+
max-doc-length = 72
|
|
68
|
+
|
|
69
|
+
[tool.ruff.format]
|
|
70
|
+
quote-style = "double"
|
|
71
|
+
indent-style = "space"
|
|
72
|
+
|
|
73
|
+
[tool.ruff.lint]
|
|
74
|
+
select = ["E", "F", "I", "W", "UP", "B"]
|
|
75
|
+
ignore = []
|
|
76
|
+
|
|
77
|
+
[tool.bumpversion]
|
|
78
|
+
current_version = "0.1.11"
|
|
79
|
+
commit = true
|
|
80
|
+
tag = true
|
|
81
|
+
tag_name = "v{new_version}"
|
|
82
|
+
|
|
83
|
+
[[tool.bumpversion.files]] # Para actualizar pyproject.toml
|
|
84
|
+
filename = "pyproject.toml"
|
|
85
|
+
search = 'version = "{current_version}"'
|
|
86
|
+
replace = 'version = "{new_version}"'
|
|
87
|
+
|
|
88
|
+
[[tool.bumpversion.files]] # Para actualizar el __init__.py
|
|
89
|
+
filename = "src/psymarkers/__init__.py"
|
|
90
|
+
search = '__version__ = "{current_version}"'
|
|
91
|
+
replace = '__version__ = "{new_version}"'
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
__version__ = "0.1.11"
|
|
@@ -0,0 +1,235 @@
|
|
|
1
|
+
import abc
|
|
2
|
+
import logging
|
|
3
|
+
import re
|
|
4
|
+
from datetime import datetime
|
|
5
|
+
from functools import lru_cache
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
import nltk
|
|
9
|
+
from nltk.sentiment import SentimentIntensityAnalyzer
|
|
10
|
+
from textstat import textstat
|
|
11
|
+
|
|
12
|
+
logger = logging.getLogger(__name__)
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@lru_cache(maxsize=1024)
|
|
16
|
+
def get_cached_tokens(text: str) -> tuple[str, ...]:
|
|
17
|
+
"""Tokenises the given text and returns a tuple of the tokens.
|
|
18
|
+
|
|
19
|
+
:param text: The text to be tokenised.
|
|
20
|
+
:return: A tuple of the tokenised text.
|
|
21
|
+
"""
|
|
22
|
+
return tuple(nltk.word_tokenize(text))
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class TextMetric(metaclass=abc.ABCMeta):
|
|
26
|
+
"""Base functor for all text metrics."""
|
|
27
|
+
|
|
28
|
+
def __call__(self, text: str, **kwargs: Any) -> float:
|
|
29
|
+
"""Executes the text metric over the specified text.
|
|
30
|
+
|
|
31
|
+
:param text: The text to be evaluated.
|
|
32
|
+
:return: The metric value.
|
|
33
|
+
"""
|
|
34
|
+
return self.compute(text, **kwargs)
|
|
35
|
+
|
|
36
|
+
@abc.abstractmethod
|
|
37
|
+
def compute(self, text: str, **kwargs: Any) -> float:
|
|
38
|
+
"""Specific logic to implement by each metric.
|
|
39
|
+
|
|
40
|
+
:param text: The text to be evaluated.
|
|
41
|
+
:return: The metric value.
|
|
42
|
+
"""
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
class RegexRatioMetric(TextMetric):
|
|
46
|
+
"""It generalises metrics that analyse word proportions.
|
|
47
|
+
|
|
48
|
+
They look for a pattern specified as a parameter and divide it by
|
|
49
|
+
the total number of words in the text.
|
|
50
|
+
"""
|
|
51
|
+
|
|
52
|
+
def __init__(self, pattern: str | re.Pattern[str]) -> None:
|
|
53
|
+
"""Initialises this object.
|
|
54
|
+
|
|
55
|
+
:param pattern: The pattern to search for.
|
|
56
|
+
"""
|
|
57
|
+
self.pattern = (
|
|
58
|
+
re.compile(pattern, re.IGNORECASE)
|
|
59
|
+
if isinstance(pattern, str)
|
|
60
|
+
else pattern
|
|
61
|
+
)
|
|
62
|
+
|
|
63
|
+
def compute(self, text: str, **kwargs: Any) -> float:
|
|
64
|
+
total_words = len(get_cached_tokens(text))
|
|
65
|
+
if total_words == 0:
|
|
66
|
+
return 0.0
|
|
67
|
+
|
|
68
|
+
matches = len(self.pattern.findall(text))
|
|
69
|
+
return matches / total_words
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
# The proportion of absolutist words in the text. Source: Al-Mosaiwi,
|
|
73
|
+
# M., & Johnstone, T. (2018). In an Absolute State: Elevated Use of
|
|
74
|
+
# Absolutist Words Is a Marker Specific to Anxiety, Depression, and
|
|
75
|
+
# Suicidal Ideation. Clinical Psychological Science.
|
|
76
|
+
absolutist_ratio = RegexRatioMetric(
|
|
77
|
+
r"\b(absolutely|all|always|complete|completely|constant|constantly|"
|
|
78
|
+
r"definitely|entire|ever|every|everyone|everything|full|must|never|"
|
|
79
|
+
r"nothing|totally|whole)\b"
|
|
80
|
+
)
|
|
81
|
+
# The proportion of 1st-person plural pronouns (we, us, our). Source:
|
|
82
|
+
# Rude, S., Gortner, E. M., & Pennebaker, J. W. (2004). Language use of
|
|
83
|
+
# depressed and depression-vulnerable college students. Cognition &
|
|
84
|
+
# Emotion. (Apparently, a decrease in first-person plural pronouns
|
|
85
|
+
# indicates social withdrawal and a lack of belongingness, which are
|
|
86
|
+
# pre-suicidal markers in something called "Interpersonal Theory of
|
|
87
|
+
# Suicide".
|
|
88
|
+
pronouns_plural_ratio = RegexRatioMetric(r"\b(we|us|our|ours|ourselves)\b")
|
|
89
|
+
|
|
90
|
+
# The proportion of words related to physical body, pain, and fatigue.
|
|
91
|
+
# Source: Kapfhammer, H. P. (2006). Somatic symptoms in depression.
|
|
92
|
+
# Dialogues in Clinical Neuroscience.
|
|
93
|
+
somatic_ratio = RegexRatioMetric(
|
|
94
|
+
r"\b(sleep|tired|ache|pain|fatigue|exhausted|exhaustion|body|"
|
|
95
|
+
r"stomach|hurt|insomnia|sick|ill|headache|nausea|weak)\b"
|
|
96
|
+
)
|
|
97
|
+
|
|
98
|
+
# The proportion of first-person singular pronouns (I, me, my, mine,
|
|
99
|
+
# myself). Source: Edwards, T., & Holtzman, N. S. (2017). A
|
|
100
|
+
# meta-analysis of correlations between depression and first person
|
|
101
|
+
# singular pronoun use. Journal of Research in Personality.
|
|
102
|
+
pronouns_ratio = RegexRatioMetric(r"\b(i|me|my|mine|myself)\b")
|
|
103
|
+
|
|
104
|
+
# The proportion of words expressing hopelessness and lack of future.
|
|
105
|
+
# Source: Beck, A. T., et al. (1985). Hopelessness and eventual suicide:
|
|
106
|
+
# a 10-year prospective study of patients hospitalized with suicidal
|
|
107
|
+
# ideation. American Journal of Psychiatry.
|
|
108
|
+
hopelessness_ratio = RegexRatioMetric(
|
|
109
|
+
r"\b(hopeless|worthless|pointless|give up|no future|too late|"
|
|
110
|
+
r"useless|trapped|despair|end it|meaningless)\b"
|
|
111
|
+
)
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
class SentimentMetric(TextMetric):
|
|
115
|
+
"""Generalises specific scores from the sentiment analyser."""
|
|
116
|
+
|
|
117
|
+
_analyzer: SentimentIntensityAnalyzer | None = None
|
|
118
|
+
|
|
119
|
+
def __init__(self, sentiment_key: str):
|
|
120
|
+
"""Initialises this object.
|
|
121
|
+
|
|
122
|
+
:param sentiment_key: The key of the sentiment analyser (for
|
|
123
|
+
SentimentIntensityAnalyzer of nltk).
|
|
124
|
+
"""
|
|
125
|
+
self.key = sentiment_key
|
|
126
|
+
|
|
127
|
+
def compute(self, text: str, **kwargs: Any) -> float:
|
|
128
|
+
if SentimentMetric._analyzer is None:
|
|
129
|
+
SentimentMetric._analyzer = SentimentIntensityAnalyzer()
|
|
130
|
+
|
|
131
|
+
assert self._analyzer is not None
|
|
132
|
+
return float(self._analyzer.polarity_scores(text)[self.key])
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
# The proportion of strictly negative emotional valence in the text.
|
|
136
|
+
# Source: Eichstaedt, J. C., et al. (2018). Facebook language predicts
|
|
137
|
+
# depression in medical records. Proceedings of the National Academy of
|
|
138
|
+
# Sciences (PNAS). (Elevated baseline negative sentiment is an indicator
|
|
139
|
+
# of depressive disorders).
|
|
140
|
+
negative_sentiment = SentimentMetric("neg")
|
|
141
|
+
|
|
142
|
+
# The overall compound emotional polarity, balancing positive, negative,
|
|
143
|
+
# and neutral markers. Source: De Choudhury, M., Gamon, M., Counts, S.,
|
|
144
|
+
# & Horvitz, E. (2013). Predicting depression via social media.
|
|
145
|
+
# Proceedings of the International AAAI Conference on Web and Social
|
|
146
|
+
# Media (ICWSM). (Captures affective shifts.
|
|
147
|
+
compound_sentiment = SentimentMetric("compound")
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
def type_token_ratio(text: str) -> float:
|
|
151
|
+
"""Lexical diversity through the Type-Token Ratio (unique / total).
|
|
152
|
+
|
|
153
|
+
Source: Tausczik, Y. R., & Pennebaker, J. W. (2010). The
|
|
154
|
+
psychological meaning of words: LIWC and computerised text analysis
|
|
155
|
+
methods. Journal of Language and Social Psychology. (A low
|
|
156
|
+
Type-Token Ratio indicates a highly repetitive vocabulary, which
|
|
157
|
+
correlates with cognitive rumination).
|
|
158
|
+
|
|
159
|
+
:param text: The text of the sentence.
|
|
160
|
+
:return: A value from 0.0 (extreme repetition) to 1.0 (every word is
|
|
161
|
+
unique).
|
|
162
|
+
"""
|
|
163
|
+
tokens = [w.lower() for w in get_cached_tokens(text) if w.isalnum()]
|
|
164
|
+
if tokens:
|
|
165
|
+
return len(set(tokens)) / len(tokens)
|
|
166
|
+
else:
|
|
167
|
+
return 0.0
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def temporal_myopia_index(text: str) -> float:
|
|
171
|
+
"""Compares the usage of past tense vs future tense verbs.
|
|
172
|
+
|
|
173
|
+
Source: Baumeister, R. F. (1990). Suicide as escape from self.
|
|
174
|
+
Psychological Review. (People in suicidal crises often experience
|
|
175
|
+
temporal narrowing, losing the ability to think of themselves into
|
|
176
|
+
the future. A high ratio indicates being 'stuck in the past' with an
|
|
177
|
+
absence of future orientation).
|
|
178
|
+
|
|
179
|
+
:param text: The text of the sentence.
|
|
180
|
+
:return: A value where higher means more past-oriented (myopia) and
|
|
181
|
+
lower means future-oriented.
|
|
182
|
+
"""
|
|
183
|
+
tokens = get_cached_tokens(text)
|
|
184
|
+
tags = nltk.pos_tag(tokens)
|
|
185
|
+
|
|
186
|
+
# VBD: Tiempo pasado, VBN: Participio
|
|
187
|
+
past_verbs = sum(1 for word, tag in tags if tag in ("VBD", "VBN"))
|
|
188
|
+
# MD: Verbos modales (will, shall, can), que indican proyección de
|
|
189
|
+
# futuro o posibilidad
|
|
190
|
+
future_modals = sum(1 for word, tag in tags if tag == "MD")
|
|
191
|
+
|
|
192
|
+
return past_verbs / (
|
|
193
|
+
future_modals + 1.0
|
|
194
|
+
) # No dividir entre 0 que se borra la Internet
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
def readability_score(text: str) -> float:
|
|
198
|
+
"""The Flesch-Kincaid Grade Level (thought disorganisation).
|
|
199
|
+
|
|
200
|
+
Source: Mota, N. B., et al. (2018). Speech analysis by graph theory
|
|
201
|
+
to assess cognitive impairment. Schizophrenia Research. (The thought
|
|
202
|
+
disorganisation is typical in depression, mania, or psychosis).
|
|
203
|
+
|
|
204
|
+
:param text: The text of the sentence.
|
|
205
|
+
:return: Float representing the U.S. school grade required to
|
|
206
|
+
understand the text.
|
|
207
|
+
"""
|
|
208
|
+
return float(textstat.flesch_kincaid_grade(text))
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
def insomnia_post(text: str, date: str = "") -> float:
|
|
212
|
+
"""Determines if the post is published during nighttime (insomnia).
|
|
213
|
+
|
|
214
|
+
:param text: The text of the sentence (unused here, but required by
|
|
215
|
+
signature).
|
|
216
|
+
:param date: The timestamp of the post.
|
|
217
|
+
:return: 1.0 if the post was between 2 AM and 5 AM, otherwise 0.0.
|
|
218
|
+
"""
|
|
219
|
+
if not date:
|
|
220
|
+
return 0.0
|
|
221
|
+
try:
|
|
222
|
+
dt = datetime.fromisoformat(date.replace("Z", "+00:00"))
|
|
223
|
+
if dt.hour in [2, 3, 4, 5]:
|
|
224
|
+
return 1.0
|
|
225
|
+
except ValueError:
|
|
226
|
+
try:
|
|
227
|
+
dt = datetime.strptime(date, "%Y-%m-%d %H:%M:%S")
|
|
228
|
+
if dt.hour in (2, 3, 4, 5):
|
|
229
|
+
return 1.0
|
|
230
|
+
except Exception as e:
|
|
231
|
+
logger.error(f"Error parsing {date}: {e}", exc_info=True)
|
|
232
|
+
except Exception as e:
|
|
233
|
+
logger.error(f"Error parsing {date}: {e}", exc_info=True)
|
|
234
|
+
|
|
235
|
+
return 0.0
|
|
File without changes
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
import os
|
|
2
|
+
|
|
3
|
+
import nltk
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
def download_nltk_resources(download_dir: str | None = None) -> None:
|
|
7
|
+
"""Downloads the required lexical resources.
|
|
8
|
+
|
|
9
|
+
:param download_dir: Directory to download the resources to.
|
|
10
|
+
"""
|
|
11
|
+
if download_dir is None:
|
|
12
|
+
download_dir = os.path.join(os.path.expanduser("~"), "nltk_data")
|
|
13
|
+
|
|
14
|
+
os.makedirs(download_dir, exist_ok=True)
|
|
15
|
+
nltk.data.path.append(download_dir)
|
|
16
|
+
|
|
17
|
+
for res in [
|
|
18
|
+
"vader_lexicon",
|
|
19
|
+
"punkt",
|
|
20
|
+
"punkt_tab",
|
|
21
|
+
"averaged_perceptron_tagger_eng",
|
|
22
|
+
]:
|
|
23
|
+
nltk.download(res, download_dir=download_dir, quiet=True)
|