temporal-normalization-spacy 2.0.2__tar.gz → 2.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {temporal_normalization_spacy-2.0.2 → temporal_normalization_spacy-2.1.0}/MANIFEST.in +1 -1
- {temporal_normalization_spacy-2.0.2/temporal_normalization_spacy.egg-info → temporal_normalization_spacy-2.1.0}/PKG-INFO +17 -5
- {temporal_normalization_spacy-2.0.2 → temporal_normalization_spacy-2.1.0}/README.md +14 -3
- {temporal_normalization_spacy-2.0.2 → temporal_normalization_spacy-2.1.0}/setup.py +2 -2
- {temporal_normalization_spacy-2.0.2 → temporal_normalization_spacy-2.1.0}/temporal_normalization/commons/temporal_models.py +39 -3
- {temporal_normalization_spacy-2.0.2 → temporal_normalization_spacy-2.1.0}/temporal_normalization/index.py +33 -10
- temporal_normalization_spacy-2.0.2/temporal_normalization/libs/temporal-normalization-2.0.jar → temporal_normalization_spacy-2.1.0/temporal_normalization/libs/temporal-normalization-2.1.0.jar +0 -0
- temporal_normalization_spacy-2.1.0/temporal_normalization/process/java_process.py +129 -0
- {temporal_normalization_spacy-2.0.2 → temporal_normalization_spacy-2.1.0/temporal_normalization_spacy.egg-info}/PKG-INFO +17 -5
- {temporal_normalization_spacy-2.0.2 → temporal_normalization_spacy-2.1.0}/temporal_normalization_spacy.egg-info/SOURCES.txt +1 -1
- temporal_normalization_spacy-2.0.2/temporal_normalization/process/java_process.py +0 -153
- {temporal_normalization_spacy-2.0.2 → temporal_normalization_spacy-2.1.0}/LICENSE +0 -0
- {temporal_normalization_spacy-2.0.2 → temporal_normalization_spacy-2.1.0}/setup.cfg +0 -0
- {temporal_normalization_spacy-2.0.2 → temporal_normalization_spacy-2.1.0}/temporal_normalization/__init__.py +0 -0
- {temporal_normalization_spacy-2.0.2 → temporal_normalization_spacy-2.1.0}/temporal_normalization/commons/__init__.py +0 -0
- {temporal_normalization_spacy-2.0.2 → temporal_normalization_spacy-2.1.0}/temporal_normalization/commons/print_utils.py +0 -0
- {temporal_normalization_spacy-2.0.2 → temporal_normalization_spacy-2.1.0}/temporal_normalization/commons/temporal_types.py +0 -0
- {temporal_normalization_spacy-2.0.2 → temporal_normalization_spacy-2.1.0}/temporal_normalization/process/__init__.py +0 -0
- {temporal_normalization_spacy-2.0.2 → temporal_normalization_spacy-2.1.0}/temporal_normalization_spacy.egg-info/dependency_links.txt +0 -0
- {temporal_normalization_spacy-2.0.2 → temporal_normalization_spacy-2.1.0}/temporal_normalization_spacy.egg-info/requires.txt +0 -0
- {temporal_normalization_spacy-2.0.2 → temporal_normalization_spacy-2.1.0}/temporal_normalization_spacy.egg-info/top_level.txt +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
Metadata-Version: 2.
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
2
|
Name: temporal_normalization_spacy
|
|
3
|
-
Version: 2.0
|
|
3
|
+
Version: 2.1.0
|
|
4
4
|
Summary: A spaCy plugin for identifying and parsing historical data in Romanian texts
|
|
5
5
|
Home-page: https://github.com/iliedorobat/timespan-normalization-spacy
|
|
6
6
|
Author: Ilie Cristian Dorobat
|
|
@@ -18,6 +18,7 @@ Dynamic: classifier
|
|
|
18
18
|
Dynamic: description
|
|
19
19
|
Dynamic: description-content-type
|
|
20
20
|
Dynamic: home-page
|
|
21
|
+
Dynamic: license-file
|
|
21
22
|
Dynamic: requires-dist
|
|
22
23
|
Dynamic: requires-python
|
|
23
24
|
Dynamic: summary
|
|
@@ -167,8 +168,17 @@ for entity in doc.ents:
|
|
|
167
168
|
still be loaded on first run,** and this process may take a few seconds/tens of seconds.
|
|
168
169
|
|
|
169
170
|
### Importing Modules & Defining Constants
|
|
171
|
+
|
|
170
172
|
```python
|
|
171
|
-
from
|
|
173
|
+
from pathlib import Path
|
|
174
|
+
|
|
175
|
+
from temporal_normalization import (
|
|
176
|
+
close_conn,
|
|
177
|
+
console,
|
|
178
|
+
extract_temporal_expressions,
|
|
179
|
+
start_conn,
|
|
180
|
+
TemporalExpression,
|
|
181
|
+
)
|
|
172
182
|
|
|
173
183
|
LANG = "ro"
|
|
174
184
|
TEXT_RO = (
|
|
@@ -183,8 +193,10 @@ TEXT_RO = (
|
|
|
183
193
|
# Display a warning if the language of the text is not Romanian.
|
|
184
194
|
console.lang_warning(TEXT_RO, target_lang=LANG)
|
|
185
195
|
|
|
186
|
-
|
|
187
|
-
|
|
196
|
+
root_path = str(Path(__file__).resolve().parent.parent.parent)
|
|
197
|
+
java_process, gateway = start_conn(root_path)
|
|
198
|
+
expressions: list[TemporalExpression] = extract_temporal_expressions(gateway, TEXT_RO)
|
|
199
|
+
close_conn(java_process, gateway)
|
|
188
200
|
```
|
|
189
201
|
|
|
190
202
|
### Accessing the Parsed Temporal Expressions
|
|
@@ -143,8 +143,17 @@ for entity in doc.ents:
|
|
|
143
143
|
still be loaded on first run,** and this process may take a few seconds/tens of seconds.
|
|
144
144
|
|
|
145
145
|
### Importing Modules & Defining Constants
|
|
146
|
+
|
|
146
147
|
```python
|
|
147
|
-
from
|
|
148
|
+
from pathlib import Path
|
|
149
|
+
|
|
150
|
+
from temporal_normalization import (
|
|
151
|
+
close_conn,
|
|
152
|
+
console,
|
|
153
|
+
extract_temporal_expressions,
|
|
154
|
+
start_conn,
|
|
155
|
+
TemporalExpression,
|
|
156
|
+
)
|
|
148
157
|
|
|
149
158
|
LANG = "ro"
|
|
150
159
|
TEXT_RO = (
|
|
@@ -159,8 +168,10 @@ TEXT_RO = (
|
|
|
159
168
|
# Display a warning if the language of the text is not Romanian.
|
|
160
169
|
console.lang_warning(TEXT_RO, target_lang=LANG)
|
|
161
170
|
|
|
162
|
-
|
|
163
|
-
|
|
171
|
+
root_path = str(Path(__file__).resolve().parent.parent.parent)
|
|
172
|
+
java_process, gateway = start_conn(root_path)
|
|
173
|
+
expressions: list[TemporalExpression] = extract_temporal_expressions(gateway, TEXT_RO)
|
|
174
|
+
close_conn(java_process, gateway)
|
|
164
175
|
```
|
|
165
176
|
|
|
166
177
|
### Accessing the Parsed Temporal Expressions
|
|
@@ -2,7 +2,7 @@ from setuptools import setup, find_packages
|
|
|
2
2
|
|
|
3
3
|
setup(
|
|
4
4
|
name="temporal_normalization_spacy",
|
|
5
|
-
version="2.0
|
|
5
|
+
version="2.1.0",
|
|
6
6
|
author="Ilie Cristian Dorobat",
|
|
7
7
|
description="A spaCy plugin for identifying and parsing historical data "
|
|
8
8
|
"in Romanian texts",
|
|
@@ -12,7 +12,7 @@ setup(
|
|
|
12
12
|
packages=find_packages(),
|
|
13
13
|
include_package_data=True,
|
|
14
14
|
package_data={
|
|
15
|
-
"temporal_normalization.libs": ["temporal-normalization-2.0.jar"],
|
|
15
|
+
"temporal_normalization.libs": ["temporal-normalization-2.1.0.jar"],
|
|
16
16
|
},
|
|
17
17
|
install_requires=["spacy>=3.0", "py4j", "langdetect"],
|
|
18
18
|
classifiers=[
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import json
|
|
2
2
|
|
|
3
|
-
from py4j.java_gateway import JavaObject
|
|
3
|
+
from py4j.java_gateway import JavaObject, JavaGateway
|
|
4
4
|
|
|
5
5
|
from temporal_normalization.commons.temporal_types import TemporalType
|
|
6
6
|
|
|
@@ -14,6 +14,7 @@ class TemporalExpression:
|
|
|
14
14
|
is_valid (bool): A flag that specifies whether the text processed
|
|
15
15
|
through timespan-normalization library is a temporal expression.
|
|
16
16
|
input_value (str or None): The original temporal expression before processing.
|
|
17
|
+
prepared_value (str or None): The temporal expression after processing.
|
|
17
18
|
time_series (list[TimeSeries]): The list of normalized temporal expressions.
|
|
18
19
|
matches (list[str]): A unique list of matched values found in the normalized
|
|
19
20
|
entities.
|
|
@@ -23,10 +24,12 @@ class TemporalExpression:
|
|
|
23
24
|
serialize = java_object.serialize()
|
|
24
25
|
json_obj = json.loads(serialize)
|
|
25
26
|
|
|
27
|
+
# fmt: off
|
|
26
28
|
self.is_valid = TemporalExpression.is_valid_json(json_obj)
|
|
27
29
|
self.input_value: str | None = json_obj["inputValue"] if self.is_valid else None
|
|
30
|
+
self.prepared_value: str | None = json_obj["preparedValue"] if self.is_valid else None
|
|
28
31
|
self.time_series: list[TimeSeries] = [
|
|
29
|
-
TimeSeries(item) for item in json_obj["timeSeries"]
|
|
32
|
+
TimeSeries(item, self.input_value, self.prepared_value) for item in json_obj["timeSeries"]
|
|
30
33
|
] if self.is_valid else []
|
|
31
34
|
self.matches: list[str] = list(
|
|
32
35
|
set(
|
|
@@ -37,6 +40,7 @@ class TemporalExpression:
|
|
|
37
40
|
]
|
|
38
41
|
)
|
|
39
42
|
)
|
|
43
|
+
# fmt: on
|
|
40
44
|
|
|
41
45
|
def __str__(self):
|
|
42
46
|
if self.input_value is None:
|
|
@@ -52,12 +56,40 @@ class TemporalExpression:
|
|
|
52
56
|
return "inputValue" in json_obj and "timeSeries" in json_obj
|
|
53
57
|
|
|
54
58
|
|
|
59
|
+
def extract_temporal_expressions(
|
|
60
|
+
gateway: JavaGateway, text: str
|
|
61
|
+
) -> list[TemporalExpression]:
|
|
62
|
+
"""
|
|
63
|
+
Extracts valid temporal expressions from the given text using the Java temporal
|
|
64
|
+
normalization gateway.
|
|
65
|
+
|
|
66
|
+
Args:
|
|
67
|
+
gateway (JavaGateway): Active Py4J gateway connected to the Java temporal
|
|
68
|
+
normalization process.
|
|
69
|
+
text (str): Input text from which to extract temporal expressions.
|
|
70
|
+
|
|
71
|
+
Returns:
|
|
72
|
+
list[TemporalExpression]: A list containing valid temporal expressions.
|
|
73
|
+
"""
|
|
74
|
+
|
|
75
|
+
expressions: list[TemporalExpression] = []
|
|
76
|
+
java_object = gateway.jvm.ro.webdata.normalization.timespan.ro.TimeExpression(text)
|
|
77
|
+
temporal_expression = TemporalExpression(java_object)
|
|
78
|
+
|
|
79
|
+
if temporal_expression.is_valid:
|
|
80
|
+
expressions.append(temporal_expression)
|
|
81
|
+
|
|
82
|
+
return expressions
|
|
83
|
+
|
|
84
|
+
|
|
55
85
|
class TimeSeries:
|
|
56
86
|
"""
|
|
57
87
|
A data structure representing a temporal expression that has been normalized
|
|
58
88
|
into a list of periods and temporal edges.
|
|
59
89
|
|
|
60
90
|
Attributes:
|
|
91
|
+
input_value (str or None): The original temporal expression before processing.
|
|
92
|
+
prepared_value (str or None): The temporal expression after processing.
|
|
61
93
|
edges (list[EdgeModel]): A list of temporal intervals represented as edges.
|
|
62
94
|
periods (list[DBpediaModel]): A list of normalized DBpedia entities
|
|
63
95
|
extracted from the expression.
|
|
@@ -65,7 +97,9 @@ class TimeSeries:
|
|
|
65
97
|
entities.
|
|
66
98
|
"""
|
|
67
99
|
|
|
68
|
-
def __init__(self, data: dict):
|
|
100
|
+
def __init__(self, data: dict, input_value: str, prepared_value: str):
|
|
101
|
+
self.input_value = input_value
|
|
102
|
+
self.prepared_value = prepared_value
|
|
69
103
|
self.edges: EdgeModel = EdgeModel(data["edges"]) if "edges" in data else None
|
|
70
104
|
self.periods: list[DBpediaModel] = (
|
|
71
105
|
[DBpediaModel(item) for item in data["periods"]]
|
|
@@ -80,10 +114,12 @@ class TimeSeries:
|
|
|
80
114
|
return f"TimeSeries(edges={self.edges}, periods={self.periods})"
|
|
81
115
|
|
|
82
116
|
def serialize(self, indent: str = ""):
|
|
117
|
+
# fmt: off
|
|
83
118
|
return (
|
|
84
119
|
f"{indent}Edges: {self.edges}\n"
|
|
85
120
|
f"{indent}Periods: {self.periods}"
|
|
86
121
|
)
|
|
122
|
+
# fmt: on
|
|
87
123
|
|
|
88
124
|
|
|
89
125
|
class DBpediaModel:
|
|
@@ -1,17 +1,26 @@
|
|
|
1
1
|
import re
|
|
2
|
+
import subprocess
|
|
3
|
+
from pathlib import Path
|
|
2
4
|
|
|
5
|
+
from py4j.java_gateway import JavaGateway
|
|
3
6
|
from spacy import Language
|
|
4
7
|
from spacy.tokens import Doc, Span
|
|
8
|
+
from spacy.tokens._retokenize import Retokenizer
|
|
5
9
|
from spacy.util import filter_spans
|
|
6
10
|
|
|
7
11
|
from temporal_normalization import TimeSeries
|
|
8
|
-
from temporal_normalization.commons.temporal_models import
|
|
9
|
-
|
|
12
|
+
from temporal_normalization.commons.temporal_models import (
|
|
13
|
+
extract_temporal_expressions,
|
|
14
|
+
TemporalExpression,
|
|
15
|
+
)
|
|
16
|
+
from temporal_normalization.process.java_process import start_conn, close_conn
|
|
10
17
|
|
|
11
18
|
try:
|
|
19
|
+
|
|
12
20
|
@Language.factory("temporal_normalization")
|
|
13
21
|
def create_normalized_component(nlp, name):
|
|
14
22
|
return TemporalNormalization(nlp, name)
|
|
23
|
+
|
|
15
24
|
except AttributeError:
|
|
16
25
|
# spaCy 2.x
|
|
17
26
|
pass
|
|
@@ -21,7 +30,7 @@ class TemporalNormalization:
|
|
|
21
30
|
"""
|
|
22
31
|
spaCy pipeline component for identifying and annotating temporal expressions in text.
|
|
23
32
|
|
|
24
|
-
This component calls the ``
|
|
33
|
+
This component calls the ``start_conn`` method to extract temporal expressions, then
|
|
25
34
|
aligns the matches with spaCy tokens using retokenization and sets a custom attribute
|
|
26
35
|
containing associated time series metadata.
|
|
27
36
|
"""
|
|
@@ -40,6 +49,11 @@ class TemporalNormalization:
|
|
|
40
49
|
Span.set_extension(TemporalNormalization.__FIELD, default=None, force=True)
|
|
41
50
|
self.nlp = nlp
|
|
42
51
|
|
|
52
|
+
root_path = str(Path(__file__).resolve().parent.parent)
|
|
53
|
+
java_process, gateway = start_conn(root_path)
|
|
54
|
+
self.java_process: subprocess.Popen = java_process
|
|
55
|
+
self.gateway: JavaGateway = gateway
|
|
56
|
+
|
|
43
57
|
def __call__(self, doc: Doc) -> Doc:
|
|
44
58
|
"""
|
|
45
59
|
Apply the component to a spaCy Doc object.
|
|
@@ -54,14 +68,17 @@ class TemporalNormalization:
|
|
|
54
68
|
Doc: The modified Doc object with temporal expressions processed.
|
|
55
69
|
"""
|
|
56
70
|
|
|
57
|
-
expressions: list[TemporalExpression] =
|
|
58
|
-
|
|
71
|
+
expressions: list[TemporalExpression] = extract_temporal_expressions(
|
|
72
|
+
self.gateway, doc.text
|
|
73
|
+
)
|
|
59
74
|
str_matches: list[str] = _prepare_str_patterns(expressions)
|
|
60
|
-
|
|
61
75
|
_retokenize(doc, str_matches, expressions)
|
|
62
76
|
|
|
63
77
|
return doc
|
|
64
78
|
|
|
79
|
+
def __del__(self):
|
|
80
|
+
close_conn(self.java_process, self.gateway)
|
|
81
|
+
|
|
65
82
|
|
|
66
83
|
def _prepare_str_patterns(expressions: list[TemporalExpression]) -> list[str]:
|
|
67
84
|
"""
|
|
@@ -101,7 +118,8 @@ def _retokenize(
|
|
|
101
118
|
each with time series metadata.
|
|
102
119
|
"""
|
|
103
120
|
|
|
104
|
-
|
|
121
|
+
# TODO: WIP
|
|
122
|
+
regex_matches: list[str] = [rf"{re.escape(item)}" for item in str_matches]
|
|
105
123
|
pattern = f"({'|'.join(regex_matches)})"
|
|
106
124
|
matches = (
|
|
107
125
|
list(re.finditer(pattern, doc.text, re.IGNORECASE))
|
|
@@ -126,6 +144,7 @@ def _retokenize(
|
|
|
126
144
|
if token.idx + len(token.text) == end_char:
|
|
127
145
|
end_token = token.i
|
|
128
146
|
|
|
147
|
+
# fmt: off
|
|
129
148
|
if start_token is not None and end_token is not None:
|
|
130
149
|
# use exact token boundaries to create a custom `Span` for well-defined
|
|
131
150
|
# time expressions with known character offsets.
|
|
@@ -144,6 +163,8 @@ def _retokenize(
|
|
|
144
163
|
matched_ts = [ts for ts in time_series if _is_substring(entity.text, ts.matches)]
|
|
145
164
|
_retokenize_entity(doc, matched_ts, entity, True, retokenized_entities, retokenizer)
|
|
146
165
|
|
|
166
|
+
# fmt: on
|
|
167
|
+
|
|
147
168
|
|
|
148
169
|
def _retokenize_entity(
|
|
149
170
|
doc: Doc,
|
|
@@ -151,7 +172,7 @@ def _retokenize_entity(
|
|
|
151
172
|
entity: Span,
|
|
152
173
|
existed_entity: bool,
|
|
153
174
|
retokenized_entities: list[Span],
|
|
154
|
-
retokenizer:
|
|
175
|
+
retokenizer: Retokenizer,
|
|
155
176
|
) -> None:
|
|
156
177
|
"""
|
|
157
178
|
Retokenizes and enriches a temporal entity span with matched time series data.
|
|
@@ -173,6 +194,8 @@ def _retokenize_entity(
|
|
|
173
194
|
_update_doc_ents(doc, entity)
|
|
174
195
|
_merge_entity(doc, entity, retokenized_entities, retokenizer)
|
|
175
196
|
|
|
197
|
+
return None
|
|
198
|
+
|
|
176
199
|
|
|
177
200
|
def _assign_time_series(
|
|
178
201
|
matched_ts: list[TimeSeries], entity: Span, existed_entity: bool
|
|
@@ -214,7 +237,7 @@ def _merge_entity(
|
|
|
214
237
|
doc: Doc,
|
|
215
238
|
entity: Span,
|
|
216
239
|
retokenized_entities: list[Span],
|
|
217
|
-
retokenizer:
|
|
240
|
+
retokenizer: Retokenizer,
|
|
218
241
|
) -> None:
|
|
219
242
|
"""
|
|
220
243
|
Merges a custom entity span into the spaCy Doc if it is not already part of
|
|
@@ -223,7 +246,7 @@ def _merge_entity(
|
|
|
223
246
|
Args:
|
|
224
247
|
entity (Span): The named entity to enrich.
|
|
225
248
|
retokenized_entities (list): Accumulator for entities that require retokenization.
|
|
226
|
-
retokenizer (
|
|
249
|
+
retokenizer (Retokenizer): The spaCy retokenizer context.
|
|
227
250
|
"""
|
|
228
251
|
|
|
229
252
|
if entity not in doc.ents:
|
|
@@ -0,0 +1,129 @@
|
|
|
1
|
+
import re
|
|
2
|
+
import shutil
|
|
3
|
+
import subprocess
|
|
4
|
+
|
|
5
|
+
from py4j.java_gateway import JavaGateway
|
|
6
|
+
from py4j.protocol import Py4JNetworkError
|
|
7
|
+
|
|
8
|
+
from temporal_normalization.commons.print_utils import console
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def start_conn(root_path: str) -> tuple[subprocess.Popen, JavaGateway]:
|
|
12
|
+
"""
|
|
13
|
+
Starts the Java temporal normalization process and establishes a Py4J gateway connection.
|
|
14
|
+
|
|
15
|
+
Args:
|
|
16
|
+
root_path (str): The root directory of the project.
|
|
17
|
+
|
|
18
|
+
Returns:
|
|
19
|
+
tuple[subprocess.Popen, JavaGateway]:
|
|
20
|
+
- The subprocess.Popen object representing the running Java process.
|
|
21
|
+
- The JavaGateway object representing the active Py4J connection.
|
|
22
|
+
|
|
23
|
+
Note:
|
|
24
|
+
- Requires Java 11 or higher to be installed and accessible in the system PATH.
|
|
25
|
+
- Requires `temporal-normalization-2.1.0.jar` to be present in the `libs` directory.
|
|
26
|
+
- The caller is responsible for closing the gateway and terminating the Java process
|
|
27
|
+
after usage to avoid orphaned processes.
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
check_java_version()
|
|
31
|
+
|
|
32
|
+
jar_path = (
|
|
33
|
+
f"{root_path}/temporal_normalization/libs/temporal-normalization-2.1.0.jar"
|
|
34
|
+
)
|
|
35
|
+
|
|
36
|
+
java_process = subprocess.Popen(
|
|
37
|
+
["java", "-jar", jar_path, "--python"],
|
|
38
|
+
stdout=subprocess.PIPE,
|
|
39
|
+
stderr=subprocess.PIPE,
|
|
40
|
+
text=True,
|
|
41
|
+
)
|
|
42
|
+
|
|
43
|
+
for line in java_process.stdout:
|
|
44
|
+
if "Gateway Server Started." in line:
|
|
45
|
+
print(line.strip())
|
|
46
|
+
break
|
|
47
|
+
|
|
48
|
+
gateway = JavaGateway()
|
|
49
|
+
print("Python connection established.")
|
|
50
|
+
|
|
51
|
+
return java_process, gateway
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def close_conn(java_process: subprocess.Popen, gateway: JavaGateway) -> None:
|
|
55
|
+
"""
|
|
56
|
+
Closes the active connection between Python and the Java process started via Py4J.
|
|
57
|
+
|
|
58
|
+
This function ensures a proper shutdown sequence:
|
|
59
|
+
1. Attempts to gracefully shut down the Py4J gateway connection.
|
|
60
|
+
- If the Java process is already closed, a Py4JNetworkError is caught and logged.
|
|
61
|
+
2. Terminates the underlying Java process.
|
|
62
|
+
3. Prints status messages for debugging/confirmation.
|
|
63
|
+
|
|
64
|
+
Args:
|
|
65
|
+
java_process (subprocess.Popen): The Java process launched with subprocess.
|
|
66
|
+
gateway (JavaGateway): The active Py4J gateway connection.
|
|
67
|
+
|
|
68
|
+
Notes:
|
|
69
|
+
- Call this function once you have finished all interactions with the Java process.
|
|
70
|
+
- It is safe to call even if the Java process has already exited.
|
|
71
|
+
"""
|
|
72
|
+
|
|
73
|
+
try:
|
|
74
|
+
# Proper way to shut down Py4J
|
|
75
|
+
gateway.shutdown()
|
|
76
|
+
print("Python connection closed.")
|
|
77
|
+
except Py4JNetworkError:
|
|
78
|
+
print("Java process already shut down.")
|
|
79
|
+
|
|
80
|
+
# Terminate Java process
|
|
81
|
+
java_process.terminate()
|
|
82
|
+
print("Java server is shutting down...")
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def check_java_version() -> None:
|
|
86
|
+
"""
|
|
87
|
+
Verifies that Java is installed and meets the minimum required version.
|
|
88
|
+
|
|
89
|
+
This function checks for the presence of the Java executable in the system PATH,
|
|
90
|
+
runs ``java -version``, and ensures that the version is at least 11. If Java is not
|
|
91
|
+
installed or the version is too low, it logs an error using ``console.error``.
|
|
92
|
+
|
|
93
|
+
Raises:
|
|
94
|
+
Logs error messages, but does not raise exceptions directly.
|
|
95
|
+
"""
|
|
96
|
+
|
|
97
|
+
min_version = 11
|
|
98
|
+
java_path = shutil.which("java")
|
|
99
|
+
|
|
100
|
+
try:
|
|
101
|
+
if java_path:
|
|
102
|
+
# Run the command to check the Java version
|
|
103
|
+
result = subprocess.run(
|
|
104
|
+
[java_path, "-version"], capture_output=True, text=True
|
|
105
|
+
)
|
|
106
|
+
|
|
107
|
+
# Print the version information (Java version is printed to stderr)
|
|
108
|
+
if result.returncode == 0:
|
|
109
|
+
version_output = result.stderr
|
|
110
|
+
match = re.search(r'version "(\d+\.\d+)', version_output)
|
|
111
|
+
|
|
112
|
+
if match:
|
|
113
|
+
crr_version = float(match.group(1))
|
|
114
|
+
if crr_version < min_version:
|
|
115
|
+
console.error(
|
|
116
|
+
f"Java {crr_version} is installed, but version {min_version} is required." # noqa 501
|
|
117
|
+
)
|
|
118
|
+
else:
|
|
119
|
+
console.error("Could not extract Java version.")
|
|
120
|
+
else:
|
|
121
|
+
console.error("Error occurred while checking the version.")
|
|
122
|
+
else:
|
|
123
|
+
console.error("Java not found.")
|
|
124
|
+
except Exception as e:
|
|
125
|
+
console.error(e.__str__())
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
if __name__ == "__main__":
|
|
129
|
+
pass
|
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
Metadata-Version: 2.
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
2
|
Name: temporal_normalization_spacy
|
|
3
|
-
Version: 2.0
|
|
3
|
+
Version: 2.1.0
|
|
4
4
|
Summary: A spaCy plugin for identifying and parsing historical data in Romanian texts
|
|
5
5
|
Home-page: https://github.com/iliedorobat/timespan-normalization-spacy
|
|
6
6
|
Author: Ilie Cristian Dorobat
|
|
@@ -18,6 +18,7 @@ Dynamic: classifier
|
|
|
18
18
|
Dynamic: description
|
|
19
19
|
Dynamic: description-content-type
|
|
20
20
|
Dynamic: home-page
|
|
21
|
+
Dynamic: license-file
|
|
21
22
|
Dynamic: requires-dist
|
|
22
23
|
Dynamic: requires-python
|
|
23
24
|
Dynamic: summary
|
|
@@ -167,8 +168,17 @@ for entity in doc.ents:
|
|
|
167
168
|
still be loaded on first run,** and this process may take a few seconds/tens of seconds.
|
|
168
169
|
|
|
169
170
|
### Importing Modules & Defining Constants
|
|
171
|
+
|
|
170
172
|
```python
|
|
171
|
-
from
|
|
173
|
+
from pathlib import Path
|
|
174
|
+
|
|
175
|
+
from temporal_normalization import (
|
|
176
|
+
close_conn,
|
|
177
|
+
console,
|
|
178
|
+
extract_temporal_expressions,
|
|
179
|
+
start_conn,
|
|
180
|
+
TemporalExpression,
|
|
181
|
+
)
|
|
172
182
|
|
|
173
183
|
LANG = "ro"
|
|
174
184
|
TEXT_RO = (
|
|
@@ -183,8 +193,10 @@ TEXT_RO = (
|
|
|
183
193
|
# Display a warning if the language of the text is not Romanian.
|
|
184
194
|
console.lang_warning(TEXT_RO, target_lang=LANG)
|
|
185
195
|
|
|
186
|
-
|
|
187
|
-
|
|
196
|
+
root_path = str(Path(__file__).resolve().parent.parent.parent)
|
|
197
|
+
java_process, gateway = start_conn(root_path)
|
|
198
|
+
expressions: list[TemporalExpression] = extract_temporal_expressions(gateway, TEXT_RO)
|
|
199
|
+
close_conn(java_process, gateway)
|
|
188
200
|
```
|
|
189
201
|
|
|
190
202
|
### Accessing the Parsed Temporal Expressions
|
|
@@ -8,7 +8,7 @@ temporal_normalization/commons/__init__.py
|
|
|
8
8
|
temporal_normalization/commons/print_utils.py
|
|
9
9
|
temporal_normalization/commons/temporal_models.py
|
|
10
10
|
temporal_normalization/commons/temporal_types.py
|
|
11
|
-
temporal_normalization/libs/temporal-normalization-2.0.jar
|
|
11
|
+
temporal_normalization/libs/temporal-normalization-2.1.0.jar
|
|
12
12
|
temporal_normalization/process/__init__.py
|
|
13
13
|
temporal_normalization/process/java_process.py
|
|
14
14
|
temporal_normalization_spacy.egg-info/PKG-INFO
|
|
@@ -1,153 +0,0 @@
|
|
|
1
|
-
import os
|
|
2
|
-
import re
|
|
3
|
-
import shutil
|
|
4
|
-
import subprocess
|
|
5
|
-
|
|
6
|
-
from py4j.java_gateway import JavaGateway
|
|
7
|
-
from py4j.protocol import Py4JNetworkError
|
|
8
|
-
|
|
9
|
-
from temporal_normalization.commons.print_utils import console
|
|
10
|
-
from temporal_normalization.commons.temporal_models import TemporalExpression
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
def start_process(text: str, expressions: list[TemporalExpression]):
|
|
14
|
-
"""
|
|
15
|
-
Launches the Java-based temporal normalization process and populates a list
|
|
16
|
-
of temporal expressions extracted from the input text.
|
|
17
|
-
|
|
18
|
-
This function starts a Java subprocess that hosts the temporal-normalization
|
|
19
|
-
server via Py4J. Once the gateway is connected, it sends the input text for
|
|
20
|
-
processing, retrieves the temporal expressions, and then shuts down both
|
|
21
|
-
the Python and Java sides of the connection.
|
|
22
|
-
|
|
23
|
-
Args:
|
|
24
|
-
text (str): The input text to be analyzed for temporal expressions.
|
|
25
|
-
expressions (list[TemporalExpression]): A list to be populated with
|
|
26
|
-
extracted temporal expressions.
|
|
27
|
-
|
|
28
|
-
Example:
|
|
29
|
-
>>> expressions: list[TemporalExpression] = []
|
|
30
|
-
>>> start_process("Sec al II-lea a.ch. a fost o perioadă de mari schimbări.", expressions)
|
|
31
|
-
>>> # ... use the list of normalized temporal expressions.
|
|
32
|
-
|
|
33
|
-
Note:
|
|
34
|
-
Requires `temporal-normalization-2.0.jar` to be present in the `libs` directory.
|
|
35
|
-
Also requires Java 11 or higher to be installed and accessible in the system PATH.
|
|
36
|
-
"""
|
|
37
|
-
|
|
38
|
-
check_java_version()
|
|
39
|
-
|
|
40
|
-
jar_path = os.path.join(
|
|
41
|
-
os.path.dirname(__file__), "../libs/temporal-normalization-2.0.jar"
|
|
42
|
-
)
|
|
43
|
-
|
|
44
|
-
java_process = subprocess.Popen(
|
|
45
|
-
["java", "-jar", jar_path, "--python"],
|
|
46
|
-
stdout=subprocess.PIPE,
|
|
47
|
-
stderr=subprocess.PIPE,
|
|
48
|
-
text=True,
|
|
49
|
-
)
|
|
50
|
-
|
|
51
|
-
for line in java_process.stdout:
|
|
52
|
-
if "Gateway Server Started..." in line:
|
|
53
|
-
print(line.strip())
|
|
54
|
-
break
|
|
55
|
-
|
|
56
|
-
gateway = gateway_conn(text, expressions)
|
|
57
|
-
|
|
58
|
-
try:
|
|
59
|
-
# Proper way to shut down Py4J
|
|
60
|
-
gateway.shutdown()
|
|
61
|
-
print("Python connection closed.")
|
|
62
|
-
except Py4JNetworkError:
|
|
63
|
-
print("Java process already shut down.")
|
|
64
|
-
|
|
65
|
-
# Terminate Java process
|
|
66
|
-
java_process.terminate()
|
|
67
|
-
print("Java server is shutting down...")
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
def gateway_conn(text: str, expressions: list[TemporalExpression]) -> JavaGateway:
|
|
71
|
-
"""
|
|
72
|
-
Establishes a connection to the Java Py4J gateway and initializes the
|
|
73
|
-
temporal expression extraction.
|
|
74
|
-
|
|
75
|
-
It creates an instance of the Java class ``TimeExpression``, wraps it in
|
|
76
|
-
a Python ``TemporalExpression``, and appends it to the given list if valid.
|
|
77
|
-
|
|
78
|
-
Args:
|
|
79
|
-
text (str): The input text to analyze.
|
|
80
|
-
expressions (list[TemporalExpression]): A list to store extracted expressions.
|
|
81
|
-
|
|
82
|
-
Returns:
|
|
83
|
-
JavaGateway: The Py4J JavaGateway object used to manage the connection.
|
|
84
|
-
|
|
85
|
-
Example:
|
|
86
|
-
>>> from py4j.java_gateway import JavaGateway
|
|
87
|
-
>>> expressions: list[TemporalExpression] = []
|
|
88
|
-
>>> gateway = gateway_conn("Sec al II-lea a.ch. a fost o perioadă de mari schimbări.", expressions)
|
|
89
|
-
>>> # ... use the list of normalized temporal expressions.
|
|
90
|
-
>>> gateway.shutdown()
|
|
91
|
-
|
|
92
|
-
Note:
|
|
93
|
-
Assumes that a Py4J-compatible Java process is already running and exposing
|
|
94
|
-
the `ro.webdata.normalization.timespan.ro.TimeExpression` class.
|
|
95
|
-
"""
|
|
96
|
-
|
|
97
|
-
gateway = JavaGateway()
|
|
98
|
-
print("Python connection established.")
|
|
99
|
-
|
|
100
|
-
java_object = gateway.jvm.ro.webdata.normalization.timespan.ro.TimeExpression(text)
|
|
101
|
-
time_expression = TemporalExpression(java_object)
|
|
102
|
-
|
|
103
|
-
if time_expression.is_valid:
|
|
104
|
-
expressions.append(time_expression)
|
|
105
|
-
|
|
106
|
-
return gateway
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
def check_java_version():
|
|
110
|
-
"""
|
|
111
|
-
Verifies that Java is installed and meets the minimum required version.
|
|
112
|
-
|
|
113
|
-
This function checks for the presence of the Java executable in the system PATH,
|
|
114
|
-
runs ``java -version``, and ensures that the version is at least 11. If Java is not
|
|
115
|
-
installed or the version is too low, it logs an error using ``console.error``.
|
|
116
|
-
|
|
117
|
-
Raises:
|
|
118
|
-
Logs error messages, but does not raise exceptions directly.
|
|
119
|
-
"""
|
|
120
|
-
|
|
121
|
-
min_version = 11
|
|
122
|
-
java_path = shutil.which("java")
|
|
123
|
-
|
|
124
|
-
try:
|
|
125
|
-
if java_path:
|
|
126
|
-
# Run the command to check the Java version
|
|
127
|
-
result = subprocess.run(
|
|
128
|
-
[java_path, "-version"], capture_output=True, text=True
|
|
129
|
-
)
|
|
130
|
-
|
|
131
|
-
# Print the version information (Java version is printed to stderr)
|
|
132
|
-
if result.returncode == 0:
|
|
133
|
-
version_output = result.stderr
|
|
134
|
-
match = re.search(r'version "(\d+\.\d+)', version_output)
|
|
135
|
-
|
|
136
|
-
if match:
|
|
137
|
-
crr_version = float(match.group(1))
|
|
138
|
-
if crr_version < min_version:
|
|
139
|
-
console.error(
|
|
140
|
-
f"Java {crr_version} is installed, but version {min_version} is required." # noqa 501
|
|
141
|
-
)
|
|
142
|
-
else:
|
|
143
|
-
console.error("Could not extract Java version.")
|
|
144
|
-
else:
|
|
145
|
-
console.error("Error occurred while checking the version.")
|
|
146
|
-
else:
|
|
147
|
-
console.error("Java not found.")
|
|
148
|
-
except Exception as e:
|
|
149
|
-
console.error(e.__str__())
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
if __name__ == "__main__":
|
|
153
|
-
pass
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|