temporal-normalization-spacy 2.0.2__tar.gz → 2.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (21) hide show
  1. {temporal_normalization_spacy-2.0.2 → temporal_normalization_spacy-2.1.0}/MANIFEST.in +1 -1
  2. {temporal_normalization_spacy-2.0.2/temporal_normalization_spacy.egg-info → temporal_normalization_spacy-2.1.0}/PKG-INFO +17 -5
  3. {temporal_normalization_spacy-2.0.2 → temporal_normalization_spacy-2.1.0}/README.md +14 -3
  4. {temporal_normalization_spacy-2.0.2 → temporal_normalization_spacy-2.1.0}/setup.py +2 -2
  5. {temporal_normalization_spacy-2.0.2 → temporal_normalization_spacy-2.1.0}/temporal_normalization/commons/temporal_models.py +39 -3
  6. {temporal_normalization_spacy-2.0.2 → temporal_normalization_spacy-2.1.0}/temporal_normalization/index.py +33 -10
  7. temporal_normalization_spacy-2.0.2/temporal_normalization/libs/temporal-normalization-2.0.jar → temporal_normalization_spacy-2.1.0/temporal_normalization/libs/temporal-normalization-2.1.0.jar +0 -0
  8. temporal_normalization_spacy-2.1.0/temporal_normalization/process/java_process.py +129 -0
  9. {temporal_normalization_spacy-2.0.2 → temporal_normalization_spacy-2.1.0/temporal_normalization_spacy.egg-info}/PKG-INFO +17 -5
  10. {temporal_normalization_spacy-2.0.2 → temporal_normalization_spacy-2.1.0}/temporal_normalization_spacy.egg-info/SOURCES.txt +1 -1
  11. temporal_normalization_spacy-2.0.2/temporal_normalization/process/java_process.py +0 -153
  12. {temporal_normalization_spacy-2.0.2 → temporal_normalization_spacy-2.1.0}/LICENSE +0 -0
  13. {temporal_normalization_spacy-2.0.2 → temporal_normalization_spacy-2.1.0}/setup.cfg +0 -0
  14. {temporal_normalization_spacy-2.0.2 → temporal_normalization_spacy-2.1.0}/temporal_normalization/__init__.py +0 -0
  15. {temporal_normalization_spacy-2.0.2 → temporal_normalization_spacy-2.1.0}/temporal_normalization/commons/__init__.py +0 -0
  16. {temporal_normalization_spacy-2.0.2 → temporal_normalization_spacy-2.1.0}/temporal_normalization/commons/print_utils.py +0 -0
  17. {temporal_normalization_spacy-2.0.2 → temporal_normalization_spacy-2.1.0}/temporal_normalization/commons/temporal_types.py +0 -0
  18. {temporal_normalization_spacy-2.0.2 → temporal_normalization_spacy-2.1.0}/temporal_normalization/process/__init__.py +0 -0
  19. {temporal_normalization_spacy-2.0.2 → temporal_normalization_spacy-2.1.0}/temporal_normalization_spacy.egg-info/dependency_links.txt +0 -0
  20. {temporal_normalization_spacy-2.0.2 → temporal_normalization_spacy-2.1.0}/temporal_normalization_spacy.egg-info/requires.txt +0 -0
  21. {temporal_normalization_spacy-2.0.2 → temporal_normalization_spacy-2.1.0}/temporal_normalization_spacy.egg-info/top_level.txt +0 -0
@@ -1,5 +1,5 @@
1
1
  include README.md
2
- include temporal_normalization/libs/temporal-normalization-2.0.jar
2
+ include temporal_normalization/libs/temporal-normalization-2.1.0.jar
3
3
 
4
4
  recursive-include temporal_normalizer/commons *
5
5
  recursive-include temporal_normalizer/libs *
@@ -1,6 +1,6 @@
1
- Metadata-Version: 2.2
1
+ Metadata-Version: 2.4
2
2
  Name: temporal_normalization_spacy
3
- Version: 2.0.2
3
+ Version: 2.1.0
4
4
  Summary: A spaCy plugin for identifying and parsing historical data in Romanian texts
5
5
  Home-page: https://github.com/iliedorobat/timespan-normalization-spacy
6
6
  Author: Ilie Cristian Dorobat
@@ -18,6 +18,7 @@ Dynamic: classifier
18
18
  Dynamic: description
19
19
  Dynamic: description-content-type
20
20
  Dynamic: home-page
21
+ Dynamic: license-file
21
22
  Dynamic: requires-dist
22
23
  Dynamic: requires-python
23
24
  Dynamic: summary
@@ -167,8 +168,17 @@ for entity in doc.ents:
167
168
  still be loaded on first run,** and this process may take a few seconds/tens of seconds.
168
169
 
169
170
  ### Importing Modules & Defining Constants
171
+
170
172
  ```python
171
- from temporal_normalization import console, TemporalExpression, start_process
173
+ from pathlib import Path
174
+
175
+ from temporal_normalization import (
176
+ close_conn,
177
+ console,
178
+ extract_temporal_expressions,
179
+ start_conn,
180
+ TemporalExpression,
181
+ )
172
182
 
173
183
  LANG = "ro"
174
184
  TEXT_RO = (
@@ -183,8 +193,10 @@ TEXT_RO = (
183
193
  # Display a warning if the language of the text is not Romanian.
184
194
  console.lang_warning(TEXT_RO, target_lang=LANG)
185
195
 
186
- expressions: list[TemporalExpression] = []
187
- start_process(TEXT_RO, expressions)
196
+ root_path = str(Path(__file__).resolve().parent.parent.parent)
197
+ java_process, gateway = start_conn(root_path)
198
+ expressions: list[TemporalExpression] = extract_temporal_expressions(gateway, TEXT_RO)
199
+ close_conn(java_process, gateway)
188
200
  ```
189
201
 
190
202
  ### Accessing the Parsed Temporal Expressions
@@ -143,8 +143,17 @@ for entity in doc.ents:
143
143
  still be loaded on first run,** and this process may take a few seconds/tens of seconds.
144
144
 
145
145
  ### Importing Modules & Defining Constants
146
+
146
147
  ```python
147
- from temporal_normalization import console, TemporalExpression, start_process
148
+ from pathlib import Path
149
+
150
+ from temporal_normalization import (
151
+ close_conn,
152
+ console,
153
+ extract_temporal_expressions,
154
+ start_conn,
155
+ TemporalExpression,
156
+ )
148
157
 
149
158
  LANG = "ro"
150
159
  TEXT_RO = (
@@ -159,8 +168,10 @@ TEXT_RO = (
159
168
  # Display a warning if the language of the text is not Romanian.
160
169
  console.lang_warning(TEXT_RO, target_lang=LANG)
161
170
 
162
- expressions: list[TemporalExpression] = []
163
- start_process(TEXT_RO, expressions)
171
+ root_path = str(Path(__file__).resolve().parent.parent.parent)
172
+ java_process, gateway = start_conn(root_path)
173
+ expressions: list[TemporalExpression] = extract_temporal_expressions(gateway, TEXT_RO)
174
+ close_conn(java_process, gateway)
164
175
  ```
165
176
 
166
177
  ### Accessing the Parsed Temporal Expressions
@@ -2,7 +2,7 @@ from setuptools import setup, find_packages
2
2
 
3
3
  setup(
4
4
  name="temporal_normalization_spacy",
5
- version="2.0.2",
5
+ version="2.1.0",
6
6
  author="Ilie Cristian Dorobat",
7
7
  description="A spaCy plugin for identifying and parsing historical data "
8
8
  "in Romanian texts",
@@ -12,7 +12,7 @@ setup(
12
12
  packages=find_packages(),
13
13
  include_package_data=True,
14
14
  package_data={
15
- "temporal_normalization.libs": ["temporal-normalization-2.0.jar"],
15
+ "temporal_normalization.libs": ["temporal-normalization-2.1.0.jar"],
16
16
  },
17
17
  install_requires=["spacy>=3.0", "py4j", "langdetect"],
18
18
  classifiers=[
@@ -1,6 +1,6 @@
1
1
  import json
2
2
 
3
- from py4j.java_gateway import JavaObject
3
+ from py4j.java_gateway import JavaObject, JavaGateway
4
4
 
5
5
  from temporal_normalization.commons.temporal_types import TemporalType
6
6
 
@@ -14,6 +14,7 @@ class TemporalExpression:
14
14
  is_valid (bool): A flag that specifies whether the text processed
15
15
  through timespan-normalization library is a temporal expression.
16
16
  input_value (str or None): The original temporal expression before processing.
17
+ prepared_value (str or None): The temporal expression after processing.
17
18
  time_series (list[TimeSeries]): The list of normalized temporal expressions.
18
19
  matches (list[str]): A unique list of matched values found in the normalized
19
20
  entities.
@@ -23,10 +24,12 @@ class TemporalExpression:
23
24
  serialize = java_object.serialize()
24
25
  json_obj = json.loads(serialize)
25
26
 
27
+ # fmt: off
26
28
  self.is_valid = TemporalExpression.is_valid_json(json_obj)
27
29
  self.input_value: str | None = json_obj["inputValue"] if self.is_valid else None
30
+ self.prepared_value: str | None = json_obj["preparedValue"] if self.is_valid else None
28
31
  self.time_series: list[TimeSeries] = [
29
- TimeSeries(item) for item in json_obj["timeSeries"]
32
+ TimeSeries(item, self.input_value, self.prepared_value) for item in json_obj["timeSeries"]
30
33
  ] if self.is_valid else []
31
34
  self.matches: list[str] = list(
32
35
  set(
@@ -37,6 +40,7 @@ class TemporalExpression:
37
40
  ]
38
41
  )
39
42
  )
43
+ # fmt: on
40
44
 
41
45
  def __str__(self):
42
46
  if self.input_value is None:
@@ -52,12 +56,40 @@ class TemporalExpression:
52
56
  return "inputValue" in json_obj and "timeSeries" in json_obj
53
57
 
54
58
 
59
+ def extract_temporal_expressions(
60
+ gateway: JavaGateway, text: str
61
+ ) -> list[TemporalExpression]:
62
+ """
63
+ Extracts valid temporal expressions from the given text using the Java temporal
64
+ normalization gateway.
65
+
66
+ Args:
67
+ gateway (JavaGateway): Active Py4J gateway connected to the Java temporal
68
+ normalization process.
69
+ text (str): Input text from which to extract temporal expressions.
70
+
71
+ Returns:
72
+ list[TemporalExpression]: A list containing valid temporal expressions.
73
+ """
74
+
75
+ expressions: list[TemporalExpression] = []
76
+ java_object = gateway.jvm.ro.webdata.normalization.timespan.ro.TimeExpression(text)
77
+ temporal_expression = TemporalExpression(java_object)
78
+
79
+ if temporal_expression.is_valid:
80
+ expressions.append(temporal_expression)
81
+
82
+ return expressions
83
+
84
+
55
85
  class TimeSeries:
56
86
  """
57
87
  A data structure representing a temporal expression that has been normalized
58
88
  into a list of periods and temporal edges.
59
89
 
60
90
  Attributes:
91
+ input_value (str or None): The original temporal expression before processing.
92
+ prepared_value (str or None): The temporal expression after processing.
61
93
  edges (list[EdgeModel]): A list of temporal intervals represented as edges.
62
94
  periods (list[DBpediaModel]): A list of normalized DBpedia entities
63
95
  extracted from the expression.
@@ -65,7 +97,9 @@ class TimeSeries:
65
97
  entities.
66
98
  """
67
99
 
68
- def __init__(self, data: dict):
100
+ def __init__(self, data: dict, input_value: str, prepared_value: str):
101
+ self.input_value = input_value
102
+ self.prepared_value = prepared_value
69
103
  self.edges: EdgeModel = EdgeModel(data["edges"]) if "edges" in data else None
70
104
  self.periods: list[DBpediaModel] = (
71
105
  [DBpediaModel(item) for item in data["periods"]]
@@ -80,10 +114,12 @@ class TimeSeries:
80
114
  return f"TimeSeries(edges={self.edges}, periods={self.periods})"
81
115
 
82
116
  def serialize(self, indent: str = ""):
117
+ # fmt: off
83
118
  return (
84
119
  f"{indent}Edges: {self.edges}\n"
85
120
  f"{indent}Periods: {self.periods}"
86
121
  )
122
+ # fmt: on
87
123
 
88
124
 
89
125
  class DBpediaModel:
@@ -1,17 +1,26 @@
1
1
  import re
2
+ import subprocess
3
+ from pathlib import Path
2
4
 
5
+ from py4j.java_gateway import JavaGateway
3
6
  from spacy import Language
4
7
  from spacy.tokens import Doc, Span
8
+ from spacy.tokens._retokenize import Retokenizer
5
9
  from spacy.util import filter_spans
6
10
 
7
11
  from temporal_normalization import TimeSeries
8
- from temporal_normalization.commons.temporal_models import TemporalExpression
9
- from temporal_normalization.process.java_process import start_process
12
+ from temporal_normalization.commons.temporal_models import (
13
+ extract_temporal_expressions,
14
+ TemporalExpression,
15
+ )
16
+ from temporal_normalization.process.java_process import start_conn, close_conn
10
17
 
11
18
  try:
19
+
12
20
  @Language.factory("temporal_normalization")
13
21
  def create_normalized_component(nlp, name):
14
22
  return TemporalNormalization(nlp, name)
23
+
15
24
  except AttributeError:
16
25
  # spaCy 2.x
17
26
  pass
@@ -21,7 +30,7 @@ class TemporalNormalization:
21
30
  """
22
31
  spaCy pipeline component for identifying and annotating temporal expressions in text.
23
32
 
24
- This component calls the ``start_process`` method to extract temporal expressions, then
33
+ This component calls the ``start_conn`` method to extract temporal expressions, then
25
34
  aligns the matches with spaCy tokens using retokenization and sets a custom attribute
26
35
  containing associated time series metadata.
27
36
  """
@@ -40,6 +49,11 @@ class TemporalNormalization:
40
49
  Span.set_extension(TemporalNormalization.__FIELD, default=None, force=True)
41
50
  self.nlp = nlp
42
51
 
52
+ root_path = str(Path(__file__).resolve().parent.parent)
53
+ java_process, gateway = start_conn(root_path)
54
+ self.java_process: subprocess.Popen = java_process
55
+ self.gateway: JavaGateway = gateway
56
+
43
57
  def __call__(self, doc: Doc) -> Doc:
44
58
  """
45
59
  Apply the component to a spaCy Doc object.
@@ -54,14 +68,17 @@ class TemporalNormalization:
54
68
  Doc: The modified Doc object with temporal expressions processed.
55
69
  """
56
70
 
57
- expressions: list[TemporalExpression] = []
58
- start_process(doc.text, expressions)
71
+ expressions: list[TemporalExpression] = extract_temporal_expressions(
72
+ self.gateway, doc.text
73
+ )
59
74
  str_matches: list[str] = _prepare_str_patterns(expressions)
60
-
61
75
  _retokenize(doc, str_matches, expressions)
62
76
 
63
77
  return doc
64
78
 
79
+ def __del__(self):
80
+ close_conn(self.java_process, self.gateway)
81
+
65
82
 
66
83
  def _prepare_str_patterns(expressions: list[TemporalExpression]) -> list[str]:
67
84
  """
@@ -101,7 +118,8 @@ def _retokenize(
101
118
  each with time series metadata.
102
119
  """
103
120
 
104
- regex_matches: list[str] = [rf"{item}" for item in str_matches]
121
+ # TODO: WIP
122
+ regex_matches: list[str] = [rf"{re.escape(item)}" for item in str_matches]
105
123
  pattern = f"({'|'.join(regex_matches)})"
106
124
  matches = (
107
125
  list(re.finditer(pattern, doc.text, re.IGNORECASE))
@@ -126,6 +144,7 @@ def _retokenize(
126
144
  if token.idx + len(token.text) == end_char:
127
145
  end_token = token.i
128
146
 
147
+ # fmt: off
129
148
  if start_token is not None and end_token is not None:
130
149
  # use exact token boundaries to create a custom `Span` for well-defined
131
150
  # time expressions with known character offsets.
@@ -144,6 +163,8 @@ def _retokenize(
144
163
  matched_ts = [ts for ts in time_series if _is_substring(entity.text, ts.matches)]
145
164
  _retokenize_entity(doc, matched_ts, entity, True, retokenized_entities, retokenizer)
146
165
 
166
+ # fmt: on
167
+
147
168
 
148
169
  def _retokenize_entity(
149
170
  doc: Doc,
@@ -151,7 +172,7 @@ def _retokenize_entity(
151
172
  entity: Span,
152
173
  existed_entity: bool,
153
174
  retokenized_entities: list[Span],
154
- retokenizer: Doc.retokenize,
175
+ retokenizer: Retokenizer,
155
176
  ) -> None:
156
177
  """
157
178
  Retokenizes and enriches a temporal entity span with matched time series data.
@@ -173,6 +194,8 @@ def _retokenize_entity(
173
194
  _update_doc_ents(doc, entity)
174
195
  _merge_entity(doc, entity, retokenized_entities, retokenizer)
175
196
 
197
+ return None
198
+
176
199
 
177
200
  def _assign_time_series(
178
201
  matched_ts: list[TimeSeries], entity: Span, existed_entity: bool
@@ -214,7 +237,7 @@ def _merge_entity(
214
237
  doc: Doc,
215
238
  entity: Span,
216
239
  retokenized_entities: list[Span],
217
- retokenizer: Doc.retokenize,
240
+ retokenizer: Retokenizer,
218
241
  ) -> None:
219
242
  """
220
243
  Merges a custom entity span into the spaCy Doc if it is not already part of
@@ -223,7 +246,7 @@ def _merge_entity(
223
246
  Args:
224
247
  entity (Span): The named entity to enrich.
225
248
  retokenized_entities (list): Accumulator for entities that require retokenization.
226
- retokenizer (Doc.retokenize): The spaCy retokenizer context.
249
+ retokenizer (Retokenizer): The spaCy retokenizer context.
227
250
  """
228
251
 
229
252
  if entity not in doc.ents:
@@ -0,0 +1,129 @@
1
+ import re
2
+ import shutil
3
+ import subprocess
4
+
5
+ from py4j.java_gateway import JavaGateway
6
+ from py4j.protocol import Py4JNetworkError
7
+
8
+ from temporal_normalization.commons.print_utils import console
9
+
10
+
11
+ def start_conn(root_path: str) -> tuple[subprocess.Popen, JavaGateway]:
12
+ """
13
+ Starts the Java temporal normalization process and establishes a Py4J gateway connection.
14
+
15
+ Args:
16
+ root_path (str): The root directory of the project.
17
+
18
+ Returns:
19
+ tuple[subprocess.Popen, JavaGateway]:
20
+ - The subprocess.Popen object representing the running Java process.
21
+ - The JavaGateway object representing the active Py4J connection.
22
+
23
+ Note:
24
+ - Requires Java 11 or higher to be installed and accessible in the system PATH.
25
+ - Requires `temporal-normalization-2.1.0.jar` to be present in the `libs` directory.
26
+ - The caller is responsible for closing the gateway and terminating the Java process
27
+ after usage to avoid orphaned processes.
28
+ """
29
+
30
+ check_java_version()
31
+
32
+ jar_path = (
33
+ f"{root_path}/temporal_normalization/libs/temporal-normalization-2.1.0.jar"
34
+ )
35
+
36
+ java_process = subprocess.Popen(
37
+ ["java", "-jar", jar_path, "--python"],
38
+ stdout=subprocess.PIPE,
39
+ stderr=subprocess.PIPE,
40
+ text=True,
41
+ )
42
+
43
+ for line in java_process.stdout:
44
+ if "Gateway Server Started." in line:
45
+ print(line.strip())
46
+ break
47
+
48
+ gateway = JavaGateway()
49
+ print("Python connection established.")
50
+
51
+ return java_process, gateway
52
+
53
+
54
+ def close_conn(java_process: subprocess.Popen, gateway: JavaGateway) -> None:
55
+ """
56
+ Closes the active connection between Python and the Java process started via Py4J.
57
+
58
+ This function ensures a proper shutdown sequence:
59
+ 1. Attempts to gracefully shut down the Py4J gateway connection.
60
+ - If the Java process is already closed, a Py4JNetworkError is caught and logged.
61
+ 2. Terminates the underlying Java process.
62
+ 3. Prints status messages for debugging/confirmation.
63
+
64
+ Args:
65
+ java_process (subprocess.Popen): The Java process launched with subprocess.
66
+ gateway (JavaGateway): The active Py4J gateway connection.
67
+
68
+ Notes:
69
+ - Call this function once you have finished all interactions with the Java process.
70
+ - It is safe to call even if the Java process has already exited.
71
+ """
72
+
73
+ try:
74
+ # Proper way to shut down Py4J
75
+ gateway.shutdown()
76
+ print("Python connection closed.")
77
+ except Py4JNetworkError:
78
+ print("Java process already shut down.")
79
+
80
+ # Terminate Java process
81
+ java_process.terminate()
82
+ print("Java server is shutting down...")
83
+
84
+
85
+ def check_java_version() -> None:
86
+ """
87
+ Verifies that Java is installed and meets the minimum required version.
88
+
89
+ This function checks for the presence of the Java executable in the system PATH,
90
+ runs ``java -version``, and ensures that the version is at least 11. If Java is not
91
+ installed or the version is too low, it logs an error using ``console.error``.
92
+
93
+ Raises:
94
+ Logs error messages, but does not raise exceptions directly.
95
+ """
96
+
97
+ min_version = 11
98
+ java_path = shutil.which("java")
99
+
100
+ try:
101
+ if java_path:
102
+ # Run the command to check the Java version
103
+ result = subprocess.run(
104
+ [java_path, "-version"], capture_output=True, text=True
105
+ )
106
+
107
+ # Print the version information (Java version is printed to stderr)
108
+ if result.returncode == 0:
109
+ version_output = result.stderr
110
+ match = re.search(r'version "(\d+\.\d+)', version_output)
111
+
112
+ if match:
113
+ crr_version = float(match.group(1))
114
+ if crr_version < min_version:
115
+ console.error(
116
+ f"Java {crr_version} is installed, but version {min_version} is required." # noqa 501
117
+ )
118
+ else:
119
+ console.error("Could not extract Java version.")
120
+ else:
121
+ console.error("Error occurred while checking the version.")
122
+ else:
123
+ console.error("Java not found.")
124
+ except Exception as e:
125
+ console.error(e.__str__())
126
+
127
+
128
+ if __name__ == "__main__":
129
+ pass
@@ -1,6 +1,6 @@
1
- Metadata-Version: 2.2
1
+ Metadata-Version: 2.4
2
2
  Name: temporal_normalization_spacy
3
- Version: 2.0.2
3
+ Version: 2.1.0
4
4
  Summary: A spaCy plugin for identifying and parsing historical data in Romanian texts
5
5
  Home-page: https://github.com/iliedorobat/timespan-normalization-spacy
6
6
  Author: Ilie Cristian Dorobat
@@ -18,6 +18,7 @@ Dynamic: classifier
18
18
  Dynamic: description
19
19
  Dynamic: description-content-type
20
20
  Dynamic: home-page
21
+ Dynamic: license-file
21
22
  Dynamic: requires-dist
22
23
  Dynamic: requires-python
23
24
  Dynamic: summary
@@ -167,8 +168,17 @@ for entity in doc.ents:
167
168
  still be loaded on first run,** and this process may take a few seconds/tens of seconds.
168
169
 
169
170
  ### Importing Modules & Defining Constants
171
+
170
172
  ```python
171
- from temporal_normalization import console, TemporalExpression, start_process
173
+ from pathlib import Path
174
+
175
+ from temporal_normalization import (
176
+ close_conn,
177
+ console,
178
+ extract_temporal_expressions,
179
+ start_conn,
180
+ TemporalExpression,
181
+ )
172
182
 
173
183
  LANG = "ro"
174
184
  TEXT_RO = (
@@ -183,8 +193,10 @@ TEXT_RO = (
183
193
  # Display a warning if the language of the text is not Romanian.
184
194
  console.lang_warning(TEXT_RO, target_lang=LANG)
185
195
 
186
- expressions: list[TemporalExpression] = []
187
- start_process(TEXT_RO, expressions)
196
+ root_path = str(Path(__file__).resolve().parent.parent.parent)
197
+ java_process, gateway = start_conn(root_path)
198
+ expressions: list[TemporalExpression] = extract_temporal_expressions(gateway, TEXT_RO)
199
+ close_conn(java_process, gateway)
188
200
  ```
189
201
 
190
202
  ### Accessing the Parsed Temporal Expressions
@@ -8,7 +8,7 @@ temporal_normalization/commons/__init__.py
8
8
  temporal_normalization/commons/print_utils.py
9
9
  temporal_normalization/commons/temporal_models.py
10
10
  temporal_normalization/commons/temporal_types.py
11
- temporal_normalization/libs/temporal-normalization-2.0.jar
11
+ temporal_normalization/libs/temporal-normalization-2.1.0.jar
12
12
  temporal_normalization/process/__init__.py
13
13
  temporal_normalization/process/java_process.py
14
14
  temporal_normalization_spacy.egg-info/PKG-INFO
@@ -1,153 +0,0 @@
1
- import os
2
- import re
3
- import shutil
4
- import subprocess
5
-
6
- from py4j.java_gateway import JavaGateway
7
- from py4j.protocol import Py4JNetworkError
8
-
9
- from temporal_normalization.commons.print_utils import console
10
- from temporal_normalization.commons.temporal_models import TemporalExpression
11
-
12
-
13
- def start_process(text: str, expressions: list[TemporalExpression]):
14
- """
15
- Launches the Java-based temporal normalization process and populates a list
16
- of temporal expressions extracted from the input text.
17
-
18
- This function starts a Java subprocess that hosts the temporal-normalization
19
- server via Py4J. Once the gateway is connected, it sends the input text for
20
- processing, retrieves the temporal expressions, and then shuts down both
21
- the Python and Java sides of the connection.
22
-
23
- Args:
24
- text (str): The input text to be analyzed for temporal expressions.
25
- expressions (list[TemporalExpression]): A list to be populated with
26
- extracted temporal expressions.
27
-
28
- Example:
29
- >>> expressions: list[TemporalExpression] = []
30
- >>> start_process("Sec al II-lea a.ch. a fost o perioadă de mari schimbări.", expressions)
31
- >>> # ... use the list of normalized temporal expressions.
32
-
33
- Note:
34
- Requires `temporal-normalization-2.0.jar` to be present in the `libs` directory.
35
- Also requires Java 11 or higher to be installed and accessible in the system PATH.
36
- """
37
-
38
- check_java_version()
39
-
40
- jar_path = os.path.join(
41
- os.path.dirname(__file__), "../libs/temporal-normalization-2.0.jar"
42
- )
43
-
44
- java_process = subprocess.Popen(
45
- ["java", "-jar", jar_path, "--python"],
46
- stdout=subprocess.PIPE,
47
- stderr=subprocess.PIPE,
48
- text=True,
49
- )
50
-
51
- for line in java_process.stdout:
52
- if "Gateway Server Started..." in line:
53
- print(line.strip())
54
- break
55
-
56
- gateway = gateway_conn(text, expressions)
57
-
58
- try:
59
- # Proper way to shut down Py4J
60
- gateway.shutdown()
61
- print("Python connection closed.")
62
- except Py4JNetworkError:
63
- print("Java process already shut down.")
64
-
65
- # Terminate Java process
66
- java_process.terminate()
67
- print("Java server is shutting down...")
68
-
69
-
70
- def gateway_conn(text: str, expressions: list[TemporalExpression]) -> JavaGateway:
71
- """
72
- Establishes a connection to the Java Py4J gateway and initializes the
73
- temporal expression extraction.
74
-
75
- It creates an instance of the Java class ``TimeExpression``, wraps it in
76
- a Python ``TemporalExpression``, and appends it to the given list if valid.
77
-
78
- Args:
79
- text (str): The input text to analyze.
80
- expressions (list[TemporalExpression]): A list to store extracted expressions.
81
-
82
- Returns:
83
- JavaGateway: The Py4J JavaGateway object used to manage the connection.
84
-
85
- Example:
86
- >>> from py4j.java_gateway import JavaGateway
87
- >>> expressions: list[TemporalExpression] = []
88
- >>> gateway = gateway_conn("Sec al II-lea a.ch. a fost o perioadă de mari schimbări.", expressions)
89
- >>> # ... use the list of normalized temporal expressions.
90
- >>> gateway.shutdown()
91
-
92
- Note:
93
- Assumes that a Py4J-compatible Java process is already running and exposing
94
- the `ro.webdata.normalization.timespan.ro.TimeExpression` class.
95
- """
96
-
97
- gateway = JavaGateway()
98
- print("Python connection established.")
99
-
100
- java_object = gateway.jvm.ro.webdata.normalization.timespan.ro.TimeExpression(text)
101
- time_expression = TemporalExpression(java_object)
102
-
103
- if time_expression.is_valid:
104
- expressions.append(time_expression)
105
-
106
- return gateway
107
-
108
-
109
- def check_java_version():
110
- """
111
- Verifies that Java is installed and meets the minimum required version.
112
-
113
- This function checks for the presence of the Java executable in the system PATH,
114
- runs ``java -version``, and ensures that the version is at least 11. If Java is not
115
- installed or the version is too low, it logs an error using ``console.error``.
116
-
117
- Raises:
118
- Logs error messages, but does not raise exceptions directly.
119
- """
120
-
121
- min_version = 11
122
- java_path = shutil.which("java")
123
-
124
- try:
125
- if java_path:
126
- # Run the command to check the Java version
127
- result = subprocess.run(
128
- [java_path, "-version"], capture_output=True, text=True
129
- )
130
-
131
- # Print the version information (Java version is printed to stderr)
132
- if result.returncode == 0:
133
- version_output = result.stderr
134
- match = re.search(r'version "(\d+\.\d+)', version_output)
135
-
136
- if match:
137
- crr_version = float(match.group(1))
138
- if crr_version < min_version:
139
- console.error(
140
- f"Java {crr_version} is installed, but version {min_version} is required." # noqa 501
141
- )
142
- else:
143
- console.error("Could not extract Java version.")
144
- else:
145
- console.error("Error occurred while checking the version.")
146
- else:
147
- console.error("Java not found.")
148
- except Exception as e:
149
- console.error(e.__str__())
150
-
151
-
152
- if __name__ == "__main__":
153
- pass