temporal-normalization-spacy 2.1.0__tar.gz → 2.2.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (23) hide show
  1. {temporal_normalization_spacy-2.1.0/temporal_normalization_spacy.egg-info → temporal_normalization_spacy-2.2.1}/PKG-INFO +16 -11
  2. {temporal_normalization_spacy-2.1.0 → temporal_normalization_spacy-2.2.1}/README.md +4 -5
  3. temporal_normalization_spacy-2.2.1/setup.py +38 -0
  4. {temporal_normalization_spacy-2.1.0 → temporal_normalization_spacy-2.2.1}/temporal_normalization/commons/print_utils.py +7 -4
  5. {temporal_normalization_spacy-2.1.0 → temporal_normalization_spacy-2.2.1}/temporal_normalization/index.py +35 -6
  6. {temporal_normalization_spacy-2.1.0 → temporal_normalization_spacy-2.2.1}/temporal_normalization/process/java_process.py +56 -9
  7. {temporal_normalization_spacy-2.1.0 → temporal_normalization_spacy-2.2.1/temporal_normalization_spacy.egg-info}/PKG-INFO +16 -11
  8. {temporal_normalization_spacy-2.1.0 → temporal_normalization_spacy-2.2.1}/temporal_normalization_spacy.egg-info/SOURCES.txt +1 -0
  9. temporal_normalization_spacy-2.2.1/temporal_normalization_spacy.egg-info/entry_points.txt +2 -0
  10. temporal_normalization_spacy-2.2.1/temporal_normalization_spacy.egg-info/requires.txt +3 -0
  11. temporal_normalization_spacy-2.1.0/setup.py +0 -24
  12. temporal_normalization_spacy-2.1.0/temporal_normalization_spacy.egg-info/requires.txt +0 -3
  13. {temporal_normalization_spacy-2.1.0 → temporal_normalization_spacy-2.2.1}/LICENSE +0 -0
  14. {temporal_normalization_spacy-2.1.0 → temporal_normalization_spacy-2.2.1}/MANIFEST.in +0 -0
  15. {temporal_normalization_spacy-2.1.0 → temporal_normalization_spacy-2.2.1}/setup.cfg +0 -0
  16. {temporal_normalization_spacy-2.1.0 → temporal_normalization_spacy-2.2.1}/temporal_normalization/__init__.py +0 -0
  17. {temporal_normalization_spacy-2.1.0 → temporal_normalization_spacy-2.2.1}/temporal_normalization/commons/__init__.py +0 -0
  18. {temporal_normalization_spacy-2.1.0 → temporal_normalization_spacy-2.2.1}/temporal_normalization/commons/temporal_models.py +0 -0
  19. {temporal_normalization_spacy-2.1.0 → temporal_normalization_spacy-2.2.1}/temporal_normalization/commons/temporal_types.py +0 -0
  20. {temporal_normalization_spacy-2.1.0 → temporal_normalization_spacy-2.2.1}/temporal_normalization/libs/temporal-normalization-2.1.0.jar +0 -0
  21. {temporal_normalization_spacy-2.1.0 → temporal_normalization_spacy-2.2.1}/temporal_normalization/process/__init__.py +0 -0
  22. {temporal_normalization_spacy-2.1.0 → temporal_normalization_spacy-2.2.1}/temporal_normalization_spacy.egg-info/dependency_links.txt +0 -0
  23. {temporal_normalization_spacy-2.1.0 → temporal_normalization_spacy-2.2.1}/temporal_normalization_spacy.egg-info/top_level.txt +0 -0
@@ -1,29 +1,35 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: temporal_normalization_spacy
3
- Version: 2.1.0
4
- Summary: A spaCy plugin for identifying and parsing historical data in Romanian texts
3
+ Version: 2.2.1
4
+ Summary: A spaCy plugin for temporal normalization and extraction of historical dates in Romanian narrative texts.
5
5
  Home-page: https://github.com/iliedorobat/timespan-normalization-spacy
6
6
  Author: Ilie Cristian Dorobat
7
+ Keywords: spacy,nlp,temporal normalization,timex,historical dates,romanian
7
8
  Classifier: Programming Language :: Python :: 3
9
+ Classifier: Programming Language :: Python :: 3.9
10
+ Classifier: Programming Language :: Python :: 3.10
11
+ Classifier: Programming Language :: Python :: 3.11
12
+ Classifier: Programming Language :: Python :: 3.12
8
13
  Classifier: License :: OSI Approved :: MIT License
9
14
  Classifier: Operating System :: OS Independent
10
- Requires-Python: >=3.6
15
+ Requires-Python: >=3.9,<3.13
11
16
  Description-Content-Type: text/markdown
12
17
  License-File: LICENSE
13
- Requires-Dist: spacy>=3.0
14
- Requires-Dist: py4j
15
- Requires-Dist: langdetect
18
+ Requires-Dist: spacy<4.0.0,>=3.8.7
19
+ Requires-Dist: py4j>=0.10.9.9
20
+ Requires-Dist: langdetect>=1.0.9
16
21
  Dynamic: author
17
22
  Dynamic: classifier
18
23
  Dynamic: description
19
24
  Dynamic: description-content-type
20
25
  Dynamic: home-page
26
+ Dynamic: keywords
21
27
  Dynamic: license-file
22
28
  Dynamic: requires-dist
23
29
  Dynamic: requires-python
24
30
  Dynamic: summary
25
31
 
26
- # Temporal Expressions Normalization for spaCy (TeNs)
32
+ # Temporal Expression Normalization for spaCy (TeNs)
27
33
 
28
34
  <b>Temporal Expressions Normalization spaCy (TeNs)</b> is a powerful pipeline component for spaCy that seamlessly
29
35
  identifies and parses date entities in text. It leverages the <b>[Temporal Expressions Normalization Framework](
@@ -75,8 +81,8 @@ quality and reliability of date-related information extracted from text.
75
81
  To integrate TeNs into spaCy pipelines you need the following:
76
82
 
77
83
  ### Prerequisites
78
- - Python 3.x
79
- - JRE 11+
84
+ - Python 3.9-3.12
85
+ - JRE 11-17
80
86
  - spaCy 3.x
81
87
  - py4j 0.10.9.9
82
88
  - langdetect 1.0.9
@@ -101,7 +107,6 @@ import subprocess
101
107
  import spacy
102
108
 
103
109
  from temporal_normalization.commons.print_utils import console
104
- from temporal_normalization.index import create_normalized_component, TemporalNormalization # noqa: F401
105
110
 
106
111
  LANG = "ro"
107
112
  MODEL = "ro_core_news_sm"
@@ -121,7 +126,7 @@ try:
121
126
  except OSError:
122
127
  console.warning(f'Started downloading {MODEL}...')
123
128
  # Download the Romanian model if it wasn't already downloaded
124
- subprocess.run(["python", "-m", "spacy", "download", MODEL])
129
+ subprocess.run(["python3", "-m", "spacy", "download", MODEL])
125
130
  # Load the spaCy model
126
131
  nlp = spacy.load(MODEL)
127
132
 
@@ -1,4 +1,4 @@
1
- # Temporal Expressions Normalization for spaCy (TeNs)
1
+ # Temporal Expression Normalization for spaCy (TeNs)
2
2
 
3
3
  <b>Temporal Expressions Normalization spaCy (TeNs)</b> is a powerful pipeline component for spaCy that seamlessly
4
4
  identifies and parses date entities in text. It leverages the <b>[Temporal Expressions Normalization Framework](
@@ -50,8 +50,8 @@ quality and reliability of date-related information extracted from text.
50
50
  To integrate TeNs into spaCy pipelines you need the following:
51
51
 
52
52
  ### Prerequisites
53
- - Python 3.x
54
- - JRE 11+
53
+ - Python 3.9-3.12
54
+ - JRE 11-17
55
55
  - spaCy 3.x
56
56
  - py4j 0.10.9.9
57
57
  - langdetect 1.0.9
@@ -76,7 +76,6 @@ import subprocess
76
76
  import spacy
77
77
 
78
78
  from temporal_normalization.commons.print_utils import console
79
- from temporal_normalization.index import create_normalized_component, TemporalNormalization # noqa: F401
80
79
 
81
80
  LANG = "ro"
82
81
  MODEL = "ro_core_news_sm"
@@ -96,7 +95,7 @@ try:
96
95
  except OSError:
97
96
  console.warning(f'Started downloading {MODEL}...')
98
97
  # Download the Romanian model if it wasn't already downloaded
99
- subprocess.run(["python", "-m", "spacy", "download", MODEL])
98
+ subprocess.run(["python3", "-m", "spacy", "download", MODEL])
100
99
  # Load the spaCy model
101
100
  nlp = spacy.load(MODEL)
102
101
 
@@ -0,0 +1,38 @@
1
+ from setuptools import setup, find_packages
2
+
3
+ setup(
4
+ name="temporal_normalization_spacy",
5
+ version="2.2.1",
6
+ author="Ilie Cristian Dorobat",
7
+ description="A spaCy plugin for temporal normalization and extraction of "
8
+ "historical dates in Romanian narrative texts.",
9
+ keywords="spacy, nlp, temporal normalization, timex, historical dates, romanian",
10
+ long_description=open("README.md").read(),
11
+ long_description_content_type="text/markdown",
12
+ url="https://github.com/iliedorobat/timespan-normalization-spacy",
13
+ packages=find_packages(),
14
+ include_package_data=True,
15
+ entry_points={
16
+ "spacy_factories": [
17
+ "temporal_normalization=temporal_normalization_spacy.factory:create_component",
18
+ ],
19
+ },
20
+ package_data={
21
+ "temporal_normalization.libs": ["temporal-normalization-2.1.0.jar"],
22
+ },
23
+ install_requires=[
24
+ "spacy>=3.8.7,<4.0.0",
25
+ "py4j>=0.10.9.9",
26
+ "langdetect>=1.0.9"
27
+ ],
28
+ classifiers=[
29
+ "Programming Language :: Python :: 3",
30
+ "Programming Language :: Python :: 3.9",
31
+ "Programming Language :: Python :: 3.10",
32
+ "Programming Language :: Python :: 3.11",
33
+ "Programming Language :: Python :: 3.12",
34
+ "License :: OSI Approved :: MIT License",
35
+ "Operating System :: OS Independent",
36
+ ],
37
+ python_requires=">=3.9,<3.13",
38
+ )
@@ -47,10 +47,13 @@ class console:
47
47
 
48
48
  @staticmethod
49
49
  def lang_warning(query: str, target_lang: str):
50
- if detect(query) != target_lang:
51
- console.warning(
52
- f'Detected language: "{detect(query)}" but required: "{target_lang}"'
53
- )
50
+ try:
51
+ if detect(query) != target_lang:
52
+ console.warning(
53
+ f'Detected language: "{detect(query)}" but required: "{target_lang}"'
54
+ )
55
+ except Exception as e:
56
+ console.warning(f'⚠️ Language could not be detected: {e}')
54
57
 
55
58
  @staticmethod
56
59
  def tokens_table(document):
@@ -1,8 +1,11 @@
1
+ import gc
1
2
  import re
2
3
  import subprocess
4
+ import time
3
5
  from pathlib import Path
4
6
 
5
7
  from py4j.java_gateway import JavaGateway
8
+ from py4j.protocol import Py4JNetworkError
6
9
  from spacy import Language
7
10
  from spacy.tokens import Doc, Span
8
11
  from spacy.tokens._retokenize import Retokenizer
@@ -48,6 +51,7 @@ class TemporalNormalization:
48
51
 
49
52
  Span.set_extension(TemporalNormalization.__FIELD, default=None, force=True)
50
53
  self.nlp = nlp
54
+ self.count = 0
51
55
 
52
56
  root_path = str(Path(__file__).resolve().parent.parent)
53
57
  java_process, gateway = start_conn(root_path)
@@ -68,15 +72,41 @@ class TemporalNormalization:
68
72
  Doc: The modified Doc object with temporal expressions processed.
69
73
  """
70
74
 
71
- expressions: list[TemporalExpression] = extract_temporal_expressions(
72
- self.gateway, doc.text
73
- )
74
- str_matches: list[str] = _prepare_str_patterns(expressions)
75
- _retokenize(doc, str_matches, expressions)
75
+ self.count += 1
76
+
77
+ if self.count % 1000 == 0:
78
+ self.count = 1
79
+
80
+ gc.collect()
81
+ time.sleep(0.01)
82
+
83
+ try:
84
+ expressions: list[TemporalExpression] = extract_temporal_expressions(
85
+ self.gateway, doc.text
86
+ )
87
+ str_matches: list[str] = _prepare_str_patterns(expressions)
88
+ _retokenize(doc, str_matches, expressions)
89
+ except Py4JNetworkError as e:
90
+ print(f"⚠️ Py4J network error: {e}")
91
+ except Exception as e:
92
+ print(f"⚠️ Unexpected error during extract_temporal_expressions: {e}")
76
93
 
77
94
  return doc
78
95
 
79
96
  def __del__(self):
97
+ """
98
+ Clean up resources when the TemporalNormalization component is destroyed.
99
+
100
+ This method is automatically called by Python's garbage collector when the
101
+ `TemporalNormalization` instance is about to be deleted. It ensures that the
102
+ external Java process and the Py4J gateway connection used for temporal
103
+ expression extraction are properly closed.
104
+
105
+ By explicitly terminating the Java subprocess and shutting down the gateway,
106
+ the method prevents resource leaks such as orphaned Java processes or open
107
+ network sockets that might otherwise persist after the Python process ends.
108
+ """
109
+
80
110
  close_conn(self.java_process, self.gateway)
81
111
 
82
112
 
@@ -118,7 +148,6 @@ def _retokenize(
118
148
  each with time series metadata.
119
149
  """
120
150
 
121
- # TODO: WIP
122
151
  regex_matches: list[str] = [rf"{re.escape(item)}" for item in str_matches]
123
152
  pattern = f"({'|'.join(regex_matches)})"
124
153
  matches = (
@@ -1,13 +1,17 @@
1
+ import io
1
2
  import re
2
3
  import shutil
3
4
  import subprocess
5
+ import threading
4
6
 
5
- from py4j.java_gateway import JavaGateway
7
+ from py4j.java_gateway import JavaGateway, GatewayParameters, CallbackServerParameters
6
8
  from py4j.protocol import Py4JNetworkError
7
9
 
8
10
  from temporal_normalization.commons.print_utils import console
9
11
 
10
12
 
13
+ gateway_started = threading.Event()
14
+
11
15
  def start_conn(root_path: str) -> tuple[subprocess.Popen, JavaGateway]:
12
16
  """
13
17
  Starts the Java temporal normalization process and establishes a Py4J gateway connection.
@@ -33,6 +37,11 @@ def start_conn(root_path: str) -> tuple[subprocess.Popen, JavaGateway]:
33
37
  f"{root_path}/temporal_normalization/libs/temporal-normalization-2.1.0.jar"
34
38
  )
35
39
 
40
+ def stdout_callback(line: str):
41
+ if "Gateway Server Started" in line:
42
+ gateway_started.set()
43
+ print(line.strip())
44
+
36
45
  java_process = subprocess.Popen(
37
46
  ["java", "-jar", jar_path, "--python"],
38
47
  stdout=subprocess.PIPE,
@@ -40,12 +49,18 @@ def start_conn(root_path: str) -> tuple[subprocess.Popen, JavaGateway]:
40
49
  text=True,
41
50
  )
42
51
 
43
- for line in java_process.stdout:
44
- if "Gateway Server Started." in line:
45
- print(line.strip())
46
- break
52
+ threading.Thread(target=drain_stream, args=(java_process.stdout, stdout_callback), daemon=True).start()
53
+ threading.Thread(target=drain_stream, args=(java_process.stderr,), daemon=True).start()
54
+
55
+ if not gateway_started.wait(timeout=10.0):
56
+ java_process.terminate()
57
+ raise RuntimeError("Java Gateway did not start within 10 seconds")
58
+
59
+ gateway = JavaGateway(
60
+ gateway_parameters=GatewayParameters(auto_convert=True, read_timeout=None),
61
+ callback_server_parameters=None,
62
+ )
47
63
 
48
- gateway = JavaGateway()
49
64
  print("Python connection established.")
50
65
 
51
66
  return java_process, gateway
@@ -73,13 +88,45 @@ def close_conn(java_process: subprocess.Popen, gateway: JavaGateway) -> None:
73
88
  try:
74
89
  # Proper way to shut down Py4J
75
90
  gateway.shutdown()
76
- print("Python connection closed.")
91
+ print("✅ Python connection closed.")
77
92
  except Py4JNetworkError:
78
- print("Java process already shut down.")
93
+ print("⚠️ Java process already shut down.")
94
+ except Exception as e:
95
+ print(f"⚠️ Error shutting down gateway: {e}")
79
96
 
80
97
  # Terminate Java process
81
98
  java_process.terminate()
82
- print("Java server is shutting down...")
99
+ java_process.wait()
100
+ print("✅ Java process terminated.")
101
+
102
+
103
+ def drain_stream(stream: io.TextIOBase, callback=None) -> None:
104
+ """
105
+ Consumes the output from a given stream until a specific marker is found,
106
+ then closes the stream.
107
+
108
+ This function is typically used to monitor the stdout or stderr of a subprocess
109
+ (e.g., a Java process started from Python) and detect when a certain event occurs,
110
+ such as the initialization of a gateway server. Once the marker line is encountered,
111
+ the function prints it (or logs it) and terminates the stream reading.
112
+
113
+ Args:
114
+ stream (io.TextIOBase): A text-based stream object to read from, usually
115
+ subprocess.stdout or subprocess.stderr.
116
+ callback (callable, optional): A function that takes a single string argument (line).
117
+ It will be called for every line read from the stream.
118
+ Useful for detecting specific markers, logging, or
119
+ triggering events when certain output appears.
120
+
121
+ Raises:
122
+ AttributeError: If the provided `stream` does not have `readline` or `close` methods.
123
+ """
124
+ for line in iter(stream.readline, ""):
125
+ line = line.strip()
126
+ if callback:
127
+ callback(line)
128
+
129
+ stream.close()
83
130
 
84
131
 
85
132
  def check_java_version() -> None:
@@ -1,29 +1,35 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: temporal_normalization_spacy
3
- Version: 2.1.0
4
- Summary: A spaCy plugin for identifying and parsing historical data in Romanian texts
3
+ Version: 2.2.1
4
+ Summary: A spaCy plugin for temporal normalization and extraction of historical dates in Romanian narrative texts.
5
5
  Home-page: https://github.com/iliedorobat/timespan-normalization-spacy
6
6
  Author: Ilie Cristian Dorobat
7
+ Keywords: spacy,nlp,temporal normalization,timex,historical dates,romanian
7
8
  Classifier: Programming Language :: Python :: 3
9
+ Classifier: Programming Language :: Python :: 3.9
10
+ Classifier: Programming Language :: Python :: 3.10
11
+ Classifier: Programming Language :: Python :: 3.11
12
+ Classifier: Programming Language :: Python :: 3.12
8
13
  Classifier: License :: OSI Approved :: MIT License
9
14
  Classifier: Operating System :: OS Independent
10
- Requires-Python: >=3.6
15
+ Requires-Python: >=3.9,<3.13
11
16
  Description-Content-Type: text/markdown
12
17
  License-File: LICENSE
13
- Requires-Dist: spacy>=3.0
14
- Requires-Dist: py4j
15
- Requires-Dist: langdetect
18
+ Requires-Dist: spacy<4.0.0,>=3.8.7
19
+ Requires-Dist: py4j>=0.10.9.9
20
+ Requires-Dist: langdetect>=1.0.9
16
21
  Dynamic: author
17
22
  Dynamic: classifier
18
23
  Dynamic: description
19
24
  Dynamic: description-content-type
20
25
  Dynamic: home-page
26
+ Dynamic: keywords
21
27
  Dynamic: license-file
22
28
  Dynamic: requires-dist
23
29
  Dynamic: requires-python
24
30
  Dynamic: summary
25
31
 
26
- # Temporal Expressions Normalization for spaCy (TeNs)
32
+ # Temporal Expression Normalization for spaCy (TeNs)
27
33
 
28
34
  <b>Temporal Expressions Normalization spaCy (TeNs)</b> is a powerful pipeline component for spaCy that seamlessly
29
35
  identifies and parses date entities in text. It leverages the <b>[Temporal Expressions Normalization Framework](
@@ -75,8 +81,8 @@ quality and reliability of date-related information extracted from text.
75
81
  To integrate TeNs into spaCy pipelines you need the following:
76
82
 
77
83
  ### Prerequisites
78
- - Python 3.x
79
- - JRE 11+
84
+ - Python 3.9-3.12
85
+ - JRE 11-17
80
86
  - spaCy 3.x
81
87
  - py4j 0.10.9.9
82
88
  - langdetect 1.0.9
@@ -101,7 +107,6 @@ import subprocess
101
107
  import spacy
102
108
 
103
109
  from temporal_normalization.commons.print_utils import console
104
- from temporal_normalization.index import create_normalized_component, TemporalNormalization # noqa: F401
105
110
 
106
111
  LANG = "ro"
107
112
  MODEL = "ro_core_news_sm"
@@ -121,7 +126,7 @@ try:
121
126
  except OSError:
122
127
  console.warning(f'Started downloading {MODEL}...')
123
128
  # Download the Romanian model if it wasn't already downloaded
124
- subprocess.run(["python", "-m", "spacy", "download", MODEL])
129
+ subprocess.run(["python3", "-m", "spacy", "download", MODEL])
125
130
  # Load the spaCy model
126
131
  nlp = spacy.load(MODEL)
127
132
 
@@ -14,5 +14,6 @@ temporal_normalization/process/java_process.py
14
14
  temporal_normalization_spacy.egg-info/PKG-INFO
15
15
  temporal_normalization_spacy.egg-info/SOURCES.txt
16
16
  temporal_normalization_spacy.egg-info/dependency_links.txt
17
+ temporal_normalization_spacy.egg-info/entry_points.txt
17
18
  temporal_normalization_spacy.egg-info/requires.txt
18
19
  temporal_normalization_spacy.egg-info/top_level.txt
@@ -0,0 +1,2 @@
1
+ [spacy_factories]
2
+ temporal_normalization = temporal_normalization_spacy.factory:create_component
@@ -0,0 +1,3 @@
1
+ spacy<4.0.0,>=3.8.7
2
+ py4j>=0.10.9.9
3
+ langdetect>=1.0.9
@@ -1,24 +0,0 @@
1
- from setuptools import setup, find_packages
2
-
3
- setup(
4
- name="temporal_normalization_spacy",
5
- version="2.1.0",
6
- author="Ilie Cristian Dorobat",
7
- description="A spaCy plugin for identifying and parsing historical data "
8
- "in Romanian texts",
9
- long_description=open("README.md").read(),
10
- long_description_content_type="text/markdown",
11
- url="https://github.com/iliedorobat/timespan-normalization-spacy",
12
- packages=find_packages(),
13
- include_package_data=True,
14
- package_data={
15
- "temporal_normalization.libs": ["temporal-normalization-2.1.0.jar"],
16
- },
17
- install_requires=["spacy>=3.0", "py4j", "langdetect"],
18
- classifiers=[
19
- "Programming Language :: Python :: 3",
20
- "License :: OSI Approved :: MIT License",
21
- "Operating System :: OS Independent",
22
- ],
23
- python_requires=">=3.6",
24
- )
@@ -1,3 +0,0 @@
1
- spacy>=3.0
2
- py4j
3
- langdetect