temporal-normalization-spacy 2.0.1__tar.gz → 2.0.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {temporal_normalization_spacy-2.0.1/temporal_normalization_spacy.egg-info → temporal_normalization_spacy-2.0.2}/PKG-INFO +2 -2
- {temporal_normalization_spacy-2.0.1 → temporal_normalization_spacy-2.0.2}/README.md +1 -1
- {temporal_normalization_spacy-2.0.1 → temporal_normalization_spacy-2.0.2}/setup.py +2 -2
- {temporal_normalization_spacy-2.0.1 → temporal_normalization_spacy-2.0.2}/temporal_normalization/commons/temporal_models.py +7 -1
- {temporal_normalization_spacy-2.0.1 → temporal_normalization_spacy-2.0.2}/temporal_normalization/index.py +60 -12
- {temporal_normalization_spacy-2.0.1 → temporal_normalization_spacy-2.0.2/temporal_normalization_spacy.egg-info}/PKG-INFO +2 -2
- {temporal_normalization_spacy-2.0.1 → temporal_normalization_spacy-2.0.2}/LICENSE +0 -0
- {temporal_normalization_spacy-2.0.1 → temporal_normalization_spacy-2.0.2}/MANIFEST.in +0 -0
- {temporal_normalization_spacy-2.0.1 → temporal_normalization_spacy-2.0.2}/setup.cfg +0 -0
- {temporal_normalization_spacy-2.0.1 → temporal_normalization_spacy-2.0.2}/temporal_normalization/__init__.py +0 -0
- {temporal_normalization_spacy-2.0.1 → temporal_normalization_spacy-2.0.2}/temporal_normalization/commons/__init__.py +0 -0
- {temporal_normalization_spacy-2.0.1 → temporal_normalization_spacy-2.0.2}/temporal_normalization/commons/print_utils.py +0 -0
- {temporal_normalization_spacy-2.0.1 → temporal_normalization_spacy-2.0.2}/temporal_normalization/commons/temporal_types.py +0 -0
- {temporal_normalization_spacy-2.0.1 → temporal_normalization_spacy-2.0.2}/temporal_normalization/libs/temporal-normalization-2.0.jar +0 -0
- {temporal_normalization_spacy-2.0.1 → temporal_normalization_spacy-2.0.2}/temporal_normalization/process/__init__.py +0 -0
- {temporal_normalization_spacy-2.0.1 → temporal_normalization_spacy-2.0.2}/temporal_normalization/process/java_process.py +0 -0
- {temporal_normalization_spacy-2.0.1 → temporal_normalization_spacy-2.0.2}/temporal_normalization_spacy.egg-info/SOURCES.txt +0 -0
- {temporal_normalization_spacy-2.0.1 → temporal_normalization_spacy-2.0.2}/temporal_normalization_spacy.egg-info/dependency_links.txt +0 -0
- {temporal_normalization_spacy-2.0.1 → temporal_normalization_spacy-2.0.2}/temporal_normalization_spacy.egg-info/requires.txt +0 -0
- {temporal_normalization_spacy-2.0.1 → temporal_normalization_spacy-2.0.2}/temporal_normalization_spacy.egg-info/top_level.txt +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.2
|
|
2
2
|
Name: temporal_normalization_spacy
|
|
3
|
-
Version: 2.0.
|
|
3
|
+
Version: 2.0.2
|
|
4
4
|
Summary: A spaCy plugin for identifying and parsing historical data in Romanian texts
|
|
5
5
|
Home-page: https://github.com/iliedorobat/timespan-normalization-spacy
|
|
6
6
|
Author: Ilie Cristian Dorobat
|
|
@@ -22,7 +22,7 @@ Dynamic: requires-dist
|
|
|
22
22
|
Dynamic: requires-python
|
|
23
23
|
Dynamic: summary
|
|
24
24
|
|
|
25
|
-
# Temporal Expressions Normalization spaCy (TeNs)
|
|
25
|
+
# Temporal Expressions Normalization for spaCy (TeNs)
|
|
26
26
|
|
|
27
27
|
<b>Temporal Expressions Normalization spaCy (TeNs)</b> is a powerful pipeline component for spaCy that seamlessly
|
|
28
28
|
identifies and parses date entities in text. It leverages the <b>[Temporal Expressions Normalization Framework](
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
# Temporal Expressions Normalization spaCy (TeNs)
|
|
1
|
+
# Temporal Expressions Normalization for spaCy (TeNs)
|
|
2
2
|
|
|
3
3
|
<b>Temporal Expressions Normalization spaCy (TeNs)</b> is a powerful pipeline component for spaCy that seamlessly
|
|
4
4
|
identifies and parses date entities in text. It leverages the <b>[Temporal Expressions Normalization Framework](
|
|
@@ -2,10 +2,10 @@ from setuptools import setup, find_packages
|
|
|
2
2
|
|
|
3
3
|
setup(
|
|
4
4
|
name="temporal_normalization_spacy",
|
|
5
|
-
version="2.0.
|
|
5
|
+
version="2.0.2",
|
|
6
6
|
author="Ilie Cristian Dorobat",
|
|
7
7
|
description="A spaCy plugin for identifying and parsing historical data "
|
|
8
|
-
|
|
8
|
+
"in Romanian texts",
|
|
9
9
|
long_description=open("README.md").read(),
|
|
10
10
|
long_description_content_type="text/markdown",
|
|
11
11
|
url="https://github.com/iliedorobat/timespan-normalization-spacy",
|
|
@@ -29,7 +29,13 @@ class TemporalExpression:
|
|
|
29
29
|
TimeSeries(item) for item in json_obj["timeSeries"]
|
|
30
30
|
] if self.is_valid else []
|
|
31
31
|
self.matches: list[str] = list(
|
|
32
|
-
set(
|
|
32
|
+
set(
|
|
33
|
+
[
|
|
34
|
+
matched_value
|
|
35
|
+
for ts in self.time_series
|
|
36
|
+
for matched_value in ts.matches
|
|
37
|
+
]
|
|
38
|
+
)
|
|
33
39
|
)
|
|
34
40
|
|
|
35
41
|
def __str__(self):
|
|
@@ -129,10 +129,10 @@ def _retokenize(
|
|
|
129
129
|
if start_token is not None and end_token is not None:
|
|
130
130
|
# use exact token boundaries to create a custom `Span` for well-defined
|
|
131
131
|
# time expressions with known character offsets.
|
|
132
|
-
entity,
|
|
132
|
+
entity, existed_entity = _create_span(doc, start_char, end_char, start_token, end_token)
|
|
133
133
|
time_series: list[TimeSeries] = [ts for expression in expressions for ts in expression.time_series]
|
|
134
134
|
matched_ts = [ts for ts in time_series if _matched(entity.text, ts.matches)]
|
|
135
|
-
|
|
135
|
+
_retokenize_entity(doc, matched_ts, entity, existed_entity, retokenized_entities, retokenizer)
|
|
136
136
|
else:
|
|
137
137
|
# For more ambiguous or loosely defined expressions, such as "martie -iunie 2013"
|
|
138
138
|
# or "dintre secolele al XV-lea și al XVIII-lea", iterates through existing entities
|
|
@@ -142,24 +142,26 @@ def _retokenize(
|
|
|
142
142
|
if entity not in retokenized_entities:
|
|
143
143
|
time_series: list[TimeSeries] = [ts for expression in expressions for ts in expression.time_series]
|
|
144
144
|
matched_ts = [ts for ts in time_series if _is_substring(entity.text, ts.matches)]
|
|
145
|
-
|
|
145
|
+
_retokenize_entity(doc, matched_ts, entity, True, retokenized_entities, retokenizer)
|
|
146
146
|
|
|
147
147
|
|
|
148
|
-
def
|
|
148
|
+
def _retokenize_entity(
|
|
149
|
+
doc: Doc,
|
|
149
150
|
matched_ts: list[TimeSeries],
|
|
151
|
+
entity: Span,
|
|
152
|
+
existed_entity: bool,
|
|
150
153
|
retokenized_entities: list[Span],
|
|
151
154
|
retokenizer: Doc.retokenize,
|
|
152
|
-
doc: Doc,
|
|
153
|
-
entity: Span,
|
|
154
|
-
exists: bool
|
|
155
155
|
) -> None:
|
|
156
156
|
"""
|
|
157
|
-
|
|
157
|
+
Retokenizes and enriches a temporal entity span with matched time series data.
|
|
158
|
+
Updates the Doc with the new entity and merges it if needed.
|
|
158
159
|
|
|
159
160
|
Args:
|
|
160
161
|
doc (Doc): The processed spaCy document.
|
|
162
|
+
matched_ts (list[TimeSeries]): The matched time series.
|
|
161
163
|
entity (Span): The named entity to enrich.
|
|
162
|
-
|
|
164
|
+
existed_entity (bool): Whether the entity already exists in doc.ents.
|
|
163
165
|
retokenized_entities (list): Accumulator for entities that require retokenization.
|
|
164
166
|
retokenizer (Doc.retokenize): The spaCy retokenizer context.
|
|
165
167
|
"""
|
|
@@ -167,11 +169,38 @@ def _assign_time_series(
|
|
|
167
169
|
if not len(matched_ts):
|
|
168
170
|
return None
|
|
169
171
|
|
|
170
|
-
|
|
172
|
+
_assign_time_series(matched_ts, entity, existed_entity)
|
|
173
|
+
_update_doc_ents(doc, entity)
|
|
174
|
+
_merge_entity(doc, entity, retokenized_entities, retokenizer)
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
def _assign_time_series(
|
|
178
|
+
matched_ts: list[TimeSeries], entity: Span, existed_entity: bool
|
|
179
|
+
) -> None:
|
|
180
|
+
"""
|
|
181
|
+
Attaches matched TimeSeries to a given entity.
|
|
182
|
+
|
|
183
|
+
Args:
|
|
184
|
+
matched_ts (list[TimeSeries]): The matched time series.
|
|
185
|
+
entity (Span): The named entity to enrich.
|
|
186
|
+
existed_entity (bool): Whether the entity already exists in doc.ents.
|
|
187
|
+
"""
|
|
188
|
+
|
|
189
|
+
if existed_entity:
|
|
171
190
|
entity._.time_series = matched_ts
|
|
172
191
|
else:
|
|
173
192
|
entity._.set("time_series", matched_ts)
|
|
174
193
|
|
|
194
|
+
|
|
195
|
+
def _update_doc_ents(doc: Doc, entity: Span) -> None:
|
|
196
|
+
"""
|
|
197
|
+
Updates the doc's entity list
|
|
198
|
+
|
|
199
|
+
Args:
|
|
200
|
+
doc (Doc): The processed spaCy document.
|
|
201
|
+
entity (Span): The named entity to enrich.
|
|
202
|
+
"""
|
|
203
|
+
|
|
175
204
|
all_ents = list(doc.ents)
|
|
176
205
|
if entity not in all_ents:
|
|
177
206
|
# E.g.: entity in all_ents => "Ecaterina Balș ( 22 iulie 1814 - august 1887 ) - născută în familia Dimachi , a doua soție a generalului Teodor Balș ( 1805-1857 ) , caimacam al Moldovei în perioada 1856-1857 , cu care se căsătorise la 30 iunie 1846 la Dimăcheni ( Dorohoi ) ."
|
|
@@ -180,7 +209,24 @@ def _assign_time_series(
|
|
|
180
209
|
|
|
181
210
|
doc.ents = filter_spans(all_ents)
|
|
182
211
|
|
|
183
|
-
|
|
212
|
+
|
|
213
|
+
def _merge_entity(
|
|
214
|
+
doc: Doc,
|
|
215
|
+
entity: Span,
|
|
216
|
+
retokenized_entities: list[Span],
|
|
217
|
+
retokenizer: Doc.retokenize,
|
|
218
|
+
) -> None:
|
|
219
|
+
"""
|
|
220
|
+
Merges a custom entity span into the spaCy Doc if it is not already part of
|
|
221
|
+
doc.ents, and tracks it in a list of retokenized entities.
|
|
222
|
+
|
|
223
|
+
Args:
|
|
224
|
+
entity (Span): The named entity to enrich.
|
|
225
|
+
retokenized_entities (list): Accumulator for entities that require retokenization.
|
|
226
|
+
retokenizer (Doc.retokenize): The spaCy retokenizer context.
|
|
227
|
+
"""
|
|
228
|
+
|
|
229
|
+
if entity not in doc.ents:
|
|
184
230
|
retokenized_entities.append(entity)
|
|
185
231
|
retokenizer.merge(entity)
|
|
186
232
|
|
|
@@ -226,7 +272,9 @@ def _is_substring(text: str, matches: list[str]) -> bool:
|
|
|
226
272
|
return False
|
|
227
273
|
|
|
228
274
|
|
|
229
|
-
def _create_span(
|
|
275
|
+
def _create_span(
|
|
276
|
+
doc: Doc, start_char: int, end_char: int, start_token: int, end_token: int
|
|
277
|
+
) -> tuple[Span, bool]:
|
|
230
278
|
"""
|
|
231
279
|
Creates a new span for a temporal expression or returns an existing overlapping entity.
|
|
232
280
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.2
|
|
2
2
|
Name: temporal_normalization_spacy
|
|
3
|
-
Version: 2.0.
|
|
3
|
+
Version: 2.0.2
|
|
4
4
|
Summary: A spaCy plugin for identifying and parsing historical data in Romanian texts
|
|
5
5
|
Home-page: https://github.com/iliedorobat/timespan-normalization-spacy
|
|
6
6
|
Author: Ilie Cristian Dorobat
|
|
@@ -22,7 +22,7 @@ Dynamic: requires-dist
|
|
|
22
22
|
Dynamic: requires-python
|
|
23
23
|
Dynamic: summary
|
|
24
24
|
|
|
25
|
-
# Temporal Expressions Normalization spaCy (TeNs)
|
|
25
|
+
# Temporal Expressions Normalization for spaCy (TeNs)
|
|
26
26
|
|
|
27
27
|
<b>Temporal Expressions Normalization spaCy (TeNs)</b> is a powerful pipeline component for spaCy that seamlessly
|
|
28
28
|
identifies and parses date entities in text. It leverages the <b>[Temporal Expressions Normalization Framework](
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|