temporal-normalization-spacy 2.0.1__tar.gz → 2.0.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (20) hide show
  1. {temporal_normalization_spacy-2.0.1/temporal_normalization_spacy.egg-info → temporal_normalization_spacy-2.0.2}/PKG-INFO +2 -2
  2. {temporal_normalization_spacy-2.0.1 → temporal_normalization_spacy-2.0.2}/README.md +1 -1
  3. {temporal_normalization_spacy-2.0.1 → temporal_normalization_spacy-2.0.2}/setup.py +2 -2
  4. {temporal_normalization_spacy-2.0.1 → temporal_normalization_spacy-2.0.2}/temporal_normalization/commons/temporal_models.py +7 -1
  5. {temporal_normalization_spacy-2.0.1 → temporal_normalization_spacy-2.0.2}/temporal_normalization/index.py +60 -12
  6. {temporal_normalization_spacy-2.0.1 → temporal_normalization_spacy-2.0.2/temporal_normalization_spacy.egg-info}/PKG-INFO +2 -2
  7. {temporal_normalization_spacy-2.0.1 → temporal_normalization_spacy-2.0.2}/LICENSE +0 -0
  8. {temporal_normalization_spacy-2.0.1 → temporal_normalization_spacy-2.0.2}/MANIFEST.in +0 -0
  9. {temporal_normalization_spacy-2.0.1 → temporal_normalization_spacy-2.0.2}/setup.cfg +0 -0
  10. {temporal_normalization_spacy-2.0.1 → temporal_normalization_spacy-2.0.2}/temporal_normalization/__init__.py +0 -0
  11. {temporal_normalization_spacy-2.0.1 → temporal_normalization_spacy-2.0.2}/temporal_normalization/commons/__init__.py +0 -0
  12. {temporal_normalization_spacy-2.0.1 → temporal_normalization_spacy-2.0.2}/temporal_normalization/commons/print_utils.py +0 -0
  13. {temporal_normalization_spacy-2.0.1 → temporal_normalization_spacy-2.0.2}/temporal_normalization/commons/temporal_types.py +0 -0
  14. {temporal_normalization_spacy-2.0.1 → temporal_normalization_spacy-2.0.2}/temporal_normalization/libs/temporal-normalization-2.0.jar +0 -0
  15. {temporal_normalization_spacy-2.0.1 → temporal_normalization_spacy-2.0.2}/temporal_normalization/process/__init__.py +0 -0
  16. {temporal_normalization_spacy-2.0.1 → temporal_normalization_spacy-2.0.2}/temporal_normalization/process/java_process.py +0 -0
  17. {temporal_normalization_spacy-2.0.1 → temporal_normalization_spacy-2.0.2}/temporal_normalization_spacy.egg-info/SOURCES.txt +0 -0
  18. {temporal_normalization_spacy-2.0.1 → temporal_normalization_spacy-2.0.2}/temporal_normalization_spacy.egg-info/dependency_links.txt +0 -0
  19. {temporal_normalization_spacy-2.0.1 → temporal_normalization_spacy-2.0.2}/temporal_normalization_spacy.egg-info/requires.txt +0 -0
  20. {temporal_normalization_spacy-2.0.1 → temporal_normalization_spacy-2.0.2}/temporal_normalization_spacy.egg-info/top_level.txt +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.2
2
2
  Name: temporal_normalization_spacy
3
- Version: 2.0.1
3
+ Version: 2.0.2
4
4
  Summary: A spaCy plugin for identifying and parsing historical data in Romanian texts
5
5
  Home-page: https://github.com/iliedorobat/timespan-normalization-spacy
6
6
  Author: Ilie Cristian Dorobat
@@ -22,7 +22,7 @@ Dynamic: requires-dist
22
22
  Dynamic: requires-python
23
23
  Dynamic: summary
24
24
 
25
- # Temporal Expressions Normalization spaCy (TeNs)
25
+ # Temporal Expressions Normalization for spaCy (TeNs)
26
26
 
27
27
  <b>Temporal Expressions Normalization spaCy (TeNs)</b> is a powerful pipeline component for spaCy that seamlessly
28
28
  identifies and parses date entities in text. It leverages the <b>[Temporal Expressions Normalization Framework](
@@ -1,4 +1,4 @@
1
- # Temporal Expressions Normalization spaCy (TeNs)
1
+ # Temporal Expressions Normalization for spaCy (TeNs)
2
2
 
3
3
  <b>Temporal Expressions Normalization spaCy (TeNs)</b> is a powerful pipeline component for spaCy that seamlessly
4
4
  identifies and parses date entities in text. It leverages the <b>[Temporal Expressions Normalization Framework](
@@ -2,10 +2,10 @@ from setuptools import setup, find_packages
2
2
 
3
3
  setup(
4
4
  name="temporal_normalization_spacy",
5
- version="2.0.1",
5
+ version="2.0.2",
6
6
  author="Ilie Cristian Dorobat",
7
7
  description="A spaCy plugin for identifying and parsing historical data "
8
- "in Romanian texts",
8
+ "in Romanian texts",
9
9
  long_description=open("README.md").read(),
10
10
  long_description_content_type="text/markdown",
11
11
  url="https://github.com/iliedorobat/timespan-normalization-spacy",
@@ -29,7 +29,13 @@ class TemporalExpression:
29
29
  TimeSeries(item) for item in json_obj["timeSeries"]
30
30
  ] if self.is_valid else []
31
31
  self.matches: list[str] = list(
32
- set([matched_value for ts in self.time_series for matched_value in ts.matches])
32
+ set(
33
+ [
34
+ matched_value
35
+ for ts in self.time_series
36
+ for matched_value in ts.matches
37
+ ]
38
+ )
33
39
  )
34
40
 
35
41
  def __str__(self):
@@ -129,10 +129,10 @@ def _retokenize(
129
129
  if start_token is not None and end_token is not None:
130
130
  # use exact token boundaries to create a custom `Span` for well-defined
131
131
  # time expressions with known character offsets.
132
- entity, exists = _create_span(doc, start_char, end_char, start_token, end_token)
132
+ entity, existed_entity = _create_span(doc, start_char, end_char, start_token, end_token)
133
133
  time_series: list[TimeSeries] = [ts for expression in expressions for ts in expression.time_series]
134
134
  matched_ts = [ts for ts in time_series if _matched(entity.text, ts.matches)]
135
- _assign_time_series(matched_ts, retokenized_entities, retokenizer, doc, entity, exists)
135
+ _retokenize_entity(doc, matched_ts, entity, existed_entity, retokenized_entities, retokenizer)
136
136
  else:
137
137
  # For more ambiguous or loosely defined expressions, such as "martie -iunie 2013"
138
138
  # or "dintre secolele al XV-lea și al XVIII-lea", iterates through existing entities
@@ -142,24 +142,26 @@ def _retokenize(
142
142
  if entity not in retokenized_entities:
143
143
  time_series: list[TimeSeries] = [ts for expression in expressions for ts in expression.time_series]
144
144
  matched_ts = [ts for ts in time_series if _is_substring(entity.text, ts.matches)]
145
- _assign_time_series(matched_ts, retokenized_entities, retokenizer, doc, entity, True)
145
+ _retokenize_entity(doc, matched_ts, entity, True, retokenized_entities, retokenizer)
146
146
 
147
147
 
148
- def _assign_time_series(
148
+ def _retokenize_entity(
149
+ doc: Doc,
149
150
  matched_ts: list[TimeSeries],
151
+ entity: Span,
152
+ existed_entity: bool,
150
153
  retokenized_entities: list[Span],
151
154
  retokenizer: Doc.retokenize,
152
- doc: Doc,
153
- entity: Span,
154
- exists: bool
155
155
  ) -> None:
156
156
  """
157
- Attaches matched TimeSeries to a given entity and updates the doc's entity list.
157
+ Retokenizes and enriches a temporal entity span with matched time series data.
158
+ Updates the Doc with the new entity and merges it if needed.
158
159
 
159
160
  Args:
160
161
  doc (Doc): The processed spaCy document.
162
+ matched_ts (list[TimeSeries]): The matched time series.
161
163
  entity (Span): The named entity to enrich.
162
- exists (bool): Whether the entity already exists in doc.ents.
164
+ existed_entity (bool): Whether the entity already exists in doc.ents.
163
165
  retokenized_entities (list): Accumulator for entities that require retokenization.
164
166
  retokenizer (Doc.retokenize): The spaCy retokenizer context.
165
167
  """
@@ -167,11 +169,38 @@ def _assign_time_series(
167
169
  if not len(matched_ts):
168
170
  return None
169
171
 
170
- if exists:
172
+ _assign_time_series(matched_ts, entity, existed_entity)
173
+ _update_doc_ents(doc, entity)
174
+ _merge_entity(doc, entity, retokenized_entities, retokenizer)
175
+
176
+
177
+ def _assign_time_series(
178
+ matched_ts: list[TimeSeries], entity: Span, existed_entity: bool
179
+ ) -> None:
180
+ """
181
+ Attaches matched TimeSeries to a given entity.
182
+
183
+ Args:
184
+ matched_ts (list[TimeSeries]): The matched time series.
185
+ entity (Span): The named entity to enrich.
186
+ existed_entity (bool): Whether the entity already exists in doc.ents.
187
+ """
188
+
189
+ if existed_entity:
171
190
  entity._.time_series = matched_ts
172
191
  else:
173
192
  entity._.set("time_series", matched_ts)
174
193
 
194
+
195
+ def _update_doc_ents(doc: Doc, entity: Span) -> None:
196
+ """
197
+ Updates the doc's entity list
198
+
199
+ Args:
200
+ doc (Doc): The processed spaCy document.
201
+ entity (Span): The named entity to enrich.
202
+ """
203
+
175
204
  all_ents = list(doc.ents)
176
205
  if entity not in all_ents:
177
206
  # E.g.: entity in all_ents => "Ecaterina Balș ( 22 iulie 1814 - august 1887 ) - născută în familia Dimachi , a doua soție a generalului Teodor Balș ( 1805-1857 ) , caimacam al Moldovei în perioada 1856-1857 , cu care se căsătorise la 30 iunie 1846 la Dimăcheni ( Dorohoi ) ."
@@ -180,7 +209,24 @@ def _assign_time_series(
180
209
 
181
210
  doc.ents = filter_spans(all_ents)
182
211
 
183
- if entity not in all_ents:
212
+
213
+ def _merge_entity(
214
+ doc: Doc,
215
+ entity: Span,
216
+ retokenized_entities: list[Span],
217
+ retokenizer: Doc.retokenize,
218
+ ) -> None:
219
+ """
220
+ Merges a custom entity span into the spaCy Doc if it is not already part of
221
+ doc.ents, and tracks it in a list of retokenized entities.
222
+
223
+ Args:
224
+ entity (Span): The named entity to enrich.
225
+ retokenized_entities (list): Accumulator for entities that require retokenization.
226
+ retokenizer (Doc.retokenize): The spaCy retokenizer context.
227
+ """
228
+
229
+ if entity not in doc.ents:
184
230
  retokenized_entities.append(entity)
185
231
  retokenizer.merge(entity)
186
232
 
@@ -226,7 +272,9 @@ def _is_substring(text: str, matches: list[str]) -> bool:
226
272
  return False
227
273
 
228
274
 
229
- def _create_span(doc: Doc, start_char: int, end_char: int, start_token: int, end_token: int) -> tuple[Span, bool]:
275
+ def _create_span(
276
+ doc: Doc, start_char: int, end_char: int, start_token: int, end_token: int
277
+ ) -> tuple[Span, bool]:
230
278
  """
231
279
  Creates a new span for a temporal expression or returns an existing overlapping entity.
232
280
 
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.2
2
2
  Name: temporal_normalization_spacy
3
- Version: 2.0.1
3
+ Version: 2.0.2
4
4
  Summary: A spaCy plugin for identifying and parsing historical data in Romanian texts
5
5
  Home-page: https://github.com/iliedorobat/timespan-normalization-spacy
6
6
  Author: Ilie Cristian Dorobat
@@ -22,7 +22,7 @@ Dynamic: requires-dist
22
22
  Dynamic: requires-python
23
23
  Dynamic: summary
24
24
 
25
- # Temporal Expressions Normalization spaCy (TeNs)
25
+ # Temporal Expressions Normalization for spaCy (TeNs)
26
26
 
27
27
  <b>Temporal Expressions Normalization spaCy (TeNs)</b> is a powerful pipeline component for spaCy that seamlessly
28
28
  identifies and parses date entities in text. It leverages the <b>[Temporal Expressions Normalization Framework](