python-dwca-reader 0.16.4__tar.gz → 0.17.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (43) hide show
  1. {python-dwca-reader-0.16.4 → python_dwca_reader-0.17.1}/CHANGES.txt +68 -1
  2. {python-dwca-reader-0.16.4 → python_dwca_reader-0.17.1}/PKG-INFO +16 -3
  3. {python-dwca-reader-0.16.4 → python_dwca_reader-0.17.1}/dwca/descriptors.py +239 -3
  4. python_dwca_reader-0.17.1/dwca/files.py +507 -0
  5. {python-dwca-reader-0.16.4 → python_dwca_reader-0.17.1}/dwca/read.py +65 -17
  6. {python-dwca-reader-0.16.4 → python_dwca_reader-0.17.1}/dwca/rows.py +105 -78
  7. {python-dwca-reader-0.16.4 → python_dwca_reader-0.17.1}/dwca/star_record.py +2 -3
  8. python_dwca_reader-0.17.1/dwca/test/archive_builder.py +188 -0
  9. python_dwca_reader-0.17.1/dwca/test/test_characterization.py +617 -0
  10. python_dwca_reader-0.17.1/dwca/test/test_datafile.py +445 -0
  11. {python-dwca-reader-0.16.4 → python_dwca_reader-0.17.1}/dwca/test/test_descriptors.py +331 -0
  12. {python-dwca-reader-0.16.4 → python_dwca_reader-0.17.1}/dwca/test/test_dwcareader.py +74 -12
  13. python_dwca_reader-0.17.1/dwca/test/test_rows.py +222 -0
  14. python_dwca_reader-0.17.1/dwca/version.py +1 -0
  15. python_dwca_reader-0.17.1/pyproject.toml +14 -0
  16. {python-dwca-reader-0.16.4 → python_dwca_reader-0.17.1}/python_dwca_reader.egg-info/PKG-INFO +16 -3
  17. {python-dwca-reader-0.16.4 → python_dwca_reader-0.17.1}/python_dwca_reader.egg-info/SOURCES.txt +3 -1
  18. {python-dwca-reader-0.16.4 → python_dwca_reader-0.17.1}/requirements-dev.txt +0 -1
  19. {python-dwca-reader-0.16.4 → python_dwca_reader-0.17.1}/setup.py +4 -1
  20. python-dwca-reader-0.16.4/dwca/files.py +0 -181
  21. python-dwca-reader-0.16.4/dwca/minibench.py +0 -61
  22. python-dwca-reader-0.16.4/dwca/test/test_datafile.py +0 -162
  23. python-dwca-reader-0.16.4/dwca/test/test_rows.py +0 -49
  24. python-dwca-reader-0.16.4/dwca/version.py +0 -1
  25. {python-dwca-reader-0.16.4 → python_dwca_reader-0.17.1}/AUTHORS +0 -0
  26. {python-dwca-reader-0.16.4 → python_dwca_reader-0.17.1}/LICENSE.txt +0 -0
  27. {python-dwca-reader-0.16.4 → python_dwca_reader-0.17.1}/MANIFEST.in +0 -0
  28. {python-dwca-reader-0.16.4 → python_dwca_reader-0.17.1}/README.rst +0 -0
  29. {python-dwca-reader-0.16.4 → python_dwca_reader-0.17.1}/code_of_conduct.txt +0 -0
  30. {python-dwca-reader-0.16.4 → python_dwca_reader-0.17.1}/dwca/__init__.py +0 -0
  31. {python-dwca-reader-0.16.4 → python_dwca_reader-0.17.1}/dwca/darwincore/__init__.py +0 -0
  32. {python-dwca-reader-0.16.4 → python_dwca_reader-0.17.1}/dwca/darwincore/build_dc_terms_list.py +0 -0
  33. {python-dwca-reader-0.16.4 → python_dwca_reader-0.17.1}/dwca/darwincore/terms.py +0 -0
  34. {python-dwca-reader-0.16.4 → python_dwca_reader-0.17.1}/dwca/darwincore/utils.py +0 -0
  35. {python-dwca-reader-0.16.4 → python_dwca_reader-0.17.1}/dwca/exceptions.py +0 -0
  36. {python-dwca-reader-0.16.4 → python_dwca_reader-0.17.1}/dwca/helpers.py +0 -0
  37. {python-dwca-reader-0.16.4 → python_dwca_reader-0.17.1}/dwca/test/__init__.py +0 -0
  38. {python-dwca-reader-0.16.4 → python_dwca_reader-0.17.1}/dwca/test/helpers.py +0 -0
  39. {python-dwca-reader-0.16.4 → python_dwca_reader-0.17.1}/dwca/test/test_star_record.py +0 -0
  40. {python-dwca-reader-0.16.4 → python_dwca_reader-0.17.1}/dwca/vendor.py +0 -0
  41. {python-dwca-reader-0.16.4 → python_dwca_reader-0.17.1}/python_dwca_reader.egg-info/dependency_links.txt +0 -0
  42. {python-dwca-reader-0.16.4 → python_dwca_reader-0.17.1}/python_dwca_reader.egg-info/top_level.txt +0 -0
  43. {python-dwca-reader-0.16.4 → python_dwca_reader-0.17.1}/setup.cfg +0 -0
@@ -1,3 +1,70 @@
1
+ v0.17.1 (2026-07-28)
2
+ --------------------
3
+
4
+ No library changes: this release exists so the packaging and documentation fixes below are
5
+ carried by a tag, since Read the Docs builds `stable` from the most recent one.
6
+
7
+ - Fixed: the Read the Docs build failed with "No module named 'pkg_resources'". The docs
8
+ requirements pinned Sphinx 2.2.0, which imports pkg_resources, and setuptools removed it in
9
+ 81.0. Only the two direct documentation dependencies are pinned now, with lower bounds.
10
+ - Fixed: the docstring of dwca.star_record.StarRecordIterator contained invalid
11
+ reStructuredText, which surfaced as a warning once v0.17.0 added the module to the API
12
+ reference.
13
+ - Packaging: pyproject.toml now declares the build backend, so builds resolve setuptools in
14
+ an isolated environment. Building v0.17.0 with an older setuptools produced an sdist
15
+ filename PyPI rejects.
16
+
17
+ v0.17.0 (2026-07-28)
18
+ --------------------
19
+
20
+ - New: DwCAReader.iter_terms() and CSVDataFile.iter_terms() yield a tuple of values per row
21
+ for a chosen list of terms, skipping both the Row object and its data dict. Measured
22
+ roughly 2.9x faster than iterating over rows and reading the same terms, and roughly 10x
23
+ faster than the pre-rewrite implementation, on a 400000-row archive reading 14 of 50
24
+ columns, on one machine; see benchmarks/README.md for the full measurement.
25
+ - New: iter_terms() accepts "id" and "coreid" to request the archive's key columns, which
26
+ often have no declared term of their own. These are the names headers() already uses.
27
+ - Performance: iterating over an archive is a single streaming pass instead of one seek and
28
+ one throwaway csv.reader per row. Measured roughly 3.8-4.0x faster on a 400000-row GBIF
29
+ download on one machine; see benchmarks/README.md for the full before/after measurement.
30
+ - Performance: the line offset index used for random access is now built on first use, so
31
+ opening an archive no longer scans every data file.
32
+ - Fixed: DataFileDescriptor.headers dropped the column at index 0 for archives without a
33
+ metafile, which also made pd_read() promote that column to the DataFrame index.
34
+ - Fixed: hash() on a CoreRow or an ExtensionRow raised TypeError. Rows have been documented
35
+ as hashable since 0.3.3 but never were.
36
+ - Fixed: comparing or hashing a CoreRow obtained without linking to its DwCAReader (e.g. via
37
+ CSVDataFile.get_row_by_position() directly rather than by iterating the reader) raised
38
+ AttributeError instead of comparing successfully.
39
+ - Fixed: rows read from two DwCAReader instances over the same archive never compared equal,
40
+ even with identical content. Row equality embeds the DataFileDescriptor, which had no
41
+ __eq__ and therefore compared by object identity. DataFileDescriptor now compares (and
42
+ hashes) by value: the data file layout it describes.
43
+ - Fixed: comparing a CoreRow to an ExtensionRow (or either to a non-row object) raised
44
+ AttributeError instead of returning False.
45
+ - Fixed: ignoreHeaderLines values above 1 only skipped a single line when iterating a
46
+ CSVDataFile directly, which let header lines leak into coreid_index and
47
+ orphaned_extension_rows.
48
+ - Fixed: an undecodable byte in a data file desynchronised the line offset index, so rows
49
+ after it were silently truncated or rejected.
50
+ - Fixed: a field whose content started or ended with the archive's fieldsEnclosedBy
51
+ character had that character stripped.
52
+ - Fixed: DwCAReader was its own iterator, so nesting two loops over the same reader silently
53
+ ran the inner one once, and calling get_corerow_by_id() or get_corerow_by_position() from
54
+ inside a loop never terminated. Each iteration now gets its own iterator.
55
+ - Fixed: in archives using fieldsEnclosedBy, a field containing the line terminator was read
56
+ as truncated at that terminator, by iteration and by random access alike, and the rows after
57
+ it were misaligned (often raising InvalidArchive on an unrelated row). Such archives are now
58
+ read correctly: the line offset index indexes CSV records rather than physical lines.
59
+ - Changed: removed the undeclared typing_extensions dependency (dwca.star_record now uses
60
+ typing.Literal). Python 3.7, which reached end of life in June 2023, is no longer
61
+ supported; the minimum is now 3.8.
62
+ - Changed: DwCAReader.next() now keeps an implicit iterator independent of any for loop over
63
+ the same reader, and starts a new pass after being exhausted. Iterate over the reader
64
+ instead; next() remains only for backwards compatibility.
65
+ - Documentation: the tutorial and the GBIF results page still called get_row_by_index(),
66
+ removed in v0.15.0. They now use get_corerow_by_position().
67
+
1
68
  v0.16.4 (2024-10-18)
2
69
  --------------------
3
70
 
@@ -264,4 +331,4 @@ v0.1.1 (2013-05-28)
264
331
  v0.1.0 (2013-05-28)
265
332
  -------------------
266
333
 
267
- - Initial release.
334
+ - Initial release.
@@ -1,20 +1,33 @@
1
- Metadata-Version: 2.1
1
+ Metadata-Version: 2.4
2
2
  Name: python-dwca-reader
3
- Version: 0.16.4
3
+ Version: 0.17.1
4
4
  Summary: A simple Python package to read Darwin Core Archive (DwC-A) files.
5
5
  Home-page: https://github.com/BelgianBiodiversityPlatform/python-dwca-reader
6
6
  Author: Nicolas Noé - Belgian Biodiversity Platform
7
7
  Author-email: n.noe@biodiversity.be
8
8
  License: BSD licence, see LICENCE.txt
9
9
  Classifier: Programming Language :: Python :: 3
10
- Classifier: Programming Language :: Python :: 3.7
11
10
  Classifier: Programming Language :: Python :: 3.8
12
11
  Classifier: Programming Language :: Python :: 3.9
13
12
  Classifier: Programming Language :: Python :: 3.10
14
13
  Classifier: Programming Language :: Python :: 3.11
14
+ Classifier: Programming Language :: Python :: 3.12
15
+ Classifier: Programming Language :: Python :: 3.13
15
16
  Classifier: Programming Language :: Python :: Implementation :: PyPy
17
+ Requires-Python: >=3.8
18
+ Description-Content-Type: text/x-rst
16
19
  License-File: LICENSE.txt
17
20
  License-File: AUTHORS
21
+ Dynamic: author
22
+ Dynamic: author-email
23
+ Dynamic: classifier
24
+ Dynamic: description
25
+ Dynamic: description-content-type
26
+ Dynamic: home-page
27
+ Dynamic: license
28
+ Dynamic: license-file
29
+ Dynamic: requires-python
30
+ Dynamic: summary
18
31
 
19
32
  .. image:: https://badge.fury.io/py/python-dwca-reader.svg
20
33
  :target: https://badge.fury.io/py/python-dwca-reader
@@ -12,6 +12,7 @@ import csv
12
12
  import io
13
13
  import os
14
14
  import re
15
+ from operator import itemgetter
15
16
  import xml.etree.ElementTree as ET
16
17
  from typing import Optional, List, Dict, Set
17
18
  from xml.etree.ElementTree import Element
@@ -210,6 +211,66 @@ class DataFileDescriptor(object):
210
211
  fields_terminated_by=fields_terminated_by,
211
212
  )
212
213
 
214
+ def __key(self):
215
+ """Return a tuple describing the data file layout. Common ground between equality and hash.
216
+
217
+ Deliberately excludes `raw_element`: it's an xml.etree.ElementTree.Element, which
218
+ compares by identity, so including it would make two descriptors built from the same
219
+ metafile section never compare equal. `lines_to_ignore` stands in for the only part of
220
+ `raw_element` that affects parsing (the ignoreHeaderLines attribute).
221
+
222
+ `created_from_file` is excluded too: it records where the descriptor came from, not how
223
+ the file is laid out, and its only behavioral effect is already covered by
224
+ `lines_to_ignore`. `represents_extension` is excluded as it's just the negation of
225
+ `represents_corefile`.
226
+ """
227
+ return (
228
+ self.file_location,
229
+ self.file_encoding,
230
+ self.type,
231
+ self.id_index,
232
+ self.coreid_index,
233
+ self.represents_corefile,
234
+ self.fields,
235
+ self.lines_terminated_by,
236
+ self.fields_enclosed_by,
237
+ self.fields_terminated_by,
238
+ self.lines_to_ignore,
239
+ )
240
+
241
+ def __eq__(self, other):
242
+ if not isinstance(other, DataFileDescriptor):
243
+ return NotImplemented
244
+
245
+ return self.__key() == other.__key()
246
+
247
+ def __hash__(self):
248
+ # __key() embeds `fields`, a list of dicts, which isn't hashable. Equal descriptors
249
+ # still hash equally because this is a subset of the equality key.
250
+ return hash(
251
+ (
252
+ self.file_location,
253
+ self.file_encoding,
254
+ self.type,
255
+ self.id_index,
256
+ self.coreid_index,
257
+ self.represents_corefile,
258
+ self.lines_terminated_by,
259
+ self.fields_enclosed_by,
260
+ self.fields_terminated_by,
261
+ self.lines_to_ignore,
262
+ )
263
+ )
264
+
265
+ @property
266
+ def field_plan(self) -> "FieldPlan":
267
+ """A cached :class:`FieldPlan` turning a split data row into a term -> value dict."""
268
+ plan = self.__dict__.get("_field_plan")
269
+ if plan is None:
270
+ plan = FieldPlan(self.fields, self.id_index, self.coreid_index)
271
+ self.__dict__["_field_plan"] = plan
272
+ return plan
273
+
213
274
  @property
214
275
  def terms(self) -> Set[str]:
215
276
  """Return a Python set containing all the Darwin Core terms appearing in file."""
@@ -229,9 +290,9 @@ class DataFileDescriptor(object):
229
290
  columns = {}
230
291
 
231
292
  for f in self.fields:
232
- if f[
233
- "index"
234
- ]: # Some (default values for example) don't have a corresponding col.
293
+ # Some fields (those carrying only a default value) have no column. Note the
294
+ # explicit None test: index 0 is a valid column and must not be dropped.
295
+ if f["index"] is not None:
235
296
  columns[f["index"]] = f["term"]
236
297
 
237
298
  # In addition to DwC terms, we may also have id (Core) or core_id (Extensions) columns
@@ -264,6 +325,181 @@ class DataFileDescriptor(object):
264
325
  return int(self.raw_element.get("ignoreHeaderLines", 0))
265
326
 
266
327
 
328
+ class FieldPlan(object):
329
+ """Precomputed mapping from a split CSV line to a term -> value dict.
330
+
331
+ Built once per :class:`DataFileDescriptor` (see its ``field_plan`` property) so that
332
+ parsing a row costs one C-level ``dict(zip(...))`` instead of a Python loop over the
333
+ field descriptors.
334
+ """
335
+
336
+ __slots__ = (
337
+ "_fields",
338
+ "_id_index",
339
+ "_coreid_index",
340
+ "_contiguous_terms",
341
+ "_terms",
342
+ "_getter",
343
+ "_single_column",
344
+ "_defaults",
345
+ "_constants",
346
+ "_ordered",
347
+ "required_columns",
348
+ )
349
+
350
+ def __init__(self, fields, id_index=None, coreid_index=None):
351
+ self._fields = fields
352
+ self._id_index = id_index
353
+ self._coreid_index = coreid_index
354
+
355
+ indexed = [f for f in fields if f["index"] is not None]
356
+ indexes = [f["index"] for f in indexed]
357
+
358
+ #: Number of columns a data row must have for this plan to apply.
359
+ self.required_columns = max(indexes) + 1 if indexes else 0
360
+
361
+ # Terms without a column: the value is always the default.
362
+ self._constants = tuple(
363
+ (f["term"], f["default"] or "") for f in fields if f["index"] is None
364
+ )
365
+ # Indexed terms that also carry a default, used when the cell is empty (issue #80).
366
+ self._defaults = tuple(
367
+ (f["term"], f["default"]) for f in indexed if f["default"]
368
+ )
369
+
370
+ # When some terms have no column, the fast paths below would append them after the
371
+ # mapped ones and change the key order of Row.data (which is user-visible through
372
+ # str(row)). Those archives use an ordered path instead; they are rare, and the
373
+ # ordered path is still free of the per-field try/except and int() it replaces.
374
+ self._ordered = (
375
+ tuple((f["term"], f["index"], f["default"]) for f in fields)
376
+ if self._constants
377
+ else None
378
+ )
379
+
380
+ if indexes and sorted(indexes) == list(range(len(indexes))):
381
+ # Fast path: the indexed fields cover columns 0..n-1 exactly once.
382
+ ordered = sorted(indexed, key=lambda f: f["index"])
383
+ self._contiguous_terms = tuple(f["term"] for f in ordered)
384
+ self._terms = None
385
+ self._getter = None
386
+ self._single_column = False
387
+ else:
388
+ self._contiguous_terms = None
389
+ self._terms = tuple(f["term"] for f in indexed)
390
+ self._getter = itemgetter(*indexes) if indexes else None
391
+ # itemgetter with a single argument returns a scalar, not a tuple.
392
+ self._single_column = len(indexes) == 1
393
+
394
+ def build_data(self, raw_fields):
395
+ """Return the term -> value dict for an already-split data row."""
396
+ if len(raw_fields) < self.required_columns:
397
+ self._raise_missing_column(raw_fields)
398
+
399
+ if self._ordered is not None:
400
+ data = {}
401
+ for term, index, default in self._ordered:
402
+ value = raw_fields[index] if index is not None else None
403
+ data[term] = value or default or ""
404
+ return data
405
+
406
+ if self._contiguous_terms is not None:
407
+ data = dict(zip(self._contiguous_terms, raw_fields))
408
+ elif self._getter is not None:
409
+ values = self._getter(raw_fields)
410
+ if self._single_column:
411
+ values = (values,)
412
+ data = dict(zip(self._terms, values))
413
+ else:
414
+ data = {}
415
+
416
+ for term, default in self._defaults:
417
+ if not data[term]:
418
+ data[term] = default
419
+ for term, constant in self._constants:
420
+ data[term] = constant
421
+
422
+ return data
423
+
424
+ def term_getter(self, terms):
425
+ """Return a callable mapping a split data row to a tuple of values for `terms`.
426
+
427
+ :raises ValueError: if any requested term is absent from the data file.
428
+ """
429
+ # `terms` is iterated three times below (missing, indexes, defaults). A generator or
430
+ # other one-shot iterable would be exhausted after the first pass, silently turning
431
+ # every later pass empty rather than raising - so normalise to a list once up front.
432
+ terms = list(terms)
433
+
434
+ by_term = {f["term"]: f for f in self._fields}
435
+
436
+ # "id" and "coreid" name the archive's key columns. These are the names this library
437
+ # already uses for them in headers(), short_headers() and pd_read(). A declared term of
438
+ # the same name wins, which matters for metafile-less archives whose terms are raw CSV
439
+ # header names, and keeps this consistent with row.data.
440
+ if "id" not in by_term and self._id_index is not None:
441
+ by_term["id"] = {"term": "id", "index": self._id_index, "default": None}
442
+ if "coreid" not in by_term and self._coreid_index is not None:
443
+ by_term["coreid"] = {"term": "coreid", "index": self._coreid_index, "default": None}
444
+
445
+ missing = [term for term in terms if term not in by_term]
446
+ if missing:
447
+ raise ValueError(
448
+ "These terms are not in this data file: {t}".format(
449
+ t=", ".join(sorted(missing))
450
+ )
451
+ )
452
+
453
+ indexes = tuple(by_term[term]["index"] for term in terms)
454
+ defaults = tuple(by_term[term]["default"] for term in terms)
455
+
456
+ if indexes and all(i is not None for i in indexes) and not any(defaults):
457
+ # Fast path: every term maps to a column and none has a default, so the whole
458
+ # tuple comes out of a single C-level call.
459
+ getter = itemgetter(*indexes)
460
+ required = max(indexes) + 1
461
+
462
+ if len(indexes) == 1:
463
+ # itemgetter with a single argument returns a scalar, not a tuple.
464
+ def get_one(raw_fields):
465
+ if len(raw_fields) < required:
466
+ self._raise_missing_column(raw_fields)
467
+ return (getter(raw_fields),)
468
+
469
+ return get_one
470
+
471
+ def get_many(raw_fields):
472
+ if len(raw_fields) < required:
473
+ self._raise_missing_column(raw_fields)
474
+ return getter(raw_fields)
475
+
476
+ return get_many
477
+
478
+ required = max([i for i in indexes if i is not None] or [-1]) + 1
479
+
480
+ def get_general(raw_fields):
481
+ if len(raw_fields) < required:
482
+ self._raise_missing_column(raw_fields)
483
+ return tuple(
484
+ (raw_fields[index] if index is not None else None) or default or ""
485
+ for index, default in zip(indexes, defaults)
486
+ )
487
+
488
+ return get_general
489
+
490
+ def _raise_missing_column(self, raw_fields):
491
+ # Slow path: report the same index the old per-field loop would have reported.
492
+ for field in self._fields:
493
+ index = field["index"]
494
+ if index is not None and index >= len(raw_fields):
495
+ raise InvalidArchive(
496
+ "The descriptor references a non-existent field (index={i})".format(
497
+ i=index
498
+ )
499
+ )
500
+ raise InvalidArchive("The descriptor references a non-existent field")
501
+
502
+
267
503
  class ArchiveDescriptor(object):
268
504
  """Class used to encapsulate the whole Metafile (`meta.xml`)."""
269
505