divejson 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
divejson/uddf.py ADDED
@@ -0,0 +1,1361 @@
1
+ """Reading UDDF into DiveJSON.
2
+
3
+ UDDF is the nominal incumbent, and the format most of a diver's history is trapped in.
4
+ This module reads one and produces a DiveJSON document plus a **report** of what the
5
+ source did not carry — which is half the output, not a diagnostic afterthought: a
6
+ converter that silently fills gaps produces a conforming document that lies, and §5.4 is
7
+ the rule it would be breaking.
8
+
9
+ `docs/uddf-mapping.md` is the prose companion: every element this module reads, every one
10
+ it deliberately does not, and the reasoning behind each heuristic. It is written for a
11
+ port in another language as much as for a reader of this file, so the *rules* live there
12
+ and only their implementation lives here.
13
+
14
+ Five decisions shape everything below.
15
+
16
+ **Tags are matched on their lowercased local name.** UDDF appears under at least four root
17
+ shapes — the `…/uddf/3.2/` namespace, `…/uddf/3.1/`, no namespace at all (divelogs.de and
18
+ APD DiveSight both emit a bare `<uddf>`), and an uppercase `<UDDF>` for 2.x. Stripping
19
+ `{uri}` and lowercasing at lookup time collapses all four into one code path, where the
20
+ alternative is the union XPaths Subsurface's own maintainers call unwieldy.
21
+
22
+ **Schema validity is never a precondition.** Both real third-party exports this converter
23
+ was built against fail the 3.2.2 XSD, and they fail structurally: ids carrying parentheses
24
+ or a leading space, `<link ref>`s pointing at them, empty `<latitude/>` elements where the
25
+ schema wants a float, a document with no namespace at all. Gating on validation would
26
+ reject the files the converter exists for. So: whitespace is stripped from ids and refs,
27
+ an empty element reads as absent, and children are taken **by name rather than by
28
+ position** — `diveType`'s child order changed between 3.2.1 and 3.2.2 without the
29
+ namespace moving, so any ordering assumption is wrong for half the corpus.
30
+
31
+ **A `<!DOCTYPE>` is refused outright.** UDDF has no legitimate use for one and spec §9
32
+ requires readers not to dereference anything found in a document. `ElementTree` blocks
33
+ external entities, but caps entity *amplification* only in recent libexpat — a several
34
+ hundredfold blowup still parses on older ones, and that is a library-version property
35
+ rather than a guarantee this package's `>=3.10` floor can make. Refusing the declaration
36
+ is the guarantee, and it costs nothing real.
37
+
38
+ **Identity is derived, never invented fresh.** UDDF ids are XML Names; DiveJSON requires a
39
+ UUID on every record and §5.3 asks that identifiers be stable across exports of the same
40
+ data. A UUIDv5 over a fixed namespace and `"{kind}:{source id}"` gives both. The kind is
41
+ in the hash because a source id is **not** unique within a file: every `<dive>` in a
42
+ Subsurface export reuses its enclosing `<repetitiongroup>`'s id, so hashing the bare id
43
+ would hand a dive and its group the same UUID. Where a source already emits real UUIDs —
44
+ this format's own reference implementation writes `dive-<uuid>` — they are reused, so a
45
+ round trip through UDDF comes back with the identities it left with.
46
+
47
+ **Absence is reported, never filled.** Every member the source did not record is omitted
48
+ and named in the report. The two *unit* ambiguities are the deliberate exceptions and are
49
+ not the same case: `<tankvolume>`'s cubic-metres-or-litres and `<o2>`'s
50
+ fraction-or-percent are values that **were** recorded, whose scale alone is in doubt, so a
51
+ magnitude test there interprets data rather than inventing it. Both fire loudly into the
52
+ report when they do.
53
+ """
54
+
55
+ from __future__ import annotations
56
+
57
+ import re
58
+ import sys
59
+ import uuid as uuid_pkg
60
+ import xml.etree.ElementTree as ET
61
+ from collections.abc import Iterator
62
+ from dataclasses import dataclass
63
+ from datetime import datetime
64
+ from decimal import ROUND_HALF_UP, Decimal, InvalidOperation
65
+ from pathlib import Path
66
+ from typing import Any
67
+
68
+ from . import SPEC_VERSION, __version__
69
+ from .validate import Issue, validate_document
70
+
71
+ # The producer key this converter writes its own provenance under (spec §5.5). The
72
+ # specification's own tools are the established product name here, the way `opendiving`
73
+ # is the reference implementation's.
74
+ PRODUCER_KEY = "divejson"
75
+
76
+ # uuid5(NAMESPACE_URL, "https://divejson.org/ns/uddf"). Fixed forever: changing it would
77
+ # renumber every document any released version of this converter has ever produced.
78
+ UDDF_ID_NAMESPACE = uuid_pkg.UUID("1b85a949-d5f7-5d67-9d04-dcc78342f907")
79
+
80
+ # UDDF is SI throughout and DiveJSON is not. Every one of these is a factor whose silent
81
+ # corruption produces a document that validates perfectly and is nonsense, so each is
82
+ # spelled once here and carries a hand-computed test in `tests/test_uddf_units.py`.
83
+ KELVIN_OFFSET = Decimal("273.15")
84
+ PASCAL_PER_BAR = Decimal(100_000)
85
+ LITRES_PER_CUBIC_METRE = Decimal(1000)
86
+ CENTIMETRES_PER_METRE = Decimal(100)
87
+ TENTHS_PER_UNIT = Decimal(10)
88
+
89
+ # At or above which a `<tankvolume>` is read as litres rather than the cubic metres UDDF
90
+ # specifies — see `_volume_litres`.
91
+ LITRES_THRESHOLD = Decimal(1)
92
+
93
+ # The largest magnitude a source number may have. Not a physical bound — the format sets
94
+ # none on a depth or a temperature, and inventing one here would be this module deciding
95
+ # how deep a dive can be. It is a *representability* bound: JSON numbers are doubles in
96
+ # every reader this format expects to meet, and a value past that range stops being a
97
+ # number on the way out. `json.dumps` writes an overflowed float as the bare token
98
+ # `Infinity`, which no RFC 8259 parser accepts, and a reader on a double-based parser turns
99
+ # an integer that large back into infinity — in both directions the document silently stops
100
+ # being readable, and this converter's own validation does not catch it, because
101
+ # `jsonschema` is happy to call infinity a number greater than zero.
102
+ #
103
+ # Divided by the largest factor any conversion below applies (litres, ×1000), so that
104
+ # checking the value on the way in also covers every value derived from it.
105
+ MAX_MAGNITUDE = Decimal(sys.float_info.max) / 1000
106
+
107
+ MAX_NOTES = 10_000
108
+ MAX_NAME = 255
109
+ MAX_LOCATION = 255
110
+ MAX_DISPLAY_NAME = 512
111
+ MIN_PO2_LIMIT = Decimal("0.4")
112
+ MAX_PO2_LIMIT = Decimal("2.0")
113
+ MIN_SURFACE_PRESSURE = Decimal("0.4")
114
+ MAX_SURFACE_PRESSURE = Decimal("1.2")
115
+ MAX_CYLINDER_PRESSURE = Decimal(350)
116
+ MIN_ALTITUDE = -450
117
+ MAX_ALTITUDE = 6500
118
+
119
+ _UUID_TEXT = re.compile(r"\A[0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12}\Z")
120
+
121
+ # Deliberately looser than any address grammar, and it is not trying to be one. `email` is
122
+ # the only member in this format whose *type* constrains the text a source can put in it,
123
+ # so a value here has to clear that bar or be omitted like anything else the source did not
124
+ # record. This admits every real address and rejects what writers actually leave in the
125
+ # field — `n/a`, `-`, a person's name, a sentence.
126
+ _EMAIL = re.compile(r"\A[^@\s]+@[^@\s]+\Z")
127
+
128
+ # `xs:dateTime`, leniently. The `T`, the seconds and the offset are each optional because
129
+ # real writers omit each of them, and the offset is accepted with or without its colon.
130
+ _DATE_TIME = re.compile(
131
+ r"\A(?P<date>\d{4}-\d{2}-\d{2})"
132
+ r"(?:T(?:(?P<hour>\d{2}):(?P<minute>\d{2})(?::(?P<second>\d{2}))?(?P<fraction>\.\d+)?)?)?"
133
+ r"(?P<offset>[Zz]|[+-]\d{2}:\d{2}|[+-]\d{4}|[+-]\d{2})?\Z"
134
+ )
135
+
136
+ # Where each of UDDF's typed equipment elements lands in §6.12's vocabulary. Everything
137
+ # landing on `other` does so because the vocabulary has no value for it, not because the
138
+ # source was silent: `<variouspieces>` is UDDF's own catch-all, and a scooter, a
139
+ # rebreather, a weight belt, a compressor and a watch are equipment this format does not
140
+ # yet name. Read the mapping off the rows rather than off any count of them.
141
+ _GEAR_TYPE: dict[str, str] = {
142
+ "boots": "boots",
143
+ "buoyancycontroldevice": "bcd",
144
+ "camera": "camera",
145
+ "compass": "compass",
146
+ "compressor": "other",
147
+ "divecomputer": "computer",
148
+ "fins": "fins",
149
+ "gloves": "gloves",
150
+ "knife": "knife",
151
+ "lead": "other",
152
+ "light": "light",
153
+ "mask": "mask",
154
+ "rebreather": "other",
155
+ "regulator": "regulator",
156
+ "scooter": "other",
157
+ "suit": "wetsuit",
158
+ "tank": "cylinder",
159
+ "variouspieces": "other",
160
+ "videocamera": "camera",
161
+ "watch": "other",
162
+ }
163
+
164
+ # The `<suittype>` values that mean a dry suit. Everything else it can hold — "wet-suit",
165
+ # "shorty", "half-suit", "two-piece" and the rest — leaves `type` at `_GEAR_TYPE`'s
166
+ # `wetsuit`, which is what every one of them is.
167
+ _DRYSUIT_TYPES = {"dry-suit", "drysuit", "hot-water-suit"}
168
+
169
+ # A `<setmarker>` whose text is exactly one of these is that event, rather than an
170
+ # `"other"` labelled with the word. It is what makes this format's own markers survive a
171
+ # round trip through UDDF, whose `<setmarker>` is a bare string with no type beside it.
172
+ _MARKER_TYPES = {"deep_stop", "safety_stop", "bookmark"}
173
+
174
+
175
+ class UddfError(Exception):
176
+ """The input could not be read as UDDF."""
177
+
178
+
179
+ class DoctypeRefusedError(UddfError):
180
+ """The document carries a `<!DOCTYPE>` declaration, which this reader refuses."""
181
+
182
+
183
+ class MalformedUddfError(UddfError):
184
+ """The input is not well-formed XML, or its root element is not `<uddf>`."""
185
+
186
+
187
+ class NonConformingOutputError(UddfError):
188
+ """The converter produced a document `divejson validate` rejects.
189
+
190
+ Always a bug in this module rather than a property of the source: every way a source
191
+ can be wrong is supposed to resolve to an omission and a note. It carries the issues,
192
+ so the failure names itself.
193
+ """
194
+
195
+ def __init__(self, issues: list[Issue]) -> None:
196
+ self.issues = issues
197
+ super().__init__("; ".join(str(issue) for issue in issues))
198
+
199
+
200
+ @dataclass(frozen=True, slots=True)
201
+ class Note:
202
+ """One thing the source did not carry, or that this converter had to interpret.
203
+
204
+ `where` is a path into the **source** document — `dive/0`, `dive/0/tankdata/1`,
205
+ `site/3`, or `$` for the file itself — because that is where a diver looking for the
206
+ missing value has to go. Indices are zero-based and count elements of that kind in
207
+ document order.
208
+ """
209
+
210
+ where: str
211
+ message: str
212
+
213
+ def __str__(self) -> str:
214
+ return f"{self.where}: {self.message}"
215
+
216
+
217
+ @dataclass(frozen=True, slots=True)
218
+ class Conversion:
219
+ """A converted document and everything the conversion could not carry."""
220
+
221
+ document: dict[str, Any]
222
+ notes: tuple[Note, ...]
223
+
224
+ def grouped(self) -> list[tuple[str, list[str]]]:
225
+ """Notes as `(message, wheres)`, in first-seen order.
226
+
227
+ One source habit produces one note per record — eight dives with no UTC offset are
228
+ eight notes — and a thousand-dive logbook would bury the interesting ones under
229
+ them. Grouping is a presentation concern, so it lives here rather than in the data.
230
+ """
231
+ grouped: dict[str, list[str]] = {}
232
+ for note in self.notes:
233
+ grouped.setdefault(note.message, []).append(note.where)
234
+ return list(grouped.items())
235
+
236
+
237
+ def convert_uddf(data: bytes, *, exported_at: datetime | None = None) -> Conversion:
238
+ """Convert one UDDF document into DiveJSON.
239
+
240
+ `data` is **bytes**, not text: an XML document declares its own encoding, and a UDDF
241
+ file that says `encoding="ISO-8859-1"` has to be decoded by the parser that read that
242
+ declaration. Handing `ElementTree` a `str` carrying one is a `ValueError` anyway.
243
+
244
+ `exported_at` defaults to now in the local zone. It is one of the two members the
245
+ document asserts about its own run rather than about the source (spec §4) — `generator`
246
+ is the other — so a caller producing documents in a fixed context, a test or a batch
247
+ import, should pass its own.
248
+
249
+ Raises `DoctypeRefusedError`, `MalformedUddfError` or `NonConformingOutputError`.
250
+ """
251
+ root = _parse(data)
252
+ return _Converter(root, exported_at=exported_at or datetime.now().astimezone()).run()
253
+
254
+
255
+ def convert_uddf_file(path: Path, *, exported_at: datetime | None = None) -> Conversion:
256
+ """`convert_uddf` on a file's bytes. `OSError` propagates."""
257
+ return convert_uddf(path.read_bytes(), exported_at=exported_at)
258
+
259
+
260
+ class _DoctypeRefusingTarget(ET.TreeBuilder):
261
+ """A parse target that stops the parse the moment a DTD is declared.
262
+
263
+ Raising from `doctype` aborts before expat has expanded a single entity reference in
264
+ the content, which is what makes this a bound on amplification rather than a check
265
+ performed after the damage. The hook is on the *target* rather than on the parser:
266
+ `XMLParser.parser`, which the equivalent expat handler would need, no longer exists on
267
+ Python 3.14, while this one behaves identically from 3.10 through 3.14.
268
+ """
269
+
270
+ def doctype(self, name: str, pubid: str | None, system: str | None) -> None:
271
+ raise DoctypeRefusedError(f"the document declares <!DOCTYPE {name}>, which this reader refuses (spec §9)")
272
+
273
+
274
+ def _parse(data: bytes) -> ET.Element:
275
+ parser = ET.XMLParser(target=_DoctypeRefusingTarget())
276
+ try:
277
+ parser.feed(data)
278
+ root = parser.close()
279
+ except DoctypeRefusedError:
280
+ raise
281
+ except ET.ParseError as error:
282
+ raise MalformedUddfError(f"not well-formed XML — {error}") from error
283
+ if root is None or _name(root) != "uddf":
284
+ found = "nothing" if root is None else f"<{_name(root)}>"
285
+ raise MalformedUddfError(f"the root element is {found}, not <uddf>")
286
+ return root
287
+
288
+
289
+ def _name(element: ET.Element) -> str:
290
+ """An element's local name, lowercased.
291
+
292
+ Both halves earn their place: the namespace is one of four, or absent, and UDDF 2.x
293
+ spelled its elements in upper case.
294
+ """
295
+ tag = element.tag
296
+ if not isinstance(tag, str): # a comment or a processing instruction
297
+ return ""
298
+ _, _, local = tag.rpartition("}")
299
+ return local.lower()
300
+
301
+
302
+ def _attr(element: ET.Element | None, name: str) -> str | None:
303
+ """An attribute by lowercased local name, whitespace stripped.
304
+
305
+ Subsurface writes one site id as `" ff47210"`, with the leading space, and the
306
+ `<link ref>`s pointing at it carry the space too — so stripping has to happen on both
307
+ sides, or the reference stops resolving.
308
+ """
309
+ if element is None:
310
+ return None
311
+ for key, value in element.attrib.items():
312
+ _, _, local = key.rpartition("}")
313
+ if local.lower() == name:
314
+ return value.strip() or None
315
+ return None
316
+
317
+
318
+ def _kids(element: ET.Element | None, name: str) -> list[ET.Element]:
319
+ if element is None:
320
+ return []
321
+ return [child for child in element if _name(child) == name]
322
+
323
+
324
+ def _kid(element: ET.Element | None, name: str) -> ET.Element | None:
325
+ if element is None:
326
+ return None
327
+ for child in element:
328
+ if _name(child) == name:
329
+ return child
330
+ return None
331
+
332
+
333
+ def _dig(element: ET.Element | None, *names: str) -> ET.Element | None:
334
+ for name in names:
335
+ element = _kid(element, name)
336
+ return element
337
+
338
+
339
+ def _text(element: ET.Element | None) -> str | None:
340
+ """An element's text, stripped. An empty element is absent, not an empty value.
341
+
342
+ Subsurface writes `<latitude/>` for a site it has no coordinates for, so this is the
343
+ single most load-bearing leniency in the parser.
344
+ """
345
+ if element is None or element.text is None:
346
+ return None
347
+ return element.text.strip() or None
348
+
349
+
350
+ def _text_of(parent: ET.Element | None, *names: str) -> str | None:
351
+ return _text(_dig(parent, *names))
352
+
353
+
354
+ def _decimal(text: str | None) -> Decimal | None:
355
+ """A number from element text, or `None` for anything that is not a usable one.
356
+
357
+ `Decimal` rather than `float` throughout: the input is decimal text and every scale
358
+ below is a decimal factor, so `Decimal("2.6") * 100` is exactly `260` where the float
359
+ route arrives at 260.00000000000003 and has to be rounded back out. `Decimal` also
360
+ accepts `"NaN"` and `"Infinity"` without complaint, which is what the finiteness check
361
+ is for.
362
+
363
+ **The magnitude bound is the other half of that check and is not optional.** `Decimal`
364
+ parses `1e999` and `1e999999999` happily and calls both finite, and neither survives
365
+ the trip out: the first becomes a float infinity, which this converter would write into
366
+ a document as a bare `Infinity` token no JSON parser accepts and its own validation
367
+ would not object to; the second overflows `Decimal`'s arithmetic on the next
368
+ multiplication, raising something that is not one of this module's errors and so
369
+ abandoning a whole batch mid-migration rather than failing the one file. Text that
370
+ cannot be carried as a number is treated as text that is not a number, which is what
371
+ it is.
372
+ """
373
+ if text is None:
374
+ return None
375
+ try:
376
+ value = Decimal(text)
377
+ except InvalidOperation:
378
+ return None
379
+ # `copy_abs`, not `abs`: the builtin is a context operation and raises `Overflow` on
380
+ # exactly the values this line exists to reject, so the guard would be the thing that
381
+ # crashed. `copy_abs` and the comparison below both leave the context alone.
382
+ if not value.is_finite() or value.copy_abs() > MAX_MAGNITUDE:
383
+ return None
384
+ return value
385
+
386
+
387
+ def _rounded(value: Decimal) -> int:
388
+ """The nearest integer, halves away from zero.
389
+
390
+ Python's own `round` is half-to-even, which is the right default for statistics and
391
+ the wrong one for a reading: 2.5 seconds of elapsed time is 3, not 2.
392
+ """
393
+ return int(value.to_integral_value(rounding=ROUND_HALF_UP))
394
+
395
+
396
+ def _integer(value: Decimal | None) -> int | None:
397
+ return None if value is None else _rounded(value)
398
+
399
+
400
+ def _is_uuid(text: str) -> bool:
401
+ return bool(_UUID_TEXT.match(text))
402
+
403
+
404
+ def _embedded_uuid(source_id: str) -> str | None:
405
+ """The UUID a source id already carries, if it carries one.
406
+
407
+ `xs:ID` is an `NCName` and cannot begin with a digit, which a hex UUID regularly does,
408
+ so a writer holding real UUIDs prefixes them — this format's reference implementation
409
+ writes `dive-019fec36-…`. Stripping a short alphabetic prefix recovers it, which is
410
+ what lets a logbook survive a round trip through UDDF with its identities intact.
411
+ """
412
+ head, dash, tail = source_id.partition("-")
413
+ if dash and head.isalpha() and len(head) <= 12 and _is_uuid(tail):
414
+ return tail.lower()
415
+ return source_id.lower() if _is_uuid(source_id) else None
416
+
417
+
418
+ def _volume_litres(raw: Decimal) -> tuple[Decimal, bool]:
419
+ """A `<tankvolume>` as litres, and whether it had to be reinterpreted.
420
+
421
+ UDDF specifies cubic metres — "not in litres, as UDDF uses SI units!" — and Subsurface
422
+ wrote litres into the field until 2025-09-30, which is in no tagged release but is in
423
+ the 6.0.x master builds people actually run. Both spellings are live in the installed
424
+ base and both are schema-valid, so no validator settles it and a fixture cannot either.
425
+
426
+ The magnitude does: a cubic metre of water capacity is a thousand-litre cylinder, and a
427
+ litre-valued 0.012 would be twelve millilitres. The same `< 1` threshold is what
428
+ Bubbletrail's UDDF importer uses.
429
+ """
430
+ if raw < LITRES_THRESHOLD:
431
+ return raw * LITRES_PER_CUBIC_METRE, False
432
+ return raw, True
433
+
434
+
435
+ def _gas_percent(raw: Decimal) -> tuple[Decimal, bool]:
436
+ """An `<o2>`/`<he>` as a percentage, and whether it had to be reinterpreted.
437
+
438
+ The UDDF documentation calls it "a real number less or equal 1.0 in percent", which
439
+ contradicts itself, and writers took both readings: pre-2017 Subsurface wrote
440
+ `<o2>34</o2>` where current writers write `0.34`. A value at or below 1 is the
441
+ documented fraction — 1.0 being pure oxygen, since a 1 % mix is not a breathing gas —
442
+ and anything above it was already a percentage.
443
+ """
444
+ if raw <= 1:
445
+ return raw * 100, False
446
+ return raw, True
447
+
448
+
449
+ def _date_time(text: str) -> tuple[str | None, str | None]:
450
+ """A DiveJSON date-time from `xs:dateTime` text, plus what had to be forgiven.
451
+
452
+ Returns `(value, note)`; `value` is `None` when the text is not a date and time at all.
453
+
454
+ **The offset is preserved exactly as recorded, and never supplied.** That is §5.2's
455
+ whole point, and converting to UTC — or assuming an offset where the source recorded
456
+ none — is the failure every tested UDDF consumer produced.
457
+
458
+ The forgiven shapes are real writer output rather than hypotheticals: Subsurface emits
459
+ a midnight dive as `<datetime>2002-06-18T</datetime>`, its XSLT building the string
460
+ with an unguarded `concat`, and a bare date is what the same bug produces one character
461
+ earlier.
462
+ """
463
+ match = _DATE_TIME.match(text.strip())
464
+ if match is None:
465
+ return None, None
466
+
467
+ parts = match.groupdict()
468
+ note = None
469
+ if parts["hour"] is None:
470
+ note = f"{text.strip()!r} records a date with no time of day; read as midnight"
471
+ hour, minute, second = "00", "00", "00"
472
+ else:
473
+ hour, minute, second = parts["hour"], parts["minute"], parts["second"]
474
+ if second is None:
475
+ note = f"{text.strip()!r} records no seconds; read as :00"
476
+ second = "00"
477
+
478
+ try:
479
+ datetime.fromisoformat(f"{parts['date']}T{hour}:{minute}:{second}")
480
+ except ValueError:
481
+ return None, None
482
+
483
+ value = f"{parts['date']}T{hour}:{minute}:{second}{parts['fraction'] or ''}"
484
+ offset = parts["offset"]
485
+ if offset is None:
486
+ return value, note
487
+ if offset in ("Z", "z"):
488
+ return value + offset, note
489
+ if len(offset) == 3: # +02
490
+ return f"{value}{offset}:00", note
491
+ if len(offset) == 5: # +0200
492
+ return f"{value}{offset[:3]}:{offset[3:]}", note
493
+ return value + offset, note
494
+
495
+
496
+ def _has_offset(value: str) -> bool:
497
+ _, _, time_part = value.partition("T")
498
+ return time_part.endswith(("Z", "z")) or "+" in time_part or "-" in time_part
499
+
500
+
501
+ class _Converter:
502
+ def __init__(self, root: ET.Element, *, exported_at: datetime) -> None:
503
+ self.root = root
504
+ self.exported_at = exported_at
505
+ self.notes: list[Note] = []
506
+ self.claimed: dict[str, str] = {}
507
+ self.source_ids: set[str] = set()
508
+ self.site_uuids: dict[str, str] = {}
509
+ self.trip_uuids: dict[str, str] = {}
510
+ self.gear_uuids: dict[str, str] = {}
511
+ self.mixes: dict[str, dict[str, Any]] = {}
512
+
513
+ # -- reporting ---------------------------------------------------------------
514
+
515
+ def note(self, where: str, message: str) -> None:
516
+ self.notes.append(Note(where, message))
517
+
518
+ # -- identity ----------------------------------------------------------------
519
+
520
+ def uuid_for(self, kind: str, source_id: str | None, where: str, index: int) -> str | None:
521
+ """A stable UUID for one source record, or `None` when it collides irreparably."""
522
+ if source_id is None:
523
+ self.note(
524
+ where,
525
+ f"the source gives this {kind} no id, so its identity is derived from its position in the "
526
+ "file and will move if the file's order changes (spec §5.3)",
527
+ )
528
+ source_id = f"#{index}"
529
+
530
+ derived = str(uuid_pkg.uuid5(UDDF_ID_NAMESPACE, f"{kind}:{source_id}"))
531
+ for candidate in dict.fromkeys((_embedded_uuid(source_id) or derived, derived)):
532
+ if candidate not in self.claimed:
533
+ self.claimed[candidate] = where
534
+ return candidate
535
+
536
+ self.note(
537
+ where,
538
+ f"a second {kind} carries the id {source_id!r}, already used by {self.claimed[derived]}; the "
539
+ "record is dropped, because two records cannot share one identity (spec §5.3)",
540
+ )
541
+ return None
542
+
543
+ # -- text --------------------------------------------------------------------
544
+
545
+ def capped(self, value: str, limit: int, where: str, member: str) -> str:
546
+ if len(value) <= limit:
547
+ return value
548
+ self.note(where, f"{member} is {len(value)} characters; the format caps it at {limit} and the rest is dropped")
549
+ return value[:limit]
550
+
551
+ def email(self, value: str | None, where: str) -> str | None:
552
+ """`<contact><email>` when it is an address, and nothing when it is not.
553
+
554
+ Every other source string reaches a member the format types as free text, where the
555
+ only limit is a length this converter caps. `email` is the exception — the schema
556
+ types it as an email address, so a `-` or an `n/a` is a value the member cannot
557
+ hold. Without this the whole conversion fails on it: the output would not validate,
558
+ which this module treats as its own bug, so one unusable header field would discard
559
+ an entire logbook instead of costing it one member and a line in the report.
560
+ """
561
+ if value is None or _EMAIL.match(value):
562
+ return value
563
+ self.note(where, f"the recorded email {value!r} is not an address; read as no email recorded")
564
+ return None
565
+
566
+ def notes_text(self, parent: ET.Element | None, where: str) -> str | None:
567
+ """A `<notes>` block as one string. Its `<link>` children carry no note text."""
568
+ notes = _kid(parent, "notes")
569
+ if notes is None:
570
+ return None
571
+ paragraphs = [text for text in (_text(para) for para in _kids(notes, "para")) if text]
572
+ if not paragraphs:
573
+ return None
574
+ return self.capped("\n\n".join(paragraphs), MAX_NOTES, where, "the note")
575
+
576
+ # -- geometry ----------------------------------------------------------------
577
+
578
+ def position(self, geography: ET.Element | None, where: str) -> dict[str, float] | None:
579
+ """A Position from `<geography>`, or `None` where there is not an honest one.
580
+
581
+ Two sources of nothing, and the second is the interesting one. An empty
582
+ `<latitude/>` is Subsurface saying it has no coordinates. An exact `0.000000` pair
583
+ is divelogs.de saying the same thing in the one spelling that looks like an answer:
584
+ every site in its export carries it, and a reader that trusts Null Island pins a
585
+ Red Sea wreck into the Atlantic.
586
+ """
587
+ latitude = _decimal(_text_of(geography, "latitude"))
588
+ longitude = _decimal(_text_of(geography, "longitude"))
589
+ if latitude is None or longitude is None:
590
+ if latitude is not None or longitude is not None:
591
+ self.note(where, "only one half of a coordinate pair was recorded, and a position needs both; dropped (spec §6)")
592
+ return None
593
+ if latitude == 0 and longitude == 0:
594
+ self.note(
595
+ where,
596
+ "the coordinates are exactly 0.000000 / 0.000000, which writers emit to mean 'unknown'; read as "
597
+ "no position rather than as Null Island",
598
+ )
599
+ return None
600
+ if not -90 <= latitude <= 90 or not -180 <= longitude <= 180:
601
+ self.note(where, f"the coordinates {latitude} / {longitude} are outside the WGS 84 range; dropped")
602
+ return None
603
+ return {"latitude": float(latitude), "longitude": float(longitude)}
604
+
605
+ # -- the run -----------------------------------------------------------------
606
+
607
+ def run(self) -> Conversion:
608
+ for element in self.root.iter():
609
+ source_id = _attr(element, "id")
610
+ if source_id is not None:
611
+ self.source_ids.add(source_id)
612
+
613
+ # Order matters: the dives resolve links into the tables the four calls above it
614
+ # fill, and the diver is last only so its identity yields to a real record's on
615
+ # the vanishingly rare id collision.
616
+ self.read_mixes()
617
+ sites = self.read_sites()
618
+ trips = self.read_trips()
619
+ gear = self.read_gear()
620
+ dives = self.read_dives()
621
+ diver = self.read_diver()
622
+
623
+ document: dict[str, Any] = {
624
+ "format": "divejson",
625
+ "version": SPEC_VERSION,
626
+ "exported_at": self.exported_at.isoformat(timespec="seconds"),
627
+ "generator": {"name": "divejson convert", "version": __version__},
628
+ }
629
+ if diver:
630
+ document["diver"] = diver
631
+ for member, rows in (("dives", dives), ("trips", trips), ("sites", sites), ("gear", gear)):
632
+ if rows:
633
+ document[member] = rows
634
+ document["extensions"] = {PRODUCER_KEY: self.provenance()}
635
+
636
+ issues = validate_document(document)
637
+ if issues:
638
+ raise NonConformingOutputError(issues)
639
+ return Conversion(document, tuple(self.notes))
640
+
641
+ def provenance(self) -> dict[str, Any]:
642
+ """What the source file said about itself.
643
+
644
+ Under a producer key rather than in `generator`, which §4 defines as what produced
645
+ *this* document — and that is the converter. The source's own identity is worth
646
+ keeping and has nowhere in the core vocabulary to go.
647
+ """
648
+ provenance: dict[str, Any] = {"converted_from": "uddf"}
649
+ version = _attr(self.root, "version")
650
+ if version:
651
+ provenance["uddf_version"] = version
652
+ generator = _kid(self.root, "generator")
653
+ name = _text_of(generator, "name")
654
+ if name:
655
+ source: dict[str, str] = {"name": name}
656
+ source_version = _text_of(generator, "version")
657
+ if source_version:
658
+ source["version"] = source_version
659
+ provenance["source_generator"] = source
660
+ return provenance
661
+
662
+ # -- diver -------------------------------------------------------------------
663
+
664
+ def read_diver(self) -> dict[str, Any] | None:
665
+ """The logbook's owner, or nothing at all.
666
+
667
+ §6.1 is explicit that a converter whose source records nothing about an owner omits
668
+ the member entirely — minting identity for one would be §5.4's fabrication applied
669
+ to people. `<owner id>` is deliberately not read as a name or a handle: it is an
670
+ XML id, and Subsurface's is the literal string "owner".
671
+ """
672
+ owner = _dig(self.root, "diver", "owner")
673
+ if owner is None:
674
+ return None
675
+ where = "diver"
676
+ names = [
677
+ text
678
+ for text in (_text_of(owner, "personal", part) for part in ("firstname", "middlename", "lastname"))
679
+ if text
680
+ ]
681
+ email = self.email(_text_of(owner, "contact", "email"), where)
682
+ if not names and not email:
683
+ self.note(where, "the source records nothing about the logbook's owner; no diver is written (spec §6.1)")
684
+ return None
685
+
686
+ diver: dict[str, Any] = {}
687
+ claimed = self.uuid_for("diver", _attr(owner, "id"), where, 0)
688
+ if claimed is not None:
689
+ diver["uuid"] = claimed
690
+ if names:
691
+ diver["name"] = self.capped(" ".join(names), MAX_NAME, where, "the diver's name")
692
+ if email:
693
+ diver["email"] = email
694
+ return diver
695
+
696
+ # -- sites -------------------------------------------------------------------
697
+
698
+ def read_sites(self) -> list[dict[str, Any]]:
699
+ sites: list[dict[str, Any]] = []
700
+ for index, element in enumerate(_kids(_kid(self.root, "divesite"), "site")):
701
+ where = f"site/{index}"
702
+ name = _text_of(element, "name")
703
+ if not name:
704
+ self.note(
705
+ where,
706
+ "the site has no name, which the format requires of one; it is dropped along with the "
707
+ "references to it, because a name cannot be invented (spec §6.10)",
708
+ )
709
+ continue
710
+ claimed = self.uuid_for("site", _attr(element, "id"), where, index)
711
+ if claimed is None:
712
+ continue
713
+
714
+ site: dict[str, Any] = {"uuid": claimed, "name": self.capped(name, MAX_NAME, where, "the site name")}
715
+ geography = _kid(element, "geography")
716
+ location = _text_of(geography, "location")
717
+ if location:
718
+ site["location"] = self.capped(location, MAX_LOCATION, where, "the site location")
719
+ position = self.position(geography, where)
720
+ if position:
721
+ site["position"] = position
722
+ notes = self.notes_text(element, where)
723
+ if notes:
724
+ site["notes"] = notes
725
+
726
+ source_id = _attr(element, "id")
727
+ if source_id:
728
+ self.site_uuids[source_id] = claimed
729
+ sites.append(site)
730
+ return sites
731
+
732
+ # -- trips -------------------------------------------------------------------
733
+
734
+ def read_trips(self) -> list[dict[str, Any]]:
735
+ """`<divetrip><trip>` as Trip records.
736
+
737
+ A `<trippart>` becomes a Trip Location: UDDF models a trip as a sequence of parts,
738
+ each with its own place and dates, and §6.9's location list is the nearest thing
739
+ this format has. The trip's own dates are the span of its parts, because `tripType`
740
+ records none of its own.
741
+ """
742
+ trips: list[dict[str, Any]] = []
743
+ for index, element in enumerate(_kids(_kid(self.root, "divetrip"), "trip")):
744
+ where = f"trip/{index}"
745
+ name = _text_of(element, "name")
746
+ if not name:
747
+ self.note(where, "the trip has no name, which the format requires of one; it is dropped (spec §6.8)")
748
+ continue
749
+
750
+ starts, ends, locations, notes = self.read_trip_parts(element, where)
751
+ if starts is None:
752
+ self.note(
753
+ where,
754
+ "the trip records no dates, and the format requires a start date; it is dropped along with "
755
+ "the dives' membership of it (spec §6.8)",
756
+ )
757
+ continue
758
+ claimed = self.uuid_for("trip", _attr(element, "id"), where, index)
759
+ if claimed is None:
760
+ continue
761
+
762
+ trip: dict[str, Any] = {"uuid": claimed, "name": self.capped(name, MAX_NAME, where, "the trip name")}
763
+ if locations:
764
+ trip["locations"] = locations
765
+ trip["starts_on"] = starts
766
+ if ends is not None and ends >= starts:
767
+ trip["ends_on"] = ends
768
+ elif ends is not None:
769
+ self.note(where, f"the trip ends on {ends}, before it starts on {starts}; the end date is dropped")
770
+ if notes:
771
+ trip["notes"] = notes
772
+
773
+ source_id = _attr(element, "id")
774
+ if source_id:
775
+ self.trip_uuids[source_id] = claimed
776
+ trips.append(trip)
777
+ return trips
778
+
779
+ def read_trip_parts(
780
+ self, element: ET.Element, where: str
781
+ ) -> tuple[str | None, str | None, list[dict[str, Any]], str | None]:
782
+ starts: list[str] = []
783
+ ends: list[str] = []
784
+ locations: list[dict[str, Any]] = []
785
+ paragraphs: list[str] = []
786
+
787
+ for part_index, part in enumerate(_kids(element, "trippart")):
788
+ part_where = f"{where}/trippart/{part_index}"
789
+ date_of_trip = _kid(part, "dateoftrip")
790
+ for attribute, collected in (("startdate", starts), ("enddate", ends)):
791
+ raw = _attr(date_of_trip, attribute)
792
+ if raw is None:
793
+ continue
794
+ value, _ = _date_time(raw)
795
+ if value is None:
796
+ self.note(part_where, f"{attribute} is {raw!r}, which is not a date; dropped")
797
+ else:
798
+ collected.append(value[:10])
799
+
800
+ geography = _kid(part, "geography")
801
+ part_name = _text_of(part, "name")
802
+ display_name = _text_of(geography, "location")
803
+ if part_name:
804
+ location: dict[str, Any] = {"name": self.capped(part_name, MAX_NAME, part_where, "the trip part's name")}
805
+ if display_name and display_name != part_name:
806
+ location["display_name"] = self.capped(display_name, MAX_DISPLAY_NAME, part_where, "the location")
807
+ position = self.position(geography, part_where)
808
+ if position:
809
+ location["position"] = position
810
+ locations.append(location)
811
+ elif geography is not None:
812
+ self.note(
813
+ part_where,
814
+ "the trip part has no name, which the format requires of a location; the place is dropped "
815
+ "(spec §6.9)",
816
+ )
817
+
818
+ part_notes = self.notes_text(part, part_where)
819
+ if part_notes:
820
+ paragraphs.append(part_notes)
821
+
822
+ joined = self.capped("\n\n".join(paragraphs), MAX_NOTES, where, "the trip note") if paragraphs else None
823
+ return (min(starts) if starts else None), (max(ends) if ends else None), locations, joined
824
+
825
+ # -- gear --------------------------------------------------------------------
826
+
827
+ def read_gear(self) -> list[dict[str, Any]]:
828
+ """The owner's kit list, from `<diver><owner><equipment>`.
829
+
830
+ `<equipmentconfiguration>` is skipped along with anything else unrecognized: it
831
+ describes how the pieces are rigged together, not a piece.
832
+ """
833
+ equipment = _dig(self.root, "diver", "owner", "equipment")
834
+ # `is not None`, never a truth test: an `Element` with no children is falsy today
835
+ # and `ElementTree` warns that it will not be.
836
+ pieces = [child for child in equipment if _name(child) in _GEAR_TYPE] if equipment is not None else []
837
+ gear: list[dict[str, Any]] = []
838
+ for index, element in enumerate(pieces):
839
+ kind = _name(element)
840
+ where = f"gear/{index}"
841
+ name = _text_of(element, "name")
842
+ if not name:
843
+ self.note(
844
+ where,
845
+ f"the <{kind}> has no name, which the format requires of a gear item; it is dropped "
846
+ "(spec §6.12)",
847
+ )
848
+ continue
849
+ claimed = self.uuid_for("gear", _attr(element, "id"), where, index)
850
+ if claimed is None:
851
+ continue
852
+
853
+ item: dict[str, Any] = {"uuid": claimed, "name": self.capped(name, MAX_NAME, where, "the gear name")}
854
+ brand = _text_of(element, "manufacturer", "name")
855
+ if brand:
856
+ item["brand"] = self.capped(brand, MAX_NAME, where, "the brand")
857
+ gear_type = _GEAR_TYPE[kind]
858
+ if kind == "suit" and (_text_of(element, "suittype") or "").lower() in _DRYSUIT_TYPES:
859
+ gear_type = "drysuit"
860
+ item["type"] = gear_type
861
+ notes = self.notes_text(element, where)
862
+ if notes:
863
+ item["notes"] = notes
864
+
865
+ source_id = _attr(element, "id")
866
+ if source_id:
867
+ self.gear_uuids[source_id] = claimed
868
+ gear.append(item)
869
+ return gear
870
+
871
+ # -- gases -------------------------------------------------------------------
872
+
873
+ def read_mixes(self) -> None:
874
+ """`<gasdefinitions><mix>` into a table the dives' `<tankdata>` link into.
875
+
876
+ Gas mixes are not a DiveJSON collection: the format carries the blend on the
877
+ cylinder that held it, so a mix nothing links to travels nowhere. A gas list with
878
+ no cylinder using it is a plan rather than a dive, which is why that is not
879
+ reported here.
880
+ """
881
+ for index, element in enumerate(_kids(_kid(self.root, "gasdefinitions"), "mix")):
882
+ source_id = _attr(element, "id")
883
+ if source_id is None:
884
+ continue
885
+ where = f"mix/{index}"
886
+ mix: dict[str, Any] = {}
887
+ for member, tag in (("oxygen", "o2"), ("helium", "he")):
888
+ raw = _decimal(_text_of(element, tag))
889
+ if raw is None:
890
+ continue
891
+ percent, reinterpreted = _gas_percent(raw)
892
+ if reinterpreted:
893
+ self.note(
894
+ where,
895
+ f"<{tag}> is {raw}, above the 1.0 the documentation describes; read as {percent} percent "
896
+ "rather than as a fraction",
897
+ )
898
+ if not 0 <= percent <= 100:
899
+ self.note(where, f"<{tag}> reads as {percent} percent, outside the 0 to 100 a fraction can be; dropped")
900
+ continue
901
+ mix[member] = float(percent)
902
+ if mix.get("oxygen", 0.0) + mix.get("helium", 0.0) > 100:
903
+ self.note(where, "oxygen and helium sum above 100 percent, which no mix can; both are dropped (spec §6.3)")
904
+ mix.pop("oxygen", None)
905
+ mix.pop("helium", None)
906
+
907
+ po2_limit = _decimal(_text_of(element, "maximumpo2"))
908
+ if po2_limit is not None:
909
+ if MIN_PO2_LIMIT <= po2_limit <= MAX_PO2_LIMIT:
910
+ mix["po2_limit"] = float(po2_limit)
911
+ else:
912
+ self.note(where, f"<maximumpo2> is {po2_limit} bar, outside the 0.4 to 2.0 the format allows; dropped")
913
+ self.mixes[source_id] = mix
914
+
915
+ # -- dives -------------------------------------------------------------------
916
+
917
+ def dive_elements(self) -> list[ET.Element]:
918
+ """Every `<dive>` under `<profiledata>`, in document order.
919
+
920
+ Taken *through* the repetition groups rather than from them: a group is a surface
921
+ interval's worth of dives and carries nothing this format records, so it is a
922
+ container the converter walks past.
923
+ """
924
+ profiledata = _kid(self.root, "profiledata")
925
+ return [dive for group in _kids(profiledata, "repetitiongroup") for dive in _kids(group, "dive")]
926
+
927
+ def read_dives(self) -> list[dict[str, Any]]:
928
+ dives = []
929
+ for index, element in enumerate(self.dive_elements()):
930
+ dive = self.read_dive(element, index)
931
+ if dive is not None:
932
+ dives.append(dive)
933
+ return dives
934
+
935
+ def read_dive(self, element: ET.Element, index: int) -> dict[str, Any] | None:
936
+ where = f"dive/{index}"
937
+ before = _kid(element, "informationbeforedive")
938
+ after = _kid(element, "informationafterdive")
939
+
940
+ started_at = self.read_started_at(before, where)
941
+ if started_at is None:
942
+ return None
943
+ claimed = self.uuid_for("dive", _attr(element, "id"), where, index)
944
+ if claimed is None:
945
+ return None
946
+
947
+ dive: dict[str, Any] = {"uuid": claimed}
948
+ number = _integer(_decimal(_text_of(before, "divenumber")))
949
+ if number is not None:
950
+ dive["dive_number"] = number
951
+ dive["started_at"] = started_at
952
+
953
+ duration = _integer(_decimal(_text_of(after, "diveduration")))
954
+ if duration is not None:
955
+ if duration > 0:
956
+ dive["duration"] = duration
957
+ else:
958
+ self.note(where, f"<diveduration> is {duration} seconds; the format records a duration only when it is positive")
959
+
960
+ notes = self.notes_text(after, where)
961
+ if notes:
962
+ dive["notes"] = notes
963
+
964
+ max_depth = self.positive(_decimal(_text_of(after, "greatestdepth")), where, "<greatestdepth>")
965
+ avg_depth = self.positive(_decimal(_text_of(after, "averagedepth")), where, "<averagedepth>")
966
+ if max_depth is not None and avg_depth is not None and avg_depth > max_depth:
967
+ self.note(
968
+ where,
969
+ f"the mean depth {avg_depth} m is deeper than the greatest depth {max_depth} m, which cannot be; "
970
+ "the mean is dropped rather than either being adjusted to fit (spec §6.2)",
971
+ )
972
+ avg_depth = None
973
+ if max_depth is not None:
974
+ dive["max_depth"] = float(max_depth)
975
+ if avg_depth is not None:
976
+ dive["avg_depth"] = float(avg_depth)
977
+
978
+ lowest = _decimal(_text_of(after, "lowesttemperature"))
979
+ if lowest is not None:
980
+ dive["bottom_temperature"] = float(lowest - KELVIN_OFFSET)
981
+
982
+ visibility = _decimal(_text_of(after, "visibility"))
983
+ if visibility is not None:
984
+ if visibility >= 0:
985
+ dive["visibility"] = float(visibility)
986
+ else:
987
+ self.note(where, f"<visibility> is {visibility} m; dropped")
988
+
989
+ used = _kid(before, "equipmentused")
990
+ weight = _decimal(_text_of(used, "leadquantity"))
991
+ if weight is not None:
992
+ if weight >= 0:
993
+ # Kept when it is zero: a recorded "no lead" is a fact about the dive, and
994
+ # absence is how "we do not know" is spelled here (spec §6.2).
995
+ dive["weight"] = float(weight)
996
+ else:
997
+ self.note(where, f"<leadquantity> is {weight} kg; dropped")
998
+
999
+ altitude = _integer(_decimal(_text_of(before, "altitude")))
1000
+ if altitude is not None:
1001
+ if MIN_ALTITUDE <= altitude <= MAX_ALTITUDE:
1002
+ dive["altitude"] = altitude
1003
+ else:
1004
+ self.note(where, f"<altitude> is {altitude} m, outside the -450 to 6500 the format allows; dropped")
1005
+
1006
+ surface_pressure = _decimal(_text_of(before, "surfacepressure"))
1007
+ if surface_pressure is not None:
1008
+ bar = surface_pressure / PASCAL_PER_BAR
1009
+ if MIN_SURFACE_PRESSURE <= bar <= MAX_SURFACE_PRESSURE:
1010
+ dive["surface_pressure"] = float(bar)
1011
+ else:
1012
+ self.note(where, f"<surfacepressure> reads as {bar} bar, outside the 0.4 to 1.2 the format allows; dropped")
1013
+
1014
+ trip_uuid = self.reference(_attr(_kid(before, "tripmembership"), "ref"), self.trip_uuids, where, "trip")
1015
+ if trip_uuid:
1016
+ dive["trip_uuid"] = trip_uuid
1017
+ site_uuids = self.references([_attr(link, "ref") for link in _kids(before, "link")], self.site_uuids, where, "dive site")
1018
+ if site_uuids:
1019
+ dive["site_uuids"] = site_uuids
1020
+ gear_uuids = self.references([_attr(link, "ref") for link in _kids(used, "link")], self.gear_uuids, where, "gear item")
1021
+ if gear_uuids:
1022
+ dive["gear_uuids"] = gear_uuids
1023
+
1024
+ cylinders, mix_refs = self.read_cylinders(element, where)
1025
+ profile, needs_gas_numbers = self.read_profile(element, where, mix_refs)
1026
+ if needs_gas_numbers:
1027
+ for gas_number, cylinder in enumerate(cylinders):
1028
+ cylinder["gas_number"] = gas_number
1029
+ self.note(
1030
+ where,
1031
+ "UDDF records no gas numbering, so the dive's cylinders are numbered from 0 in the order the "
1032
+ "file lists them, to tie each pressure channel and gas switch to its cylinder (spec §6.5)",
1033
+ )
1034
+ if cylinders:
1035
+ dive["cylinders"] = cylinders
1036
+ if profile:
1037
+ dive["profile"] = profile
1038
+ return dive
1039
+
1040
+ def read_started_at(self, before: ET.Element | None, where: str) -> str | None:
1041
+ raw = _text_of(before, "datetime")
1042
+ if raw is None:
1043
+ self.note(
1044
+ where,
1045
+ "the dive records no <datetime>, and the format requires a start time; the dive is dropped "
1046
+ "(spec §6.2)",
1047
+ )
1048
+ return None
1049
+ started_at, forgiven = _date_time(raw)
1050
+ if started_at is None:
1051
+ self.note(where, f"<datetime> is {raw!r}, which is not a date and time; the dive is dropped (spec §6.2)")
1052
+ return None
1053
+ if forgiven:
1054
+ self.note(where, forgiven)
1055
+ if not _has_offset(started_at):
1056
+ self.note(where, "the source recorded no UTC offset on the dive's start time; the wall clock travels alone (spec §5.2)")
1057
+ return started_at
1058
+
1059
+ def positive(self, value: Decimal | None, where: str, member: str) -> Decimal | None:
1060
+ """A measurement the format records only when it is above zero.
1061
+
1062
+ Zero is what this format's own UDDF writer emits for a depth it never had —
1063
+ `<greatestdepth>` is mandatory in UDDF and optional here — so reading it back as a
1064
+ measurement would turn "not recorded" into "the surface".
1065
+ """
1066
+ if value is None or value > 0:
1067
+ return value
1068
+ self.note(where, f"{member} is {value}, which the format records only when positive; read as not recorded")
1069
+ return None
1070
+
1071
+ def reference(self, ref: str | None, table: dict[str, str], where: str, kind: str) -> str | None:
1072
+ resolved = self.references([ref], table, where, kind)
1073
+ return resolved[0] if resolved else None
1074
+
1075
+ def references(self, refs: list[str | None], table: dict[str, str], where: str, kind: str) -> list[str]:
1076
+ """Resolved references, in source order, without repeats.
1077
+
1078
+ Order is meaningful — §5.3 makes the first `site_uuids` entry the primary site —
1079
+ and the schema forbids the same uuid twice in one list.
1080
+ """
1081
+ resolved: list[str] = []
1082
+ for ref in refs:
1083
+ if ref is None:
1084
+ continue
1085
+ if ref in table:
1086
+ if table[ref] not in resolved:
1087
+ resolved.append(table[ref])
1088
+ elif ref not in self.source_ids:
1089
+ self.note(where, f"a link points at {ref!r}, which nothing in the file defines; the reference is dropped")
1090
+ elif ref not in self.mixes:
1091
+ # A `<link>` under `informationbeforedive` addresses a site here, but the
1092
+ # schema lets it address a buddy or a shop too, and one under
1093
+ # `<equipmentused>` addresses a piece of kit. A reference to a record this
1094
+ # converter carries nowhere is worth a note; a gas reference is not.
1095
+ self.note(where, f"a link points at {ref!r}, which is not a {kind} this converter carries; the reference is dropped")
1096
+ return resolved
1097
+
1098
+ # -- cylinders ---------------------------------------------------------------
1099
+
1100
+ def read_cylinders(self, element: ET.Element, where: str) -> tuple[list[dict[str, Any]], list[str | None]]:
1101
+ """A dive's `<tankdata>` as Cylinders, plus the mix each one links to.
1102
+
1103
+ The second return value is what the profile needs: `<tankpressure ref>` and
1104
+ `<switchmix ref>` both address a *mix*, so tying a channel to its cylinder means
1105
+ going back through the link that cylinder made.
1106
+ """
1107
+ cylinders: list[dict[str, Any]] = []
1108
+ mix_refs: list[str | None] = []
1109
+ for index, tank in enumerate(_kids(element, "tankdata")):
1110
+ tank_where = f"{where}/tankdata/{index}"
1111
+ cylinder: dict[str, Any] = {}
1112
+
1113
+ raw_volume = _decimal(_text_of(tank, "tankvolume"))
1114
+ if raw_volume is None:
1115
+ self.note(
1116
+ tank_where,
1117
+ "the source records no cylinder size; the cylinder carries its gas and pressures without "
1118
+ "one (spec §6.3)",
1119
+ )
1120
+ else:
1121
+ litres, reinterpreted = _volume_litres(raw_volume)
1122
+ if reinterpreted:
1123
+ self.note(
1124
+ tank_where,
1125
+ f"<tankvolume> is {raw_volume}, too large to be the cubic metres UDDF specifies; read as "
1126
+ f"{litres} litres, which is how some builds of Subsurface write it",
1127
+ )
1128
+ if litres > 0:
1129
+ cylinder["volume"] = float(litres)
1130
+ else:
1131
+ self.note(tank_where, f"<tankvolume> reads as {litres} litres; the format records a size only when positive")
1132
+
1133
+ start = self.pressure_bar(_text_of(tank, "tankpressurebegin"), tank_where, "<tankpressurebegin>")
1134
+ end = self.pressure_bar(_text_of(tank, "tankpressureend"), tank_where, "<tankpressureend>")
1135
+ if start is not None and start == 0:
1136
+ # §6.3 is explicit: a recorded zero start pressure is a device's
1137
+ # absent-marker rather than a measurement, and writers must not emit it.
1138
+ self.note(
1139
+ tank_where,
1140
+ "the start pressure is 0 bar, which devices write to mean 'not recorded'; read as not "
1141
+ "recorded (spec §6.3)",
1142
+ )
1143
+ start = None
1144
+ if start is not None and end is not None and end > start:
1145
+ self.note(
1146
+ tank_where,
1147
+ f"the end pressure {end} bar is above the start pressure {start} bar, which cannot be; the "
1148
+ "end pressure is dropped",
1149
+ )
1150
+ end = None
1151
+ if start is not None:
1152
+ cylinder["start_pressure"] = float(start)
1153
+ if end is not None:
1154
+ cylinder["end_pressure"] = float(end)
1155
+
1156
+ mix_ref = _attr(_kid(tank, "link"), "ref")
1157
+ if mix_ref is None:
1158
+ self.note(tank_where, "the source records no gas for this cylinder; absent means not recorded, never air (spec §6.3)")
1159
+ elif mix_ref in self.mixes:
1160
+ cylinder.update(self.mixes[mix_ref])
1161
+ else:
1162
+ self.note(
1163
+ tank_where,
1164
+ f"the cylinder links to the gas {mix_ref!r}, which <gasdefinitions> does not define; its mix "
1165
+ "is not recorded",
1166
+ )
1167
+
1168
+ cylinders.append(cylinder)
1169
+ mix_refs.append(mix_ref)
1170
+ return cylinders, mix_refs
1171
+
1172
+ def pressure_bar(self, text: str | None, where: str, member: str) -> Decimal | None:
1173
+ value = _decimal(text)
1174
+ if value is None:
1175
+ return None
1176
+ bar = value / PASCAL_PER_BAR
1177
+ if not 0 <= bar <= MAX_CYLINDER_PRESSURE:
1178
+ self.note(where, f"{member} reads as {bar} bar, outside the 0 to 350 the format allows; dropped")
1179
+ return None
1180
+ return bar
1181
+
1182
+ # -- profile -----------------------------------------------------------------
1183
+
1184
+ def read_profile(
1185
+ self, element: ET.Element, where: str, mix_refs: list[str | None]
1186
+ ) -> tuple[dict[str, Any] | None, bool]:
1187
+ """`<samples><waypoint>` as a Profile, and whether the dive needs gas numbers.
1188
+
1189
+ UDDF puts every reading taken at one instant inside one `<waypoint>`; DiveJSON
1190
+ splits them into channels sampled on their own axes. So the waypoints set the time
1191
+ axis and each channel takes only the waypoints that actually carried a reading for
1192
+ it — which is why a converted Subsurface dive keeps 431 depth samples and 29
1193
+ temperatures rather than padding the second to match the first.
1194
+ """
1195
+ samples = _kid(element, "samples")
1196
+ if samples is None:
1197
+ return None, False
1198
+
1199
+ timed: list[tuple[int, ET.Element]] = []
1200
+ for index, waypoint in enumerate(_kids(samples, "waypoint")):
1201
+ second = _integer(_decimal(_text_of(waypoint, "divetime")))
1202
+ if second is None:
1203
+ self.note(
1204
+ f"{where}/waypoint/{index}",
1205
+ "the waypoint records no <divetime>, so it has no place on the profile's time axis; dropped",
1206
+ )
1207
+ elif second < 0:
1208
+ self.note(f"{where}/waypoint/{index}", f"the waypoint is at {second} s, before the dive began; dropped")
1209
+ else:
1210
+ timed.append((second, waypoint))
1211
+
1212
+ timed.sort(key=lambda pair: pair[0])
1213
+ ordered: list[tuple[int, ET.Element]] = []
1214
+ for second, waypoint in timed:
1215
+ if ordered and ordered[-1][0] == second:
1216
+ self.note(
1217
+ where,
1218
+ f"two waypoints share the second {second}; the later one is dropped, because the format's "
1219
+ "sample times are strictly increasing (spec §6.5)",
1220
+ )
1221
+ continue
1222
+ ordered.append((second, waypoint))
1223
+ if not ordered:
1224
+ return None, False
1225
+
1226
+ cylinders_of_mix: dict[str, list[int]] = {}
1227
+ for index, ref in enumerate(mix_refs):
1228
+ if ref is not None:
1229
+ cylinders_of_mix.setdefault(ref, []).append(index)
1230
+
1231
+ depth: dict[str, list[int]] = {"times": [], "values": []}
1232
+ temperature: dict[str, list[int]] = {"times": [], "values": []}
1233
+ pressures: dict[int, dict[str, list[int]]] = {}
1234
+ events: list[dict[str, Any]] = []
1235
+ needs_gas_numbers = False
1236
+
1237
+ for second, waypoint in ordered:
1238
+ metres = _decimal(_text_of(waypoint, "depth"))
1239
+ if metres is not None:
1240
+ depth["times"].append(second)
1241
+ depth["values"].append(_rounded(metres * CENTIMETRES_PER_METRE))
1242
+
1243
+ kelvin = _decimal(_text_of(waypoint, "temperature"))
1244
+ if kelvin is not None:
1245
+ temperature["times"].append(second)
1246
+ temperature["values"].append(_rounded((kelvin - KELVIN_OFFSET) * TENTHS_PER_UNIT))
1247
+
1248
+ for cylinder_index, tenths in self.waypoint_pressures(waypoint, where, second, cylinders_of_mix, len(mix_refs)):
1249
+ channel = pressures.setdefault(cylinder_index, {"times": [], "values": []})
1250
+ if channel["times"] and channel["times"][-1] == second:
1251
+ self.note(where, f"two tank pressures at {second} s resolve to the same cylinder; the later one is dropped")
1252
+ continue
1253
+ needs_gas_numbers = True
1254
+ channel["times"].append(second)
1255
+ channel["values"].append(tenths)
1256
+
1257
+ marker = _text_of(waypoint, "setmarker")
1258
+ if marker is not None:
1259
+ if marker in _MARKER_TYPES:
1260
+ events.append({"time": second, "type": marker})
1261
+ else:
1262
+ events.append({"time": second, "type": "other", "label": marker})
1263
+
1264
+ switch = _kid(waypoint, "switchmix")
1265
+ if switch is not None:
1266
+ event: dict[str, Any] = {"time": second, "type": "gas_switch"}
1267
+ ref = _attr(switch, "ref")
1268
+ if ref is not None and ref in cylinders_of_mix:
1269
+ event["gas_number"] = cylinders_of_mix[ref][0]
1270
+ needs_gas_numbers = True
1271
+ elif ref is not None:
1272
+ self.note(
1273
+ where,
1274
+ f"a gas switch at {second} s names the gas {ref!r}, which no cylinder on this dive links "
1275
+ "to; the switch is kept without saying what it was to (spec §6.6)",
1276
+ )
1277
+ events.append(event)
1278
+
1279
+ if not (depth["times"] or temperature["times"] or pressures or events):
1280
+ # Waypoints whose every reading was unusable are not a profile. Emitting the
1281
+ # bare `duration: 0` the members below would leave behind asserts a sampled
1282
+ # record of zero length, which is a thing the source did not say.
1283
+ #
1284
+ # Reported, unlike a dive that simply has no `<samples>`: the source *did*
1285
+ # record a profile here, and this is the converter unable to carry it. That is
1286
+ # the same class as a dropped waypoint or a dropped coordinate pair, and every
1287
+ # one of those says so.
1288
+ subject = "waypoint carries" if len(ordered) == 1 else "waypoints carry"
1289
+ self.note(
1290
+ where,
1291
+ f"the dive's {len(ordered)} {subject} a time but no reading this format can hold, so it "
1292
+ "arrives with no profile at all rather than one of zero length",
1293
+ )
1294
+ return None, False
1295
+
1296
+ latest = max(
1297
+ (channel["times"][-1] for channel in (depth, temperature, *pressures.values()) if channel["times"]),
1298
+ default=0,
1299
+ )
1300
+ profile: dict[str, Any] = {"duration": latest}
1301
+ if depth["times"]:
1302
+ profile["depth"] = depth
1303
+ if temperature["times"]:
1304
+ profile["temperature"] = temperature
1305
+ if pressures:
1306
+ profile["pressures"] = [
1307
+ {"times": channel["times"], "values": channel["values"], "gas_number": cylinder_index}
1308
+ for cylinder_index, channel in sorted(pressures.items())
1309
+ ]
1310
+ if events:
1311
+ events.sort(key=lambda event: event["time"])
1312
+ profile["events"] = events
1313
+ return profile, needs_gas_numbers
1314
+
1315
+ def waypoint_pressures(
1316
+ self,
1317
+ waypoint: ET.Element,
1318
+ where: str,
1319
+ second: int,
1320
+ cylinders_of_mix: dict[str, list[int]],
1321
+ tank_count: int,
1322
+ ) -> Iterator[tuple[int, int]]:
1323
+ """Each `<tankpressure>` on one waypoint, as `(cylinder index, tenths of a bar)`.
1324
+
1325
+ Two cylinders on one blend link the same `<mix>`, so a reference resolves to a
1326
+ *list* of cylinders and repeated references on one waypoint take them in order —
1327
+ the sidemount pair the format's own §6.3 describes, whose two channels would
1328
+ otherwise collapse onto one cylinder.
1329
+
1330
+ `@ref` is optional, the UDDF documentation noting that a linked double measured at
1331
+ one pressure may omit it, so a reference-less reading is taken as the dive's
1332
+ cylinder when there is exactly one and dropped when there is a choice to get wrong.
1333
+ """
1334
+ seen: dict[str, int] = {}
1335
+ for element in _kids(waypoint, "tankpressure"):
1336
+ pascal = _decimal(_text(element))
1337
+ if pascal is None:
1338
+ continue
1339
+ ref = _attr(element, "ref")
1340
+ if ref is None:
1341
+ if tank_count != 1:
1342
+ self.note(
1343
+ where,
1344
+ f"a tank pressure at {second} s names no cylinder, and the dive has {tank_count}; the "
1345
+ "reading is dropped rather than guessed onto one",
1346
+ )
1347
+ continue
1348
+ index = 0
1349
+ else:
1350
+ candidates = cylinders_of_mix.get(ref, [])
1351
+ position = seen.get(ref, 0)
1352
+ seen[ref] = position + 1
1353
+ if position >= len(candidates):
1354
+ self.note(
1355
+ where,
1356
+ f"a tank pressure at {second} s names the gas {ref!r}, which no further cylinder on this "
1357
+ "dive links to; the reading is dropped",
1358
+ )
1359
+ continue
1360
+ index = candidates[position]
1361
+ yield index, _rounded(pascal / PASCAL_PER_BAR * TENTHS_PER_UNIT)