triplot 1.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,1311 @@
1
+ # The TRIOS reader. Fix it in this file; the format is docs/TRI-FORMAT.md,
2
+ # and tests/test_reader.py checks it against TRIOS's own exports of real
3
+ # runs.
4
+
5
+ """
6
+ trios_io.py -- read TA Instruments TRIOS measurements (DSC, SDT, TGA ...).
7
+
8
+ This program's one TRIOS reader. It takes either a native binary ``.tri``
9
+ or a TRIOS ``.txt`` export and returns the same structure, so the plotting
10
+ code never has to care which it was handed::
11
+
12
+ data['head'] -> metadata (filename, sample name, operator, mass ...)
13
+ data['numdata'] -> [ {prog, dims, units, nums (N x M ndarray)}, ... ]
14
+ data['analyses'] -> { step_name: { model: [ {field: value}, ... ] } }
15
+
16
+ The binary layout is documented in TRI-FORMAT.md. Read that before changing
17
+ anything here; it also records how each constant was found, so a future TRIOS
18
+ release that shifts an offset can be re-derived rather than guessed at.
19
+
20
+ Three things this reader learned the hard way, all of which bit earlier
21
+ versions and are worth keeping in mind:
22
+
23
+ 1. Segments are delimited by their *step objects* (the "Ramp 10,00 °C/min to
24
+ 250 °C" program strings), NOT by a fixed number of arrays per segment. An
25
+ analysis can attach a derived curve (a running integral, a polynomial fit)
26
+ to one segment only, and a partial final segment can record fewer signals
27
+ than the rest -- both break any fixed-stride chunking.
28
+
29
+ 2. The signal list is stored in the file, so signals are looked up BY NAME.
30
+ Index 2 is Heat Flow on a DSC25 but Sample Flow on an SDT650: hard-coding
31
+ the index silently plots gas flow as heat flow.
32
+
33
+ 3. Time is stored in seconds (TRIOS displays minutes) and Heat Flow in watts;
34
+ the "(Normalized)" signals are per gram of sample.
35
+
36
+ 4. Every array carries a 16-byte signal id, and arrays are named by it. The
37
+ final segment of a run looked as if it stored only the raw sensors: no
38
+ Temperature and no Heat Flow (see 5 for why). An earlier version guessed
39
+ that segment's signals by shape and picked Set Point Temperature, which
40
+ starts at the programmed 30 degC, not at the ~53 degC the sample had
41
+ actually cooled to. The curve came out flattened and stretched back to
42
+ 30 degC.
43
+
44
+ 5. An array can carry a FLAGS LIST, one uint32 per sample, in front of its
45
+ values; a plain array is the same layout with an empty list. A signal
46
+ with samples that hold no measurement is stored that way - the last
47
+ samples of a run, and in some runs the first ones: Temperature, Heat
48
+ Flow, Heat Flow Phase and Total Heat Capacity on a DSC25 (5 to 35
49
+ samples), Temperature, Temperature Rate, Heat Flow, Weight Corrected Heat
50
+ Flow and Temperature Difference on an SDT650 (25 to 50). Point 4's "raw
51
+ sensors only" final segment, and the indium ramp with "no heat flow",
52
+ were these arrays not being read. An earlier fix matched the flagged form
53
+ by 8 fixed bytes, 01102101 080d0200, and the 080d0200 in it is the list's
54
+ byte LENGTH, 4 + 4 * 33601: it found the arrays of a 33601-sample segment
55
+ and of no other (a 39001-sample SDT run drew its Weight in kg as the heat
56
+ flow). Flagged samples are NaN. SDT signals are also stored in SI units -
57
+ Weight in kg, Weight Corrected Heat Flow in W/kg, the gas flows in L/s -
58
+ and the file has no sample-size field: the sample mass is the reference
59
+ the Weight Change (%) is taken against, when that is a mass at all (see
60
+ `_mass_from_weight`). docs/TRI-FORMAT.md section 3 has the layout.
61
+ """
62
+ from __future__ import annotations
63
+
64
+ import re
65
+ import struct
66
+ from collections import Counter
67
+ from pathlib import Path
68
+
69
+ import numpy as np
70
+
71
+
72
+ # --------------------------------------------------------------------------- #
73
+ # Binary layout constants (see TRI-FORMAT.md)
74
+ # --------------------------------------------------------------------------- #
75
+ # Every signal is a float32 column in ONE layout, plain or flagged:
76
+ #
77
+ # <n:i32> 01 10 21 01 <size:u32> <m:i32> <m x u32 flags> 01 00 <n:i32> <n x f32>
78
+ #
79
+ # <size> is the byte length of what follows it up to the 01 00, 4 + 4 * m.
80
+ # A plain array has an empty list (m = 0, size = 4); its first 14 bytes are
81
+ # the old fixed signature 01102101 04000000 00000000 0100. A flagged array
82
+ # has one flag per sample (m = n). Nothing about the length is assumed: the
83
+ # three counts and the size are checked against each other (`_signal_array`).
84
+ VALUE_TAG = bytes.fromhex('01102101')
85
+ ARRAY_SIG = bytes.fromhex('0110210104000000000000000100') # the plain case
86
+
87
+ # The flags on a RECORDED signal (DSC25 and SDT650, TRIOS 5.1.1, 5.11 and
88
+ # 6.0): 0 on a measured sample, 0x08000008 on a sample with no measurement
89
+ # in it (stored as 0.0). A list with 0x10 in it belongs to a curve TRIOS
90
+ # CALCULATED - the points of an analysis, an SDT run's Heat Flow
91
+ # (Normalized) and Weight (%) - and those sit in the document region, where
92
+ # one read as a signal would move the start of the analysis search past
93
+ # the analyses (the reference file went from 18 analyses to 0).
94
+ FLAG_CALCULATED = 0x10
95
+
96
+ # The tag that introduces a step (segment) object, just before its program name.
97
+ STEP_TAG = bytes.fromhex('07200134')
98
+
99
+ # Program verbs that begin a step name.
100
+ STEP_WORDS = ('Ramp', 'Equilibrate', 'Isothermal', 'Modulate', 'Jump', 'Mark',
101
+ 'Repeat', 'Abort', 'Increment', 'Sampling', 'Data storage')
102
+
103
+ PRINTABLE = re.compile(rb'[ -~\xc2\xb0\xb5\xc2\xb2\xc2\xb3]{6,}')
104
+
105
+ # Analysis models whose record layout has been decoded. Everything else in a
106
+ # file is still reported (name + cursors) but without its result fields.
107
+ ONSET_MODELS = ('Onset point', 'Endset point')
108
+ INTEGRAL_MODELS = ('Peak Integration (enthalpy)',)
109
+
110
+
111
+ # --------------------------------------------------------------------------- #
112
+ # Small helpers
113
+ # --------------------------------------------------------------------------- #
114
+ def _trapz(y, x):
115
+ fn = getattr(np, 'trapezoid', None) or np.trapz
116
+ return float(fn(y, x))
117
+
118
+
119
+ def _meta_string(raw, key):
120
+ """Value of a .NET length-prefixed metadata string (single-byte length)."""
121
+ k = key if isinstance(key, bytes) else key.encode()
122
+ j = raw.find(k)
123
+ if j == -1:
124
+ return None
125
+ p = j + len(k)
126
+ ln = raw[p]
127
+ if ln >= 0x80:
128
+ return None
129
+ return raw[p + 1:p + 1 + ln].decode('utf-8', 'replace')
130
+
131
+
132
+ def _sample_mass_g(raw):
133
+ """Sample mass in grams. TRIOS stores it in mg. None when the file has
134
+ no such field (an SDT run: see `_mass_from_weight`)."""
135
+ for key in ('samplesize', 'samplemass'):
136
+ v = _meta_string(raw, key)
137
+ try:
138
+ return float(v.replace(',', '.')) / 1000.0
139
+ except (TypeError, ValueError, AttributeError):
140
+ continue
141
+ return None
142
+
143
+
144
+ def signal_list(raw, limit=400_000):
145
+ """The instrument's signal list, as stored in the file header.
146
+
147
+ TRIOS writes it as one '; '-joined run of names near the top. Returns them
148
+ in acquisition order, which is the order the float32 arrays follow."""
149
+ best = None
150
+ for m in PRINTABLE.finditer(raw, 0, limit):
151
+ s = m.group().decode('utf-8', 'replace')
152
+ if s.count('; ') >= 4 and 'Temperature' in s:
153
+ if best is None or len(s) > len(best):
154
+ best = s
155
+ if not best:
156
+ return []
157
+ return [x.strip() for x in best.split(';') if x.strip()]
158
+
159
+
160
+ # --------------------------------------------------------------------------- #
161
+ # Binary reader
162
+ # --------------------------------------------------------------------------- #
163
+ def _arrays(raw):
164
+ """Every float32 signal array: (tag_offset, values), in file order,
165
+ plain or flagged (flagged samples NaN). `tag_offset` is where the
166
+ array's leading count sits, which is what `_signal_id` counts back
167
+ from. TRIOS's later copies are still in here; `_recordings` drops
168
+ them."""
169
+ out = []
170
+ i = raw.find(VALUE_TAG, 4)
171
+ while i != -1:
172
+ values = _signal_array(raw, i)
173
+ if values is not None:
174
+ out.append((i - 4, values))
175
+ i = raw.find(VALUE_TAG, i + 1)
176
+ return out
177
+
178
+
179
+ def _signal_array(raw, i):
180
+ """The values of the array whose tag is at `i`, flagged samples NaN, or
181
+ None when the bytes there are not a recorded signal (see VALUE_TAG for
182
+ the layout).
183
+
184
+ Every length in the layout is checked against the others, so nothing
185
+ about the array's size is assumed. What is NOT a recorded signal: a
186
+ curve TRIOS calculated (its flags carry FLAG_CALCULATED), and, in most
187
+ files, TRIOS's own copy of each flagged signal of the final segment,
188
+ which has three more bytes between the count and the tag. Where a copy
189
+ has no such gap (some real runs) it is read here and `_recordings`
190
+ drops it."""
191
+ n = len(raw)
192
+ if i < 4 or i + 12 > n:
193
+ return None
194
+ count = struct.unpack_from('<i', raw, i - 4)[0]
195
+ size, m = struct.unpack_from('<Ii', raw, i + 4)
196
+ if not 0 < count < 50_000_000 or m not in (0, count) or size != 4 + 4 * m:
197
+ return None
198
+ body = i + 8 + size
199
+ if (body + 6 + 4 * count > n
200
+ or raw[body:body + 2] != b'\x01\x00'
201
+ or struct.unpack_from('<i', raw, body + 2)[0] != count):
202
+ return None
203
+ values = np.frombuffer(raw, '<f4', count=count,
204
+ offset=body + 6).astype(float)
205
+ if m:
206
+ flags = np.frombuffer(raw, '<u4', count=m, offset=i + 12)
207
+ if np.any(flags & FLAG_CALCULATED):
208
+ return None
209
+ values[flags != 0] = np.nan
210
+ return values
211
+
212
+
213
+ def _is_copy(values, earlier):
214
+ """True when `values` is TRIOS's later copy of the recording `earlier`:
215
+ one sample longer, a 1.0 in front, and the rest the same bit for bit
216
+ (NaN where the recording is flagged). Seen for every flagged signal of
217
+ the final segment, in the document region after the last segment."""
218
+ return (len(values) == len(earlier) + 1 and values[0] == 1.0
219
+ and np.array_equal(values[1:], earlier, equal_nan=True))
220
+
221
+
222
+ def _recordings(raw):
223
+ """`(arrays, copies)`: the recorded signal arrays, in file order, and how
224
+ many later copies of them were left out.
225
+
226
+ A copy carries its signal's id, so it would be named like the recording;
227
+ worse, sitting in the document region, it would move `doc_start` - the
228
+ start of the analysis search - past analyses that come before it, and a
229
+ step-name string between the last segment and it would open a segment
230
+ made of nothing but copies. So copies go before anything else looks at
231
+ the arrays."""
232
+ out, copies, last = [], 0, {}
233
+ for off, values in _arrays(raw):
234
+ sid = _signal_id(raw, off)
235
+ earlier = last.get(sid)
236
+ if earlier is not None and _is_copy(values, earlier):
237
+ copies += 1
238
+ continue
239
+ last[sid] = values
240
+ out.append((off, values))
241
+ return out, copies
242
+
243
+
244
+ def _step_name(run):
245
+ """The step name inside a printable run, without its .NET length byte.
246
+
247
+ A step name is a length-prefixed .NET string, and the length byte is
248
+ itself PRINTABLE whenever the name is 32..126 bytes long: ' ' for 32, '!'
249
+ for 33, '"' for 34 and so on, so the regex swallows it with the name. Only
250
+ ' ' and '!' used to be stripped, which worked for exactly the 32- and
251
+ 33-byte names of the files this was written on. Another run says "Ramp
252
+ 10.00 degC/min to 210.0000 degC" - 34 bytes, prefix '"' - so every
253
+ heating step was invisible and its arrays were merged into the cooling
254
+ segment before it: 3 segments read out of 7. The byte is recognised by
255
+ what it IS, the length of the name that follows it (in UTF-8 bytes, which
256
+ is how the degree sign counts two)."""
257
+ if len(run) > 1 and run[0] <= len(run) - 1:
258
+ name = run[1:1 + run[0]]
259
+ if name.decode('utf-8', 'replace').startswith(STEP_WORDS):
260
+ return name.decode('utf-8', 'replace').strip()
261
+ return run.lstrip(b'!').decode('utf-8', 'replace').strip()
262
+
263
+
264
+ def _step_objects(raw, array_offsets):
265
+ """(offset, program name) for each measured segment, in order.
266
+
267
+ Segments are delimited by their step objects. The step tag is a good hint
268
+ but not universal -- an "Isothermal 5,0 min" step can carry a different
269
+ preamble -- so a candidate is accepted when it is a step-verb string that
270
+ actually separates signal arrays. The procedure summary near the top of the
271
+ file lists every step in ONE string, which lands before the first array and
272
+ so collapses harmlessly into the opening boundary."""
273
+ if not array_offsets:
274
+ return []
275
+ first, last = array_offsets[0], array_offsets[-1]
276
+ cand = []
277
+ for m in PRINTABLE.finditer(raw, 0, last):
278
+ txt = _step_name(m.group())
279
+ if not txt.startswith(STEP_WORDS):
280
+ continue
281
+ # a name is one step, not the whole procedure listing
282
+ if txt.count(';') > 1:
283
+ continue
284
+ cand.append((m.start(), txt))
285
+ # keep the boundaries that actually have arrays on both sides, plus the
286
+ # opening one; drop near-duplicates (the same name written twice)
287
+ steps = []
288
+ for off, txt in cand:
289
+ if off > first and not any(a > off for a in array_offsets):
290
+ continue
291
+ if steps and off - steps[-1][0] < 64 and txt == steps[-1][1]:
292
+ continue
293
+ steps.append((off, txt))
294
+ # anything before the first array belongs to segment 1: keep only the last
295
+ pre = [x for x in steps if x[0] <= first]
296
+ post = [x for x in steps if x[0] > first]
297
+ return (pre[-1:] if pre else []) + post
298
+
299
+
300
+ def _cache_for_chain(raw, q, max_back=8_000_000):
301
+ """The cached analysed curve belonging to the chain whose index array
302
+ starts at ``q``.
303
+
304
+ Layout is <rows:int32><cols:int32> then rows*cols float64, and the block
305
+ ends exactly where the index array begins -- that adjacency is the link
306
+ between a cache and its analysis. Searching BACKWARDS from ``q`` for a
307
+ header satisfying ``h + 8 + rows*cols*8 == q`` finds it directly.
308
+
309
+ A forward scan that consumes blocks as it goes does NOT work: it locks onto
310
+ the first plausible header and steps over later ones, which is how an
311
+ earlier version of this reader found 4 caches where the file holds 8.
312
+
313
+ Note the cache can be SHORTER than the segment it came from -- TRIOS
314
+ stores the samples that have a heat flow, e.g. 2605 rows for a
315
+ 2640-sample segment whose last 35 heat-flow samples are flagged (NaN
316
+ here) -- so comparisons against it must be prefix-wise."""
317
+ n = len(raw)
318
+ lo = max(0, q - max_back)
319
+ for h in range(q - 8, lo, -4):
320
+ r = struct.unpack_from('<i', raw, h)[0]
321
+ c = struct.unpack_from('<i', raw, h + 4)[0]
322
+ if not (50 < r < 5_000_000 and 1 <= c <= 8):
323
+ continue
324
+ if h + 8 + r * c * 8 != q:
325
+ continue
326
+ blk = np.frombuffer(raw, '<f8', count=r * c, offset=h + 8)
327
+ if np.all(np.isfinite(blk)) and np.abs(blk).max() < 1e7:
328
+ return blk.reshape(-1, c)
329
+ return None
330
+
331
+
332
+ def _chain_caches(raw, doc_start):
333
+ """{chain_offset: cache} for every analysis chain in the file."""
334
+ out = {}
335
+ for ch in _analysis_chains(raw, doc_start):
336
+ blk = _cache_for_chain(raw, ch['chain'])
337
+ if blk is not None:
338
+ out[ch['chain']] = blk
339
+ return out
340
+
341
+
342
+ def _repair_partial(seg, caches, mass_g):
343
+ """Pin the signals of a segment whose array count does not match the list.
344
+
345
+ A partial segment records a different subset of signals, in an order that
346
+ is neither a prefix nor a fixed shift of the full list -- on the reference
347
+ file its Temperature sits at array 12 and Heat Flow at 15, with several
348
+ other arrays spanning a plausible temperature range. Guessing by shape
349
+ picks the wrong one. An analysis cache resolves it exactly: the cached
350
+ column equals the segment's own samples bit for bit."""
351
+ vals = seg['raw_signals']
352
+ for blk in caches:
353
+ # col 0 is the x channel (temperature); a later column is the analysed
354
+ # signal, stored normalized (W/g) where the raw array is in watts.
355
+ ref = blk[:, 0]
356
+ for a in vals:
357
+ if len(a) < len(ref) or not _cache_eq(ref, a):
358
+ continue
359
+ if _is_temperature(a):
360
+ seg.setdefault('pinned', {})['Temperature'] = a
361
+ if blk.shape[1] > 1 and mass_g:
362
+ sig = blk[:, 1] * mass_g
363
+ for b in vals:
364
+ if len(b) >= len(sig) and _cache_eq(sig, b):
365
+ seg['pinned']['Heat Flow'] = b
366
+ break
367
+ break
368
+ return seg.get('pinned', {})
369
+
370
+
371
+ def _is_temperature(a):
372
+ a = a[np.isfinite(a)] # a flagged array's NaN samples
373
+ return (a.size > 2 and -200.0 < float(a.min())
374
+ and float(a.max()) < 2000.0 and float(a.max() - a.min()) > 2.0)
375
+
376
+
377
+ def _signal_id(raw, tag_off):
378
+ """The 16-byte id of the signal an array holds.
379
+
380
+ It sits 38..22 bytes before the array's leading count and is the same for
381
+ one signal in every segment of a file (Temperature, Heat Flow T1, ...), so
382
+ it names an array exactly, whatever its position in the segment."""
383
+ return bytes(raw[tag_off - 38:tag_off - 22])
384
+
385
+
386
+ def _learn_ids(raw, groups, names):
387
+ """{signal id: name}, learned from the segments that store the full list.
388
+
389
+ In a full segment the first len(names) arrays follow the signal list in
390
+ order, so position gives the name. Returns None when there is no full
391
+ segment or two segments disagree -- then the ids cannot be trusted and the
392
+ caller falls back to the older position/shape logic."""
393
+ ids = {}
394
+ for g in groups:
395
+ if len(g) < len(names):
396
+ continue
397
+ row = [_signal_id(raw, off) for off, _ in g[:len(names)]]
398
+ if len(set(row)) != len(row):
399
+ return None
400
+ for sid, nm in zip(row, names):
401
+ if ids.setdefault(sid, nm) != nm:
402
+ return None
403
+ return ids if len(ids) == len(names) else None
404
+
405
+
406
+ def _learn_aliases(full, names):
407
+ """{name: [other names]} for signals that are one array stored twice.
408
+
409
+ TRIOS writes the displayed signals as copies of the sensor they come from:
410
+ Temperature is Sample Sensor Temperature and Heat Flow is the selected
411
+ Heat Flow T1, bit for bit. A partial final segment stores only the
412
+ sensors, so this is what lets its Temperature and Heat Flow be recovered
413
+ exactly. An alias is accepted only when the two arrays are identical in
414
+ EVERY full segment and actually vary (two all-zero arrays are equal
415
+ without being the same signal)."""
416
+ out = {}
417
+ for a in names:
418
+ for b in names:
419
+ if a == b:
420
+ continue
421
+ same = [np.array_equal(sg[a], sg[b]) and float(np.ptp(sg[a])) > 0
422
+ for sg in full if a in sg and b in sg]
423
+ if same and all(same):
424
+ out.setdefault(a, []).append(b)
425
+ return out
426
+
427
+
428
+ def _segments(raw):
429
+ """Split the signal arrays into per-segment records.
430
+
431
+ Boundaries come from the step objects, so a segment carrying an extra
432
+ analysis curve, or a partial final segment with a shorter signal list, is
433
+ still delimited correctly."""
434
+ # Later copies are dropped silently: they are part of the format, not a
435
+ # problem with the file, and the note line is for problems.
436
+ arrs, _copies = _recordings(raw)
437
+ if not arrs:
438
+ raise ValueError('no TRIOS signal arrays found (not a .tri?)')
439
+ names = signal_list(raw)
440
+ steps = _step_objects(raw, [off for off, _ in arrs])
441
+ if not steps: # fall back to one segment
442
+ steps = [(0, 'segment')]
443
+
444
+ bounds = [p for p, _ in steps]
445
+ groups = [[] for _ in steps]
446
+ for tag_off, vals in arrs:
447
+ k = sum(1 for b in bounds if b <= tag_off) - 1
448
+ groups[max(k, 0)].append((tag_off, vals))
449
+
450
+ doc_start = max(off for off, _ in arrs)
451
+ ids = _learn_ids(raw, groups, names) if names else None
452
+
453
+ segs = []
454
+ for g, (_, prog) in zip(groups, steps):
455
+ if not g:
456
+ continue
457
+ vals = [v for _, v in g]
458
+ if ids:
459
+ # Name every array by its signal id. Exact in every segment,
460
+ # including a partial one, which stores a different subset in a
461
+ # different order. Arrays with an unknown id are curves an
462
+ # analysis (running integral, polynomial fit) attached to the
463
+ # segment, not recorded signals. A second array with a known id
464
+ # is a copy the copy check did not catch; the first one, in
465
+ # signal-list order, is the recording.
466
+ by_name = {}
467
+ unknown = 0
468
+ for off, v in g:
469
+ nm = ids.get(_signal_id(raw, off))
470
+ if nm is None:
471
+ unknown += 1
472
+ else:
473
+ by_name.setdefault(nm, v)
474
+ sig = {nm: by_name[nm] for nm in names if nm in by_name}
475
+ exact = True
476
+ twice = len(g) - len(by_name) - unknown
477
+ if unknown:
478
+ print(f"[trios_io] segment '{prog[:34]}': dropped {unknown} "
479
+ "analysis-generated curve(s) appended after the "
480
+ "recorded signals.")
481
+ if twice:
482
+ print(f"[trios_io] segment '{prog[:34]}': {twice} signal(s) "
483
+ "stored twice; the first of each was kept.")
484
+ elif names and len(vals) >= len(names):
485
+ # No usable ids: map by position when the counts line up. Extra
486
+ # arrays at the end are analysis curves and are dropped.
487
+ sig = dict(zip(names, vals[:len(names)]))
488
+ exact = True
489
+ if len(vals) > len(names):
490
+ print(f"[trios_io] segment '{prog[:34]}': dropped "
491
+ f"{len(vals) - len(names)} analysis-generated curve(s) "
492
+ "appended after the recorded signals.")
493
+ else:
494
+ # A partial segment without ids is not a prefix of the list, so
495
+ # its signals have to be identified by shape (_resolve_signals).
496
+ sig = {}
497
+ exact = False
498
+ segs.append({'prog': prog, 'signals': sig, 'raw_signals': vals,
499
+ 'exact': exact, 'names': names, 'ids': ids})
500
+
501
+ # Fill signals a partial segment did not store from the sensor they are a
502
+ # copy of (see _learn_aliases).
503
+ full = [s['signals'] for s in segs
504
+ if s['exact'] and len(s['signals']) == len(names)]
505
+ aliases = _learn_aliases(full, names) if full else {}
506
+ for s in segs:
507
+ if not s['exact'] or len(s['signals']) == len(names):
508
+ continue
509
+ filled = []
510
+ for nm in names:
511
+ if nm in s['signals']:
512
+ continue
513
+ src = next((b for b in aliases.get(nm, ()) if b in s['signals']),
514
+ None)
515
+ if src is not None:
516
+ s['signals'][nm] = s['signals'][src]
517
+ filled.append(f'{nm} = {src}')
518
+ s['signals'] = {nm: s['signals'][nm] for nm in names
519
+ if nm in s['signals']}
520
+ missing = [nm for nm in names if nm not in s['signals']]
521
+ note = f"; filled {', '.join(filled)}" if filled else ''
522
+ gone = f"; not recorded: {', '.join(missing)}" if missing else ''
523
+ print(f"[trios_io] segment '{s['prog'][:34]}' is partial "
524
+ f"({len(s['raw_signals'])} of {len(names)} signals stored)"
525
+ f"{note}{gone}.")
526
+ return segs, doc_start
527
+
528
+
529
+ def _resolve_signals(seg, mass_g):
530
+ """Return an ordered {name: array} for one segment, plus derived columns."""
531
+ if seg['exact']:
532
+ sig = dict(seg['signals'])
533
+ else:
534
+ vals = seg['raw_signals']
535
+ sig = {}
536
+ t = vals[0]
537
+ sig['Time'] = t
538
+ pinned = seg.get('pinned') or {}
539
+ sig.update(pinned)
540
+ if 'Temperature' not in sig:
541
+ temps = [a for a in vals[1:] if _is_temperature(a)]
542
+ if temps:
543
+ sig['Temperature'] = max(temps,
544
+ key=lambda a: float(a.max() - a.min()))
545
+ if 'Heat Flow' not in sig:
546
+ hf = [a for a in vals
547
+ if a is not t and a is not sig.get('Temperature')
548
+ and a.size > 2 and float(np.abs(a).max()) < 50.0
549
+ and float(np.abs(a).mean()) < 1.0]
550
+ if hf:
551
+ sig['Heat Flow'] = max(hf, key=lambda a: float(a.std()))
552
+ how = ('pinned exactly by an analysis cache' if pinned
553
+ else 'identified by shape -- UNVERIFIED, check against a .txt '
554
+ 'export before quoting numbers from this segment')
555
+ print(f"[trios_io] segment '{seg['prog'][:34]}': {len(vals)} arrays vs "
556
+ f"{len(seg['names'])} named signals; {how}.")
557
+
558
+ out = {}
559
+ if 'Time' in sig:
560
+ out['Time'] = sig['Time'] / 60.0 # seconds -> minutes
561
+ for k, v in sig.items():
562
+ if k != 'Time':
563
+ out[k] = v * SI_TO_UNITS.get(k, 1.0)
564
+
565
+ # A normalized heat flow is what a plot actually wants on the y axis:
566
+ # watts over the sample mass, for DSC and SDT alike. TRIOS's own export
567
+ # of an SDT run says so (0.4163 W/g = 2.0107 mW / 4.830 mg); "Weight
568
+ # Corrected Heat Flow" divides by the weight LEFT at each moment
569
+ # instead, and is only the fallback when there is no mass - and not even
570
+ # then when the recorded weight is not positive, because divided by a
571
+ # negative weight it is the heat flow upside down.
572
+ if 'Heat Flow' in out and mass_g:
573
+ out['Heat Flow (Normalized)'] = out['Heat Flow'] / mass_g
574
+ elif ('Weight Corrected Heat Flow' in out
575
+ and _weight_is_positive(out.get('Weight'))):
576
+ out['Heat Flow (Normalized)'] = out['Weight Corrected Heat Flow']
577
+ return out
578
+
579
+
580
+ # Signals a .tri stores in SI that UNITS names otherwise (SDT650). Checked
581
+ # on two real runs: Heat Flow (W) / Weight (kg) equals Weight Corrected Heat
582
+ # Flow to 1e-4, so that one is W/kg; both gas flows read 1/600 L/s, the
583
+ # instrument's 100 mL/min purge.
584
+ SI_TO_UNITS = {
585
+ 'Weight': 1e6, # kg -> mg
586
+ 'Weight Corrected Heat Flow': 1e-3, # W/kg -> W/g
587
+ 'Sample Flow': 60_000.0, # L/s -> mL/min
588
+ 'Balance Flow': 60_000.0,
589
+ # A DSC25's purge is in L/s too: the reference file's Full export
590
+ # writes it in mL/min, exactly 60000 times the stored value.
591
+ 'Cell Purge': 60_000.0,
592
+ }
593
+
594
+
595
+ # How far Weight / (Weight Change / 100) may wander over a segment and still
596
+ # be ONE reference mass: 1e-4 of it. On every SDT run it was checked on
597
+ # (124 with a mass) it wanders by under 3.2e-7 of it, and it equals
598
+ # the export's Sample Mass to the 6 figures the reader writes.
599
+ MASS_SPREAD = 1e-4
600
+
601
+
602
+ def _mass_from_weight(sig):
603
+ """`(grams, why_not)`: the sample mass from an SDT segment's Weight (mg)
604
+ and Weight Change (%), which is the reference the percentage is taken
605
+ against - the same at every sample.
606
+
607
+ `(None, None)` when the segment has not both. `(None, reason)` when it
608
+ has both and they do not describe a sample mass: a ratio that is not
609
+ positive at every sample - three real runs record -99.9 mg against a
610
+ Weight Change of +99.99 %, and TRIOS's own normalised curve is upside
611
+ down with them - or one that is not constant. Golden rule 4: such a file
612
+ has NO sample mass, and the panel says so where one would be used. Both
613
+ may end slightly below zero TOGETHER when the sample is all gone
614
+ (a sample that sublimes: 16.67 to -0.19 mg, 99.98 to -1.11 %); the
615
+ ratio is still the one mass, 16.6726 mg."""
616
+ w, pct = sig.get('Weight'), sig.get('Weight Change')
617
+ if w is None or pct is None:
618
+ return None, None
619
+ ok = np.isfinite(w) & np.isfinite(pct) & (np.abs(pct) > 1.0)
620
+ if ok.sum() < 3:
621
+ return None, None
622
+ w, pct = w[ok], pct[ok]
623
+ ratio = w / (pct / 100.0)
624
+ if not np.all(ratio > 0):
625
+ return None, ('the recorded Weight ({:.6g} to {:.6g} mg) and Weight '
626
+ 'Change ({:.6g} to {:.6g} %) have opposite signs'
627
+ .format(float(w.min()), float(w.max()),
628
+ float(pct.min()), float(pct.max())))
629
+ mass = float(np.median(ratio))
630
+ spread = float(ratio.max() - ratio.min())
631
+ if spread > MASS_SPREAD * mass:
632
+ return None, ('Weight / Weight Change is not one reference mass '
633
+ '({:.6g} to {:.6g} mg)'.format(float(ratio.min()),
634
+ float(ratio.max())))
635
+ return mass / 1000.0, None
636
+
637
+
638
+ def _weight_is_positive(w):
639
+ """True when a segment's Weight (mg) is recorded and positive wherever
640
+ it is recorded."""
641
+ if w is None:
642
+ return False
643
+ w = w[np.isfinite(w)]
644
+ return bool(w.size) and bool(np.all(w > 0))
645
+
646
+
647
+ UNITS = {
648
+ 'Time': 'min', 'Temperature': '°C', 'Heat Flow': 'W',
649
+ 'Heat Flow (Normalized)': 'W/g', 'Weight': 'mg', 'Weight Change': '%',
650
+ 'Weight Corrected Heat Flow': 'W/g', 'Temperature Rate': '°C/min',
651
+ 'Sample Flow': 'mL/min', 'Balance Flow': 'mL/min',
652
+ 'Temperature Difference': '°C',
653
+ # As TRIOS's own signal list names them (the [Signal List] of an SDT
654
+ # export: "Set Point (degC)", "Power Requested (W)").
655
+ 'Set Point': '\u00b0C', 'Power Requested': 'W', 'Power Delivered': 'W',
656
+ 'Cell Purge': 'mL/min',
657
+ }
658
+
659
+
660
+ # --------------------------------------------------------------------------- #
661
+ # Analyses
662
+ # --------------------------------------------------------------------------- #
663
+ def _cache_eq(a, b):
664
+ for off in range(0, 65):
665
+ m = min(len(a), len(b) - off)
666
+ if m < 50:
667
+ break
668
+ probe = min(m, 200)
669
+ if np.abs(a[:probe] - b[off:off + probe]).max() < 1e-6:
670
+ if np.abs(a[:m] - b[off:off + m]).max() < 1e-6:
671
+ return True
672
+ return False
673
+
674
+
675
+ def _analysis_chains(raw, doc_start):
676
+ """Locate every user-added analysis in the document region.
677
+
678
+ Each is stored as a chain
679
+ [cached analysed curve (f64)] [row-index array (u32 0,1,2,...)] [record]
680
+ The record holds the cursor positions as float64 at fixed offsets; the
681
+ cached curve repeats the analysed segment's own float32 samples exactly,
682
+ which is what ties an analysis to its scan (the .txt export only names the
683
+ step *program*, and three segments can share one).
684
+
685
+ Returns [{model, cursors, stored, points, chain, variable}] in creation
686
+ order: `cursors` in the record's order (an endset's transition cursor
687
+ first), `stored` TRIOS's result at +16, `points` TRIOS's construction as
688
+ (x, y) pairs for an onset, endset or Tg (y in TRIOS's display unit of
689
+ the analysed curve, see RECORD_Y_SCALE), `variable` the analysed curve's
690
+ 16-byte signal id or None. Other results are NOT read back -- they are
691
+ recomputed from the curve by trios_analysis.
692
+ """
693
+ n = len(raw)
694
+ needle = struct.pack('<8I', *range(8))
695
+ known = ('Onset point', 'Endset point', 'Peak Integration (enthalpy)',
696
+ 'Glass transition', 'Peak height', 'Signal min', 'Signal max',
697
+ 'Signal change', 'Curve Y at X', 'Curve X at Y', 'Statistics',
698
+ 'Polynomial', 'Running Integral', 'Oxidation temperature',
699
+ 'Area under the curve', 'Find peaks')
700
+
701
+ chains = []
702
+ pos = raw.find(needle, doc_start)
703
+ while pos != -1:
704
+ k = 8
705
+ while pos + 4 * k + 4 <= n and struct.unpack_from('<I', raw, pos + 4 * k)[0] == k:
706
+ k += 1
707
+ if k >= 200:
708
+ chains.append((pos, pos + 4 * k))
709
+ pos = raw.find(needle, pos + 4 * k)
710
+
711
+ out, seen = [], set()
712
+ named = headed = 0
713
+ starts = [c[0] for c in chains] + [n]
714
+ for ci, (q, e) in enumerate(chains):
715
+ win = raw[e:min(e + 2500, starts[ci + 1])]
716
+ model = next((m for m in known if (' - ' + m).encode() in win), None)
717
+ if model is None:
718
+ continue
719
+ named += 1
720
+ # The record's float64 fields start at a FIXED offset behind a fixed
721
+ # header; see _record_start. Field layout (TRI-FORMAT.md section 5):
722
+ # onset / endset : TRIOS's construction, three (x, y) points at
723
+ # +0, +16, +32 (+16 is the RESULT), and the two
724
+ # cursors as (x, curve y) at +86 and +132 - for
725
+ # an onset the flat one first, for an endset
726
+ # the transition first
727
+ # integration : +0 first baseline cursor, +96 second
728
+ # glass transition : four (x, y) PAIRS at +0, +16, +32, +48
729
+ # the rest : +0 cursor, +132 second cursor
730
+ # +16 is deliberately not used as a cursor -- it is TRIOS's own answer.
731
+ rec = _record_start(raw, e)
732
+ if rec is None:
733
+ continue # a display copy, not the record itself
734
+ headed += 1
735
+ off1 = 0
736
+ if model.startswith('Peak Integration'):
737
+ off2 = 96
738
+ elif model == 'Glass transition':
739
+ # A Tg record is four points down the transition, not a cursor
740
+ # pair with a result between them: onset cursor, onset, end,
741
+ # end cursor, each as (x, y) at a 16-byte stride. Reading +132
742
+ # as the second cursor (the tangent-model layout) gave 0.0, which
743
+ # is why the Tg drawing could not be written before.
744
+ off2 = 48
745
+ elif model in ONSET_MODELS:
746
+ # Not +0: that is the construction's first point, which is the
747
+ # flat cursor for an onset but a point on the inflection tangent
748
+ # for an endset (the reference file: 93.8469 where the cursor
749
+ # is 93.4654).
750
+ off1, off2 = 86, 132
751
+ else:
752
+ off2 = 132
753
+ c0 = struct.unpack_from('<d', raw, rec + off1)[0]
754
+ c1 = struct.unpack_from('<d', raw, rec + off2)[0]
755
+ stored = struct.unpack_from('<d', raw, rec + 16)[0]
756
+ points = None
757
+ if model == 'Glass transition':
758
+ points = [struct.unpack_from('<2d', raw, rec + off)
759
+ for off in (0, 16, 32, 48)]
760
+ elif model in ONSET_MODELS:
761
+ points = [struct.unpack_from('<2d', raw, rec + off)
762
+ for off in (0, 16, 32)]
763
+ ok = [np.isfinite(v) and -200.0 < v < 2000.0 for v in (c0, c1)]
764
+ if model in ONSET_MODELS + INTEGRAL_MODELS and not all(ok):
765
+ print(f"[trios_io] skipped a '{model}' record whose cursors are not "
766
+ f"temperatures ({c0:.4g}, {c1:.4g}); the record layout may "
767
+ "have changed, see TRI-FORMAT.md section 5.")
768
+ continue
769
+ if not ok[1]:
770
+ c1 = float('nan') # a one-cursor model (Peak height, ...)
771
+ key = (model, round(c0, 4), None if not ok[1] else round(c1, 4))
772
+ if key in seen:
773
+ continue
774
+ seen.add(key)
775
+ out.append({'model': model, 'cursors': (c0, c1),
776
+ 'stored': stored, 'points': points, 'chain': q,
777
+ 'variable': _record_variable(raw, rec, starts[ci + 1])})
778
+ if named and not headed:
779
+ print('[trios_io] found analyses but no record with the known header; '
780
+ 'this TRIOS version may store them differently (TRI-FORMAT.md '
781
+ 'section 5). Analyses were not recovered.')
782
+ return out
783
+
784
+
785
+ # Every analysis record's float64 fields start exactly 30 bytes after its index
786
+ # array ends, behind this header (identical in TRIOS 5.1.1 and 6.0, for all 14
787
+ # analysis models tried):
788
+ # 01 00 01 00 <u32> 0f 2f 01 <u32> 10 2f 02 <u32> <u32> <u32>
789
+ # The display copies of an analysis carry a different header (24 2f 01 ...) and
790
+ # no fields. An earlier version searched byte by byte for the first pair of
791
+ # plausible temperatures instead; it matched 4 bytes early whenever a cursor's
792
+ # low mantissa bytes, read together with the header's last u32 (2), happened to
793
+ # decode as -2.0, and drew that analysis at 0 °C.
794
+ RECORD_HEAD = 30
795
+ REC_TAG0 = bytes.fromhex('01000100') # at +0
796
+ REC_TAG1 = bytes.fromhex('0f2f01') # at +8
797
+ REC_TAG2 = bytes.fromhex('102f02') # at +15
798
+
799
+
800
+ # TRIOS's own ids for the curves it CALCULATES (not in the signal list, so
801
+ # not learned from the segments like the recorded ones). The same in every
802
+ # file tried, DSC25 and SDT650, TRIOS 5.1.1 to 6.0.
803
+ # The names are this reader's: the Weight (%) curve is its "Weight Change".
804
+ CALCULATED_IDS = {
805
+ bytes.fromhex('2f85cc58bf1cb343a3f97135b826d88a'): 'Heat Flow (Normalized)',
806
+ bytes.fromhex('ba6bb3c0fdeeab47935c906a39544545'): 'Weight Change',
807
+ }
808
+
809
+ # How far behind a record's fields its point arrays may start (seen: +851 to
810
+ # +1343; the record's name strings sit in between).
811
+ POINTS_WINDOW = 4000
812
+
813
+ # A record's y values are in TRIOS's DISPLAY unit of the analysed curve, and
814
+ # this turns them into the unit of the reader's column of that name. Checked
815
+ # on every onset/endset record in the files tried, by the record's
816
+ # "curve y at the cursor" (+94, +140) against the curve there: Heat Flow
817
+ # (Normalized) in W/g (764 cursors, within 6e-4 relative, the nearest sample
818
+ # being up to 0.04 K off), Weight Change in % (366, within 2e-5), Heat Flow
819
+ # in mW (a run with no sample mass: 8 cursors, exactly 1000 x watts).
820
+ RECORD_Y_SCALE = {'Heat Flow': 1e-3} # mW -> W
821
+
822
+
823
+ def _record_variable(raw, rec, stop):
824
+ """The 16-byte signal id of the curve an analysis was made ON, or None.
825
+
826
+ Behind every record come the analysis's points as two small CALCULATED
827
+ arrays (flags 0x10, VALUE_TAG layout): x, then y, each carrying the id of
828
+ its signal. x is Temperature's id; y's id is the analysed variable - the
829
+ thing TRIOS's export calls "Analysed variables: Weight vs. Temperature".
830
+ Decoded on an SDT run where two onsets were made on the weight and the
831
+ integration on the heat flow."""
832
+ n = len(raw)
833
+ stop = min(stop, rec + POINTS_WINDOW, n)
834
+ found = []
835
+ i = raw.find(VALUE_TAG, rec)
836
+ while i != -1 and i < stop and len(found) < 2:
837
+ if i + 12 <= n:
838
+ count = struct.unpack_from('<i', raw, i - 4)[0]
839
+ size, m = struct.unpack_from('<Ii', raw, i + 4)
840
+ body = i + 8 + size
841
+ if (0 < count < 100_000 and m == count and size == 4 + 4 * m
842
+ and body + 6 <= n and raw[body:body + 2] == b'\x01\x00'
843
+ and struct.unpack_from('<i', raw, body + 2)[0] == count):
844
+ flags = np.frombuffer(raw, '<u4', count=m, offset=i + 12)
845
+ if np.all(flags & FLAG_CALCULATED):
846
+ found.append(_signal_id(raw, i - 4))
847
+ i = raw.find(VALUE_TAG, i + 1)
848
+ return found[1] if len(found) == 2 else None
849
+
850
+
851
+ def _record_start(raw, e):
852
+ """Offset of the record fields for the index array ending at ``e``, or None
853
+ when what follows is not a record header."""
854
+ if (raw[e:e + 4] == REC_TAG0
855
+ and raw[e + 8:e + 11] == REC_TAG1
856
+ and raw[e + 15:e + 18] == REC_TAG2
857
+ and e + RECORD_HEAD + 140 <= len(raw)):
858
+ return e + RECORD_HEAD
859
+ return None
860
+
861
+
862
+ def _attribute(blk, segs_xy):
863
+ """Which segment was this analysis run on?
864
+
865
+ The cache's x column repeats that segment's own float32 samples bit for
866
+ bit, so an exact (prefix-wise) comparison names the scan -- including
867
+ between repeat scans whose results differ by less than 0.02 K and which no
868
+ geometric reconstruction can separate.
869
+
870
+ Returns None for a cache-less record; the caller reuses the previous
871
+ attribution, which is the order TRIOS writes them in."""
872
+ if blk is None:
873
+ return None
874
+ ref = blk[:, 0]
875
+ for j, (T, _) in enumerate(segs_xy):
876
+ if len(T) >= len(ref) and _cache_eq(ref, T):
877
+ return j
878
+ return None
879
+
880
+
881
+ def attach_analyses(data, path, cache_by_chain=None, doc_start=None,
882
+ raw=None, ids=None):
883
+ """Recover the analyses from a .tri and recompute their results.
884
+
885
+ Populates ``data['analyses']`` in the same shape the .txt export gives, so
886
+ the annotation helpers work identically for either source. Values
887
+ are recomputed from the curve (see trios_analysis), not read back from the
888
+ binary -- only the model, the cursors and the scan attribution come from
889
+ the file. `doc_start` (where the last recorded array starts), `raw` and
890
+ `ids` (the learned {signal id: name}) are passed by `read_tri_binary`,
891
+ which has them already.
892
+
893
+ Besides the '<value> <unit>' strings, an entry can carry two keys that
894
+ are NOT text: `segment` (int, 1-based) and `construction` (a list of
895
+ [x, y] floats, see below). And `variable`, the name of the analysed
896
+ curve ('Heat Flow (Normalized)', 'Weight Change'), when the record says
897
+ it."""
898
+ if raw is None:
899
+ raw = Path(path).read_bytes()
900
+ if doc_start is None:
901
+ arrs, _copies = _recordings(raw)
902
+ if not arrs:
903
+ return data
904
+ doc_start = max(off for off, _ in arrs)
905
+ chains = _analysis_chains(raw, doc_start)
906
+ if not chains:
907
+ return data
908
+
909
+ if cache_by_chain is None:
910
+ cache_by_chain = _chain_caches(raw, doc_start)
911
+
912
+ # (numdata index, temperature) of every segment that has one; the
913
+ # attribution counts in this list, so it is mapped back to numdata
914
+ # (an index into it used to be taken for a numdata index directly).
915
+ with_t = []
916
+ for j, d in enumerate(data['numdata']):
917
+ if 'Temperature' in d['dims']:
918
+ with_t.append((j, d['nums'][:, d['dims'].index('Temperature')]))
919
+ segs_xy = [(T, None) for _j, T in with_t]
920
+
921
+ last = None
922
+ for ch in chains:
923
+ k = _attribute(cache_by_chain.get(ch['chain']), segs_xy)
924
+ j = with_t[k][0] if k is not None else None
925
+ if j is None:
926
+ j = last
927
+ else:
928
+ last = j
929
+ if j is None or j >= len(data['numdata']):
930
+ continue
931
+ d = data['numdata'][j]
932
+ i = {k: m for m, k in enumerate(d['dims'])}
933
+ if 'Time' not in i or 'Temperature' not in i:
934
+ continue
935
+ # The analysed variable, when the record's points name it: the curve
936
+ # the Python check is run on, and the unit of the construction's y.
937
+ sid = ch.get('variable')
938
+ variable = None
939
+ if sid is not None:
940
+ variable = CALCULATED_IDS.get(sid) or (ids or {}).get(sid)
941
+ y_name = variable if variable in i else (
942
+ 'Heat Flow (Normalized)' if variable is None else None)
943
+ if (ch['model'].startswith('Peak Integration')
944
+ and y_name != 'Heat Flow (Normalized)'):
945
+ y_name = None # an enthalpy in J/g needs the curve in W/g
946
+ t = d['nums'][:, i['Time']]
947
+ T = d['nums'][:, i['Temperature']]
948
+ Q = d['nums'][:, i[y_name]] if y_name in i else None
949
+ # A flagged sample is NaN, and one NaN inside a window turns every
950
+ # least-squares tangent into NaN: the check runs on the samples
951
+ # that hold a measurement.
952
+ keep = np.isfinite(t) & np.isfinite(T)
953
+ if Q is not None:
954
+ keep &= np.isfinite(Q)
955
+ Q = Q[keep]
956
+ t, T = t[keep], T[keep]
957
+ c0, c1 = ch['cursors']
958
+ info = _recompute(ch['model'], t, T, Q, c0, c1, ch.get('stored'),
959
+ ch.get('points'),
960
+ y_unit=UNITS.get(variable or 'Heat Flow (Normalized)',
961
+ ''))
962
+ info['Model'] = ch['model']
963
+ if variable:
964
+ info['variable'] = variable
965
+ if ch.get('points'):
966
+ # TRIOS's own construction, as numbers (never text: nothing
967
+ # may list it as a result). x in degC; y in the unit of the
968
+ # reader's column called `variable` - W/g, % for the weight,
969
+ # W for Heat Flow (the record has TRIOS's display unit, mW).
970
+ # Only with a known variable: without one the unit of y is
971
+ # not known either.
972
+ scale = RECORD_Y_SCALE.get(variable, 1.0)
973
+ info['construction'] = [[float(x), float(y) * scale]
974
+ for x, y in ch['points']]
975
+
976
+ info['segment'] = j + 1
977
+ data['analyses'].setdefault(d['prog'], {}) \
978
+ .setdefault(ch['model'], []).append(info)
979
+ return data
980
+
981
+
982
+ def _analysis_fn(name):
983
+ """Find an analysis routine whether it is vendored into this file or lives
984
+ in a sibling trios_analysis module. No import of `sys` -- the vendoring
985
+ step strips module headers, so this has to work on globals alone."""
986
+ fn = globals().get(name)
987
+ if fn is not None:
988
+ return fn
989
+ try:
990
+ import trios_analysis
991
+ except ImportError:
992
+ try:
993
+ from achdsc import trios_analysis
994
+ except ImportError:
995
+ return None
996
+ return getattr(trios_analysis, name, None)
997
+
998
+
999
+ def _tg_fields(T, Q, points, y_unit='W/g'):
1000
+ """Glass-transition results from the four points TRIOS stored.
1001
+
1002
+ The record holds (x, y) for the onset cursor, the ONSET, the END and the
1003
+ end cursor. The onset and end points are TRIOS's own tangent construction,
1004
+ so they are reported as they stand; the midpoint is the half-height
1005
+ crossing between them, which is what "Midpoint type: Half height" means
1006
+ and is NOT the mean of the two (0.06 K apart on the reference file).
1007
+
1008
+ Validated on the reference file (TRIOS 5.1.1, 50 K/min up-scan): the
1009
+ crossing comes out at 78.911 degC and TRIOS's own export says 78,911
1010
+ degC. The onset point's y matches the curve to 2e-5 W/g; the end point's
1011
+ y is 0.025 W/g off the curve, as it must be, because it sits on the END
1012
+ TANGENT rather than on the data.
1013
+ """
1014
+ if not points or len(points) != 4:
1015
+ return {}
1016
+ (cur0, _y0), (on_x, on_y), (end_x, end_y), (cur1, _y1) = points
1017
+ for value in (cur0, on_x, end_x, cur1):
1018
+ if not np.isfinite(value) or not -200.0 < value < 2000.0:
1019
+ return {}
1020
+ out = {'Onset cursor x': f'{cur0:.4f} °C',
1021
+ 'End cursor x': f'{cur1:.4f} °C',
1022
+ 'Onset x': f'{on_x:.4f} °C',
1023
+ 'End x': f'{end_x:.4f} °C',
1024
+ 'Step height': f'{end_y - on_y:.4f} {y_unit}'.rstrip()}
1025
+ if Q is None:
1026
+ return out
1027
+ order = np.argsort(T)
1028
+ ts, qs = np.asarray(T)[order], np.asarray(Q)[order]
1029
+ lo, hi = min(on_x, end_x), max(on_x, end_x)
1030
+ window = (ts >= lo) & (ts <= hi)
1031
+ if window.sum() >= 2:
1032
+ half = 0.5 * (on_y + end_y)
1033
+ tw, qw = ts[window], qs[window]
1034
+ if qw[-1] < qw[0]: # np.interp needs an increasing x
1035
+ tw, qw = tw[::-1], qw[::-1]
1036
+ out['Midpoint'] = f'{float(np.interp(half, qw, tw)):.4f} °C'
1037
+ return out
1038
+
1039
+
1040
+ def _python(fn, *args, **kwargs):
1041
+ """An analysis routine's result, or {} when it cannot be computed.
1042
+
1043
+ One analysis that cannot be recomputed (too few samples, a degenerate
1044
+ fit) must not cost the file ALL its analyses: `read_tri_binary` catches
1045
+ whatever `attach_analyses` raises, and loses the lot."""
1046
+ if fn is None or any(a is None for a in args):
1047
+ return {}
1048
+ try:
1049
+ with np.errstate(all='ignore'):
1050
+ return fn(*args, **kwargs) or {}
1051
+ except (ValueError, FloatingPointError, np.linalg.LinAlgError,
1052
+ IndexError, ZeroDivisionError):
1053
+ return {}
1054
+
1055
+
1056
+ def _recompute(model, t, T, Q, c0, c1, stored=None, points=None,
1057
+ y_unit='W/g'):
1058
+ """Cursor positions -> result fields, using the analysis routines.
1059
+
1060
+ Emitted as '<value> <unit>' strings so the annotation helpers parse them
1061
+ exactly as they parse a .txt export. `c0, c1` are the cursors in the
1062
+ RECORD's order (`_analysis_chains`): for an endset the transition cursor
1063
+ comes first. `Q` is None when the segment has no curve to recompute on;
1064
+ then only what the record itself holds is reported."""
1065
+ out = {}
1066
+ if model == 'Glass transition':
1067
+ fields = _tg_fields(T, Q, points, y_unit)
1068
+ if fields:
1069
+ return fields
1070
+ # No usable points: fall through to the two cursors, so the analysis
1071
+ # is still reported rather than lost.
1072
+ if model in ('Onset point', 'Endset point'):
1073
+ endset = 'End' in model
1074
+ # TRIOS's names, from its export: "Onset cursor x" is the cursor on
1075
+ # the FLAT side for both models - for an endset the one stored
1076
+ # second (the reference file: Transition cursor x 93,465, Onset
1077
+ # cursor x 116,637).
1078
+ flat, transition = (c1, c0) if endset else (c0, c1)
1079
+ if endset:
1080
+ out['Transition cursor x'] = f'{transition:.4f} °C'
1081
+ out['Onset cursor x'] = f'{flat:.4f} °C'
1082
+ else:
1083
+ out['Onset cursor x'] = f'{flat:.4f} °C'
1084
+ out['Transition cursor x'] = f'{transition:.4f} °C'
1085
+ k = 'Endset x' if endset else 'Onset x'
1086
+ # TRIOS's own answer is stored in the record at +16 and is exact, so it
1087
+ # is what gets reported. The Python reconstruction is kept alongside for
1088
+ # comparison -- it agrees to ~0.2 K on a clean step but is only an
1089
+ # approximation of TRIOS's internal tangent algorithm. It takes the
1090
+ # flat cursor first: that is the one its baseline tangent is fitted
1091
+ # at (an endset passed transition-first came out as the onset).
1092
+ if stored is not None:
1093
+ out[k] = f'{stored:.4f} °C'
1094
+ r = _python(_analysis_fn('onset_point'), T, Q, flat, transition,
1095
+ kind='endset' if endset else 'onset')
1096
+ if k in r and np.isfinite(r[k]):
1097
+ out[k + ' (python)'] = f'{r[k]:.4f} °C'
1098
+ if stored is None:
1099
+ out[k] = f'{r[k]:.4f} °C'
1100
+ elif model.startswith('Peak Integration'):
1101
+ out['Baseline cursor x'] = f'{c0:.4f} °C'
1102
+ out['Baseline cursor x1'] = f'{c1:.4f} °C'
1103
+ r = _python(_analysis_fn('peak_integration'), t, T, Q, c0, c1)
1104
+ if r:
1105
+ out['Enthalpy (normalized)'] = \
1106
+ f"{abs(r['Enthalpy (normalized)']):.4f} J/g"
1107
+ out['Peak temperature'] = f"{r['Peak temperature']:.4f} °C"
1108
+ else:
1109
+ out['Cursor x'] = f'{c0:.4f} °C'
1110
+ out['Cursor x1'] = f'{c1:.4f} °C'
1111
+ return out
1112
+
1113
+
1114
+ def read_tri_binary(path):
1115
+ """Native TRIOS ``.tri`` reader. Works for DSC, SDT and TGA files."""
1116
+ raw = Path(path).read_bytes()
1117
+ mass_g = _sample_mass_g(raw)
1118
+ mass_source = 'recorded' if mass_g else None
1119
+ segs, doc_start = _segments(raw)
1120
+
1121
+ why_not = None
1122
+ if not mass_g:
1123
+ # An SDT file has no sample-size field; its Weight Change (%) is
1124
+ # taken against the sample mass, so that is where it is read - from
1125
+ # the first segment that has both, and BEFORE a partial segment is
1126
+ # repaired, because the repair pins its heat flow by the mass.
1127
+ for seg in segs:
1128
+ if not seg['exact']:
1129
+ continue
1130
+ mass_g, why_not = _mass_from_weight(_resolve_signals(seg, None))
1131
+ if mass_g:
1132
+ mass_source = 'derived from the weight'
1133
+ if mass_g or why_not:
1134
+ break
1135
+ if not mass_g:
1136
+ print('[trios_io] no sample mass {}; heat flow left un-normalised.'
1137
+ .format('found' if not why_not else '(' + why_not + ')'))
1138
+
1139
+ cache_by_chain = _chain_caches(raw, doc_start)
1140
+ if any(not sg['exact'] for sg in segs):
1141
+ caches = list(cache_by_chain.values())
1142
+ for sg in segs:
1143
+ if not sg['exact']:
1144
+ _repair_partial(sg, caches, mass_g)
1145
+
1146
+ names = [s['prog'] for s in segs]
1147
+ progs = [f'{nm} #{j + 1}' for j, nm in enumerate(names)]
1148
+
1149
+ numdata = []
1150
+ for j, seg in enumerate(segs):
1151
+ sig = _resolve_signals(seg, mass_g)
1152
+ if not sig:
1153
+ continue
1154
+ dims = list(sig.keys())
1155
+ cols = [sig[d] for d in dims]
1156
+ m = min(len(c) for c in cols)
1157
+ numdata.append({
1158
+ 'prog': progs[j],
1159
+ 'dims': dims,
1160
+ 'units': [UNITS.get(d, '') for d in dims],
1161
+ 'nums': np.column_stack([c[:m] for c in cols]),
1162
+ })
1163
+
1164
+ head = {'Filename': Path(path).stem}
1165
+ for key in ('instrumenttype', 'samplename', 'operator', 'project',
1166
+ 'rundate', 'samplesize'):
1167
+ v = _meta_string(raw, key)
1168
+ if v:
1169
+ head[key] = v
1170
+ if mass_g:
1171
+ head['Sample Mass'] = f'{mass_g * 1000:g} mg'
1172
+ # 'recorded' (the file's sample-size field) or 'derived from the
1173
+ # weight' (an SDT run: Weight / Weight Change), so a program can say
1174
+ # which - the second is an inference, however exact.
1175
+ head['mass_source'] = mass_source
1176
+
1177
+ data = {'head': head, 'numdata': numdata, 'analyses': {}}
1178
+ try:
1179
+ attach_analyses(data, path, cache_by_chain, doc_start, raw,
1180
+ ids=segs[0].get('ids') if segs else None)
1181
+ except Exception as e: # never let this break a plot
1182
+ print(f'[trios_io] analyses not recovered: {type(e).__name__}: {e}')
1183
+ return data
1184
+
1185
+
1186
+ # --------------------------------------------------------------------------- #
1187
+ # Text reader (TRIOS .txt export)
1188
+ # --------------------------------------------------------------------------- #
1189
+ def _text_dim(name, unit):
1190
+ """An export column's name as the binary reader names that signal."""
1191
+ if name.strip() == 'Weight' and unit.strip() == '%':
1192
+ return 'Weight Change'
1193
+ return name
1194
+
1195
+
1196
+ def read_tri_text(path):
1197
+ """TRIOS ``.txt`` export -> the same structure as read_tri_binary."""
1198
+ text = Path(path).read_text(encoding='utf-8', errors='replace')
1199
+ text = text.replace('Â', '') # cp1252-read-as-utf8 mojibake
1200
+
1201
+ data = {'numdata': [], 'analyses': {}}
1202
+ matches = list(re.finditer(r'\n\[([\w\s]+)\]\n', text))
1203
+ sections = []
1204
+ if matches:
1205
+ sections.append(('head', text[:matches[0].start()]))
1206
+ for k, m in enumerate(matches):
1207
+ end = matches[k + 1].start() if k + 1 < len(matches) else len(text)
1208
+ sections.append((m.group(1), text[m.end():end]))
1209
+ else:
1210
+ sections.append(('head', text))
1211
+
1212
+ for name, body in sections:
1213
+ body = body.strip('\n')
1214
+ # TRIOS writes two different text schemas: the plain export uses
1215
+ # '[step]', the Full export '[Step]' plus '[Header]' / '[Parameters: x]'.
1216
+ if name.lower() == 'step':
1217
+ lines = body.split('\n')
1218
+ if len(lines) < 4:
1219
+ continue
1220
+ dims, units = lines[1].split('\t'), lines[2].split('\t')
1221
+ # An SDT export calls its percentage "Weight" too, with the unit
1222
+ # '%' (and some carry a second "Weight" in mg beside it). The
1223
+ # binary reader's names are the signal list's - "Weight" is mg,
1224
+ # "Weight Change" is % - and a consumer must not have to read the
1225
+ # unit to know which it was handed: one export's 99.7 % came out
1226
+ # as 99.7 mg, and as 462 % of a 21.5 mg sample.
1227
+ dims = [_text_dim(d, u) for d, u in zip(dims, units)] \
1228
+ + dims[len(units):]
1229
+ rows = []
1230
+ for ln in lines[3:]:
1231
+ parts = ln.strip().replace(',', '.').split('\t')
1232
+ if len(parts) != len(dims) or '' in parts:
1233
+ continue
1234
+ try:
1235
+ rows.append([float(p) for p in parts])
1236
+ except ValueError:
1237
+ continue
1238
+ if rows:
1239
+ data['numdata'].append({
1240
+ 'prog': lines[0], 'dims': dims, 'units': units,
1241
+ 'nums': np.array(rows, dtype=float)})
1242
+ elif name == 'Analysis':
1243
+ counts, analysis = Counter(), {}
1244
+ for ln in body.split('\n'):
1245
+ parts = ln.rstrip('\n').split('\t')
1246
+ if len(parts) < 2:
1247
+ continue
1248
+ key, val = parts[0], parts[1]
1249
+ analysis[f'{key}{counts[key]}' if counts[key] else key] = val
1250
+ counts[key] += 1
1251
+ said = analysis.get('Analysed variables', '')
1252
+ if ' vs. ' in said:
1253
+ # The analysed curve, named as the binary reader names it
1254
+ # (`attach_analyses`): the weight an SDT onset is made on is
1255
+ # TRIOS's Weight (%), the reader's "Weight Change".
1256
+ y = said.split(' vs. ', 1)[0].strip()
1257
+ analysis['variable'] = 'Weight Change' if y == 'Weight' else y
1258
+ if 'Analyzed step' in analysis and 'Model' in analysis:
1259
+ full = analysis['Analyzed step']
1260
+ prog = full.split(' - ', 1)[1] if ' - ' in full else full
1261
+ analysis['prog'] = prog
1262
+ data['analyses'].setdefault(prog, {}) \
1263
+ .setdefault(analysis['Model'], []).append(analysis)
1264
+ elif name == 'Procedure':
1265
+ kv = {}
1266
+ for ln in body.split('\n'):
1267
+ parts = ln.split('\t')
1268
+ if len(parts) == 2:
1269
+ kv[parts[0]] = parts[1]
1270
+ data['Procedure'] = kv
1271
+ else:
1272
+ kv = {}
1273
+ for ln in body.split('\n'):
1274
+ parts = ln.split('\t')
1275
+ if len(parts) == 2:
1276
+ kv[parts[0]] = parts[1]
1277
+ if name == 'head':
1278
+ data['head'] = kv
1279
+ else:
1280
+ data[name] = kv
1281
+
1282
+ # The sample mass is what a W/g axis is made of, and every consumer looks
1283
+ # for it in `head` -- but TRIOS writes it in [Procedure] in a plain export
1284
+ # and in [Sample] in a Full one, so it is folded in from wherever it
1285
+ # turned up. Without this a .txt-only file has no mass, and a plotter
1286
+ # either refuses it or invents one.
1287
+ head = data.setdefault('head', {})
1288
+ for section in ('Procedure', 'Sample'):
1289
+ block = data.get(section) or {}
1290
+ for key in ('Sample Mass', 'Sample Name', 'Pan Type', 'Project Name',
1291
+ 'Operator'):
1292
+ if block.get(key) and not head.get(key):
1293
+ head[key] = block[key]
1294
+ if head.get('Sample Mass'):
1295
+ head.setdefault('mass_source', 'recorded') # as the export says
1296
+ if not head.get('samplename'):
1297
+ head['samplename'] = (head.get('Sample name')
1298
+ or head.get('Sample Name') or '')
1299
+ return data
1300
+
1301
+
1302
+ def read_tri(path):
1303
+ """Read a TRIOS measurement: native ``.tri`` binary or ``.txt`` export.
1304
+
1305
+ Dispatch is by content, not extension -- TRIOS binaries begin with NUL
1306
+ bytes and carry an 'instrumenttype' key, the text export starts with
1307
+ 'Filename'."""
1308
+ with open(path, 'rb') as f:
1309
+ head = f.read(64)
1310
+ is_binary = b'\x00' in head[:16] or b'instrumenttype' in head
1311
+ return read_tri_binary(path) if is_binary else read_tri_text(path)