triplot 1.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
dscpanel/core/chem.py ADDED
@@ -0,0 +1,252 @@
1
+ """A skeletal structure from a SMILES: the atoms and bonds to draw.
2
+
3
+ RDKit lays the molecule out in 2D (CoordGen when it has it, which draws
4
+ rings and chains the way a chemist would), and this hands back plain data -
5
+ elements, positions, hydrogens, charges, bond orders - which the plot draws
6
+ itself with its own painter (`ui/plot.py`). Drawn that way a structure is
7
+ VECTOR in every export, its bond width and label size are settings, and its
8
+ labels can stay upright when the structure is rotated.
9
+ RDKit's own drawing would give a picture, none of that.
10
+
11
+ The layout is STORED with the figure, so a session with a structure opens
12
+ without RDKit; RDKit is needed only to make a new one. It is optional: see
13
+ `available`.
14
+
15
+ UI-free: no Qt here.
16
+ """
17
+
18
+ import math
19
+ import re
20
+
21
+ #: The characters a SMILES is made of. Anything else - a space, a comma in
22
+ #: running text - means the clipboard holds words, not a structure.
23
+ _SMILES = re.compile(r"^[A-Za-z0-9@+\-\[\]\(\)=#$:/\\%.*]+$")
24
+
25
+
26
+ def available():
27
+ """True when RDKit is there to lay a structure out."""
28
+ try:
29
+ from rdkit import Chem # noqa: F401
30
+ from rdkit.Chem import rdDepictor # noqa: F401
31
+ except Exception:
32
+ return False
33
+ return True
34
+
35
+
36
+ #: One SMILES token: a bracket atom, a two-letter halogen, an organic-
37
+ #: subset atom (aromatic in lower case), a bond, a branch, a ring closure.
38
+ _TOKEN = re.compile(r"\[[^\[\]]+\]|Br|Cl|[BCNOPSFI]|[bcnops]|\*"
39
+ r"|[-=#$:/\\.]|[()]|%\d\d|\d")
40
+ _ATOM = re.compile(r"\[[^\[\]]+\]|Br|Cl|[BCNOPSFI]|[bcnops]|\*")
41
+
42
+
43
+ def plausible_smiles(text):
44
+ """True when `text` reads as a SMILES of two atoms or more, WITHOUT
45
+ RDKit: every character belongs to a token of the grammar, brackets and
46
+ branches balance, and every ring-closure number is opened and closed.
47
+ Not proof - RDKit alone parses it - but enough to tell "O=C(O)CCCCC(O)=O"
48
+ from a word, which is what the program must know to say RDKit is
49
+ missing (pasting one otherwise did nothing visible)."""
50
+ text = str(text or "").strip()
51
+ if not text or "\n" in text or not _SMILES.match(text):
52
+ return False
53
+ tokens = _TOKEN.findall(text)
54
+ if "".join(tokens) != text:
55
+ return False
56
+ depth, rings, atoms = 0, {}, 0
57
+ for token in tokens:
58
+ if token == "(":
59
+ depth += 1
60
+ elif token == ")":
61
+ depth -= 1
62
+ if depth < 0:
63
+ return False
64
+ elif token.isdigit() or token.startswith("%"):
65
+ rings[token] = rings.get(token, 0) + 1
66
+ elif _ATOM.fullmatch(token):
67
+ atoms += 1
68
+ return (depth == 0 and atoms >= 2
69
+ and all(count % 2 == 0 for count in rings.values()))
70
+
71
+
72
+ def looks_like_smiles(text):
73
+ """True when `text` is one line that parses as a molecule of at least
74
+ two atoms. A single atom ("C", "N") is far more likely a letter somebody
75
+ copied than a structure."""
76
+ text = str(text or "").strip()
77
+ if not text or "\n" in text or not _SMILES.match(text):
78
+ return False
79
+ molecule = _parse(text)
80
+ return molecule is not None and molecule.GetNumAtoms() >= 2
81
+
82
+
83
+ def _parse(smiles):
84
+ """The RDKit molecule, or None - quietly: RDKit prints a parse error for
85
+ every string that is not a SMILES, and most pasted text is not."""
86
+ if not available():
87
+ return None
88
+ from rdkit import Chem, RDLogger
89
+ RDLogger.DisableLog("rdApp.*")
90
+ try:
91
+ return Chem.MolFromSmiles(str(smiles))
92
+ except Exception:
93
+ return None
94
+ finally:
95
+ RDLogger.EnableLog("rdApp.*")
96
+
97
+
98
+ def layout(smiles):
99
+ """`{"atoms": [...], "bonds": [...]}` for a SMILES, or None.
100
+
101
+ Atoms: `{"el", "x", "y", "h", "charge", "show"}`, positions in BOND
102
+ LENGTHS (the average bond is 1) with y UP, centred on the origin.
103
+ `show` is False for a carbon that is a vertex, as in a skeletal
104
+ formula. Bonds: `{"a", "b", "order", "ring", "stereo"}`, `ring` the
105
+ centre of the smallest ring a double bond is in (its second line is
106
+ drawn towards it), or None; `stereo` "wedge" (towards the viewer) or
107
+ "hash" (away) for a bond RDKit wedges at a stereocentre of the SMILES
108
+ (`@` / `@@`), with `a` the stereocentre - the narrow end - or None.
109
+ """
110
+ molecule = _parse(smiles)
111
+ if molecule is None or molecule.GetNumAtoms() == 0:
112
+ return None
113
+ from rdkit import Chem
114
+ from rdkit.Chem import rdDepictor
115
+ try:
116
+ rdDepictor.SetPreferCoordGen(True)
117
+ except Exception:
118
+ pass
119
+ rdDepictor.Compute2DCoords(molecule)
120
+ # Which single bond at each stereocentre is drawn as a wedge or a hash,
121
+ # chosen by RDKit from the layout. It puts the stereocentre first in
122
+ # each such bond.
123
+ try:
124
+ Chem.WedgeMolBonds(molecule, molecule.GetConformer())
125
+ except Exception:
126
+ pass
127
+ try:
128
+ Chem.Kekulize(molecule, clearAromaticFlags=True)
129
+ except Exception:
130
+ pass
131
+ conformer = molecule.GetConformer()
132
+ points = [conformer.GetAtomPosition(i)
133
+ for i in range(molecule.GetNumAtoms())]
134
+ lengths = [math.hypot(points[b.GetBeginAtomIdx()].x
135
+ - points[b.GetEndAtomIdx()].x,
136
+ points[b.GetBeginAtomIdx()].y
137
+ - points[b.GetEndAtomIdx()].y)
138
+ for b in molecule.GetBonds()]
139
+ unit = (sum(lengths) / len(lengths)) if lengths else 1.0
140
+ unit = unit or 1.0
141
+ cx = sum(p.x for p in points) / len(points)
142
+ cy = sum(p.y for p in points) / len(points)
143
+ atoms = []
144
+ for atom, point in zip(molecule.GetAtoms(), points):
145
+ symbol = atom.GetSymbol()
146
+ charge = atom.GetFormalCharge()
147
+ shown = (symbol != "C" or charge != 0 or atom.GetDegree() == 0
148
+ or atom.GetIsotope() != 0)
149
+ atoms.append({"el": symbol, "x": (point.x - cx) / unit,
150
+ "y": (point.y - cy) / unit,
151
+ "h": int(atom.GetTotalNumHs()), "charge": int(charge),
152
+ "show": bool(shown)})
153
+ _separate_fragments(molecule, atoms)
154
+ rings = [list(ring) for ring in molecule.GetRingInfo().AtomRings()]
155
+ bonds = []
156
+ for bond in molecule.GetBonds():
157
+ a, b = bond.GetBeginAtomIdx(), bond.GetEndAtomIdx()
158
+ order = {1.0: 1, 2.0: 2, 3.0: 3}.get(bond.GetBondTypeAsDouble(), 1)
159
+ ring = None
160
+ if order == 2:
161
+ holding = [r for r in rings if a in r and b in r]
162
+ if holding:
163
+ smallest = min(holding, key=len)
164
+ ring = [sum(atoms[i]["x"] for i in smallest) / len(smallest),
165
+ sum(atoms[i]["y"] for i in smallest) / len(smallest)]
166
+ stereo = None
167
+ if order == 1:
168
+ stereo = {Chem.BondDir.BEGINWEDGE: "wedge",
169
+ Chem.BondDir.BEGINDASH: "hash"}.get(bond.GetBondDir())
170
+ bonds.append({"a": a, "b": b, "order": order, "ring": ring,
171
+ "stereo": stereo})
172
+ return {"atoms": atoms, "bonds": bonds}
173
+
174
+
175
+ def _separate_fragments(molecule, atoms):
176
+ """Put the pieces of a salt or a solvate side by side, in the order the
177
+ SMILES names them: RDKit's layout can drop a counter-ion onto the atom
178
+ it balances (Na+ on a carboxylate's O-). A bond length and a half
179
+ apart, level with the first piece; then everything centred again."""
180
+ from rdkit import Chem
181
+ pieces = Chem.GetMolFrags(molecule)
182
+ if len(pieces) < 2:
183
+ return
184
+ right = None
185
+ level = None
186
+ for piece in pieces:
187
+ xs = [atoms[i]["x"] for i in piece]
188
+ ys = [atoms[i]["y"] for i in piece]
189
+ middle = (min(ys) + max(ys)) / 2.0
190
+ if right is None:
191
+ right, level = max(xs), middle
192
+ continue
193
+ shift_x = right + 1.5 - min(xs)
194
+ shift_y = level - middle
195
+ for i in piece:
196
+ atoms[i]["x"] += shift_x
197
+ atoms[i]["y"] += shift_y
198
+ right = max(atoms[i]["x"] for i in piece)
199
+ cx = sum(a["x"] for a in atoms) / len(atoms)
200
+ cy = sum(a["y"] for a in atoms) / len(atoms)
201
+ for atom in atoms:
202
+ atom["x"] -= cx
203
+ atom["y"] -= cy
204
+
205
+
206
+ def label_of(atom, hydrogens_left=False):
207
+ """An atom's label in the figure markup: `OH`, `H_{2}N`, `N^{+}`."""
208
+ text = atom["el"]
209
+ count = int(atom.get("h", 0))
210
+ if count:
211
+ hydrogens = "H" if count == 1 else "H_{%d}" % count
212
+ text = hydrogens + text if hydrogens_left else text + hydrogens
213
+ charge = int(atom.get("charge", 0))
214
+ if charge:
215
+ sign = "+" if charge > 0 else "−"
216
+ text += "^{%s%s}" % ("" if abs(charge) == 1 else abs(charge), sign)
217
+ return text
218
+
219
+
220
+ def mirrored_layout(atoms, bonds, horizontal=True):
221
+ """A drawn structure mirrored left-right (or top-bottom) about its
222
+ middle, as NEW lists: `(atoms, bonds)`. Wedges and hashes swap, so it
223
+ is the same molecule drawn the other way round and not its mirror
224
+ image. Labels stay upright: they are drawn at the atoms' places,
225
+ never mirrored themselves."""
226
+ key = "x" if horizontal else "y"
227
+ values = [float(atom.get(key, 0.0)) for atom in atoms]
228
+ if not values:
229
+ return list(atoms), list(bonds)
230
+ middle = (min(values) + max(values)) / 2.0
231
+ new_atoms = []
232
+ for atom in atoms:
233
+ moved = dict(atom)
234
+ moved[key] = 2.0 * middle - float(atom.get(key, 0.0))
235
+ new_atoms.append(moved)
236
+ swap = {"wedge": "hash", "hash": "wedge"}
237
+ index = 0 if horizontal else 1
238
+ new_bonds = []
239
+ for bond in bonds:
240
+ turned = dict(bond)
241
+ if turned.get("stereo") in swap:
242
+ turned["stereo"] = swap[turned["stereo"]]
243
+ # A ring's double bond keeps the centre of its ring, which says
244
+ # which side its inner line is on: it mirrors with the atoms, or
245
+ # the line is drawn outside the ring.
246
+ ring = turned.get("ring")
247
+ if ring and len(ring) == 2:
248
+ ring = [float(ring[0]), float(ring[1])]
249
+ ring[index] = 2.0 * middle - ring[index]
250
+ turned["ring"] = ring
251
+ new_bonds.append(turned)
252
+ return new_atoms, new_bonds
dscpanel/core/dtg.py ADDED
@@ -0,0 +1,135 @@
1
+ """DTG: the derivative of a thermogravimetric mass curve.
2
+
3
+ Worked out here from the m% the file records; the derivative arrays TRIOS
4
+ stores are flagged "calculated" (TRI-FORMAT.md 3b) and are never read as a
5
+ signal.
6
+
7
+ * **Against time first.** The temperature jitters sample to sample and
8
+ doubles back at a segment's start (CLAUDE.md, "a DSC curve is a
9
+ PARAMETRIC curve"), so dividing by dT sample by sample would divide by
10
+ noise. The slope is taken against TIME, and per degree it is that slope
11
+ over the segment's heating rate, fitted once for the whole segment.
12
+ * **Smoothed by a local straight line**: at each sample, the least-squares
13
+ slope of the samples within half the window either side. That is the
14
+ Savitzky-Golay first derivative for a quadratic on even spacing, and it
15
+ stays right where the spacing is not even and at the ends, where the
16
+ window is simply cut short. The window is given in KELVIN of the ramp
17
+ (`Scan.dtg_window`), turned into samples with the heating rate.
18
+ * **A loss is positive**: DTG = -dm/dt in %/min, and -dm/dt / |beta| in
19
+ %/degC, whichever way the segment runs.
20
+
21
+ UI-free: numpy only.
22
+ """
23
+
24
+ import numpy as np
25
+
26
+ #: The two units a DTG is drawn in.
27
+ PER_DEGREE = "%/\u00b0C"
28
+ PER_MINUTE = "%/min"
29
+ UNITS = (PER_DEGREE, PER_MINUTE)
30
+
31
+ #: The smoothing window a new DTG curve starts with, in kelvin.
32
+ WINDOW_K = 2.0
33
+ #: Below this heating rate (K/min) a segment is isothermal: no per-degree
34
+ #: derivative exists, and a window in kelvin means nothing.
35
+ ISOTHERMAL_RATE = 0.05
36
+ #: Half-width in samples where a window in kelvin cannot be converted.
37
+ FALLBACK_HALF = 5
38
+
39
+
40
+ def heating_rate(time_min, temp_c):
41
+ """The segment's heating rate in K/min, fitted over its measured
42
+ samples (a straight line through T(t)), or None."""
43
+ if time_min is None or temp_c is None:
44
+ return None
45
+ t = np.asarray(time_min, dtype=float)
46
+ T = np.asarray(temp_c, dtype=float)
47
+ ok = np.isfinite(t) & np.isfinite(T)
48
+ if ok.sum() < 3 or np.ptp(t[ok]) <= 0:
49
+ return None
50
+ slope = np.polyfit(t[ok], T[ok], 1)[0]
51
+ return float(slope)
52
+
53
+
54
+ def half_window(time_min, rate, window_k):
55
+ """Samples either side of each point that `window_k` kelvin spans."""
56
+ t = np.asarray(time_min, dtype=float)
57
+ steps = np.diff(t[np.isfinite(t)])
58
+ steps = steps[steps > 0]
59
+ if not len(steps) or rate is None or abs(rate) < ISOTHERMAL_RATE:
60
+ return FALLBACK_HALF
61
+ per_sample = abs(rate) * float(np.median(steps)) # kelvin
62
+ if window_k <= 0 or per_sample <= 0:
63
+ return 1
64
+ return max(1, int(round(window_k / 2.0 / per_sample)))
65
+
66
+
67
+ def local_slope(x, y, half):
68
+ """At every sample, the least-squares slope dy/dx of the samples within
69
+ `half` either side (fewer at the ends). NaN where fewer than three
70
+ measured samples are in reach."""
71
+ x = np.asarray(x, dtype=float)
72
+ y = np.asarray(y, dtype=float)
73
+ n = len(x)
74
+ ok = np.isfinite(x) & np.isfinite(y)
75
+ # Centre x and y (on the measured samples) before summing, or the
76
+ # sums of squares lose every digit that matters to cancellation.
77
+ if ok.any():
78
+ x = x - np.mean(x[ok])
79
+ y = y - np.mean(y[ok])
80
+ w = ok.astype(float)
81
+ xs = np.where(ok, x, 0.0)
82
+ ys = np.where(ok, y, 0.0)
83
+
84
+ def windowed(values):
85
+ total = np.concatenate([[0.0], np.cumsum(values)])
86
+ lo = np.clip(np.arange(n) - half, 0, n)
87
+ hi = np.clip(np.arange(n) + half + 1, 0, n)
88
+ return total[hi] - total[lo]
89
+
90
+ count = windowed(w)
91
+ sx, sy = windowed(xs), windowed(ys)
92
+ sxx, sxy = windowed(xs * xs), windowed(xs * ys)
93
+ with np.errstate(invalid="ignore", divide="ignore"):
94
+ spread = sxx - sx * sx / count
95
+ slope = (sxy - sx * sy / count) / spread
96
+ bad = (count < 3) | ~(spread > 0) | ~np.isfinite(slope)
97
+ slope[bad] = np.nan
98
+ return slope
99
+
100
+
101
+ def dtg(time_min, temp_c, percent, unit=PER_DEGREE, window_k=WINDOW_K):
102
+ """The DTG of a mass curve in `unit`, one value per sample, or None
103
+ when it cannot be worked out (`missing` says why)."""
104
+ if missing(time_min, temp_c, percent, unit) is not None:
105
+ return None
106
+ rate = heating_rate(time_min, temp_c)
107
+ half = half_window(time_min, rate, float(window_k))
108
+ per_minute = -local_slope(time_min, percent, half)
109
+ if unit == PER_MINUTE:
110
+ return per_minute
111
+ return per_minute / abs(rate)
112
+
113
+
114
+ def missing(time_min, temp_c, percent, unit=PER_DEGREE):
115
+ """What stops a DTG in `unit`, or None."""
116
+ if percent is None or not np.isfinite(
117
+ np.asarray(percent, dtype=float)).any():
118
+ return "mass in this segment"
119
+ if time_min is None or not np.isfinite(
120
+ np.asarray(time_min, dtype=float)).any():
121
+ return "time in this segment"
122
+ if unit == PER_DEGREE:
123
+ rate = heating_rate(time_min, temp_c)
124
+ if rate is None or abs(rate) < ISOTHERMAL_RATE:
125
+ return "heating rate (an isothermal segment has no %/\u00b0C)"
126
+ return None
127
+
128
+
129
+ def factor(unit, rate):
130
+ """What one %/degC of DTG is in `unit` (an offset converts by it)."""
131
+ if unit == PER_DEGREE:
132
+ return 1.0
133
+ if rate is None or abs(rate) < ISOTHERMAL_RATE:
134
+ return None
135
+ return abs(rate)