@motionscript/molecule 0.0.0-stage → 0.1.0-alpha.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (53) hide show
  1. package/CHANGELOG.md +5 -0
  2. package/LICENSE +201 -0
  3. package/dist/browser/index.js +4 -0
  4. package/dist/browser/index.js.map +7 -0
  5. package/dist/browser/manifest.json +11 -0
  6. package/dist/index.d.ts +3 -0
  7. package/dist/index.d.ts.map +1 -0
  8. package/dist/index.js +3 -0
  9. package/dist/index.js.map +1 -0
  10. package/dist/nodes.d.ts +18 -0
  11. package/dist/nodes.d.ts.map +1 -0
  12. package/dist/nodes.js +18 -0
  13. package/dist/nodes.js.map +1 -0
  14. package/dist/protein/chemistry.d.ts +52 -0
  15. package/dist/protein/chemistry.d.ts.map +1 -0
  16. package/dist/protein/chemistry.js +208 -0
  17. package/dist/protein/chemistry.js.map +1 -0
  18. package/dist/protein/index.d.ts +35 -0
  19. package/dist/protein/index.d.ts.map +1 -0
  20. package/dist/protein/index.js +35 -0
  21. package/dist/protein/index.js.map +1 -0
  22. package/dist/protein/parse.d.ts +32 -0
  23. package/dist/protein/parse.d.ts.map +1 -0
  24. package/dist/protein/parse.js +387 -0
  25. package/dist/protein/parse.js.map +1 -0
  26. package/dist/protein/protein.d.ts +265 -0
  27. package/dist/protein/protein.d.ts.map +1 -0
  28. package/dist/protein/protein.js +645 -0
  29. package/dist/protein/protein.js.map +1 -0
  30. package/dist/protein/ribbon.d.ts +83 -0
  31. package/dist/protein/ribbon.d.ts.map +1 -0
  32. package/dist/protein/ribbon.js +468 -0
  33. package/dist/protein/ribbon.js.map +1 -0
  34. package/dist/protein/shared.d.ts +221 -0
  35. package/dist/protein/shared.d.ts.map +1 -0
  36. package/dist/protein/shared.js +478 -0
  37. package/dist/protein/shared.js.map +1 -0
  38. package/dist/protein/structure.d.ts +184 -0
  39. package/dist/protein/structure.d.ts.map +1 -0
  40. package/dist/protein/structure.js +324 -0
  41. package/dist/protein/structure.js.map +1 -0
  42. package/package.json +64 -3
  43. package/registry.json +22 -0
  44. package/src/index.ts +2 -0
  45. package/src/nodes.ts +18 -0
  46. package/src/protein/chemistry.ts +223 -0
  47. package/src/protein/index.ts +34 -0
  48. package/src/protein/parse.ts +427 -0
  49. package/src/protein/protein.ts +897 -0
  50. package/src/protein/ribbon.ts +658 -0
  51. package/src/protein/shared.ts +622 -0
  52. package/src/protein/structure.ts +491 -0
  53. package/README.md +0 -4
@@ -0,0 +1,491 @@
1
+ /**
2
+ * What a parsed molecule *is*, and the three things derived from it that no
3
+ * coordinate file states outright: which atoms are bonded, which curve each
4
+ * chain traces, and how big the whole thing is.
5
+ *
6
+ * The shape here is the seam between the parser (`parse.ts`) and the node
7
+ * (`protein.ts`), and it is deliberately flat and numeric: parallel arrays of
8
+ * plain numbers rather than a graph of objects. A mid-sized structure is tens of
9
+ * thousands of atoms, the node rebuilds its command list every frame, and an
10
+ * object per atom would put a few hundred thousand allocations between the
11
+ * scrubber and the picture.
12
+ *
13
+ * Parsing happens **once, ahead of the build** — the same arrangement a chart's
14
+ * CSV gets (see `SceneAsset.rows`), and for the same reason: `buildScene` is
15
+ * synchronous, so anything that has to be read off a network or a file has to
16
+ * already be here by the time it runs.
17
+ */
18
+
19
+ import { covalentRadius, traceAtomName, type ResidueKind } from "./chemistry"
20
+
21
+ /** One atom, as the file gave it. Coordinates are in Ångströms. */
22
+ export interface ProteinAtom {
23
+ /** The file's own serial number, kept so `CONECT` records can be resolved. */
24
+ serial: number
25
+ /** The atom's name within its residue — `CA`, `OG1`, `FE`. */
26
+ name: string
27
+ /** The element symbol, upper-cased. Derived when the file omits it. */
28
+ element: string
29
+ /** The residue's three-letter code — `ALA`, `HOH`, `HEM`. */
30
+ residue: string
31
+ /** The residue's number within its chain, as the file numbers it. */
32
+ residueSeq: number
33
+ /** The chain the residue belongs to — `A`, `B`, … */
34
+ chain: string
35
+ x: number
36
+ y: number
37
+ z: number
38
+ /** Whether the file filed this under `HETATM` — a ligand, an ion, a water. */
39
+ hetero: boolean
40
+ /** What the residue is, resolved once here so nothing downstream re-derives it. */
41
+ kind: ResidueKind
42
+ }
43
+
44
+ /**
45
+ * What a stretch of backbone is doing, which is the one property of a protein
46
+ * that a picture of it is usually *about*.
47
+ *
48
+ * Taken from the file's own `HELIX`/`SHEET` annotations rather than computed:
49
+ * assigning secondary structure from coordinates alone is DSSP, which is a
50
+ * hydrogen-bond model and a paper of its own. A file that annotates none draws
51
+ * as an even tube, which is the honest picture of "nobody said".
52
+ */
53
+ export type SecondaryStructure = "helix" | "sheet" | "coil"
54
+
55
+ /** One residue's place on a chain's backbone curve. */
56
+ export interface TracePoint {
57
+ x: number
58
+ y: number
59
+ z: number
60
+ /** What this residue is part of, for the ribbon's shape and its colouring. */
61
+ secondary: SecondaryStructure
62
+ /**
63
+ * The residue's position along **this trace**, 0-based. Against the trace's
64
+ * own length this is what a spectrum colouring needs: N terminus to C
65
+ * terminus, per chain, independent of how long the others are.
66
+ */
67
+ index: number
68
+ /**
69
+ * The residue's number as the file gives it, which is what ties a trace point
70
+ * back to the atoms of the same residue.
71
+ *
72
+ * Both are needed and neither substitutes for the other: {@link index} is
73
+ * dense and ordered, so it is what a spectrum is computed against, while this
74
+ * is sparse and arbitrary — files skip numbers, start at 17, and number
75
+ * insertions — so it is what a *lookup* has to be keyed on.
76
+ */
77
+ residueSeq: number
78
+ /**
79
+ * Which way this residue's ribbon faces — the unit vector its **wide**
80
+ * direction is swept along.
81
+ *
82
+ * A tube needs a point and nothing else; a cartoon ribbon needs to know which
83
+ * way round it is, and that is not something a curve through α-carbons can
84
+ * answer. The residue's own carbonyl does: CA→O is roughly perpendicular to
85
+ * the chain and turns with the fold, so sweeping the ribbon's width along it
86
+ * is what makes a helix read as a twisted band and a strand as a flat one.
87
+ * This is the same vector every molecular viewer builds its cartoon frame
88
+ * from.
89
+ *
90
+ * `null` when the file has no carbonyl for the residue — a Cα-only model, a
91
+ * nucleotide, the last residue of a truncated chain — and the ribbon builder
92
+ * falls back to the curve's own binormal there. See `ribbonFrames`.
93
+ *
94
+ * Not orthogonalized here: it is a *direction*, and the frame that has to be
95
+ * square is built against the tangent, which only exists once the neighbours
96
+ * are known.
97
+ */
98
+ normal: { x: number; y: number; z: number } | null
99
+ }
100
+
101
+ /** One polymer chain: its identity and the curve its backbone follows. */
102
+ export interface ProteinChain {
103
+ /** The chain identifier the file uses — `A`, `B`, … */
104
+ id: string
105
+ /** The α-carbon (or phosphorus) trace, in residue order. */
106
+ trace: TracePoint[]
107
+ }
108
+
109
+ /**
110
+ * A parsed molecule.
111
+ *
112
+ * Bonds and chains are *derived* (see {@link deriveStructure}) rather than read,
113
+ * because coordinate files mostly don't state them: `CONECT` records cover the
114
+ * ligands and little else, and the chain break between two residues is implied
115
+ * by their distance.
116
+ */
117
+ export interface ProteinStructure {
118
+ /** The accession or filename this came from, for labelling. */
119
+ id: string
120
+ /** The file's `TITLE`, when it has one. */
121
+ title: string
122
+ atoms: ProteinAtom[]
123
+ /** Bonded pairs, as indices into {@link atoms}. */
124
+ bonds: ProteinBond[]
125
+ chains: ProteinChain[]
126
+ /** The centroid of every atom — what the node translates to the origin. */
127
+ center: { x: number; y: number; z: number }
128
+ /** Distance from {@link center} to the furthest atom, in Ångströms. */
129
+ radius: number
130
+ }
131
+
132
+ /** One bond, as a pair of indices into {@link ProteinStructure.atoms}. */
133
+ export interface ProteinBond {
134
+ a: number
135
+ b: number
136
+ }
137
+
138
+ /** An empty molecule — what an unresolved or unparseable source draws as. */
139
+ export const EMPTY_STRUCTURE: ProteinStructure = {
140
+ id: "",
141
+ title: "",
142
+ atoms: [],
143
+ bonds: [],
144
+ chains: [],
145
+ center: { x: 0, y: 0, z: 0 },
146
+ radius: 1,
147
+ }
148
+
149
+ /**
150
+ * A secondary-structure span as the file states it: a run of residue numbers on
151
+ * one chain. Collected by the parser, applied to the traces here.
152
+ */
153
+ export interface SecondarySpan {
154
+ chain: string
155
+ start: number
156
+ end: number
157
+ kind: Exclude<SecondaryStructure, "coil">
158
+ }
159
+
160
+ /**
161
+ * Completes a parsed atom list into a {@link ProteinStructure}: infers the
162
+ * bonds, walks out each chain's backbone, and measures the whole thing.
163
+ *
164
+ * Split from the parsers because both formats reach the same place — an atom
165
+ * list and a set of annotated spans — and everything past that point is
166
+ * geometry rather than syntax.
167
+ */
168
+ export function deriveStructure(
169
+ id: string,
170
+ title: string,
171
+ atoms: ProteinAtom[],
172
+ spans: SecondarySpan[]
173
+ ): ProteinStructure {
174
+ if (atoms.length === 0) return { ...EMPTY_STRUCTURE, id, title }
175
+
176
+ return {
177
+ id,
178
+ title,
179
+ atoms,
180
+ bonds: inferBonds(atoms),
181
+ chains: traceChains(atoms, spans),
182
+ ...measure(atoms),
183
+ }
184
+ }
185
+
186
+ // --- Bonds -----------------------------------------------------------------
187
+
188
+ /**
189
+ * How much further apart than the sum of their covalent radii two atoms may sit
190
+ * and still count as bonded, in Ångströms.
191
+ *
192
+ * The conventional tolerance. It has to be generous enough to survive a
193
+ * moderate-resolution structure's coordinate error and mean enough not to bond
194
+ * a residue to the one packed against it — 0.45 Å is where every viewer has
195
+ * settled, and the gap between a real bond (~1.5 Å) and the closest non-bonded
196
+ * contact (~2.8 Å) is wide enough that the exact number rarely decides anything.
197
+ */
198
+ const BOND_TOLERANCE = 0.45
199
+
200
+ /**
201
+ * The shortest separation treated as a bond. Two atoms closer than this are
202
+ * alternate conformations of the same one, or a duplicate record — bonding them
203
+ * would draw a stick of no length and a mesh with no orientation.
204
+ */
205
+ const MIN_BOND_LENGTH = 0.4
206
+
207
+ /**
208
+ * The furthest two atoms can be bonded, which sets the neighbour grid's cell
209
+ * size: the largest pair of covalent radii here (potassium's, twice) plus the
210
+ * tolerance.
211
+ */
212
+ const MAX_BOND_LENGTH = 2.5
213
+
214
+ /**
215
+ * Infers bonds by distance, over a uniform grid.
216
+ *
217
+ * The grid is what makes this affordable. Bonding is a nearest-neighbour
218
+ * question and the naïve form is every atom against every other — 20,000 atoms
219
+ * is 200 million comparisons, which is a visible freeze on the frame someone
220
+ * picks a structure. Bucketed at the maximum bond length, each atom only looks
221
+ * at the 27 cells around it, and since the cells are that size no bond can span
222
+ * further, so nothing is missed. The work becomes linear in the atom count.
223
+ *
224
+ * Waters are skipped outright: they are a single oxygen with nothing to bond to
225
+ * (their hydrogens are almost never in the file), so every comparison against
226
+ * one is wasted, and there can be more of them than there are protein atoms.
227
+ */
228
+ export function inferBonds(atoms: ProteinAtom[]): ProteinBond[] {
229
+ const bonds: ProteinBond[] = []
230
+ const cells = new Map<string, number[]>()
231
+ const radii = new Float32Array(atoms.length)
232
+
233
+ for (let i = 0; i < atoms.length; i++) {
234
+ const atom = atoms[i]!
235
+ if (atom.kind === "water") continue
236
+ radii[i] = covalentRadius(atom.element)
237
+ const key = cellKey(atom.x, atom.y, atom.z)
238
+ const bucket = cells.get(key)
239
+ if (bucket) bucket.push(i)
240
+ else cells.set(key, [i])
241
+ }
242
+
243
+ for (const [key, bucket] of cells) {
244
+ const [cx, cy, cz] = key.split(",").map(Number) as [number, number, number]
245
+
246
+ for (let dx = -1; dx <= 1; dx++) {
247
+ for (let dy = -1; dy <= 1; dy++) {
248
+ for (let dz = -1; dz <= 1; dz++) {
249
+ const neighbours = cells.get(`${cx + dx},${cy + dy},${cz + dz}`)
250
+ if (!neighbours) continue
251
+
252
+ for (const i of bucket) {
253
+ for (const j of neighbours) {
254
+ // Each unordered pair is reached from both cells, so keep only one
255
+ // ordering. This also excludes an atom against itself.
256
+ if (j <= i) continue
257
+ const a = atoms[i]!
258
+ const b = atoms[j]!
259
+ const limit = radii[i]! + radii[j]! + BOND_TOLERANCE
260
+ const dxx = a.x - b.x
261
+ const dyy = a.y - b.y
262
+ const dzz = a.z - b.z
263
+ const squared = dxx * dxx + dyy * dyy + dzz * dzz
264
+ if (squared > limit * limit) continue
265
+ if (squared < MIN_BOND_LENGTH * MIN_BOND_LENGTH) continue
266
+ bonds.push({ a: i, b: j })
267
+ }
268
+ }
269
+ }
270
+ }
271
+ }
272
+ }
273
+
274
+ return bonds
275
+ }
276
+
277
+ /** The grid cell a coordinate falls in, as a map key. */
278
+ function cellKey(x: number, y: number, z: number): string {
279
+ const cx = Math.floor(x / MAX_BOND_LENGTH)
280
+ const cy = Math.floor(y / MAX_BOND_LENGTH)
281
+ const cz = Math.floor(z / MAX_BOND_LENGTH)
282
+ return `${cx},${cy},${cz}`
283
+ }
284
+
285
+ // --- Chains ----------------------------------------------------------------
286
+
287
+ /**
288
+ * How far apart two consecutive α-carbons may sit and still be the same chain,
289
+ * in Ångströms.
290
+ *
291
+ * Consecutive α-carbons are 3.8 Å apart — the distance is fixed by the peptide
292
+ * bond's geometry, not by what the protein is doing — so a larger gap is a
293
+ * *chain break*: a disordered loop the crystallographer could not resolve, which
294
+ * the file records by simply skipping those residues. Drawing through one would
295
+ * run a ribbon across the middle of the molecule between two ends that are not
296
+ * joined. Nucleotide phosphorus atoms sit further apart (~6 Å), which is why the
297
+ * threshold is well above 3.8 rather than snug against it.
298
+ */
299
+ const CHAIN_BREAK_DISTANCE = 7.5
300
+
301
+ /**
302
+ * The backbone curve of every chain, in residue order, with each residue's
303
+ * secondary structure attached.
304
+ *
305
+ * A chain break starts a **new trace** rather than being drawn through — see
306
+ * {@link CHAIN_BREAK_DISTANCE} — so one chain identifier can produce several
307
+ * entries. That is the right unit downstream anyway: each entry is one
308
+ * continuous curve, which is exactly what a tube is swept along.
309
+ */
310
+ export function traceChains(
311
+ atoms: ProteinAtom[],
312
+ spans: SecondarySpan[]
313
+ ): ProteinChain[] {
314
+ const assign = secondaryLookup(spans)
315
+ // Built ahead of the walk rather than during it, because the carbonyl comes
316
+ // *after* the α-carbon in a residue's records and the trace point is written
317
+ // as the α-carbon is reached. One pass over the atoms costs nothing beside
318
+ // the bond inference already running over them.
319
+ const carbonyls = carbonylDirections(atoms)
320
+ const chains: ProteinChain[] = []
321
+
322
+ let current: ProteinChain | null = null
323
+ let previous: ProteinAtom | null = null
324
+
325
+ for (const atom of atoms) {
326
+ if (atom.name !== traceAtomName(atom.kind)) continue
327
+
328
+ const broken =
329
+ previous !== null &&
330
+ (previous.chain !== atom.chain ||
331
+ distance(previous, atom) > CHAIN_BREAK_DISTANCE)
332
+
333
+ if (current === null || broken) {
334
+ current = { id: atom.chain, trace: [] }
335
+ chains.push(current)
336
+ }
337
+
338
+ current.trace.push({
339
+ x: atom.x,
340
+ y: atom.y,
341
+ z: atom.z,
342
+ secondary: assign(atom.chain, atom.residueSeq),
343
+ index: current.trace.length,
344
+ residueSeq: atom.residueSeq,
345
+ normal: carbonyls.get(residueKey(atom)) ?? null,
346
+ })
347
+ previous = atom
348
+ }
349
+
350
+ // A single point is not a curve — nothing can be swept along it and no
351
+ // direction can be taken from it — so a lone resolved residue is dropped
352
+ // rather than left for the tube builder to divide by zero over.
353
+ return chains.filter((chain) => chain.trace.length > 1)
354
+ }
355
+
356
+ /**
357
+ * Builds the residue → secondary-structure lookup.
358
+ *
359
+ * A map per chain of residue number to kind, rather than a scan of the span list
360
+ * per residue: a large structure has thousands of residues and hundreds of
361
+ * spans, and the product of the two is the sort of quiet quadratic that only
362
+ * shows up on the file somebody actually cares about.
363
+ */
364
+ function secondaryLookup(
365
+ spans: SecondarySpan[]
366
+ ): (chain: string, residue: number) => SecondaryStructure {
367
+ const byChain = new Map<string, Map<number, SecondaryStructure>>()
368
+
369
+ for (const span of spans) {
370
+ let residues = byChain.get(span.chain)
371
+ if (!residues) {
372
+ residues = new Map()
373
+ byChain.set(span.chain, residues)
374
+ }
375
+ // A malformed span (end before start, or one covering the whole file) would
376
+ // otherwise fill the map with junk; the loop simply doesn't run backwards.
377
+ for (let seq = span.start; seq <= span.end; seq++) {
378
+ residues.set(seq, span.kind)
379
+ }
380
+ }
381
+
382
+ return (chain, residue) => byChain.get(chain)?.get(residue) ?? "coil"
383
+ }
384
+
385
+ /** A residue's identity across the whole file — its chain and its number. */
386
+ function residueKey(atom: ProteinAtom): string {
387
+ return `${atom.chain}:${atom.residueSeq}`
388
+ }
389
+
390
+ /**
391
+ * The unit CA→O vector of every amino-acid residue that has both atoms.
392
+ *
393
+ * This is the one piece of chemistry a *cartoon* needs and a tube does not. A
394
+ * ribbon has a width, so it has a side, and nothing about a curve through
395
+ * α-carbons says which way round it should be: sweep a flat band along that
396
+ * curve with an arbitrary frame and a helix comes out as a twisted ribbon in
397
+ * the wrong plane, its face turning at random rather than with the fold.
398
+ *
399
+ * The carbonyl is what every molecular viewer resolves that with. It points
400
+ * roughly across the chain, it is what hydrogen-bonds a helix to itself and a
401
+ * strand to its neighbour, and so it turns *with* the structure — which makes a
402
+ * band swept along it lie the way the fold does.
403
+ *
404
+ * Amino acids only. A nucleotide's ribbon has no equivalent (its trace atom is
405
+ * the phosphorus and its "width" is the base pair), and a residue whose file
406
+ * gives no `O` — a Cα-only model, a truncated terminus — simply has no entry;
407
+ * both fall back to the curve's own binormal in the ribbon builder.
408
+ */
409
+ function carbonylDirections(
410
+ atoms: ProteinAtom[]
411
+ ): Map<string, { x: number; y: number; z: number }> {
412
+ const alpha = new Map<string, ProteinAtom>()
413
+ const oxygen = new Map<string, ProteinAtom>()
414
+
415
+ for (const atom of atoms) {
416
+ if (atom.kind !== "amino") continue
417
+ // The first record of each name wins, which is what picks conformation A
418
+ // out of a residue the file gives two of: alternates are written in order
419
+ // and the ribbon needs one frame, not an average of two.
420
+ if (atom.name === "CA") {
421
+ if (!alpha.has(residueKey(atom))) alpha.set(residueKey(atom), atom)
422
+ } else if (atom.name === "O") {
423
+ if (!oxygen.has(residueKey(atom))) oxygen.set(residueKey(atom), atom)
424
+ }
425
+ }
426
+
427
+ const directions = new Map<string, { x: number; y: number; z: number }>()
428
+ for (const [key, ca] of alpha) {
429
+ const o = oxygen.get(key)
430
+ if (!o) continue
431
+ const x = o.x - ca.x
432
+ const y = o.y - ca.y
433
+ const z = o.z - ca.z
434
+ const length = Math.hypot(x, y, z)
435
+ // A zero-length carbonyl is a duplicated record rather than a direction.
436
+ if (length < 1e-6) continue
437
+ directions.set(key, { x: x / length, y: y / length, z: z / length })
438
+ }
439
+ return directions
440
+ }
441
+
442
+ /** Straight-line distance between two atoms, in Ångströms. */
443
+ function distance(a: ProteinAtom, b: ProteinAtom): number {
444
+ return Math.hypot(a.x - b.x, a.y - b.y, a.z - b.z)
445
+ }
446
+
447
+ // --- Bounds ----------------------------------------------------------------
448
+
449
+ /**
450
+ * The molecule's centroid and the radius of the sphere about it that holds every
451
+ * atom.
452
+ *
453
+ * The centroid rather than the centre of the bounding box: a structure with one
454
+ * long tail — a coiled-coil, a nucleic acid strand hanging off a complex —
455
+ * has a box whose centre is nowhere near the mass of it, and the camera would
456
+ * frame the empty half.
457
+ *
458
+ * The radius is floored above zero so a single-atom file (or one whose
459
+ * coordinates are all identical) still gives the camera a distance to work from
460
+ * rather than parking it inside the atom.
461
+ */
462
+ function measure(atoms: ProteinAtom[]): {
463
+ center: { x: number; y: number; z: number }
464
+ radius: number
465
+ } {
466
+ let sx = 0
467
+ let sy = 0
468
+ let sz = 0
469
+ for (const atom of atoms) {
470
+ sx += atom.x
471
+ sy += atom.y
472
+ sz += atom.z
473
+ }
474
+ const center = {
475
+ x: sx / atoms.length,
476
+ y: sy / atoms.length,
477
+ z: sz / atoms.length,
478
+ }
479
+
480
+ let radius = 0
481
+ for (const atom of atoms) {
482
+ const d = Math.hypot(
483
+ atom.x - center.x,
484
+ atom.y - center.y,
485
+ atom.z - center.z
486
+ )
487
+ if (d > radius) radius = d
488
+ }
489
+
490
+ return { center, radius: Math.max(radius, 1) }
491
+ }
package/README.md DELETED
@@ -1,4 +0,0 @@
1
- # Temporary Holding Version
2
-
3
- This version is a temporary placeholder for this package. An operational version to replace this has been submitted for review and is awaiting a staged release.
4
- If no other versions are published within 30 days, this package and version will be deleted.