@motionscript/molecule 0.0.0-stage → 0.1.0-alpha.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (53) hide show
  1. package/CHANGELOG.md +5 -0
  2. package/LICENSE +201 -0
  3. package/dist/browser/index.js +4 -0
  4. package/dist/browser/index.js.map +7 -0
  5. package/dist/browser/manifest.json +11 -0
  6. package/dist/index.d.ts +3 -0
  7. package/dist/index.d.ts.map +1 -0
  8. package/dist/index.js +3 -0
  9. package/dist/index.js.map +1 -0
  10. package/dist/nodes.d.ts +18 -0
  11. package/dist/nodes.d.ts.map +1 -0
  12. package/dist/nodes.js +18 -0
  13. package/dist/nodes.js.map +1 -0
  14. package/dist/protein/chemistry.d.ts +52 -0
  15. package/dist/protein/chemistry.d.ts.map +1 -0
  16. package/dist/protein/chemistry.js +208 -0
  17. package/dist/protein/chemistry.js.map +1 -0
  18. package/dist/protein/index.d.ts +35 -0
  19. package/dist/protein/index.d.ts.map +1 -0
  20. package/dist/protein/index.js +35 -0
  21. package/dist/protein/index.js.map +1 -0
  22. package/dist/protein/parse.d.ts +32 -0
  23. package/dist/protein/parse.d.ts.map +1 -0
  24. package/dist/protein/parse.js +387 -0
  25. package/dist/protein/parse.js.map +1 -0
  26. package/dist/protein/protein.d.ts +265 -0
  27. package/dist/protein/protein.d.ts.map +1 -0
  28. package/dist/protein/protein.js +645 -0
  29. package/dist/protein/protein.js.map +1 -0
  30. package/dist/protein/ribbon.d.ts +83 -0
  31. package/dist/protein/ribbon.d.ts.map +1 -0
  32. package/dist/protein/ribbon.js +468 -0
  33. package/dist/protein/ribbon.js.map +1 -0
  34. package/dist/protein/shared.d.ts +221 -0
  35. package/dist/protein/shared.d.ts.map +1 -0
  36. package/dist/protein/shared.js +478 -0
  37. package/dist/protein/shared.js.map +1 -0
  38. package/dist/protein/structure.d.ts +184 -0
  39. package/dist/protein/structure.d.ts.map +1 -0
  40. package/dist/protein/structure.js +324 -0
  41. package/dist/protein/structure.js.map +1 -0
  42. package/package.json +64 -3
  43. package/registry.json +22 -0
  44. package/src/index.ts +2 -0
  45. package/src/nodes.ts +18 -0
  46. package/src/protein/chemistry.ts +223 -0
  47. package/src/protein/index.ts +34 -0
  48. package/src/protein/parse.ts +427 -0
  49. package/src/protein/protein.ts +897 -0
  50. package/src/protein/ribbon.ts +658 -0
  51. package/src/protein/shared.ts +622 -0
  52. package/src/protein/structure.ts +491 -0
  53. package/README.md +0 -4
@@ -0,0 +1,184 @@
1
+ /**
2
+ * What a parsed molecule *is*, and the three things derived from it that no
3
+ * coordinate file states outright: which atoms are bonded, which curve each
4
+ * chain traces, and how big the whole thing is.
5
+ *
6
+ * The shape here is the seam between the parser (`parse.ts`) and the node
7
+ * (`protein.ts`), and it is deliberately flat and numeric: parallel arrays of
8
+ * plain numbers rather than a graph of objects. A mid-sized structure is tens of
9
+ * thousands of atoms, the node rebuilds its command list every frame, and an
10
+ * object per atom would put a few hundred thousand allocations between the
11
+ * scrubber and the picture.
12
+ *
13
+ * Parsing happens **once, ahead of the build** — the same arrangement a chart's
14
+ * CSV gets (see `SceneAsset.rows`), and for the same reason: `buildScene` is
15
+ * synchronous, so anything that has to be read off a network or a file has to
16
+ * already be here by the time it runs.
17
+ */
18
+ import { type ResidueKind } from "./chemistry.js";
19
+ /** One atom, as the file gave it. Coordinates are in Ångströms. */
20
+ export interface ProteinAtom {
21
+ /** The file's own serial number, kept so `CONECT` records can be resolved. */
22
+ serial: number;
23
+ /** The atom's name within its residue — `CA`, `OG1`, `FE`. */
24
+ name: string;
25
+ /** The element symbol, upper-cased. Derived when the file omits it. */
26
+ element: string;
27
+ /** The residue's three-letter code — `ALA`, `HOH`, `HEM`. */
28
+ residue: string;
29
+ /** The residue's number within its chain, as the file numbers it. */
30
+ residueSeq: number;
31
+ /** The chain the residue belongs to — `A`, `B`, … */
32
+ chain: string;
33
+ x: number;
34
+ y: number;
35
+ z: number;
36
+ /** Whether the file filed this under `HETATM` — a ligand, an ion, a water. */
37
+ hetero: boolean;
38
+ /** What the residue is, resolved once here so nothing downstream re-derives it. */
39
+ kind: ResidueKind;
40
+ }
41
+ /**
42
+ * What a stretch of backbone is doing, which is the one property of a protein
43
+ * that a picture of it is usually *about*.
44
+ *
45
+ * Taken from the file's own `HELIX`/`SHEET` annotations rather than computed:
46
+ * assigning secondary structure from coordinates alone is DSSP, which is a
47
+ * hydrogen-bond model and a paper of its own. A file that annotates none draws
48
+ * as an even tube, which is the honest picture of "nobody said".
49
+ */
50
+ export type SecondaryStructure = "helix" | "sheet" | "coil";
51
+ /** One residue's place on a chain's backbone curve. */
52
+ export interface TracePoint {
53
+ x: number;
54
+ y: number;
55
+ z: number;
56
+ /** What this residue is part of, for the ribbon's shape and its colouring. */
57
+ secondary: SecondaryStructure;
58
+ /**
59
+ * The residue's position along **this trace**, 0-based. Against the trace's
60
+ * own length this is what a spectrum colouring needs: N terminus to C
61
+ * terminus, per chain, independent of how long the others are.
62
+ */
63
+ index: number;
64
+ /**
65
+ * The residue's number as the file gives it, which is what ties a trace point
66
+ * back to the atoms of the same residue.
67
+ *
68
+ * Both are needed and neither substitutes for the other: {@link index} is
69
+ * dense and ordered, so it is what a spectrum is computed against, while this
70
+ * is sparse and arbitrary — files skip numbers, start at 17, and number
71
+ * insertions — so it is what a *lookup* has to be keyed on.
72
+ */
73
+ residueSeq: number;
74
+ /**
75
+ * Which way this residue's ribbon faces — the unit vector its **wide**
76
+ * direction is swept along.
77
+ *
78
+ * A tube needs a point and nothing else; a cartoon ribbon needs to know which
79
+ * way round it is, and that is not something a curve through α-carbons can
80
+ * answer. The residue's own carbonyl does: CA→O is roughly perpendicular to
81
+ * the chain and turns with the fold, so sweeping the ribbon's width along it
82
+ * is what makes a helix read as a twisted band and a strand as a flat one.
83
+ * This is the same vector every molecular viewer builds its cartoon frame
84
+ * from.
85
+ *
86
+ * `null` when the file has no carbonyl for the residue — a Cα-only model, a
87
+ * nucleotide, the last residue of a truncated chain — and the ribbon builder
88
+ * falls back to the curve's own binormal there. See `ribbonFrames`.
89
+ *
90
+ * Not orthogonalized here: it is a *direction*, and the frame that has to be
91
+ * square is built against the tangent, which only exists once the neighbours
92
+ * are known.
93
+ */
94
+ normal: {
95
+ x: number;
96
+ y: number;
97
+ z: number;
98
+ } | null;
99
+ }
100
+ /** One polymer chain: its identity and the curve its backbone follows. */
101
+ export interface ProteinChain {
102
+ /** The chain identifier the file uses — `A`, `B`, … */
103
+ id: string;
104
+ /** The α-carbon (or phosphorus) trace, in residue order. */
105
+ trace: TracePoint[];
106
+ }
107
+ /**
108
+ * A parsed molecule.
109
+ *
110
+ * Bonds and chains are *derived* (see {@link deriveStructure}) rather than read,
111
+ * because coordinate files mostly don't state them: `CONECT` records cover the
112
+ * ligands and little else, and the chain break between two residues is implied
113
+ * by their distance.
114
+ */
115
+ export interface ProteinStructure {
116
+ /** The accession or filename this came from, for labelling. */
117
+ id: string;
118
+ /** The file's `TITLE`, when it has one. */
119
+ title: string;
120
+ atoms: ProteinAtom[];
121
+ /** Bonded pairs, as indices into {@link atoms}. */
122
+ bonds: ProteinBond[];
123
+ chains: ProteinChain[];
124
+ /** The centroid of every atom — what the node translates to the origin. */
125
+ center: {
126
+ x: number;
127
+ y: number;
128
+ z: number;
129
+ };
130
+ /** Distance from {@link center} to the furthest atom, in Ångströms. */
131
+ radius: number;
132
+ }
133
+ /** One bond, as a pair of indices into {@link ProteinStructure.atoms}. */
134
+ export interface ProteinBond {
135
+ a: number;
136
+ b: number;
137
+ }
138
+ /** An empty molecule — what an unresolved or unparseable source draws as. */
139
+ export declare const EMPTY_STRUCTURE: ProteinStructure;
140
+ /**
141
+ * A secondary-structure span as the file states it: a run of residue numbers on
142
+ * one chain. Collected by the parser, applied to the traces here.
143
+ */
144
+ export interface SecondarySpan {
145
+ chain: string;
146
+ start: number;
147
+ end: number;
148
+ kind: Exclude<SecondaryStructure, "coil">;
149
+ }
150
+ /**
151
+ * Completes a parsed atom list into a {@link ProteinStructure}: infers the
152
+ * bonds, walks out each chain's backbone, and measures the whole thing.
153
+ *
154
+ * Split from the parsers because both formats reach the same place — an atom
155
+ * list and a set of annotated spans — and everything past that point is
156
+ * geometry rather than syntax.
157
+ */
158
+ export declare function deriveStructure(id: string, title: string, atoms: ProteinAtom[], spans: SecondarySpan[]): ProteinStructure;
159
+ /**
160
+ * Infers bonds by distance, over a uniform grid.
161
+ *
162
+ * The grid is what makes this affordable. Bonding is a nearest-neighbour
163
+ * question and the naïve form is every atom against every other — 20,000 atoms
164
+ * is 200 million comparisons, which is a visible freeze on the frame someone
165
+ * picks a structure. Bucketed at the maximum bond length, each atom only looks
166
+ * at the 27 cells around it, and since the cells are that size no bond can span
167
+ * further, so nothing is missed. The work becomes linear in the atom count.
168
+ *
169
+ * Waters are skipped outright: they are a single oxygen with nothing to bond to
170
+ * (their hydrogens are almost never in the file), so every comparison against
171
+ * one is wasted, and there can be more of them than there are protein atoms.
172
+ */
173
+ export declare function inferBonds(atoms: ProteinAtom[]): ProteinBond[];
174
+ /**
175
+ * The backbone curve of every chain, in residue order, with each residue's
176
+ * secondary structure attached.
177
+ *
178
+ * A chain break starts a **new trace** rather than being drawn through — see
179
+ * {@link CHAIN_BREAK_DISTANCE} — so one chain identifier can produce several
180
+ * entries. That is the right unit downstream anyway: each entry is one
181
+ * continuous curve, which is exactly what a tube is swept along.
182
+ */
183
+ export declare function traceChains(atoms: ProteinAtom[], spans: SecondarySpan[]): ProteinChain[];
184
+ //# sourceMappingURL=structure.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"structure.d.ts","sourceRoot":"","sources":["../../src/protein/structure.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;GAgBG;AAEH,OAAO,EAAiC,KAAK,WAAW,EAAE,MAAM,aAAa,CAAA;AAE7E,mEAAmE;AACnE,MAAM,WAAW,WAAW;IAC1B,8EAA8E;IAC9E,MAAM,EAAE,MAAM,CAAA;IACd,8DAA8D;IAC9D,IAAI,EAAE,MAAM,CAAA;IACZ,uEAAuE;IACvE,OAAO,EAAE,MAAM,CAAA;IACf,6DAA6D;IAC7D,OAAO,EAAE,MAAM,CAAA;IACf,qEAAqE;IACrE,UAAU,EAAE,MAAM,CAAA;IAClB,qDAAqD;IACrD,KAAK,EAAE,MAAM,CAAA;IACb,CAAC,EAAE,MAAM,CAAA;IACT,CAAC,EAAE,MAAM,CAAA;IACT,CAAC,EAAE,MAAM,CAAA;IACT,8EAA8E;IAC9E,MAAM,EAAE,OAAO,CAAA;IACf,mFAAmF;IACnF,IAAI,EAAE,WAAW,CAAA;CAClB;AAED;;;;;;;;GAQG;AACH,MAAM,MAAM,kBAAkB,GAAG,OAAO,GAAG,OAAO,GAAG,MAAM,CAAA;AAE3D,uDAAuD;AACvD,MAAM,WAAW,UAAU;IACzB,CAAC,EAAE,MAAM,CAAA;IACT,CAAC,EAAE,MAAM,CAAA;IACT,CAAC,EAAE,MAAM,CAAA;IACT,8EAA8E;IAC9E,SAAS,EAAE,kBAAkB,CAAA;IAC7B;;;;OAIG;IACH,KAAK,EAAE,MAAM,CAAA;IACb;;;;;;;;OAQG;IACH,UAAU,EAAE,MAAM,CAAA;IAClB;;;;;;;;;;;;;;;;;;;OAmBG;IACH,MAAM,EAAE;QAAE,CAAC,EAAE,MAAM,CAAC;QAAC,CAAC,EAAE,MAAM,CAAC;QAAC,CAAC,EAAE,MAAM,CAAA;KAAE,GAAG,IAAI,CAAA;CACnD;AAED,0EAA0E;AAC1E,MAAM,WAAW,YAAY;IAC3B,uDAAuD;IACvD,EAAE,EAAE,MAAM,CAAA;IACV,4DAA4D;IAC5D,KAAK,EAAE,UAAU,EAAE,CAAA;CACpB;AAED;;;;;;;GAOG;AACH,MAAM,WAAW,gBAAgB;IAC/B,+DAA+D;IAC/D,EAAE,EAAE,MAAM,CAAA;IACV,2CAA2C;IAC3C,KAAK,EAAE,MAAM,CAAA;IACb,KAAK,EAAE,WAAW,EAAE,CAAA;IACpB,mDAAmD;IACnD,KAAK,EAAE,WAAW,EAAE,CAAA;IACpB,MAAM,EAAE,YAAY,EAAE,CAAA;IACtB,2EAA2E;IAC3E,MAAM,EAAE;QAAE,CAAC,EAAE,MAAM,CAAC;QAAC,CAAC,EAAE,MAAM,CAAC;QAAC,CAAC,EAAE,MAAM,CAAA;KAAE,CAAA;IAC3C,uEAAuE;IACvE,MAAM,EAAE,MAAM,CAAA;CACf;AAED,0EAA0E;AAC1E,MAAM,WAAW,WAAW;IAC1B,CAAC,EAAE,MAAM,CAAA;IACT,CAAC,EAAE,MAAM,CAAA;CACV;AAED,6EAA6E;AAC7E,eAAO,MAAM,eAAe,EAAE,gBAQ7B,CAAA;AAED;;;GAGG;AACH,MAAM,WAAW,aAAa;IAC5B,KAAK,EAAE,MAAM,CAAA;IACb,KAAK,EAAE,MAAM,CAAA;IACb,GAAG,EAAE,MAAM,CAAA;IACX,IAAI,EAAE,OAAO,CAAC,kBAAkB,EAAE,MAAM,CAAC,CAAA;CAC1C;AAED;;;;;;;GAOG;AACH,wBAAgB,eAAe,CAC7B,EAAE,EAAE,MAAM,EACV,KAAK,EAAE,MAAM,EACb,KAAK,EAAE,WAAW,EAAE,EACpB,KAAK,EAAE,aAAa,EAAE,GACrB,gBAAgB,CAWlB;AA8BD;;;;;;;;;;;;;GAaG;AACH,wBAAgB,UAAU,CAAC,KAAK,EAAE,WAAW,EAAE,GAAG,WAAW,EAAE,CA+C9D;AA0BD;;;;;;;;GAQG;AACH,wBAAgB,WAAW,CACzB,KAAK,EAAE,WAAW,EAAE,EACpB,KAAK,EAAE,aAAa,EAAE,GACrB,YAAY,EAAE,CAyChB"}
@@ -0,0 +1,324 @@
1
+ /**
2
+ * What a parsed molecule *is*, and the three things derived from it that no
3
+ * coordinate file states outright: which atoms are bonded, which curve each
4
+ * chain traces, and how big the whole thing is.
5
+ *
6
+ * The shape here is the seam between the parser (`parse.ts`) and the node
7
+ * (`protein.ts`), and it is deliberately flat and numeric: parallel arrays of
8
+ * plain numbers rather than a graph of objects. A mid-sized structure is tens of
9
+ * thousands of atoms, the node rebuilds its command list every frame, and an
10
+ * object per atom would put a few hundred thousand allocations between the
11
+ * scrubber and the picture.
12
+ *
13
+ * Parsing happens **once, ahead of the build** — the same arrangement a chart's
14
+ * CSV gets (see `SceneAsset.rows`), and for the same reason: `buildScene` is
15
+ * synchronous, so anything that has to be read off a network or a file has to
16
+ * already be here by the time it runs.
17
+ */
18
+ import { covalentRadius, traceAtomName } from "./chemistry.js";
19
+ /** An empty molecule — what an unresolved or unparseable source draws as. */
20
+ export const EMPTY_STRUCTURE = {
21
+ id: "",
22
+ title: "",
23
+ atoms: [],
24
+ bonds: [],
25
+ chains: [],
26
+ center: { x: 0, y: 0, z: 0 },
27
+ radius: 1,
28
+ };
29
+ /**
30
+ * Completes a parsed atom list into a {@link ProteinStructure}: infers the
31
+ * bonds, walks out each chain's backbone, and measures the whole thing.
32
+ *
33
+ * Split from the parsers because both formats reach the same place — an atom
34
+ * list and a set of annotated spans — and everything past that point is
35
+ * geometry rather than syntax.
36
+ */
37
+ export function deriveStructure(id, title, atoms, spans) {
38
+ if (atoms.length === 0)
39
+ return { ...EMPTY_STRUCTURE, id, title };
40
+ return {
41
+ id,
42
+ title,
43
+ atoms,
44
+ bonds: inferBonds(atoms),
45
+ chains: traceChains(atoms, spans),
46
+ ...measure(atoms),
47
+ };
48
+ }
49
+ // --- Bonds -----------------------------------------------------------------
50
+ /**
51
+ * How much further apart than the sum of their covalent radii two atoms may sit
52
+ * and still count as bonded, in Ångströms.
53
+ *
54
+ * The conventional tolerance. It has to be generous enough to survive a
55
+ * moderate-resolution structure's coordinate error and mean enough not to bond
56
+ * a residue to the one packed against it — 0.45 Å is where every viewer has
57
+ * settled, and the gap between a real bond (~1.5 Å) and the closest non-bonded
58
+ * contact (~2.8 Å) is wide enough that the exact number rarely decides anything.
59
+ */
60
+ const BOND_TOLERANCE = 0.45;
61
+ /**
62
+ * The shortest separation treated as a bond. Two atoms closer than this are
63
+ * alternate conformations of the same one, or a duplicate record — bonding them
64
+ * would draw a stick of no length and a mesh with no orientation.
65
+ */
66
+ const MIN_BOND_LENGTH = 0.4;
67
+ /**
68
+ * The furthest two atoms can be bonded, which sets the neighbour grid's cell
69
+ * size: the largest pair of covalent radii here (potassium's, twice) plus the
70
+ * tolerance.
71
+ */
72
+ const MAX_BOND_LENGTH = 2.5;
73
+ /**
74
+ * Infers bonds by distance, over a uniform grid.
75
+ *
76
+ * The grid is what makes this affordable. Bonding is a nearest-neighbour
77
+ * question and the naïve form is every atom against every other — 20,000 atoms
78
+ * is 200 million comparisons, which is a visible freeze on the frame someone
79
+ * picks a structure. Bucketed at the maximum bond length, each atom only looks
80
+ * at the 27 cells around it, and since the cells are that size no bond can span
81
+ * further, so nothing is missed. The work becomes linear in the atom count.
82
+ *
83
+ * Waters are skipped outright: they are a single oxygen with nothing to bond to
84
+ * (their hydrogens are almost never in the file), so every comparison against
85
+ * one is wasted, and there can be more of them than there are protein atoms.
86
+ */
87
+ export function inferBonds(atoms) {
88
+ const bonds = [];
89
+ const cells = new Map();
90
+ const radii = new Float32Array(atoms.length);
91
+ for (let i = 0; i < atoms.length; i++) {
92
+ const atom = atoms[i];
93
+ if (atom.kind === "water")
94
+ continue;
95
+ radii[i] = covalentRadius(atom.element);
96
+ const key = cellKey(atom.x, atom.y, atom.z);
97
+ const bucket = cells.get(key);
98
+ if (bucket)
99
+ bucket.push(i);
100
+ else
101
+ cells.set(key, [i]);
102
+ }
103
+ for (const [key, bucket] of cells) {
104
+ const [cx, cy, cz] = key.split(",").map(Number);
105
+ for (let dx = -1; dx <= 1; dx++) {
106
+ for (let dy = -1; dy <= 1; dy++) {
107
+ for (let dz = -1; dz <= 1; dz++) {
108
+ const neighbours = cells.get(`${cx + dx},${cy + dy},${cz + dz}`);
109
+ if (!neighbours)
110
+ continue;
111
+ for (const i of bucket) {
112
+ for (const j of neighbours) {
113
+ // Each unordered pair is reached from both cells, so keep only one
114
+ // ordering. This also excludes an atom against itself.
115
+ if (j <= i)
116
+ continue;
117
+ const a = atoms[i];
118
+ const b = atoms[j];
119
+ const limit = radii[i] + radii[j] + BOND_TOLERANCE;
120
+ const dxx = a.x - b.x;
121
+ const dyy = a.y - b.y;
122
+ const dzz = a.z - b.z;
123
+ const squared = dxx * dxx + dyy * dyy + dzz * dzz;
124
+ if (squared > limit * limit)
125
+ continue;
126
+ if (squared < MIN_BOND_LENGTH * MIN_BOND_LENGTH)
127
+ continue;
128
+ bonds.push({ a: i, b: j });
129
+ }
130
+ }
131
+ }
132
+ }
133
+ }
134
+ }
135
+ return bonds;
136
+ }
137
+ /** The grid cell a coordinate falls in, as a map key. */
138
+ function cellKey(x, y, z) {
139
+ const cx = Math.floor(x / MAX_BOND_LENGTH);
140
+ const cy = Math.floor(y / MAX_BOND_LENGTH);
141
+ const cz = Math.floor(z / MAX_BOND_LENGTH);
142
+ return `${cx},${cy},${cz}`;
143
+ }
144
+ // --- Chains ----------------------------------------------------------------
145
+ /**
146
+ * How far apart two consecutive α-carbons may sit and still be the same chain,
147
+ * in Ångströms.
148
+ *
149
+ * Consecutive α-carbons are 3.8 Å apart — the distance is fixed by the peptide
150
+ * bond's geometry, not by what the protein is doing — so a larger gap is a
151
+ * *chain break*: a disordered loop the crystallographer could not resolve, which
152
+ * the file records by simply skipping those residues. Drawing through one would
153
+ * run a ribbon across the middle of the molecule between two ends that are not
154
+ * joined. Nucleotide phosphorus atoms sit further apart (~6 Å), which is why the
155
+ * threshold is well above 3.8 rather than snug against it.
156
+ */
157
+ const CHAIN_BREAK_DISTANCE = 7.5;
158
+ /**
159
+ * The backbone curve of every chain, in residue order, with each residue's
160
+ * secondary structure attached.
161
+ *
162
+ * A chain break starts a **new trace** rather than being drawn through — see
163
+ * {@link CHAIN_BREAK_DISTANCE} — so one chain identifier can produce several
164
+ * entries. That is the right unit downstream anyway: each entry is one
165
+ * continuous curve, which is exactly what a tube is swept along.
166
+ */
167
+ export function traceChains(atoms, spans) {
168
+ const assign = secondaryLookup(spans);
169
+ // Built ahead of the walk rather than during it, because the carbonyl comes
170
+ // *after* the α-carbon in a residue's records and the trace point is written
171
+ // as the α-carbon is reached. One pass over the atoms costs nothing beside
172
+ // the bond inference already running over them.
173
+ const carbonyls = carbonylDirections(atoms);
174
+ const chains = [];
175
+ let current = null;
176
+ let previous = null;
177
+ for (const atom of atoms) {
178
+ if (atom.name !== traceAtomName(atom.kind))
179
+ continue;
180
+ const broken = previous !== null &&
181
+ (previous.chain !== atom.chain ||
182
+ distance(previous, atom) > CHAIN_BREAK_DISTANCE);
183
+ if (current === null || broken) {
184
+ current = { id: atom.chain, trace: [] };
185
+ chains.push(current);
186
+ }
187
+ current.trace.push({
188
+ x: atom.x,
189
+ y: atom.y,
190
+ z: atom.z,
191
+ secondary: assign(atom.chain, atom.residueSeq),
192
+ index: current.trace.length,
193
+ residueSeq: atom.residueSeq,
194
+ normal: carbonyls.get(residueKey(atom)) ?? null,
195
+ });
196
+ previous = atom;
197
+ }
198
+ // A single point is not a curve — nothing can be swept along it and no
199
+ // direction can be taken from it — so a lone resolved residue is dropped
200
+ // rather than left for the tube builder to divide by zero over.
201
+ return chains.filter((chain) => chain.trace.length > 1);
202
+ }
203
+ /**
204
+ * Builds the residue → secondary-structure lookup.
205
+ *
206
+ * A map per chain of residue number to kind, rather than a scan of the span list
207
+ * per residue: a large structure has thousands of residues and hundreds of
208
+ * spans, and the product of the two is the sort of quiet quadratic that only
209
+ * shows up on the file somebody actually cares about.
210
+ */
211
+ function secondaryLookup(spans) {
212
+ const byChain = new Map();
213
+ for (const span of spans) {
214
+ let residues = byChain.get(span.chain);
215
+ if (!residues) {
216
+ residues = new Map();
217
+ byChain.set(span.chain, residues);
218
+ }
219
+ // A malformed span (end before start, or one covering the whole file) would
220
+ // otherwise fill the map with junk; the loop simply doesn't run backwards.
221
+ for (let seq = span.start; seq <= span.end; seq++) {
222
+ residues.set(seq, span.kind);
223
+ }
224
+ }
225
+ return (chain, residue) => byChain.get(chain)?.get(residue) ?? "coil";
226
+ }
227
+ /** A residue's identity across the whole file — its chain and its number. */
228
+ function residueKey(atom) {
229
+ return `${atom.chain}:${atom.residueSeq}`;
230
+ }
231
+ /**
232
+ * The unit CA→O vector of every amino-acid residue that has both atoms.
233
+ *
234
+ * This is the one piece of chemistry a *cartoon* needs and a tube does not. A
235
+ * ribbon has a width, so it has a side, and nothing about a curve through
236
+ * α-carbons says which way round it should be: sweep a flat band along that
237
+ * curve with an arbitrary frame and a helix comes out as a twisted ribbon in
238
+ * the wrong plane, its face turning at random rather than with the fold.
239
+ *
240
+ * The carbonyl is what every molecular viewer resolves that with. It points
241
+ * roughly across the chain, it is what hydrogen-bonds a helix to itself and a
242
+ * strand to its neighbour, and so it turns *with* the structure — which makes a
243
+ * band swept along it lie the way the fold does.
244
+ *
245
+ * Amino acids only. A nucleotide's ribbon has no equivalent (its trace atom is
246
+ * the phosphorus and its "width" is the base pair), and a residue whose file
247
+ * gives no `O` — a Cα-only model, a truncated terminus — simply has no entry;
248
+ * both fall back to the curve's own binormal in the ribbon builder.
249
+ */
250
+ function carbonylDirections(atoms) {
251
+ const alpha = new Map();
252
+ const oxygen = new Map();
253
+ for (const atom of atoms) {
254
+ if (atom.kind !== "amino")
255
+ continue;
256
+ // The first record of each name wins, which is what picks conformation A
257
+ // out of a residue the file gives two of: alternates are written in order
258
+ // and the ribbon needs one frame, not an average of two.
259
+ if (atom.name === "CA") {
260
+ if (!alpha.has(residueKey(atom)))
261
+ alpha.set(residueKey(atom), atom);
262
+ }
263
+ else if (atom.name === "O") {
264
+ if (!oxygen.has(residueKey(atom)))
265
+ oxygen.set(residueKey(atom), atom);
266
+ }
267
+ }
268
+ const directions = new Map();
269
+ for (const [key, ca] of alpha) {
270
+ const o = oxygen.get(key);
271
+ if (!o)
272
+ continue;
273
+ const x = o.x - ca.x;
274
+ const y = o.y - ca.y;
275
+ const z = o.z - ca.z;
276
+ const length = Math.hypot(x, y, z);
277
+ // A zero-length carbonyl is a duplicated record rather than a direction.
278
+ if (length < 1e-6)
279
+ continue;
280
+ directions.set(key, { x: x / length, y: y / length, z: z / length });
281
+ }
282
+ return directions;
283
+ }
284
+ /** Straight-line distance between two atoms, in Ångströms. */
285
+ function distance(a, b) {
286
+ return Math.hypot(a.x - b.x, a.y - b.y, a.z - b.z);
287
+ }
288
+ // --- Bounds ----------------------------------------------------------------
289
+ /**
290
+ * The molecule's centroid and the radius of the sphere about it that holds every
291
+ * atom.
292
+ *
293
+ * The centroid rather than the centre of the bounding box: a structure with one
294
+ * long tail — a coiled-coil, a nucleic acid strand hanging off a complex —
295
+ * has a box whose centre is nowhere near the mass of it, and the camera would
296
+ * frame the empty half.
297
+ *
298
+ * The radius is floored above zero so a single-atom file (or one whose
299
+ * coordinates are all identical) still gives the camera a distance to work from
300
+ * rather than parking it inside the atom.
301
+ */
302
+ function measure(atoms) {
303
+ let sx = 0;
304
+ let sy = 0;
305
+ let sz = 0;
306
+ for (const atom of atoms) {
307
+ sx += atom.x;
308
+ sy += atom.y;
309
+ sz += atom.z;
310
+ }
311
+ const center = {
312
+ x: sx / atoms.length,
313
+ y: sy / atoms.length,
314
+ z: sz / atoms.length,
315
+ };
316
+ let radius = 0;
317
+ for (const atom of atoms) {
318
+ const d = Math.hypot(atom.x - center.x, atom.y - center.y, atom.z - center.z);
319
+ if (d > radius)
320
+ radius = d;
321
+ }
322
+ return { center, radius: Math.max(radius, 1) };
323
+ }
324
+ //# sourceMappingURL=structure.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"structure.js","sourceRoot":"","sources":["../../src/protein/structure.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;GAgBG;AAEH,OAAO,EAAE,cAAc,EAAE,aAAa,EAAoB,MAAM,aAAa,CAAA;AAuH7E,6EAA6E;AAC7E,MAAM,CAAC,MAAM,eAAe,GAAqB;IAC/C,EAAE,EAAE,EAAE;IACN,KAAK,EAAE,EAAE;IACT,KAAK,EAAE,EAAE;IACT,KAAK,EAAE,EAAE;IACT,MAAM,EAAE,EAAE;IACV,MAAM,EAAE,EAAE,CAAC,EAAE,CAAC,EAAE,CAAC,EAAE,CAAC,EAAE,CAAC,EAAE,CAAC,EAAE;IAC5B,MAAM,EAAE,CAAC;CACV,CAAA;AAaD;;;;;;;GAOG;AACH,MAAM,UAAU,eAAe,CAC7B,EAAU,EACV,KAAa,EACb,KAAoB,EACpB,KAAsB;IAEtB,IAAI,KAAK,CAAC,MAAM,KAAK,CAAC;QAAE,OAAO,EAAE,GAAG,eAAe,EAAE,EAAE,EAAE,KAAK,EAAE,CAAA;IAEhE,OAAO;QACL,EAAE;QACF,KAAK;QACL,KAAK;QACL,KAAK,EAAE,UAAU,CAAC,KAAK,CAAC;QACxB,MAAM,EAAE,WAAW,CAAC,KAAK,EAAE,KAAK,CAAC;QACjC,GAAG,OAAO,CAAC,KAAK,CAAC;KAClB,CAAA;AACH,CAAC;AAED,8EAA8E;AAE9E;;;;;;;;;GASG;AACH,MAAM,cAAc,GAAG,IAAI,CAAA;AAE3B;;;;GAIG;AACH,MAAM,eAAe,GAAG,GAAG,CAAA;AAE3B;;;;GAIG;AACH,MAAM,eAAe,GAAG,GAAG,CAAA;AAE3B;;;;;;;;;;;;;GAaG;AACH,MAAM,UAAU,UAAU,CAAC,KAAoB;IAC7C,MAAM,KAAK,GAAkB,EAAE,CAAA;IAC/B,MAAM,KAAK,GAAG,IAAI,GAAG,EAAoB,CAAA;IACzC,MAAM,KAAK,GAAG,IAAI,YAAY,CAAC,KAAK,CAAC,MAAM,CAAC,CAAA;IAE5C,KAAK,IAAI,CAAC,GAAG,CAAC,EAAE,CAAC,GAAG,KAAK,CAAC,MAAM,EAAE,CAAC,EAAE,EAAE,CAAC;QACtC,MAAM,IAAI,GAAG,KAAK,CAAC,CAAC,CAAE,CAAA;QACtB,IAAI,IAAI,CAAC,IAAI,KAAK,OAAO;YAAE,SAAQ;QACnC,KAAK,CAAC,CAAC,CAAC,GAAG,cAAc,CAAC,IAAI,CAAC,OAAO,CAAC,CAAA;QACvC,MAAM,GAAG,GAAG,OAAO,CAAC,IAAI,CAAC,CAAC,EAAE,IAAI,CAAC,CAAC,EAAE,IAAI,CAAC,CAAC,CAAC,CAAA;QAC3C,MAAM,MAAM,GAAG,KAAK,CAAC,GAAG,CAAC,GAAG,CAAC,CAAA;QAC7B,IAAI,MAAM;YAAE,MAAM,CAAC,IAAI,CAAC,CAAC,CAAC,CAAA;;YACrB,KAAK,CAAC,GAAG,CAAC,GAAG,EAAE,CAAC,CAAC,CAAC,CAAC,CAAA;IAC1B,CAAC;IAED,KAAK,MAAM,CAAC,GAAG,EAAE,MAAM,CAAC,IAAI,KAAK,EAAE,CAAC;QAClC,MAAM,CAAC,EAAE,EAAE,EAAE,EAAE,EAAE,CAAC,GAAG,GAAG,CAAC,KAAK,CAAC,GAAG,CAAC,CAAC,GAAG,CAAC,MAAM,CAA6B,CAAA;QAE3E,KAAK,IAAI,EAAE,GAAG,CAAC,CAAC,EAAE,EAAE,IAAI,CAAC,EAAE,EAAE,EAAE,EAAE,CAAC;YAChC,KAAK,IAAI,EAAE,GAAG,CAAC,CAAC,EAAE,EAAE,IAAI,CAAC,EAAE,EAAE,EAAE,EAAE,CAAC;gBAChC,KAAK,IAAI,EAAE,GAAG,CAAC,CAAC,EAAE,EAAE,IAAI,CAAC,EAAE,EAAE,EAAE,EAAE,CAAC;oBAChC,MAAM,UAAU,GAAG,KAAK,CAAC,GAAG,CAAC,GAAG,EAAE,GAAG,EAAE,IAAI,EAAE,GAAG,EAAE,IAAI,EAAE,GAAG,EAAE,EAAE,CAAC,CAAA;oBAChE,IAAI,CAAC,UAAU;wBAAE,SAAQ;oBAEzB,KAAK,MAAM,CAAC,IAAI,MAAM,EAAE,CAAC;wBACvB,KAAK,MAAM,CAAC,IAAI,UAAU,EAAE,CAAC;4BAC3B,mEAAmE;4BACnE,uDAAuD;4BACvD,IAAI,CAAC,IAAI,CAAC;gCAAE,SAAQ;4BACpB,MAAM,CAAC,GAAG,KAAK,CAAC,CAAC,CAAE,CAAA;4BACnB,MAAM,CAAC,GAAG,KAAK,CAAC,CAAC,CAAE,CAAA;4BACnB,MAAM,KAAK,GAAG,KAAK,CAAC,CAAC,CAAE,GAAG,KAAK,CAAC,CAAC,CAAE,GAAG,cAAc,CAAA;4BACpD,MAAM,GAAG,GAAG,CAAC,CAAC,CAAC,GAAG,CAAC,CAAC,CAAC,CAAA;4BACrB,MAAM,GAAG,GAAG,CAAC,CAAC,CAAC,GAAG,CAAC,CAAC,CAAC,CAAA;4BACrB,MAAM,GAAG,GAAG,CAAC,CAAC,CAAC,GAAG,CAAC,CAAC,CAAC,CAAA;4BACrB,MAAM,OAAO,GAAG,GAAG,GAAG,GAAG,GAAG,GAAG,GAAG,GAAG,GAAG,GAAG,GAAG,GAAG,CAAA;4BACjD,IAAI,OAAO,GAAG,KAAK,GAAG,KAAK;gCAAE,SAAQ;4BACrC,IAAI,OAAO,GAAG,eAAe,GAAG,eAAe;gCAAE,SAAQ;4BACzD,KAAK,CAAC,IAAI,CAAC,EAAE,CAAC,EAAE,CAAC,EAAE,CAAC,EAAE,CAAC,EAAE,CAAC,CAAA;wBAC5B,CAAC;oBACH,CAAC;gBACH,CAAC;YACH,CAAC;QACH,CAAC;IACH,CAAC;IAED,OAAO,KAAK,CAAA;AACd,CAAC;AAED,yDAAyD;AACzD,SAAS,OAAO,CAAC,CAAS,EAAE,CAAS,EAAE,CAAS;IAC9C,MAAM,EAAE,GAAG,IAAI,CAAC,KAAK,CAAC,CAAC,GAAG,eAAe,CAAC,CAAA;IAC1C,MAAM,EAAE,GAAG,IAAI,CAAC,KAAK,CAAC,CAAC,GAAG,eAAe,CAAC,CAAA;IAC1C,MAAM,EAAE,GAAG,IAAI,CAAC,KAAK,CAAC,CAAC,GAAG,eAAe,CAAC,CAAA;IAC1C,OAAO,GAAG,EAAE,IAAI,EAAE,IAAI,EAAE,EAAE,CAAA;AAC5B,CAAC;AAED,8EAA8E;AAE9E;;;;;;;;;;;GAWG;AACH,MAAM,oBAAoB,GAAG,GAAG,CAAA;AAEhC;;;;;;;;GAQG;AACH,MAAM,UAAU,WAAW,CACzB,KAAoB,EACpB,KAAsB;IAEtB,MAAM,MAAM,GAAG,eAAe,CAAC,KAAK,CAAC,CAAA;IACrC,4EAA4E;IAC5E,6EAA6E;IAC7E,2EAA2E;IAC3E,gDAAgD;IAChD,MAAM,SAAS,GAAG,kBAAkB,CAAC,KAAK,CAAC,CAAA;IAC3C,MAAM,MAAM,GAAmB,EAAE,CAAA;IAEjC,IAAI,OAAO,GAAwB,IAAI,CAAA;IACvC,IAAI,QAAQ,GAAuB,IAAI,CAAA;IAEvC,KAAK,MAAM,IAAI,IAAI,KAAK,EAAE,CAAC;QACzB,IAAI,IAAI,CAAC,IAAI,KAAK,aAAa,CAAC,IAAI,CAAC,IAAI,CAAC;YAAE,SAAQ;QAEpD,MAAM,MAAM,GACV,QAAQ,KAAK,IAAI;YACjB,CAAC,QAAQ,CAAC,KAAK,KAAK,IAAI,CAAC,KAAK;gBAC5B,QAAQ,CAAC,QAAQ,EAAE,IAAI,CAAC,GAAG,oBAAoB,CAAC,CAAA;QAEpD,IAAI,OAAO,KAAK,IAAI,IAAI,MAAM,EAAE,CAAC;YAC/B,OAAO,GAAG,EAAE,EAAE,EAAE,IAAI,CAAC,KAAK,EAAE,KAAK,EAAE,EAAE,EAAE,CAAA;YACvC,MAAM,CAAC,IAAI,CAAC,OAAO,CAAC,CAAA;QACtB,CAAC;QAED,OAAO,CAAC,KAAK,CAAC,IAAI,CAAC;YACjB,CAAC,EAAE,IAAI,CAAC,CAAC;YACT,CAAC,EAAE,IAAI,CAAC,CAAC;YACT,CAAC,EAAE,IAAI,CAAC,CAAC;YACT,SAAS,EAAE,MAAM,CAAC,IAAI,CAAC,KAAK,EAAE,IAAI,CAAC,UAAU,CAAC;YAC9C,KAAK,EAAE,OAAO,CAAC,KAAK,CAAC,MAAM;YAC3B,UAAU,EAAE,IAAI,CAAC,UAAU;YAC3B,MAAM,EAAE,SAAS,CAAC,GAAG,CAAC,UAAU,CAAC,IAAI,CAAC,CAAC,IAAI,IAAI;SAChD,CAAC,CAAA;QACF,QAAQ,GAAG,IAAI,CAAA;IACjB,CAAC;IAED,uEAAuE;IACvE,yEAAyE;IACzE,gEAAgE;IAChE,OAAO,MAAM,CAAC,MAAM,CAAC,CAAC,KAAK,EAAE,EAAE,CAAC,KAAK,CAAC,KAAK,CAAC,MAAM,GAAG,CAAC,CAAC,CAAA;AACzD,CAAC;AAED;;;;;;;GAOG;AACH,SAAS,eAAe,CACtB,KAAsB;IAEtB,MAAM,OAAO,GAAG,IAAI,GAAG,EAA2C,CAAA;IAElE,KAAK,MAAM,IAAI,IAAI,KAAK,EAAE,CAAC;QACzB,IAAI,QAAQ,GAAG,OAAO,CAAC,GAAG,CAAC,IAAI,CAAC,KAAK,CAAC,CAAA;QACtC,IAAI,CAAC,QAAQ,EAAE,CAAC;YACd,QAAQ,GAAG,IAAI,GAAG,EAAE,CAAA;YACpB,OAAO,CAAC,GAAG,CAAC,IAAI,CAAC,KAAK,EAAE,QAAQ,CAAC,CAAA;QACnC,CAAC;QACD,4EAA4E;QAC5E,2EAA2E;QAC3E,KAAK,IAAI,GAAG,GAAG,IAAI,CAAC,KAAK,EAAE,GAAG,IAAI,IAAI,CAAC,GAAG,EAAE,GAAG,EAAE,EAAE,CAAC;YAClD,QAAQ,CAAC,GAAG,CAAC,GAAG,EAAE,IAAI,CAAC,IAAI,CAAC,CAAA;QAC9B,CAAC;IACH,CAAC;IAED,OAAO,CAAC,KAAK,EAAE,OAAO,EAAE,EAAE,CAAC,OAAO,CAAC,GAAG,CAAC,KAAK,CAAC,EAAE,GAAG,CAAC,OAAO,CAAC,IAAI,MAAM,CAAA;AACvE,CAAC;AAED,6EAA6E;AAC7E,SAAS,UAAU,CAAC,IAAiB;IACnC,OAAO,GAAG,IAAI,CAAC,KAAK,IAAI,IAAI,CAAC,UAAU,EAAE,CAAA;AAC3C,CAAC;AAED;;;;;;;;;;;;;;;;;;GAkBG;AACH,SAAS,kBAAkB,CACzB,KAAoB;IAEpB,MAAM,KAAK,GAAG,IAAI,GAAG,EAAuB,CAAA;IAC5C,MAAM,MAAM,GAAG,IAAI,GAAG,EAAuB,CAAA;IAE7C,KAAK,MAAM,IAAI,IAAI,KAAK,EAAE,CAAC;QACzB,IAAI,IAAI,CAAC,IAAI,KAAK,OAAO;YAAE,SAAQ;QACnC,yEAAyE;QACzE,0EAA0E;QAC1E,yDAAyD;QACzD,IAAI,IAAI,CAAC,IAAI,KAAK,IAAI,EAAE,CAAC;YACvB,IAAI,CAAC,KAAK,CAAC,GAAG,CAAC,UAAU,CAAC,IAAI,CAAC,CAAC;gBAAE,KAAK,CAAC,GAAG,CAAC,UAAU,CAAC,IAAI,CAAC,EAAE,IAAI,CAAC,CAAA;QACrE,CAAC;aAAM,IAAI,IAAI,CAAC,IAAI,KAAK,GAAG,EAAE,CAAC;YAC7B,IAAI,CAAC,MAAM,CAAC,GAAG,CAAC,UAAU,CAAC,IAAI,CAAC,CAAC;gBAAE,MAAM,CAAC,GAAG,CAAC,UAAU,CAAC,IAAI,CAAC,EAAE,IAAI,CAAC,CAAA;QACvE,CAAC;IACH,CAAC;IAED,MAAM,UAAU,GAAG,IAAI,GAAG,EAA+C,CAAA;IACzE,KAAK,MAAM,CAAC,GAAG,EAAE,EAAE,CAAC,IAAI,KAAK,EAAE,CAAC;QAC9B,MAAM,CAAC,GAAG,MAAM,CAAC,GAAG,CAAC,GAAG,CAAC,CAAA;QACzB,IAAI,CAAC,CAAC;YAAE,SAAQ;QAChB,MAAM,CAAC,GAAG,CAAC,CAAC,CAAC,GAAG,EAAE,CAAC,CAAC,CAAA;QACpB,MAAM,CAAC,GAAG,CAAC,CAAC,CAAC,GAAG,EAAE,CAAC,CAAC,CAAA;QACpB,MAAM,CAAC,GAAG,CAAC,CAAC,CAAC,GAAG,EAAE,CAAC,CAAC,CAAA;QACpB,MAAM,MAAM,GAAG,IAAI,CAAC,KAAK,CAAC,CAAC,EAAE,CAAC,EAAE,CAAC,CAAC,CAAA;QAClC,yEAAyE;QACzE,IAAI,MAAM,GAAG,IAAI;YAAE,SAAQ;QAC3B,UAAU,CAAC,GAAG,CAAC,GAAG,EAAE,EAAE,CAAC,EAAE,CAAC,GAAG,MAAM,EAAE,CAAC,EAAE,CAAC,GAAG,MAAM,EAAE,CAAC,EAAE,CAAC,GAAG,MAAM,EAAE,CAAC,CAAA;IACtE,CAAC;IACD,OAAO,UAAU,CAAA;AACnB,CAAC;AAED,8DAA8D;AAC9D,SAAS,QAAQ,CAAC,CAAc,EAAE,CAAc;IAC9C,OAAO,IAAI,CAAC,KAAK,CAAC,CAAC,CAAC,CAAC,GAAG,CAAC,CAAC,CAAC,EAAE,CAAC,CAAC,CAAC,GAAG,CAAC,CAAC,CAAC,EAAE,CAAC,CAAC,CAAC,GAAG,CAAC,CAAC,CAAC,CAAC,CAAA;AACpD,CAAC;AAED,8EAA8E;AAE9E;;;;;;;;;;;;GAYG;AACH,SAAS,OAAO,CAAC,KAAoB;IAInC,IAAI,EAAE,GAAG,CAAC,CAAA;IACV,IAAI,EAAE,GAAG,CAAC,CAAA;IACV,IAAI,EAAE,GAAG,CAAC,CAAA;IACV,KAAK,MAAM,IAAI,IAAI,KAAK,EAAE,CAAC;QACzB,EAAE,IAAI,IAAI,CAAC,CAAC,CAAA;QACZ,EAAE,IAAI,IAAI,CAAC,CAAC,CAAA;QACZ,EAAE,IAAI,IAAI,CAAC,CAAC,CAAA;IACd,CAAC;IACD,MAAM,MAAM,GAAG;QACb,CAAC,EAAE,EAAE,GAAG,KAAK,CAAC,MAAM;QACpB,CAAC,EAAE,EAAE,GAAG,KAAK,CAAC,MAAM;QACpB,CAAC,EAAE,EAAE,GAAG,KAAK,CAAC,MAAM;KACrB,CAAA;IAED,IAAI,MAAM,GAAG,CAAC,CAAA;IACd,KAAK,MAAM,IAAI,IAAI,KAAK,EAAE,CAAC;QACzB,MAAM,CAAC,GAAG,IAAI,CAAC,KAAK,CAClB,IAAI,CAAC,CAAC,GAAG,MAAM,CAAC,CAAC,EACjB,IAAI,CAAC,CAAC,GAAG,MAAM,CAAC,CAAC,EACjB,IAAI,CAAC,CAAC,GAAG,MAAM,CAAC,CAAC,CAClB,CAAA;QACD,IAAI,CAAC,GAAG,MAAM;YAAE,MAAM,GAAG,CAAC,CAAA;IAC5B,CAAC;IAED,OAAO,EAAE,MAAM,EAAE,MAAM,EAAE,IAAI,CAAC,GAAG,CAAC,MAAM,EAAE,CAAC,CAAC,EAAE,CAAA;AAChD,CAAC","sourcesContent":["/**\n * What a parsed molecule *is*, and the three things derived from it that no\n * coordinate file states outright: which atoms are bonded, which curve each\n * chain traces, and how big the whole thing is.\n *\n * The shape here is the seam between the parser (`parse.ts`) and the node\n * (`protein.ts`), and it is deliberately flat and numeric: parallel arrays of\n * plain numbers rather than a graph of objects. A mid-sized structure is tens of\n * thousands of atoms, the node rebuilds its command list every frame, and an\n * object per atom would put a few hundred thousand allocations between the\n * scrubber and the picture.\n *\n * Parsing happens **once, ahead of the build** — the same arrangement a chart's\n * CSV gets (see `SceneAsset.rows`), and for the same reason: `buildScene` is\n * synchronous, so anything that has to be read off a network or a file has to\n * already be here by the time it runs.\n */\n\nimport { covalentRadius, traceAtomName, type ResidueKind } from \"./chemistry\"\n\n/** One atom, as the file gave it. Coordinates are in Ångströms. */\nexport interface ProteinAtom {\n /** The file's own serial number, kept so `CONECT` records can be resolved. */\n serial: number\n /** The atom's name within its residue — `CA`, `OG1`, `FE`. */\n name: string\n /** The element symbol, upper-cased. Derived when the file omits it. */\n element: string\n /** The residue's three-letter code — `ALA`, `HOH`, `HEM`. */\n residue: string\n /** The residue's number within its chain, as the file numbers it. */\n residueSeq: number\n /** The chain the residue belongs to — `A`, `B`, … */\n chain: string\n x: number\n y: number\n z: number\n /** Whether the file filed this under `HETATM` — a ligand, an ion, a water. */\n hetero: boolean\n /** What the residue is, resolved once here so nothing downstream re-derives it. */\n kind: ResidueKind\n}\n\n/**\n * What a stretch of backbone is doing, which is the one property of a protein\n * that a picture of it is usually *about*.\n *\n * Taken from the file's own `HELIX`/`SHEET` annotations rather than computed:\n * assigning secondary structure from coordinates alone is DSSP, which is a\n * hydrogen-bond model and a paper of its own. A file that annotates none draws\n * as an even tube, which is the honest picture of \"nobody said\".\n */\nexport type SecondaryStructure = \"helix\" | \"sheet\" | \"coil\"\n\n/** One residue's place on a chain's backbone curve. */\nexport interface TracePoint {\n x: number\n y: number\n z: number\n /** What this residue is part of, for the ribbon's shape and its colouring. */\n secondary: SecondaryStructure\n /**\n * The residue's position along **this trace**, 0-based. Against the trace's\n * own length this is what a spectrum colouring needs: N terminus to C\n * terminus, per chain, independent of how long the others are.\n */\n index: number\n /**\n * The residue's number as the file gives it, which is what ties a trace point\n * back to the atoms of the same residue.\n *\n * Both are needed and neither substitutes for the other: {@link index} is\n * dense and ordered, so it is what a spectrum is computed against, while this\n * is sparse and arbitrary — files skip numbers, start at 17, and number\n * insertions — so it is what a *lookup* has to be keyed on.\n */\n residueSeq: number\n /**\n * Which way this residue's ribbon faces — the unit vector its **wide**\n * direction is swept along.\n *\n * A tube needs a point and nothing else; a cartoon ribbon needs to know which\n * way round it is, and that is not something a curve through α-carbons can\n * answer. The residue's own carbonyl does: CA→O is roughly perpendicular to\n * the chain and turns with the fold, so sweeping the ribbon's width along it\n * is what makes a helix read as a twisted band and a strand as a flat one.\n * This is the same vector every molecular viewer builds its cartoon frame\n * from.\n *\n * `null` when the file has no carbonyl for the residue — a Cα-only model, a\n * nucleotide, the last residue of a truncated chain — and the ribbon builder\n * falls back to the curve's own binormal there. See `ribbonFrames`.\n *\n * Not orthogonalized here: it is a *direction*, and the frame that has to be\n * square is built against the tangent, which only exists once the neighbours\n * are known.\n */\n normal: { x: number; y: number; z: number } | null\n}\n\n/** One polymer chain: its identity and the curve its backbone follows. */\nexport interface ProteinChain {\n /** The chain identifier the file uses — `A`, `B`, … */\n id: string\n /** The α-carbon (or phosphorus) trace, in residue order. */\n trace: TracePoint[]\n}\n\n/**\n * A parsed molecule.\n *\n * Bonds and chains are *derived* (see {@link deriveStructure}) rather than read,\n * because coordinate files mostly don't state them: `CONECT` records cover the\n * ligands and little else, and the chain break between two residues is implied\n * by their distance.\n */\nexport interface ProteinStructure {\n /** The accession or filename this came from, for labelling. */\n id: string\n /** The file's `TITLE`, when it has one. */\n title: string\n atoms: ProteinAtom[]\n /** Bonded pairs, as indices into {@link atoms}. */\n bonds: ProteinBond[]\n chains: ProteinChain[]\n /** The centroid of every atom — what the node translates to the origin. */\n center: { x: number; y: number; z: number }\n /** Distance from {@link center} to the furthest atom, in Ångströms. */\n radius: number\n}\n\n/** One bond, as a pair of indices into {@link ProteinStructure.atoms}. */\nexport interface ProteinBond {\n a: number\n b: number\n}\n\n/** An empty molecule — what an unresolved or unparseable source draws as. */\nexport const EMPTY_STRUCTURE: ProteinStructure = {\n id: \"\",\n title: \"\",\n atoms: [],\n bonds: [],\n chains: [],\n center: { x: 0, y: 0, z: 0 },\n radius: 1,\n}\n\n/**\n * A secondary-structure span as the file states it: a run of residue numbers on\n * one chain. Collected by the parser, applied to the traces here.\n */\nexport interface SecondarySpan {\n chain: string\n start: number\n end: number\n kind: Exclude<SecondaryStructure, \"coil\">\n}\n\n/**\n * Completes a parsed atom list into a {@link ProteinStructure}: infers the\n * bonds, walks out each chain's backbone, and measures the whole thing.\n *\n * Split from the parsers because both formats reach the same place — an atom\n * list and a set of annotated spans — and everything past that point is\n * geometry rather than syntax.\n */\nexport function deriveStructure(\n id: string,\n title: string,\n atoms: ProteinAtom[],\n spans: SecondarySpan[]\n): ProteinStructure {\n if (atoms.length === 0) return { ...EMPTY_STRUCTURE, id, title }\n\n return {\n id,\n title,\n atoms,\n bonds: inferBonds(atoms),\n chains: traceChains(atoms, spans),\n ...measure(atoms),\n }\n}\n\n// --- Bonds -----------------------------------------------------------------\n\n/**\n * How much further apart than the sum of their covalent radii two atoms may sit\n * and still count as bonded, in Ångströms.\n *\n * The conventional tolerance. It has to be generous enough to survive a\n * moderate-resolution structure's coordinate error and mean enough not to bond\n * a residue to the one packed against it — 0.45 Å is where every viewer has\n * settled, and the gap between a real bond (~1.5 Å) and the closest non-bonded\n * contact (~2.8 Å) is wide enough that the exact number rarely decides anything.\n */\nconst BOND_TOLERANCE = 0.45\n\n/**\n * The shortest separation treated as a bond. Two atoms closer than this are\n * alternate conformations of the same one, or a duplicate record — bonding them\n * would draw a stick of no length and a mesh with no orientation.\n */\nconst MIN_BOND_LENGTH = 0.4\n\n/**\n * The furthest two atoms can be bonded, which sets the neighbour grid's cell\n * size: the largest pair of covalent radii here (potassium's, twice) plus the\n * tolerance.\n */\nconst MAX_BOND_LENGTH = 2.5\n\n/**\n * Infers bonds by distance, over a uniform grid.\n *\n * The grid is what makes this affordable. Bonding is a nearest-neighbour\n * question and the naïve form is every atom against every other — 20,000 atoms\n * is 200 million comparisons, which is a visible freeze on the frame someone\n * picks a structure. Bucketed at the maximum bond length, each atom only looks\n * at the 27 cells around it, and since the cells are that size no bond can span\n * further, so nothing is missed. The work becomes linear in the atom count.\n *\n * Waters are skipped outright: they are a single oxygen with nothing to bond to\n * (their hydrogens are almost never in the file), so every comparison against\n * one is wasted, and there can be more of them than there are protein atoms.\n */\nexport function inferBonds(atoms: ProteinAtom[]): ProteinBond[] {\n const bonds: ProteinBond[] = []\n const cells = new Map<string, number[]>()\n const radii = new Float32Array(atoms.length)\n\n for (let i = 0; i < atoms.length; i++) {\n const atom = atoms[i]!\n if (atom.kind === \"water\") continue\n radii[i] = covalentRadius(atom.element)\n const key = cellKey(atom.x, atom.y, atom.z)\n const bucket = cells.get(key)\n if (bucket) bucket.push(i)\n else cells.set(key, [i])\n }\n\n for (const [key, bucket] of cells) {\n const [cx, cy, cz] = key.split(\",\").map(Number) as [number, number, number]\n\n for (let dx = -1; dx <= 1; dx++) {\n for (let dy = -1; dy <= 1; dy++) {\n for (let dz = -1; dz <= 1; dz++) {\n const neighbours = cells.get(`${cx + dx},${cy + dy},${cz + dz}`)\n if (!neighbours) continue\n\n for (const i of bucket) {\n for (const j of neighbours) {\n // Each unordered pair is reached from both cells, so keep only one\n // ordering. This also excludes an atom against itself.\n if (j <= i) continue\n const a = atoms[i]!\n const b = atoms[j]!\n const limit = radii[i]! + radii[j]! + BOND_TOLERANCE\n const dxx = a.x - b.x\n const dyy = a.y - b.y\n const dzz = a.z - b.z\n const squared = dxx * dxx + dyy * dyy + dzz * dzz\n if (squared > limit * limit) continue\n if (squared < MIN_BOND_LENGTH * MIN_BOND_LENGTH) continue\n bonds.push({ a: i, b: j })\n }\n }\n }\n }\n }\n }\n\n return bonds\n}\n\n/** The grid cell a coordinate falls in, as a map key. */\nfunction cellKey(x: number, y: number, z: number): string {\n const cx = Math.floor(x / MAX_BOND_LENGTH)\n const cy = Math.floor(y / MAX_BOND_LENGTH)\n const cz = Math.floor(z / MAX_BOND_LENGTH)\n return `${cx},${cy},${cz}`\n}\n\n// --- Chains ----------------------------------------------------------------\n\n/**\n * How far apart two consecutive α-carbons may sit and still be the same chain,\n * in Ångströms.\n *\n * Consecutive α-carbons are 3.8 Å apart — the distance is fixed by the peptide\n * bond's geometry, not by what the protein is doing — so a larger gap is a\n * *chain break*: a disordered loop the crystallographer could not resolve, which\n * the file records by simply skipping those residues. Drawing through one would\n * run a ribbon across the middle of the molecule between two ends that are not\n * joined. Nucleotide phosphorus atoms sit further apart (~6 Å), which is why the\n * threshold is well above 3.8 rather than snug against it.\n */\nconst CHAIN_BREAK_DISTANCE = 7.5\n\n/**\n * The backbone curve of every chain, in residue order, with each residue's\n * secondary structure attached.\n *\n * A chain break starts a **new trace** rather than being drawn through — see\n * {@link CHAIN_BREAK_DISTANCE} — so one chain identifier can produce several\n * entries. That is the right unit downstream anyway: each entry is one\n * continuous curve, which is exactly what a tube is swept along.\n */\nexport function traceChains(\n atoms: ProteinAtom[],\n spans: SecondarySpan[]\n): ProteinChain[] {\n const assign = secondaryLookup(spans)\n // Built ahead of the walk rather than during it, because the carbonyl comes\n // *after* the α-carbon in a residue's records and the trace point is written\n // as the α-carbon is reached. One pass over the atoms costs nothing beside\n // the bond inference already running over them.\n const carbonyls = carbonylDirections(atoms)\n const chains: ProteinChain[] = []\n\n let current: ProteinChain | null = null\n let previous: ProteinAtom | null = null\n\n for (const atom of atoms) {\n if (atom.name !== traceAtomName(atom.kind)) continue\n\n const broken =\n previous !== null &&\n (previous.chain !== atom.chain ||\n distance(previous, atom) > CHAIN_BREAK_DISTANCE)\n\n if (current === null || broken) {\n current = { id: atom.chain, trace: [] }\n chains.push(current)\n }\n\n current.trace.push({\n x: atom.x,\n y: atom.y,\n z: atom.z,\n secondary: assign(atom.chain, atom.residueSeq),\n index: current.trace.length,\n residueSeq: atom.residueSeq,\n normal: carbonyls.get(residueKey(atom)) ?? null,\n })\n previous = atom\n }\n\n // A single point is not a curve — nothing can be swept along it and no\n // direction can be taken from it — so a lone resolved residue is dropped\n // rather than left for the tube builder to divide by zero over.\n return chains.filter((chain) => chain.trace.length > 1)\n}\n\n/**\n * Builds the residue → secondary-structure lookup.\n *\n * A map per chain of residue number to kind, rather than a scan of the span list\n * per residue: a large structure has thousands of residues and hundreds of\n * spans, and the product of the two is the sort of quiet quadratic that only\n * shows up on the file somebody actually cares about.\n */\nfunction secondaryLookup(\n spans: SecondarySpan[]\n): (chain: string, residue: number) => SecondaryStructure {\n const byChain = new Map<string, Map<number, SecondaryStructure>>()\n\n for (const span of spans) {\n let residues = byChain.get(span.chain)\n if (!residues) {\n residues = new Map()\n byChain.set(span.chain, residues)\n }\n // A malformed span (end before start, or one covering the whole file) would\n // otherwise fill the map with junk; the loop simply doesn't run backwards.\n for (let seq = span.start; seq <= span.end; seq++) {\n residues.set(seq, span.kind)\n }\n }\n\n return (chain, residue) => byChain.get(chain)?.get(residue) ?? \"coil\"\n}\n\n/** A residue's identity across the whole file — its chain and its number. */\nfunction residueKey(atom: ProteinAtom): string {\n return `${atom.chain}:${atom.residueSeq}`\n}\n\n/**\n * The unit CA→O vector of every amino-acid residue that has both atoms.\n *\n * This is the one piece of chemistry a *cartoon* needs and a tube does not. A\n * ribbon has a width, so it has a side, and nothing about a curve through\n * α-carbons says which way round it should be: sweep a flat band along that\n * curve with an arbitrary frame and a helix comes out as a twisted ribbon in\n * the wrong plane, its face turning at random rather than with the fold.\n *\n * The carbonyl is what every molecular viewer resolves that with. It points\n * roughly across the chain, it is what hydrogen-bonds a helix to itself and a\n * strand to its neighbour, and so it turns *with* the structure — which makes a\n * band swept along it lie the way the fold does.\n *\n * Amino acids only. A nucleotide's ribbon has no equivalent (its trace atom is\n * the phosphorus and its \"width\" is the base pair), and a residue whose file\n * gives no `O` — a Cα-only model, a truncated terminus — simply has no entry;\n * both fall back to the curve's own binormal in the ribbon builder.\n */\nfunction carbonylDirections(\n atoms: ProteinAtom[]\n): Map<string, { x: number; y: number; z: number }> {\n const alpha = new Map<string, ProteinAtom>()\n const oxygen = new Map<string, ProteinAtom>()\n\n for (const atom of atoms) {\n if (atom.kind !== \"amino\") continue\n // The first record of each name wins, which is what picks conformation A\n // out of a residue the file gives two of: alternates are written in order\n // and the ribbon needs one frame, not an average of two.\n if (atom.name === \"CA\") {\n if (!alpha.has(residueKey(atom))) alpha.set(residueKey(atom), atom)\n } else if (atom.name === \"O\") {\n if (!oxygen.has(residueKey(atom))) oxygen.set(residueKey(atom), atom)\n }\n }\n\n const directions = new Map<string, { x: number; y: number; z: number }>()\n for (const [key, ca] of alpha) {\n const o = oxygen.get(key)\n if (!o) continue\n const x = o.x - ca.x\n const y = o.y - ca.y\n const z = o.z - ca.z\n const length = Math.hypot(x, y, z)\n // A zero-length carbonyl is a duplicated record rather than a direction.\n if (length < 1e-6) continue\n directions.set(key, { x: x / length, y: y / length, z: z / length })\n }\n return directions\n}\n\n/** Straight-line distance between two atoms, in Ångströms. */\nfunction distance(a: ProteinAtom, b: ProteinAtom): number {\n return Math.hypot(a.x - b.x, a.y - b.y, a.z - b.z)\n}\n\n// --- Bounds ----------------------------------------------------------------\n\n/**\n * The molecule's centroid and the radius of the sphere about it that holds every\n * atom.\n *\n * The centroid rather than the centre of the bounding box: a structure with one\n * long tail — a coiled-coil, a nucleic acid strand hanging off a complex —\n * has a box whose centre is nowhere near the mass of it, and the camera would\n * frame the empty half.\n *\n * The radius is floored above zero so a single-atom file (or one whose\n * coordinates are all identical) still gives the camera a distance to work from\n * rather than parking it inside the atom.\n */\nfunction measure(atoms: ProteinAtom[]): {\n center: { x: number; y: number; z: number }\n radius: number\n} {\n let sx = 0\n let sy = 0\n let sz = 0\n for (const atom of atoms) {\n sx += atom.x\n sy += atom.y\n sz += atom.z\n }\n const center = {\n x: sx / atoms.length,\n y: sy / atoms.length,\n z: sz / atoms.length,\n }\n\n let radius = 0\n for (const atom of atoms) {\n const d = Math.hypot(\n atom.x - center.x,\n atom.y - center.y,\n atom.z - center.z\n )\n if (d > radius) radius = d\n }\n\n return { center, radius: Math.max(radius, 1) }\n}\n"]}
package/package.json CHANGED
@@ -1,6 +1,67 @@
1
1
  {
2
2
  "name": "@motionscript/molecule",
3
- "version": "0.0.0-stage",
4
- "stub": true,
5
- "description": "Temporary package placeholder for staged publishing"
3
+ "version": "0.1.0-alpha.0",
4
+ "type": "module",
5
+ "description": "Animated molecular structures for Motion Script: cartoon ribbons, sticks and surfaces from a PDB entry.",
6
+ "repository": {
7
+ "type": "git",
8
+ "url": "git+https://github.com/motionscript-dev/motionscript.git",
9
+ "directory": "packages/components/molecule"
10
+ },
11
+ "homepage": "https://motionscript.dev",
12
+ "bugs": {
13
+ "url": "https://github.com/motionscript-dev/motionscript/issues"
14
+ },
15
+ "main": "dist/index.js",
16
+ "types": "dist/index.d.ts",
17
+ "sideEffects": false,
18
+ "files": [
19
+ "dist",
20
+ "src",
21
+ "registry.json",
22
+ "CHANGELOG.md",
23
+ "!dist/**/*.tsbuildinfo",
24
+ "!src/**/tests/**",
25
+ "!src/**/*.test.ts",
26
+ "!src/**/*.test.tsx",
27
+ "!src/**/*.fixtures.ts"
28
+ ],
29
+ "publishConfig": {
30
+ "access": "public"
31
+ },
32
+ "exports": {
33
+ ".": {
34
+ "types": "./dist/index.d.ts",
35
+ "default": "./dist/index.js"
36
+ }
37
+ },
38
+ "keywords": [
39
+ "motion-script",
40
+ "motion script",
41
+ "molecule",
42
+ "animation"
43
+ ],
44
+ "license": "Apache-2.0",
45
+ "peerDependencies": {
46
+ "@motionscript/core": "^0.1.0-alpha.0"
47
+ },
48
+ "devDependencies": {
49
+ "@types/node": "^25.6.2",
50
+ "eslint": "^9.39.4",
51
+ "eslint-import-resolver-typescript": "^4.4.4",
52
+ "eslint-plugin-import-x": "^4.16.1",
53
+ "tsc-alias": "^1.8.16",
54
+ "typescript": "^6.0.3",
55
+ "typescript-eslint": "^8.48.0",
56
+ "vitest": "^4.1.5",
57
+ "@motionscript/core": "0.1.0-alpha.0"
58
+ },
59
+ "scripts": {
60
+ "build": "tsc -p tsconfig.build.json && tsc-alias -p tsconfig.build.json --resolve-full-paths && node ../../../scripts/build-browser.mjs",
61
+ "dev": "tsc -p tsconfig.build.json --watch & tsc-alias -p tsconfig.build.json --resolve-full-paths -w",
62
+ "clean": "rimraf --glob dist .turbo *.tsbuildinfo",
63
+ "test": "vitest --passWithNoTests",
64
+ "lint": "eslint .",
65
+ "typecheck": "tsc -p tsconfig.json --noEmit"
66
+ }
6
67
  }
package/registry.json ADDED
@@ -0,0 +1,22 @@
1
+ {
2
+ "kit": {
3
+ "barrel": "src/index.ts",
4
+ "specifier": "@motionscript/molecule"
5
+ },
6
+ "items": [
7
+ {
8
+ "name": "protein",
9
+ "description": "A molecular structure from a PDB entry, as cartoon ribbons or sticks.",
10
+ "root": "src/protein",
11
+ "files": [
12
+ "chemistry.ts",
13
+ "index.ts",
14
+ "parse.ts",
15
+ "protein.ts",
16
+ "ribbon.ts",
17
+ "shared.ts",
18
+ "structure.ts"
19
+ ]
20
+ }
21
+ ]
22
+ }