@motionscript/molecule 0.0.0-stage → 0.1.0-alpha.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (53) hide show
  1. package/CHANGELOG.md +5 -0
  2. package/LICENSE +201 -0
  3. package/dist/browser/index.js +4 -0
  4. package/dist/browser/index.js.map +7 -0
  5. package/dist/browser/manifest.json +11 -0
  6. package/dist/index.d.ts +3 -0
  7. package/dist/index.d.ts.map +1 -0
  8. package/dist/index.js +3 -0
  9. package/dist/index.js.map +1 -0
  10. package/dist/nodes.d.ts +18 -0
  11. package/dist/nodes.d.ts.map +1 -0
  12. package/dist/nodes.js +18 -0
  13. package/dist/nodes.js.map +1 -0
  14. package/dist/protein/chemistry.d.ts +52 -0
  15. package/dist/protein/chemistry.d.ts.map +1 -0
  16. package/dist/protein/chemistry.js +208 -0
  17. package/dist/protein/chemistry.js.map +1 -0
  18. package/dist/protein/index.d.ts +35 -0
  19. package/dist/protein/index.d.ts.map +1 -0
  20. package/dist/protein/index.js +35 -0
  21. package/dist/protein/index.js.map +1 -0
  22. package/dist/protein/parse.d.ts +32 -0
  23. package/dist/protein/parse.d.ts.map +1 -0
  24. package/dist/protein/parse.js +387 -0
  25. package/dist/protein/parse.js.map +1 -0
  26. package/dist/protein/protein.d.ts +265 -0
  27. package/dist/protein/protein.d.ts.map +1 -0
  28. package/dist/protein/protein.js +645 -0
  29. package/dist/protein/protein.js.map +1 -0
  30. package/dist/protein/ribbon.d.ts +83 -0
  31. package/dist/protein/ribbon.d.ts.map +1 -0
  32. package/dist/protein/ribbon.js +468 -0
  33. package/dist/protein/ribbon.js.map +1 -0
  34. package/dist/protein/shared.d.ts +221 -0
  35. package/dist/protein/shared.d.ts.map +1 -0
  36. package/dist/protein/shared.js +478 -0
  37. package/dist/protein/shared.js.map +1 -0
  38. package/dist/protein/structure.d.ts +184 -0
  39. package/dist/protein/structure.d.ts.map +1 -0
  40. package/dist/protein/structure.js +324 -0
  41. package/dist/protein/structure.js.map +1 -0
  42. package/package.json +64 -3
  43. package/registry.json +22 -0
  44. package/src/index.ts +2 -0
  45. package/src/nodes.ts +18 -0
  46. package/src/protein/chemistry.ts +223 -0
  47. package/src/protein/index.ts +34 -0
  48. package/src/protein/parse.ts +427 -0
  49. package/src/protein/protein.ts +897 -0
  50. package/src/protein/ribbon.ts +658 -0
  51. package/src/protein/shared.ts +622 -0
  52. package/src/protein/structure.ts +491 -0
  53. package/README.md +0 -4
@@ -0,0 +1,427 @@
1
+ /**
2
+ * Reads a coordinate file into a {@link ProteinStructure}.
3
+ *
4
+ * Two formats, because the Protein Data Bank serves two and which one an entry
5
+ * has is not the author's choice: the legacy **PDB** format numbers atoms in
6
+ * five columns and names chains in one, so an entry that outgrew either — a
7
+ * ribosome, a capsid — exists only as **mmCIF**. Refusing the second would mean
8
+ * refusing exactly the structures most worth looking at.
9
+ *
10
+ * Both are read the same way: pull out an atom list and whatever secondary
11
+ * structure the file annotates, then hand both to `deriveStructure`, which does
12
+ * the geometry. {@link parseStructure} sniffs which is which, so a caller
13
+ * holding a downloaded file never has to.
14
+ *
15
+ * **The first model only.** An NMR ensemble is twenty superposed conformations
16
+ * of the same molecule; drawing all of them at once produces a blur, and
17
+ * choosing between them is a question this node does not ask. The first is the
18
+ * conventional representative.
19
+ */
20
+
21
+ import { elementOf, residueKind } from "./chemistry"
22
+ import {
23
+ deriveStructure,
24
+ EMPTY_STRUCTURE,
25
+ type ProteinAtom,
26
+ type ProteinStructure,
27
+ type SecondarySpan,
28
+ } from "./structure"
29
+
30
+ /**
31
+ * Reads a coordinate file, detecting its format from the contents.
32
+ *
33
+ * `id` labels the result — an accession, or the asset's name. Anything that
34
+ * can't be read at all comes back as {@link EMPTY_STRUCTURE} rather than
35
+ * throwing: this runs against a file someone just uploaded or an accession
36
+ * someone is halfway through typing, and both of those are ordinary states
37
+ * rather than errors. An empty structure draws nothing, which is the honest
38
+ * picture of a file with no atoms in it.
39
+ */
40
+ export function parseStructure(text: string, id = ""): ProteinStructure {
41
+ if (typeof text !== "string" || text.trim() === "") {
42
+ return { ...EMPTY_STRUCTURE, id }
43
+ }
44
+ return isMmcif(text) ? parseMmcif(text, id) : parsePdb(text, id)
45
+ }
46
+
47
+ /**
48
+ * Whether a file is mmCIF.
49
+ *
50
+ * The `data_` block header is mmCIF's first line and appears in no PDB file, so
51
+ * it decides on its own — but only over the first stretch, since `data_` is also
52
+ * an ordinary substring that could turn up in a PDB `REMARK`. The category
53
+ * prefix is the belt-and-braces check for a fragment handed over without its
54
+ * header.
55
+ */
56
+ function isMmcif(text: string): boolean {
57
+ const head = text.slice(0, 4096)
58
+ return /^\s*data_/.test(head) || head.includes("_atom_site.")
59
+ }
60
+
61
+ // --- PDB -------------------------------------------------------------------
62
+
63
+ /**
64
+ * Reads the legacy PDB format, which is **fixed-column**: every field is at a
65
+ * known offset and the whitespace between them is padding rather than a
66
+ * separator. Splitting on spaces is the classic way to get this wrong — a
67
+ * residue number that runs into its insertion code, or a `-` sign that eats the
68
+ * gap between two coordinates, and the record silently comes apart.
69
+ *
70
+ * Columns, 1-based as the specification numbers them:
71
+ *
72
+ * 7–11 serial 13–16 name 18–20 resName 22 chainID
73
+ * 23–26 resSeq 31–38 x 39–46 y 47–54 z
74
+ * 77–78 element
75
+ */
76
+ function parsePdb(text: string, id: string): ProteinStructure {
77
+ const atoms: ProteinAtom[] = []
78
+ const spans: SecondarySpan[] = []
79
+ let title = ""
80
+
81
+ for (const line of text.split("\n")) {
82
+ const record = line.slice(0, 6).trim()
83
+
84
+ // Everything after the first `ENDMDL` is another conformation of what we
85
+ // already have. Stopping at it rather than filtering by model number also
86
+ // ends the read early on a large ensemble.
87
+ if (record === "ENDMDL") break
88
+
89
+ if (record === "ATOM" || record === "HETATM") {
90
+ const atom = parsePdbAtom(line, record === "HETATM")
91
+ if (atom) atoms.push(atom)
92
+ continue
93
+ }
94
+
95
+ if (record === "TITLE") {
96
+ // A long title is continued across several records, with the continuation
97
+ // number in columns 9–10 and the text always from column 11.
98
+ title = `${title} ${line.slice(10).trim()}`.trim()
99
+ continue
100
+ }
101
+
102
+ // The two annotation records. Their first-residue fields sit at *different*
103
+ // offsets from each other — a genuine wart of the format rather than a
104
+ // mistake here — while their last-residue fields agree.
105
+ if (record === "HELIX") {
106
+ const span = pdbSpan(line, 19, 21, "helix")
107
+ if (span) spans.push(span)
108
+ continue
109
+ }
110
+ if (record === "SHEET") {
111
+ const span = pdbSpan(line, 21, 22, "sheet")
112
+ if (span) spans.push(span)
113
+ }
114
+ }
115
+
116
+ return deriveStructure(id, title, atoms, spans)
117
+ }
118
+
119
+ /** One `ATOM`/`HETATM` record, or `null` when its coordinates don't read. */
120
+ function parsePdbAtom(line: string, hetero: boolean): ProteinAtom | null {
121
+ // Truncated before the coordinates. Worth checking outright: a short slice of
122
+ // a fixed-column record reads as the empty string, and `Number("")` is 0
123
+ // rather than `NaN` — so without this a mangled line becomes an atom sitting
124
+ // at the origin, which is far harder to notice than a missing one.
125
+ if (line.length < 54) return null
126
+
127
+ // An alternate location: the same atom modelled twice because the side chain
128
+ // is disordered. Keeping both would double the sticks through that residue,
129
+ // so take the first (blank or `A`), which is the convention.
130
+ const altLoc = line.slice(16, 17).trim()
131
+ if (altLoc !== "" && altLoc !== "A") return null
132
+
133
+ const x = Number(line.slice(30, 38))
134
+ const y = Number(line.slice(38, 46))
135
+ const z = Number(line.slice(46, 54))
136
+ if (!Number.isFinite(x) || !Number.isFinite(y) || !Number.isFinite(z)) {
137
+ return null
138
+ }
139
+
140
+ const name = line.slice(12, 16).trim()
141
+ const residue = line.slice(17, 20).trim()
142
+ const kind = residueKind(residue)
143
+
144
+ return {
145
+ serial: Number(line.slice(6, 11)) || 0,
146
+ name,
147
+ element: elementOf(line.slice(76, 78), name, kind),
148
+ residue,
149
+ residueSeq: Number(line.slice(22, 26)) || 0,
150
+ // A file with a single unnamed chain leaves the column blank; calling that
151
+ // `A` keeps "group by chain" from producing one group named nothing.
152
+ chain: line.slice(21, 22).trim() || "A",
153
+ x,
154
+ y,
155
+ z,
156
+ hetero,
157
+ kind,
158
+ }
159
+ }
160
+
161
+ /** Where the last residue's number sits on both `HELIX` and `SHEET` — columns 34–37. */
162
+ const PDB_SPAN_END_AT = 33
163
+
164
+ /**
165
+ * One `HELIX`/`SHEET` record as a residue span.
166
+ *
167
+ * The first-residue offsets differ between the two records, so they are passed
168
+ * in rather than hard-coded: a helix names its chain in column 20 and its first
169
+ * residue in 22–25, a sheet names its chain in 22 and its first residue in
170
+ * 23–26. Their *last* residue is in the same place in both, which is the one
171
+ * thing they agree on.
172
+ */
173
+ function pdbSpan(
174
+ line: string,
175
+ chainAt: number,
176
+ startAt: number,
177
+ kind: SecondarySpan["kind"]
178
+ ): SecondarySpan | null {
179
+ const chain = line.slice(chainAt, chainAt + 1).trim() || "A"
180
+ // Read as text first: a blank fixed-column field slices to `""`, and
181
+ // `Number("")` is 0, so a `Number.isFinite` check alone would accept a record
182
+ // that states nothing as a span over residue 0.
183
+ const startText = line.slice(startAt, startAt + 4).trim()
184
+ const endText = line.slice(PDB_SPAN_END_AT, PDB_SPAN_END_AT + 4).trim()
185
+ if (startText === "" || endText === "") return null
186
+
187
+ const start = Number(startText)
188
+ const end = Number(endText)
189
+ if (!Number.isFinite(start) || !Number.isFinite(end)) return null
190
+ return { chain, start, end, kind }
191
+ }
192
+
193
+ // --- mmCIF -----------------------------------------------------------------
194
+
195
+ /**
196
+ * Reads mmCIF, which is a tagged format rather than a positional one: values are
197
+ * whitespace-separated and their meaning comes from the column headers declared
198
+ * above them, so the reader has to learn the layout before it can read a row.
199
+ *
200
+ * Only three categories are consulted — the atoms, and the two that annotate
201
+ * secondary structure. mmCIF carries dozens more, and every one this doesn't
202
+ * read is one this can't be broken by.
203
+ */
204
+ function parseMmcif(text: string, id: string): ProteinStructure {
205
+ const lines = text.split("\n")
206
+ const atoms = mmcifAtoms(lines)
207
+ const spans = [
208
+ ...mmcifSpans(lines, "_struct_conf", "helix"),
209
+ ...mmcifSpans(lines, "_struct_sheet_range", "sheet"),
210
+ ]
211
+ return deriveStructure(id, mmcifTitle(lines), atoms, spans)
212
+ }
213
+
214
+ /** The atoms of the first model in an `_atom_site` loop. */
215
+ function mmcifAtoms(lines: string[]): ProteinAtom[] {
216
+ const atoms: ProteinAtom[] = []
217
+ let model: string | null = null
218
+
219
+ for (const row of mmcifRows(lines, "_atom_site")) {
220
+ // `auth_*` is the numbering the literature uses and the one a PDB file would
221
+ // have carried; `label_*` is the internal, re-derived scheme. Preferring
222
+ // auth means a residue number quoted in a paper matches what is drawn, and
223
+ // falling back keeps a file that omits it readable.
224
+ const chain = row("auth_asym_id") || row("label_asym_id") || "A"
225
+ const residue = row("auth_comp_id") || row("label_comp_id")
226
+ const name = row("auth_atom_id") || row("label_atom_id")
227
+
228
+ const thisModel = row("pdbx_PDB_model_num")
229
+ if (model === null) model = thisModel
230
+ else if (thisModel !== model) break
231
+
232
+ const altLoc = row("label_alt_id")
233
+ if (altLoc !== "" && altLoc !== "." && altLoc !== "?" && altLoc !== "A") {
234
+ continue
235
+ }
236
+
237
+ const x = Number(row("Cartn_x"))
238
+ const y = Number(row("Cartn_y"))
239
+ const z = Number(row("Cartn_z"))
240
+ if (!Number.isFinite(x) || !Number.isFinite(y) || !Number.isFinite(z)) {
241
+ continue
242
+ }
243
+
244
+ const kind = residueKind(residue)
245
+ atoms.push({
246
+ serial: Number(row("id")) || 0,
247
+ name,
248
+ element: elementOf(row("type_symbol"), name, kind),
249
+ residue,
250
+ residueSeq: Number(row("auth_seq_id") || row("label_seq_id")) || 0,
251
+ chain,
252
+ x,
253
+ y,
254
+ z,
255
+ hetero: row("group_PDB") === "HETATM",
256
+ kind,
257
+ })
258
+ }
259
+
260
+ return atoms
261
+ }
262
+
263
+ /** The residue spans of one annotation category. */
264
+ function mmcifSpans(
265
+ lines: string[],
266
+ category: string,
267
+ kind: SecondarySpan["kind"]
268
+ ): SecondarySpan[] {
269
+ const spans: SecondarySpan[] = []
270
+
271
+ for (const row of mmcifRows(lines, category)) {
272
+ // `_struct_conf` covers turns and bends as well as helices, all under one
273
+ // category and told apart by this tag. A sheet range has no such tag, so an
274
+ // absent one passes.
275
+ const type = row("conf_type_id")
276
+ if (type !== "" && !type.startsWith("HELX")) continue
277
+
278
+ const chain = row("beg_auth_asym_id") || row("beg_label_asym_id") || "A"
279
+ const start = Number(row("beg_auth_seq_id") || row("beg_label_seq_id"))
280
+ const end = Number(row("end_auth_seq_id") || row("end_label_seq_id"))
281
+ if (!Number.isFinite(start) || !Number.isFinite(end)) continue
282
+ spans.push({ chain, start, end, kind })
283
+ }
284
+
285
+ return spans
286
+ }
287
+
288
+ /** The entry's title, from the `_struct.title` item. */
289
+ function mmcifTitle(lines: string[]): string {
290
+ for (let i = 0; i < lines.length; i++) {
291
+ const line = lines[i]!
292
+ if (!line.startsWith("_struct.title")) continue
293
+ const inline = line.slice("_struct.title".length).trim()
294
+ // A value too long for the line is carried on the next one, or in a
295
+ // semicolon-delimited block below it; the inline form covers the rest.
296
+ if (inline !== "") return unquote(inline)
297
+ const next = lines[i + 1]?.trim() ?? ""
298
+ return unquote(next.startsWith(";") ? next.slice(1) : next)
299
+ }
300
+ return ""
301
+ }
302
+
303
+ /**
304
+ * Walks the rows of one mmCIF category, yielding a field reader for each.
305
+ *
306
+ * The reader is a closure over the current row rather than an object, so
307
+ * nothing allocates a record per atom — this runs over hundreds of thousands of
308
+ * rows on a large entry, and the caller only ever wants a handful of the fields.
309
+ *
310
+ * Handles both forms a category takes: a `loop_` with a header block and many
311
+ * rows, and the flat `_category.field value` form mmCIF uses when there is
312
+ * exactly one row.
313
+ */
314
+ function* mmcifRows(
315
+ lines: string[],
316
+ category: string
317
+ ): Generator<(field: string) => string> {
318
+ const prefix = `${category}.`
319
+
320
+ for (let i = 0; i < lines.length; i++) {
321
+ const line = lines[i]!.trim()
322
+
323
+ // The flat, single-row form: consecutive `_category.field value` lines.
324
+ if (line.startsWith(prefix)) {
325
+ const single = new Map<string, string>()
326
+ while (i < lines.length) {
327
+ const item = lines[i]!.trim()
328
+ if (!item.startsWith(prefix)) break
329
+ const gap = item.search(/\s/)
330
+ if (gap === -1) {
331
+ // A field whose value is on the following line.
332
+ single.set(item.slice(prefix.length), unquote(lines[++i]?.trim() ?? ""))
333
+ } else {
334
+ single.set(
335
+ item.slice(prefix.length, gap),
336
+ unquote(item.slice(gap).trim())
337
+ )
338
+ }
339
+ i++
340
+ }
341
+ yield (field) => single.get(field) ?? ""
342
+ return
343
+ }
344
+
345
+ if (line !== "loop_") continue
346
+
347
+ // The header block: every `_category.field` line, in the order the values
348
+ // will arrive in.
349
+ const columns = new Map<string, number>()
350
+ let cursor = i + 1
351
+ while (cursor < lines.length) {
352
+ const header = lines[cursor]!.trim()
353
+ if (!header.startsWith("_")) break
354
+ if (header.startsWith(prefix)) {
355
+ columns.set(header.slice(prefix.length), columns.size)
356
+ }
357
+ cursor++
358
+ }
359
+ // A `loop_` for some other category. Its rows carry no leading underscore
360
+ // and aren't `loop_`, so the outer scan walks past them harmlessly.
361
+ if (columns.size === 0) continue
362
+
363
+ for (; cursor < lines.length; cursor++) {
364
+ const row = lines[cursor]!
365
+ const trimmed = row.trim()
366
+ // A loop ends at the next block, the next loop, or a blank line.
367
+ if (trimmed === "" || trimmed === "#" || trimmed.startsWith("_")) break
368
+ if (trimmed === "loop_" || trimmed.startsWith("data_")) break
369
+
370
+ const values = splitCifRow(trimmed)
371
+ yield (field) => {
372
+ const index = columns.get(field)
373
+ if (index === undefined) return ""
374
+ const value = values[index] ?? ""
375
+ return value === "." || value === "?" ? "" : value
376
+ }
377
+ }
378
+ return
379
+ }
380
+ }
381
+
382
+ /**
383
+ * Splits one mmCIF row into values, respecting quotes.
384
+ *
385
+ * Quoting is not decoration here: a chain identifier can be `'A'`, and a residue
386
+ * name can contain a space that a plain `split` would turn into two columns and
387
+ * shift every field after it by one.
388
+ */
389
+ function splitCifRow(row: string): string[] {
390
+ const values: string[] = []
391
+ let i = 0
392
+
393
+ while (i < row.length) {
394
+ const char = row[i]!
395
+ if (char === " " || char === "\t") {
396
+ i++
397
+ continue
398
+ }
399
+ if (char === "'" || char === '"') {
400
+ const end = row.indexOf(char, i + 1)
401
+ if (end === -1) {
402
+ values.push(row.slice(i + 1))
403
+ break
404
+ }
405
+ values.push(row.slice(i + 1, end))
406
+ i = end + 1
407
+ continue
408
+ }
409
+ let end = i
410
+ while (end < row.length && row[end] !== " " && row[end] !== "\t") end++
411
+ values.push(row.slice(i, end))
412
+ i = end
413
+ }
414
+
415
+ return values
416
+ }
417
+
418
+ /** Strips the quotes mmCIF wraps a value in, and the null placeholders. */
419
+ function unquote(value: string): string {
420
+ const text = value.trim()
421
+ if (text === "." || text === "?") return ""
422
+ const first = text[0]
423
+ if ((first === "'" || first === '"') && text.endsWith(first) && text.length > 1) {
424
+ return text.slice(1, -1)
425
+ }
426
+ return text
427
+ }